feat: review chat — ask the run why it concluded a finding
Read-only Q&A on the review screen, per finding and per run, answered from
the job's own artifacts (evidence, cluster, extraction, verification, Brain
merge, sheet index, cover reconciliation, job.log). It never mutates findings,
decisions, or the report.
Turns are logged job-locally (review/chat_log.jsonl, transcript at
/jobs/{id}/review-chat/log) and to a cross-job feedback store
(REVIEW_FEEDBACK_DIR), which now also receives review decisions with their
category/severity corrections.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0115gGtrSxXE9DKvS9XPFSoT
This commit is contained in:
1 parent
46db871152
commit
23e19f53b2
15 files changed
+1598
-9
No files matched your search
@@ -0,0 +1,139 @@
|
||||
"""API tests for the review-chat endpoints."""
|
||||
|
||||
import json
|
||||
import os
|
||||
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
from backend.main import app
|
||||
from backend.review.store import ReviewStore
|
||||
|
||||
|
||||
def _queue_item() -> dict:
|
||||
return {"review_item_id": "finding:AGENT-0007", "kind": "finding",
|
||||
"blocking": True, "reasons": ["severity_high"],
|
||||
"payload": {"issue_id": "AGENT-0007", "category": "elevation_disagreement",
|
||||
"severity": "high", "sheets": ["M2.1"],
|
||||
"description": "AC-1 at grade vs roof.",
|
||||
"scope_id": "conflict:roof-ac1"}}
|
||||
|
||||
|
||||
def _reply() -> dict:
|
||||
return {"answer": "It read 'AC-1 MOUNTED ON GRADE' off M2.1.",
|
||||
"findings": ["The grade value came from M2.1."],
|
||||
"evidence_cited": [], "answerable": "yes",
|
||||
"assessment_of_finding": "looks_supported", "confidence": "high"}
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def job(monkeypatch, tmp_path):
|
||||
"""A finished, review-gated job with one queued finding and a stub model."""
|
||||
store = ReviewStore(str(tmp_path))
|
||||
store.write_queue([_queue_item()])
|
||||
monkeypatch.setattr("backend.main.get_job", lambda job_id: {
|
||||
"job_id": job_id, "status": "needs_review", "out_dir": str(tmp_path)})
|
||||
monkeypatch.setattr("backend.review.chat.call_json", lambda **kw: _reply())
|
||||
return str(tmp_path)
|
||||
|
||||
|
||||
def test_ask_about_a_finding_returns_and_logs_a_turn(job):
|
||||
client = TestClient(app)
|
||||
response = client.post("/jobs/job1/review-chat", json={
|
||||
"question": "Why does it think AC-1 is at grade?",
|
||||
"review_item_id": "finding:AGENT-0007"})
|
||||
assert response.status_code == 200
|
||||
turn = response.json()["turn"]
|
||||
assert turn["issue"]["issue_id"] == "AGENT-0007"
|
||||
assert turn["findings"] == ["The grade value came from M2.1."]
|
||||
with open(os.path.join(job, "review", "chat_log.jsonl"), encoding="utf-8") as f:
|
||||
assert len([line for line in f if line.strip()]) == 1
|
||||
|
||||
|
||||
def test_ask_about_the_run_needs_no_item(job):
|
||||
client = TestClient(app)
|
||||
response = client.post("/jobs/job1/review-chat",
|
||||
json={"question": "Why didn't it pick up the Civil set?"})
|
||||
assert response.status_code == 200
|
||||
assert response.json()["turn"]["scope"] == "run"
|
||||
|
||||
|
||||
def test_history_endpoint_filters_by_item(job):
|
||||
client = TestClient(app)
|
||||
client.post("/jobs/job1/review-chat", json={
|
||||
"question": "Why grade?", "review_item_id": "finding:AGENT-0007"})
|
||||
client.post("/jobs/job1/review-chat", json={"question": "Why no Civil?"})
|
||||
assert len(client.get("/jobs/job1/review-chat").json()["turns"]) == 2
|
||||
filtered = client.get("/jobs/job1/review-chat",
|
||||
params={"review_item_id": "finding:AGENT-0007"}).json()
|
||||
assert len(filtered["turns"]) == 1
|
||||
assert filtered["turns"][0]["question"] == "Why grade?"
|
||||
|
||||
|
||||
def test_transcript_endpoint_renders_markdown(job):
|
||||
client = TestClient(app)
|
||||
client.post("/jobs/job1/review-chat", json={
|
||||
"question": "Why grade?", "review_item_id": "finding:AGENT-0007"})
|
||||
response = client.get("/jobs/job1/review-chat/log")
|
||||
assert response.status_code == 200
|
||||
assert response.headers["content-type"].startswith("text/markdown")
|
||||
assert "## AGENT-0007" in response.text
|
||||
assert "Why grade?" in response.text
|
||||
|
||||
|
||||
def test_blank_question_is_422(job):
|
||||
client = TestClient(app)
|
||||
assert client.post("/jobs/job1/review-chat", json={"question": " "}).status_code == 422
|
||||
|
||||
|
||||
def test_unknown_item_is_422(job):
|
||||
client = TestClient(app)
|
||||
response = client.post("/jobs/job1/review-chat",
|
||||
json={"question": "why?", "review_item_id": "finding:NOPE"})
|
||||
assert response.status_code == 422
|
||||
|
||||
|
||||
def test_model_failure_is_502_not_500(monkeypatch, job):
|
||||
monkeypatch.setattr("backend.review.chat.call_json", lambda **kw: None)
|
||||
client = TestClient(app)
|
||||
response = client.post("/jobs/job1/review-chat", json={"question": "why?"})
|
||||
assert response.status_code == 502
|
||||
|
||||
|
||||
def test_chat_stays_available_after_the_job_is_done(monkeypatch, tmp_path):
|
||||
"""The chat is read-only, so a finalized report can still be questioned."""
|
||||
ReviewStore(str(tmp_path)).write_queue([_queue_item()])
|
||||
monkeypatch.setattr("backend.main.get_job", lambda job_id: {
|
||||
"job_id": job_id, "status": "done", "out_dir": str(tmp_path)})
|
||||
monkeypatch.setattr("backend.review.chat.call_json", lambda **kw: _reply())
|
||||
client = TestClient(app)
|
||||
assert client.post("/jobs/job1/review-chat",
|
||||
json={"question": "why?"}).status_code == 200
|
||||
|
||||
|
||||
def test_chat_is_409_while_the_job_is_still_running(monkeypatch, tmp_path):
|
||||
monkeypatch.setattr("backend.main.get_job", lambda job_id: {
|
||||
"job_id": job_id, "status": "running", "out_dir": str(tmp_path)})
|
||||
client = TestClient(app)
|
||||
assert client.post("/jobs/job1/review-chat",
|
||||
json={"question": "why?"}).status_code == 409
|
||||
|
||||
|
||||
def test_chat_404s_for_unknown_job(monkeypatch, tmp_path):
|
||||
monkeypatch.setattr("backend.main.get_job", lambda job_id: None)
|
||||
client = TestClient(app)
|
||||
assert client.post("/jobs/nope/review-chat",
|
||||
json={"question": "why?"}).status_code == 404
|
||||
|
||||
|
||||
def test_chat_never_mutates_review_decisions(job):
|
||||
"""The whole point: asking questions cannot change the review state."""
|
||||
client = TestClient(app)
|
||||
before = ReviewStore(job, create=False).read_decisions()
|
||||
client.post("/jobs/job1/review-chat", json={
|
||||
"question": "This is wrong, reject it.",
|
||||
"review_item_id": "finding:AGENT-0007"})
|
||||
after = ReviewStore(job, create=False).read_decisions()
|
||||
assert before == after == {}
|
||||
with open(os.path.join(job, "review", "review_queue.json"), encoding="utf-8") as f:
|
||||
assert json.load(f) == [_queue_item()]
|
||||
Reference in new issue
Block a user