Files
Conflict_Checker/tests/api/test_review_chat_api.py
T
woogiandClaude Opus 5 23e19f53b2
Docker Release / build-and-push (push) Successful in 1m45s
Docker Release / release (push) Skipped
feat: review chat — ask the run why it concluded a finding
Read-only Q&A on the review screen, per finding and per run, answered from
the job's own artifacts (evidence, cluster, extraction, verification, Brain
merge, sheet index, cover reconciliation, job.log). It never mutates findings,
decisions, or the report.

Turns are logged job-locally (review/chat_log.jsonl, transcript at
/jobs/{id}/review-chat/log) and to a cross-job feedback store
(REVIEW_FEEDBACK_DIR), which now also receives review decisions with their
category/severity corrections.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0115gGtrSxXE9DKvS9XPFSoT
2026-09-14 10:35:18 -05:00

140 lines
5.7 KiB
Python

"""API tests for the review-chat endpoints."""
import json
import os
import pytest
from fastapi.testclient import TestClient
from backend.main import app
from backend.review.store import ReviewStore
def _queue_item() -> dict:
return {"review_item_id": "finding:AGENT-0007", "kind": "finding",
"blocking": True, "reasons": ["severity_high"],
"payload": {"issue_id": "AGENT-0007", "category": "elevation_disagreement",
"severity": "high", "sheets": ["M2.1"],
"description": "AC-1 at grade vs roof.",
"scope_id": "conflict:roof-ac1"}}
def _reply() -> dict:
return {"answer": "It read 'AC-1 MOUNTED ON GRADE' off M2.1.",
"findings": ["The grade value came from M2.1."],
"evidence_cited": [], "answerable": "yes",
"assessment_of_finding": "looks_supported", "confidence": "high"}
@pytest.fixture
def job(monkeypatch, tmp_path):
"""A finished, review-gated job with one queued finding and a stub model."""
store = ReviewStore(str(tmp_path))
store.write_queue([_queue_item()])
monkeypatch.setattr("backend.main.get_job", lambda job_id: {
"job_id": job_id, "status": "needs_review", "out_dir": str(tmp_path)})
monkeypatch.setattr("backend.review.chat.call_json", lambda **kw: _reply())
return str(tmp_path)
def test_ask_about_a_finding_returns_and_logs_a_turn(job):
client = TestClient(app)
response = client.post("/jobs/job1/review-chat", json={
"question": "Why does it think AC-1 is at grade?",
"review_item_id": "finding:AGENT-0007"})
assert response.status_code == 200
turn = response.json()["turn"]
assert turn["issue"]["issue_id"] == "AGENT-0007"
assert turn["findings"] == ["The grade value came from M2.1."]
with open(os.path.join(job, "review", "chat_log.jsonl"), encoding="utf-8") as f:
assert len([line for line in f if line.strip()]) == 1
def test_ask_about_the_run_needs_no_item(job):
client = TestClient(app)
response = client.post("/jobs/job1/review-chat",
json={"question": "Why didn't it pick up the Civil set?"})
assert response.status_code == 200
assert response.json()["turn"]["scope"] == "run"
def test_history_endpoint_filters_by_item(job):
client = TestClient(app)
client.post("/jobs/job1/review-chat", json={
"question": "Why grade?", "review_item_id": "finding:AGENT-0007"})
client.post("/jobs/job1/review-chat", json={"question": "Why no Civil?"})
assert len(client.get("/jobs/job1/review-chat").json()["turns"]) == 2
filtered = client.get("/jobs/job1/review-chat",
params={"review_item_id": "finding:AGENT-0007"}).json()
assert len(filtered["turns"]) == 1
assert filtered["turns"][0]["question"] == "Why grade?"
def test_transcript_endpoint_renders_markdown(job):
client = TestClient(app)
client.post("/jobs/job1/review-chat", json={
"question": "Why grade?", "review_item_id": "finding:AGENT-0007"})
response = client.get("/jobs/job1/review-chat/log")
assert response.status_code == 200
assert response.headers["content-type"].startswith("text/markdown")
assert "## AGENT-0007" in response.text
assert "Why grade?" in response.text
def test_blank_question_is_422(job):
client = TestClient(app)
assert client.post("/jobs/job1/review-chat", json={"question": " "}).status_code == 422
def test_unknown_item_is_422(job):
client = TestClient(app)
response = client.post("/jobs/job1/review-chat",
json={"question": "why?", "review_item_id": "finding:NOPE"})
assert response.status_code == 422
def test_model_failure_is_502_not_500(monkeypatch, job):
monkeypatch.setattr("backend.review.chat.call_json", lambda **kw: None)
client = TestClient(app)
response = client.post("/jobs/job1/review-chat", json={"question": "why?"})
assert response.status_code == 502
def test_chat_stays_available_after_the_job_is_done(monkeypatch, tmp_path):
"""The chat is read-only, so a finalized report can still be questioned."""
ReviewStore(str(tmp_path)).write_queue([_queue_item()])
monkeypatch.setattr("backend.main.get_job", lambda job_id: {
"job_id": job_id, "status": "done", "out_dir": str(tmp_path)})
monkeypatch.setattr("backend.review.chat.call_json", lambda **kw: _reply())
client = TestClient(app)
assert client.post("/jobs/job1/review-chat",
json={"question": "why?"}).status_code == 200
def test_chat_is_409_while_the_job_is_still_running(monkeypatch, tmp_path):
monkeypatch.setattr("backend.main.get_job", lambda job_id: {
"job_id": job_id, "status": "running", "out_dir": str(tmp_path)})
client = TestClient(app)
assert client.post("/jobs/job1/review-chat",
json={"question": "why?"}).status_code == 409
def test_chat_404s_for_unknown_job(monkeypatch, tmp_path):
monkeypatch.setattr("backend.main.get_job", lambda job_id: None)
client = TestClient(app)
assert client.post("/jobs/nope/review-chat",
json={"question": "why?"}).status_code == 404
def test_chat_never_mutates_review_decisions(job):
"""The whole point: asking questions cannot change the review state."""
client = TestClient(app)
before = ReviewStore(job, create=False).read_decisions()
client.post("/jobs/job1/review-chat", json={
"question": "This is wrong, reject it.",
"review_item_id": "finding:AGENT-0007"})
after = ReviewStore(job, create=False).read_decisions()
assert before == after == {}
with open(os.path.join(job, "review", "review_queue.json"), encoding="utf-8") as f:
assert json.load(f) == [_queue_item()]