Files
woogiandClaude Opus 5 23e19f53b2
Docker Release / build-and-push (push) Successful in 1m45s
Docker Release / release (push) Skipped
feat: review chat — ask the run why it concluded a finding
Read-only Q&A on the review screen, per finding and per run, answered from
the job's own artifacts (evidence, cluster, extraction, verification, Brain
merge, sheet index, cover reconciliation, job.log). It never mutates findings,
decisions, or the report.

Turns are logged job-locally (review/chat_log.jsonl, transcript at
/jobs/{id}/review-chat/log) and to a cross-job feedback store
(REVIEW_FEEDBACK_DIR), which now also receives review decisions with their
category/severity corrections.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0115gGtrSxXE9DKvS9XPFSoT
2026-09-14 10:35:18 -05:00

173 lines
7.0 KiB
Python

"""Review chat: answer normalization, logging, and the feedback roll-up."""
import json
import os
import pytest
from backend import config
from backend.review import chat
def _queue() -> list:
return [{
"review_item_id": "finding:AGENT-0007", "kind": "finding", "blocking": True,
"reasons": ["severity_high"],
"payload": {"issue_id": "AGENT-0007", "source_stage": "conflict",
"category": "elevation_disagreement", "severity": "high",
"confidence": "medium", "location": "Roof / AC-1",
"sheets": ["M2.1"], "description": "AC-1 at grade vs roof.",
"evidence": [], "scope_id": "conflict:roof-ac1"},
}]
def _model_reply(**overrides) -> dict:
reply = {
"answer": "The extractor read 'AC-1 MOUNTED ON GRADE' off M2.1.",
"findings": ["The grade reading came from M2.1's text layer."],
"evidence_cited": [{"artifact": "source_sheets[M2.1]", "sheet": "M2.1",
"quote": "AC-1 MOUNTED ON GRADE",
"why_it_matters": "It is the sole basis for 'grade'."}],
"answerable": "yes",
"missing_information": None,
"assessment_of_finding": "looks_supported",
"suggested_category_correction": None,
"confidence": "high",
}
reply.update(overrides)
return reply
@pytest.fixture
def fake_llm(monkeypatch):
"""Stub the model; the chat must never need a network to be tested."""
calls = []
def _call(**kwargs):
calls.append(kwargs)
return calls_reply[0]
calls_reply = [_model_reply()]
monkeypatch.setattr("backend.review.chat.call_json", lambda **kw: _call(**kw))
return calls, calls_reply
def test_ask_logs_issue_question_and_findings(tmp_path, fake_llm):
turn = chat.ask("job1", str(tmp_path), "Why is AC-1 at grade?",
review_item_id="finding:AGENT-0007", queue=_queue())
assert turn["question"] == "Why is AC-1 at grade?"
assert turn["findings"] == ["The grade reading came from M2.1's text layer."]
# The log's "issue in question" is a snapshot, not a bare id.
assert turn["issue"]["issue_id"] == "AGENT-0007"
assert turn["issue"]["severity"] == "high"
path = os.path.join(str(tmp_path), "review", "chat_log.jsonl")
with open(path, encoding="utf-8") as f:
logged = [json.loads(line) for line in f if line.strip()]
assert len(logged) == 1
assert logged[0]["turn_id"] == turn["turn_id"]
def test_ask_appends_to_the_cross_job_feedback_store(tmp_path, fake_llm):
chat.ask("job1", str(tmp_path), "Is this really a floor drain?",
review_item_id="finding:AGENT-0007", queue=_queue())
path = os.path.join(config.REVIEW_FEEDBACK_DIR, "chat_turns.jsonl")
with open(path, encoding="utf-8") as f:
records = [json.loads(line) for line in f if line.strip()]
assert records[0]["kind"] == "review_chat_turn"
assert records[0]["issue_id"] == "AGENT-0007"
assert records[0]["job_id"] == "job1"
def test_correction_signal_is_captured_as_structured_data(tmp_path, fake_llm):
"""A misidentification correction survives as a field, not free text."""
_, reply = fake_llm
reply[0] = _model_reply(suggested_category_correction="power floor box",
assessment_of_finding="looks_unsupported")
turn = chat.ask("job1", str(tmp_path), "That is not a floor drain.",
review_item_id="finding:AGENT-0007", queue=_queue())
assert turn["suggested_category_correction"] == "power floor box"
path = os.path.join(config.REVIEW_FEEDBACK_DIR, "chat_turns.jsonl")
with open(path, encoding="utf-8") as f:
record = json.loads(f.readline())
assert record["suggested_category_correction"] == "power floor box"
assert record["assessment_of_finding"] == "looks_unsupported"
def test_run_scope_question_needs_no_item(tmp_path, fake_llm):
turn = chat.ask("job1", str(tmp_path), "Why didn't it pick up the Civil set?")
assert turn["review_item_id"] is None
assert turn["scope"] == "run"
assert turn["issue"] is None
def test_history_is_replayed_for_the_same_thread(tmp_path, fake_llm):
calls, _ = fake_llm
chat.ask("job1", str(tmp_path), "First question?",
review_item_id="finding:AGENT-0007", queue=_queue())
chat.ask("job1", str(tmp_path), "Follow-up?",
review_item_id="finding:AGENT-0007", queue=_queue())
assert "First question?" in calls[1]["user_text"]
# A run-scope turn must not inherit an item thread's history.
chat.ask("job1", str(tmp_path), "Unrelated run question?")
assert "First question?" not in calls[2]["user_text"]
def test_blank_and_oversized_questions_are_rejected(tmp_path, fake_llm):
with pytest.raises(chat.ChatError):
chat.ask("job1", str(tmp_path), " ")
with pytest.raises(chat.ChatError):
chat.ask("job1", str(tmp_path),
"x" * (config.REVIEW_CHAT_MAX_QUESTION_CHARS + 1))
def test_unknown_review_item_is_rejected(tmp_path, fake_llm):
with pytest.raises(chat.ChatError):
chat.ask("job1", str(tmp_path), "why?", review_item_id="finding:NOPE",
queue=_queue())
def test_unusable_model_reply_raises_and_logs_nothing(tmp_path, monkeypatch):
monkeypatch.setattr("backend.review.chat.call_json", lambda **kw: None)
with pytest.raises(RuntimeError):
chat.ask("job1", str(tmp_path), "why?")
assert not os.path.exists(os.path.join(str(tmp_path), "review", "chat_log.jsonl"))
def test_bad_enum_values_fall_back_instead_of_failing(tmp_path, fake_llm):
_, reply = fake_llm
reply[0] = _model_reply(answerable="probably", confidence="",
assessment_of_finding="made_up")
turn = chat.ask("job1", str(tmp_path), "why?")
assert turn["answerable"] == "partial"
assert turn["confidence"] == "low"
assert turn["assessment_of_finding"] == "cannot_tell"
def test_disabled_chat_refuses(tmp_path, monkeypatch, fake_llm):
monkeypatch.setattr("backend.config.ENABLE_REVIEW_CHAT", False)
with pytest.raises(chat.ChatError):
chat.ask("job1", str(tmp_path), "why?")
def test_read_log_skips_corrupt_lines(tmp_path, fake_llm):
chat.ask("job1", str(tmp_path), "why?")
path = os.path.join(str(tmp_path), "review", "chat_log.jsonl")
with open(path, "a", encoding="utf-8") as f:
f.write("{not json\n")
assert len(chat.read_log(str(tmp_path))) == 1
def test_markdown_transcript_groups_by_issue(tmp_path, fake_llm):
chat.ask("job1", str(tmp_path), "Why is AC-1 at grade?",
review_item_id="finding:AGENT-0007", queue=_queue())
chat.ask("job1", str(tmp_path), "Why no Civil?")
markdown = chat.render_log_markdown(chat.read_log(str(tmp_path)))
assert "## AGENT-0007" in markdown
assert "## Run-scope questions" in markdown
assert "Why is AC-1 at grade?" in markdown
assert "**Findings**" in markdown
def test_markdown_transcript_handles_empty_log():
assert "No questions" in chat.render_log_markdown([])