feat: review chat — ask the run why it concluded a finding
Read-only Q&A on the review screen, per finding and per run, answered from
the job's own artifacts (evidence, cluster, extraction, verification, Brain
merge, sheet index, cover reconciliation, job.log). It never mutates findings,
decisions, or the report.
Turns are logged job-locally (review/chat_log.jsonl, transcript at
/jobs/{id}/review-chat/log) and to a cross-job feedback store
(REVIEW_FEEDBACK_DIR), which now also receives review decisions with their
category/severity corrections.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0115gGtrSxXE9DKvS9XPFSoT
This commit is contained in:
1 parent
46db871152
commit
23e19f53b2
15 files changed
+1598
-9
No files matched your search
@@ -0,0 +1,137 @@
|
||||
"""Context bundles for the review chat: what the model is allowed to see."""
|
||||
|
||||
import json
|
||||
import os
|
||||
|
||||
from backend.review.chat_context import build_context
|
||||
|
||||
|
||||
def _write(out_dir: str, name: str, value) -> None:
|
||||
path = os.path.join(out_dir, name)
|
||||
os.makedirs(os.path.dirname(path), exist_ok=True)
|
||||
with open(path, "w", encoding="utf-8") as f:
|
||||
json.dump(value, f)
|
||||
|
||||
|
||||
def _finding() -> dict:
|
||||
return {
|
||||
"issue_id": "AGENT-0007",
|
||||
"source_stage": "conflict",
|
||||
"category": "elevation_disagreement",
|
||||
"severity": "high",
|
||||
"confidence": "medium",
|
||||
"location": "Roof / AC-1",
|
||||
"disciplines": ["Mechanical"],
|
||||
"sheets": ["M2.1"],
|
||||
"description": "AC-1 shown at grade on M2.1 but on the roof elsewhere.",
|
||||
"evidence": [{"discipline": "Mechanical", "sheet": "M2.1",
|
||||
"source_text": "AC-1 MOUNTED ON GRADE", "asserted_value": "grade"}],
|
||||
"scope_id": "conflict:roof-ac1",
|
||||
"verification": {"status": "unverified", "verdicts": []},
|
||||
}
|
||||
|
||||
|
||||
def _queue() -> list:
|
||||
return [{"review_item_id": "finding:AGENT-0007", "kind": "finding",
|
||||
"blocking": True, "reasons": ["severity_high"], "payload": _finding()}]
|
||||
|
||||
|
||||
def _job_dir(tmp_path) -> str:
|
||||
out_dir = str(tmp_path)
|
||||
_write(out_dir, "conflicts.json", {
|
||||
"source": "set.pdf",
|
||||
"summary": {"pipeline_mode": "agent", "agent_status": "needs_review",
|
||||
"by_stage": {"conflicts": 3}, "conflicts_found": 3},
|
||||
"sheet_index": {"sheet_index": [
|
||||
{"sheet_number": "M2.1", "discipline": "Mechanical"},
|
||||
{"sheet_number": "A1.1", "discipline": "Architectural"},
|
||||
]},
|
||||
"sheet_reconciliation": {"declared_total": 4, "found_total": 2,
|
||||
"declared_not_in_set": ["C-001", "C-101"],
|
||||
"in_set_not_declared": []},
|
||||
})
|
||||
_write(out_dir, "agent/memory.json", {
|
||||
"sheets": [{
|
||||
"sheet_number": "M2.1", "discipline": "Mechanical", "page_number": 7,
|
||||
"assertions": [{"attribute": "mounting", "value": "grade",
|
||||
"source_text": "AC-1 MOUNTED ON GRADE",
|
||||
"base64": "SHOULD-NOT-APPEAR"}],
|
||||
}],
|
||||
"clusters": [{"key": "roof-ac1", "location": "Roof / AC-1",
|
||||
"disciplines": ["Mechanical", "Architectural"],
|
||||
"assertions": [{"sheet_number": "M2.1", "attribute": "mounting",
|
||||
"value": "grade", "base64": "SHOULD-NOT-APPEAR"}]}],
|
||||
"decisions": [{"finding_refs": ["AGENT-0007"], "action": "kept",
|
||||
"reason": "supported", "kept_issue_id": "AGENT-0007"}],
|
||||
"suppressed": [],
|
||||
})
|
||||
return out_dir
|
||||
|
||||
|
||||
def test_item_scope_carries_the_reasoning_chain(tmp_path):
|
||||
context = build_context(_job_dir(tmp_path), "finding:AGENT-0007", _queue())
|
||||
assert context["scope"] == "item"
|
||||
assert context["finding"]["issue_id"] == "AGENT-0007"
|
||||
# The chain a "why does it think X" answer has to walk.
|
||||
assert context["originating_cluster"]["key"] == "roof-ac1"
|
||||
assert context["source_sheets"][0]["sheet_number"] == "M2.1"
|
||||
assert context["brain_decisions"][0]["action"] == "kept"
|
||||
assert context["finding"]["verification"]["status"] == "unverified"
|
||||
|
||||
|
||||
def test_context_never_leaks_base64(tmp_path):
|
||||
"""Page images blow up the prompt and are useless as quotable evidence."""
|
||||
context = build_context(_job_dir(tmp_path), "finding:AGENT-0007", _queue())
|
||||
assert "SHOULD-NOT-APPEAR" not in json.dumps(context)
|
||||
|
||||
|
||||
def test_run_scope_carries_coverage_material(tmp_path):
|
||||
"""The 'why didn't it pick up the Civil set' inputs are all present."""
|
||||
context = build_context(_job_dir(tmp_path), None, _queue(),
|
||||
question="why didn't it pick up the Civil set?")
|
||||
assert context["scope"] == "run"
|
||||
assert set(context["sheets_by_discipline"]) == {"Mechanical", "Architectural"}
|
||||
assert context["sheet_reconciliation"]["declared_not_in_set"] == ["C-001", "C-101"]
|
||||
assert "finding" not in context
|
||||
assert context["run"]["code_review_enabled"] in (True, False)
|
||||
|
||||
|
||||
def test_unknown_item_falls_back_to_run_scope(tmp_path):
|
||||
context = build_context(_job_dir(tmp_path), "finding:NOPE", _queue())
|
||||
assert context["scope"] == "run"
|
||||
|
||||
|
||||
def test_missing_artifacts_degrade_to_empty(tmp_path):
|
||||
context = build_context(str(tmp_path), None, [])
|
||||
assert context["scope"] == "run"
|
||||
assert context["artifacts_available"] == {
|
||||
"conflicts.json": False, "agent/memory.json": False, "job.log": False}
|
||||
|
||||
|
||||
def test_log_excerpt_matches_question_terms(tmp_path):
|
||||
out_dir = _job_dir(tmp_path)
|
||||
with open(os.path.join(out_dir, "job.log"), "w", encoding="utf-8") as f:
|
||||
f.write("[Extract] page 3 Civil sheet unreadable, skipped\n")
|
||||
f.write("[Brain] merged 2 findings\n")
|
||||
context = build_context(out_dir, None, [], question="why no Civil sheets?")
|
||||
assert any("Civil" in line for line in context["log_excerpt"])
|
||||
assert context["artifacts_available"]["job.log"] is True
|
||||
|
||||
|
||||
def test_reviewer_decision_so_far_is_included(tmp_path):
|
||||
decisions = {"finding:AGENT-0007": {"decision": "reject",
|
||||
"reason_code": "extraction_misread",
|
||||
"comment": "that is a power floor box"}}
|
||||
context = build_context(_job_dir(tmp_path), "finding:AGENT-0007", _queue(), decisions)
|
||||
assert context["reviewer_decision_so_far"]["reason_code"] == "extraction_misread"
|
||||
|
||||
|
||||
def test_clean_cluster_item_uses_cluster_scope(tmp_path):
|
||||
queue = [{"review_item_id": "clean_cluster:roof-ac1", "kind": "clean_cluster",
|
||||
"blocking": False, "reasons": ["audit_sample"],
|
||||
"payload": {"key": "roof-ac1", "location": "Roof / AC-1",
|
||||
"assertions": [{"sheet_number": "M2.1", "value": "grade"}]}}]
|
||||
context = build_context(_job_dir(tmp_path), "clean_cluster:roof-ac1", queue)
|
||||
assert context["scope"] == "item"
|
||||
assert context["cluster"]["key"] == "roof-ac1"
|
||||
assert "finding" not in context
|
||||
Reference in new issue
Block a user