feat: review chat — ask the run why it concluded a finding
Read-only Q&A on the review screen, per finding and per run, answered from
the job's own artifacts (evidence, cluster, extraction, verification, Brain
merge, sheet index, cover reconciliation, job.log). It never mutates findings,
decisions, or the report.
Turns are logged job-locally (review/chat_log.jsonl, transcript at
/jobs/{id}/review-chat/log) and to a cross-job feedback store
(REVIEW_FEEDBACK_DIR), which now also receives review decisions with their
category/severity corrections.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0115gGtrSxXE9DKvS9XPFSoT
This commit is contained in:
1 parent
46db871152
commit
23e19f53b2
15 files changed
+1598
-9
No files matched your search
@@ -0,0 +1,139 @@
|
||||
"""API tests for the review-chat endpoints."""
|
||||
|
||||
import json
|
||||
import os
|
||||
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
from backend.main import app
|
||||
from backend.review.store import ReviewStore
|
||||
|
||||
|
||||
def _queue_item() -> dict:
|
||||
return {"review_item_id": "finding:AGENT-0007", "kind": "finding",
|
||||
"blocking": True, "reasons": ["severity_high"],
|
||||
"payload": {"issue_id": "AGENT-0007", "category": "elevation_disagreement",
|
||||
"severity": "high", "sheets": ["M2.1"],
|
||||
"description": "AC-1 at grade vs roof.",
|
||||
"scope_id": "conflict:roof-ac1"}}
|
||||
|
||||
|
||||
def _reply() -> dict:
|
||||
return {"answer": "It read 'AC-1 MOUNTED ON GRADE' off M2.1.",
|
||||
"findings": ["The grade value came from M2.1."],
|
||||
"evidence_cited": [], "answerable": "yes",
|
||||
"assessment_of_finding": "looks_supported", "confidence": "high"}
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def job(monkeypatch, tmp_path):
|
||||
"""A finished, review-gated job with one queued finding and a stub model."""
|
||||
store = ReviewStore(str(tmp_path))
|
||||
store.write_queue([_queue_item()])
|
||||
monkeypatch.setattr("backend.main.get_job", lambda job_id: {
|
||||
"job_id": job_id, "status": "needs_review", "out_dir": str(tmp_path)})
|
||||
monkeypatch.setattr("backend.review.chat.call_json", lambda **kw: _reply())
|
||||
return str(tmp_path)
|
||||
|
||||
|
||||
def test_ask_about_a_finding_returns_and_logs_a_turn(job):
|
||||
client = TestClient(app)
|
||||
response = client.post("/jobs/job1/review-chat", json={
|
||||
"question": "Why does it think AC-1 is at grade?",
|
||||
"review_item_id": "finding:AGENT-0007"})
|
||||
assert response.status_code == 200
|
||||
turn = response.json()["turn"]
|
||||
assert turn["issue"]["issue_id"] == "AGENT-0007"
|
||||
assert turn["findings"] == ["The grade value came from M2.1."]
|
||||
with open(os.path.join(job, "review", "chat_log.jsonl"), encoding="utf-8") as f:
|
||||
assert len([line for line in f if line.strip()]) == 1
|
||||
|
||||
|
||||
def test_ask_about_the_run_needs_no_item(job):
|
||||
client = TestClient(app)
|
||||
response = client.post("/jobs/job1/review-chat",
|
||||
json={"question": "Why didn't it pick up the Civil set?"})
|
||||
assert response.status_code == 200
|
||||
assert response.json()["turn"]["scope"] == "run"
|
||||
|
||||
|
||||
def test_history_endpoint_filters_by_item(job):
|
||||
client = TestClient(app)
|
||||
client.post("/jobs/job1/review-chat", json={
|
||||
"question": "Why grade?", "review_item_id": "finding:AGENT-0007"})
|
||||
client.post("/jobs/job1/review-chat", json={"question": "Why no Civil?"})
|
||||
assert len(client.get("/jobs/job1/review-chat").json()["turns"]) == 2
|
||||
filtered = client.get("/jobs/job1/review-chat",
|
||||
params={"review_item_id": "finding:AGENT-0007"}).json()
|
||||
assert len(filtered["turns"]) == 1
|
||||
assert filtered["turns"][0]["question"] == "Why grade?"
|
||||
|
||||
|
||||
def test_transcript_endpoint_renders_markdown(job):
|
||||
client = TestClient(app)
|
||||
client.post("/jobs/job1/review-chat", json={
|
||||
"question": "Why grade?", "review_item_id": "finding:AGENT-0007"})
|
||||
response = client.get("/jobs/job1/review-chat/log")
|
||||
assert response.status_code == 200
|
||||
assert response.headers["content-type"].startswith("text/markdown")
|
||||
assert "## AGENT-0007" in response.text
|
||||
assert "Why grade?" in response.text
|
||||
|
||||
|
||||
def test_blank_question_is_422(job):
|
||||
client = TestClient(app)
|
||||
assert client.post("/jobs/job1/review-chat", json={"question": " "}).status_code == 422
|
||||
|
||||
|
||||
def test_unknown_item_is_422(job):
|
||||
client = TestClient(app)
|
||||
response = client.post("/jobs/job1/review-chat",
|
||||
json={"question": "why?", "review_item_id": "finding:NOPE"})
|
||||
assert response.status_code == 422
|
||||
|
||||
|
||||
def test_model_failure_is_502_not_500(monkeypatch, job):
|
||||
monkeypatch.setattr("backend.review.chat.call_json", lambda **kw: None)
|
||||
client = TestClient(app)
|
||||
response = client.post("/jobs/job1/review-chat", json={"question": "why?"})
|
||||
assert response.status_code == 502
|
||||
|
||||
|
||||
def test_chat_stays_available_after_the_job_is_done(monkeypatch, tmp_path):
|
||||
"""The chat is read-only, so a finalized report can still be questioned."""
|
||||
ReviewStore(str(tmp_path)).write_queue([_queue_item()])
|
||||
monkeypatch.setattr("backend.main.get_job", lambda job_id: {
|
||||
"job_id": job_id, "status": "done", "out_dir": str(tmp_path)})
|
||||
monkeypatch.setattr("backend.review.chat.call_json", lambda **kw: _reply())
|
||||
client = TestClient(app)
|
||||
assert client.post("/jobs/job1/review-chat",
|
||||
json={"question": "why?"}).status_code == 200
|
||||
|
||||
|
||||
def test_chat_is_409_while_the_job_is_still_running(monkeypatch, tmp_path):
|
||||
monkeypatch.setattr("backend.main.get_job", lambda job_id: {
|
||||
"job_id": job_id, "status": "running", "out_dir": str(tmp_path)})
|
||||
client = TestClient(app)
|
||||
assert client.post("/jobs/job1/review-chat",
|
||||
json={"question": "why?"}).status_code == 409
|
||||
|
||||
|
||||
def test_chat_404s_for_unknown_job(monkeypatch, tmp_path):
|
||||
monkeypatch.setattr("backend.main.get_job", lambda job_id: None)
|
||||
client = TestClient(app)
|
||||
assert client.post("/jobs/nope/review-chat",
|
||||
json={"question": "why?"}).status_code == 404
|
||||
|
||||
|
||||
def test_chat_never_mutates_review_decisions(job):
|
||||
"""The whole point: asking questions cannot change the review state."""
|
||||
client = TestClient(app)
|
||||
before = ReviewStore(job, create=False).read_decisions()
|
||||
client.post("/jobs/job1/review-chat", json={
|
||||
"question": "This is wrong, reject it.",
|
||||
"review_item_id": "finding:AGENT-0007"})
|
||||
after = ReviewStore(job, create=False).read_decisions()
|
||||
assert before == after == {}
|
||||
with open(os.path.join(job, "review", "review_queue.json"), encoding="utf-8") as f:
|
||||
assert json.load(f) == [_queue_item()]
|
||||
@@ -0,0 +1,15 @@
|
||||
"""Shared test fixtures."""
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def isolated_feedback_store(tmp_path, monkeypatch):
|
||||
"""Keep the cross-job feedback store out of the real outputs directory.
|
||||
|
||||
Saving a review decision or asking a chat question appends to
|
||||
config.REVIEW_FEEDBACK_DIR, which is process-wide rather than job-local.
|
||||
Without this, running the suite would accumulate junk in backend/outputs.
|
||||
"""
|
||||
monkeypatch.setattr("backend.config.REVIEW_FEEDBACK_DIR",
|
||||
str(tmp_path / "_feedback"))
|
||||
@@ -0,0 +1,172 @@
|
||||
"""Review chat: answer normalization, logging, and the feedback roll-up."""
|
||||
|
||||
import json
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
from backend import config
|
||||
from backend.review import chat
|
||||
|
||||
|
||||
def _queue() -> list:
|
||||
return [{
|
||||
"review_item_id": "finding:AGENT-0007", "kind": "finding", "blocking": True,
|
||||
"reasons": ["severity_high"],
|
||||
"payload": {"issue_id": "AGENT-0007", "source_stage": "conflict",
|
||||
"category": "elevation_disagreement", "severity": "high",
|
||||
"confidence": "medium", "location": "Roof / AC-1",
|
||||
"sheets": ["M2.1"], "description": "AC-1 at grade vs roof.",
|
||||
"evidence": [], "scope_id": "conflict:roof-ac1"},
|
||||
}]
|
||||
|
||||
|
||||
def _model_reply(**overrides) -> dict:
|
||||
reply = {
|
||||
"answer": "The extractor read 'AC-1 MOUNTED ON GRADE' off M2.1.",
|
||||
"findings": ["The grade reading came from M2.1's text layer."],
|
||||
"evidence_cited": [{"artifact": "source_sheets[M2.1]", "sheet": "M2.1",
|
||||
"quote": "AC-1 MOUNTED ON GRADE",
|
||||
"why_it_matters": "It is the sole basis for 'grade'."}],
|
||||
"answerable": "yes",
|
||||
"missing_information": None,
|
||||
"assessment_of_finding": "looks_supported",
|
||||
"suggested_category_correction": None,
|
||||
"confidence": "high",
|
||||
}
|
||||
reply.update(overrides)
|
||||
return reply
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def fake_llm(monkeypatch):
|
||||
"""Stub the model; the chat must never need a network to be tested."""
|
||||
calls = []
|
||||
|
||||
def _call(**kwargs):
|
||||
calls.append(kwargs)
|
||||
return calls_reply[0]
|
||||
|
||||
calls_reply = [_model_reply()]
|
||||
monkeypatch.setattr("backend.review.chat.call_json", lambda **kw: _call(**kw))
|
||||
return calls, calls_reply
|
||||
|
||||
|
||||
def test_ask_logs_issue_question_and_findings(tmp_path, fake_llm):
|
||||
turn = chat.ask("job1", str(tmp_path), "Why is AC-1 at grade?",
|
||||
review_item_id="finding:AGENT-0007", queue=_queue())
|
||||
assert turn["question"] == "Why is AC-1 at grade?"
|
||||
assert turn["findings"] == ["The grade reading came from M2.1's text layer."]
|
||||
# The log's "issue in question" is a snapshot, not a bare id.
|
||||
assert turn["issue"]["issue_id"] == "AGENT-0007"
|
||||
assert turn["issue"]["severity"] == "high"
|
||||
path = os.path.join(str(tmp_path), "review", "chat_log.jsonl")
|
||||
with open(path, encoding="utf-8") as f:
|
||||
logged = [json.loads(line) for line in f if line.strip()]
|
||||
assert len(logged) == 1
|
||||
assert logged[0]["turn_id"] == turn["turn_id"]
|
||||
|
||||
|
||||
def test_ask_appends_to_the_cross_job_feedback_store(tmp_path, fake_llm):
|
||||
chat.ask("job1", str(tmp_path), "Is this really a floor drain?",
|
||||
review_item_id="finding:AGENT-0007", queue=_queue())
|
||||
path = os.path.join(config.REVIEW_FEEDBACK_DIR, "chat_turns.jsonl")
|
||||
with open(path, encoding="utf-8") as f:
|
||||
records = [json.loads(line) for line in f if line.strip()]
|
||||
assert records[0]["kind"] == "review_chat_turn"
|
||||
assert records[0]["issue_id"] == "AGENT-0007"
|
||||
assert records[0]["job_id"] == "job1"
|
||||
|
||||
|
||||
def test_correction_signal_is_captured_as_structured_data(tmp_path, fake_llm):
|
||||
"""A misidentification correction survives as a field, not free text."""
|
||||
_, reply = fake_llm
|
||||
reply[0] = _model_reply(suggested_category_correction="power floor box",
|
||||
assessment_of_finding="looks_unsupported")
|
||||
turn = chat.ask("job1", str(tmp_path), "That is not a floor drain.",
|
||||
review_item_id="finding:AGENT-0007", queue=_queue())
|
||||
assert turn["suggested_category_correction"] == "power floor box"
|
||||
path = os.path.join(config.REVIEW_FEEDBACK_DIR, "chat_turns.jsonl")
|
||||
with open(path, encoding="utf-8") as f:
|
||||
record = json.loads(f.readline())
|
||||
assert record["suggested_category_correction"] == "power floor box"
|
||||
assert record["assessment_of_finding"] == "looks_unsupported"
|
||||
|
||||
|
||||
def test_run_scope_question_needs_no_item(tmp_path, fake_llm):
|
||||
turn = chat.ask("job1", str(tmp_path), "Why didn't it pick up the Civil set?")
|
||||
assert turn["review_item_id"] is None
|
||||
assert turn["scope"] == "run"
|
||||
assert turn["issue"] is None
|
||||
|
||||
|
||||
def test_history_is_replayed_for_the_same_thread(tmp_path, fake_llm):
|
||||
calls, _ = fake_llm
|
||||
chat.ask("job1", str(tmp_path), "First question?",
|
||||
review_item_id="finding:AGENT-0007", queue=_queue())
|
||||
chat.ask("job1", str(tmp_path), "Follow-up?",
|
||||
review_item_id="finding:AGENT-0007", queue=_queue())
|
||||
assert "First question?" in calls[1]["user_text"]
|
||||
# A run-scope turn must not inherit an item thread's history.
|
||||
chat.ask("job1", str(tmp_path), "Unrelated run question?")
|
||||
assert "First question?" not in calls[2]["user_text"]
|
||||
|
||||
|
||||
def test_blank_and_oversized_questions_are_rejected(tmp_path, fake_llm):
|
||||
with pytest.raises(chat.ChatError):
|
||||
chat.ask("job1", str(tmp_path), " ")
|
||||
with pytest.raises(chat.ChatError):
|
||||
chat.ask("job1", str(tmp_path),
|
||||
"x" * (config.REVIEW_CHAT_MAX_QUESTION_CHARS + 1))
|
||||
|
||||
|
||||
def test_unknown_review_item_is_rejected(tmp_path, fake_llm):
|
||||
with pytest.raises(chat.ChatError):
|
||||
chat.ask("job1", str(tmp_path), "why?", review_item_id="finding:NOPE",
|
||||
queue=_queue())
|
||||
|
||||
|
||||
def test_unusable_model_reply_raises_and_logs_nothing(tmp_path, monkeypatch):
|
||||
monkeypatch.setattr("backend.review.chat.call_json", lambda **kw: None)
|
||||
with pytest.raises(RuntimeError):
|
||||
chat.ask("job1", str(tmp_path), "why?")
|
||||
assert not os.path.exists(os.path.join(str(tmp_path), "review", "chat_log.jsonl"))
|
||||
|
||||
|
||||
def test_bad_enum_values_fall_back_instead_of_failing(tmp_path, fake_llm):
|
||||
_, reply = fake_llm
|
||||
reply[0] = _model_reply(answerable="probably", confidence="",
|
||||
assessment_of_finding="made_up")
|
||||
turn = chat.ask("job1", str(tmp_path), "why?")
|
||||
assert turn["answerable"] == "partial"
|
||||
assert turn["confidence"] == "low"
|
||||
assert turn["assessment_of_finding"] == "cannot_tell"
|
||||
|
||||
|
||||
def test_disabled_chat_refuses(tmp_path, monkeypatch, fake_llm):
|
||||
monkeypatch.setattr("backend.config.ENABLE_REVIEW_CHAT", False)
|
||||
with pytest.raises(chat.ChatError):
|
||||
chat.ask("job1", str(tmp_path), "why?")
|
||||
|
||||
|
||||
def test_read_log_skips_corrupt_lines(tmp_path, fake_llm):
|
||||
chat.ask("job1", str(tmp_path), "why?")
|
||||
path = os.path.join(str(tmp_path), "review", "chat_log.jsonl")
|
||||
with open(path, "a", encoding="utf-8") as f:
|
||||
f.write("{not json\n")
|
||||
assert len(chat.read_log(str(tmp_path))) == 1
|
||||
|
||||
|
||||
def test_markdown_transcript_groups_by_issue(tmp_path, fake_llm):
|
||||
chat.ask("job1", str(tmp_path), "Why is AC-1 at grade?",
|
||||
review_item_id="finding:AGENT-0007", queue=_queue())
|
||||
chat.ask("job1", str(tmp_path), "Why no Civil?")
|
||||
markdown = chat.render_log_markdown(chat.read_log(str(tmp_path)))
|
||||
assert "## AGENT-0007" in markdown
|
||||
assert "## Run-scope questions" in markdown
|
||||
assert "Why is AC-1 at grade?" in markdown
|
||||
assert "**Findings**" in markdown
|
||||
|
||||
|
||||
def test_markdown_transcript_handles_empty_log():
|
||||
assert "No questions" in chat.render_log_markdown([])
|
||||
@@ -0,0 +1,137 @@
|
||||
"""Context bundles for the review chat: what the model is allowed to see."""
|
||||
|
||||
import json
|
||||
import os
|
||||
|
||||
from backend.review.chat_context import build_context
|
||||
|
||||
|
||||
def _write(out_dir: str, name: str, value) -> None:
|
||||
path = os.path.join(out_dir, name)
|
||||
os.makedirs(os.path.dirname(path), exist_ok=True)
|
||||
with open(path, "w", encoding="utf-8") as f:
|
||||
json.dump(value, f)
|
||||
|
||||
|
||||
def _finding() -> dict:
|
||||
return {
|
||||
"issue_id": "AGENT-0007",
|
||||
"source_stage": "conflict",
|
||||
"category": "elevation_disagreement",
|
||||
"severity": "high",
|
||||
"confidence": "medium",
|
||||
"location": "Roof / AC-1",
|
||||
"disciplines": ["Mechanical"],
|
||||
"sheets": ["M2.1"],
|
||||
"description": "AC-1 shown at grade on M2.1 but on the roof elsewhere.",
|
||||
"evidence": [{"discipline": "Mechanical", "sheet": "M2.1",
|
||||
"source_text": "AC-1 MOUNTED ON GRADE", "asserted_value": "grade"}],
|
||||
"scope_id": "conflict:roof-ac1",
|
||||
"verification": {"status": "unverified", "verdicts": []},
|
||||
}
|
||||
|
||||
|
||||
def _queue() -> list:
|
||||
return [{"review_item_id": "finding:AGENT-0007", "kind": "finding",
|
||||
"blocking": True, "reasons": ["severity_high"], "payload": _finding()}]
|
||||
|
||||
|
||||
def _job_dir(tmp_path) -> str:
|
||||
out_dir = str(tmp_path)
|
||||
_write(out_dir, "conflicts.json", {
|
||||
"source": "set.pdf",
|
||||
"summary": {"pipeline_mode": "agent", "agent_status": "needs_review",
|
||||
"by_stage": {"conflicts": 3}, "conflicts_found": 3},
|
||||
"sheet_index": {"sheet_index": [
|
||||
{"sheet_number": "M2.1", "discipline": "Mechanical"},
|
||||
{"sheet_number": "A1.1", "discipline": "Architectural"},
|
||||
]},
|
||||
"sheet_reconciliation": {"declared_total": 4, "found_total": 2,
|
||||
"declared_not_in_set": ["C-001", "C-101"],
|
||||
"in_set_not_declared": []},
|
||||
})
|
||||
_write(out_dir, "agent/memory.json", {
|
||||
"sheets": [{
|
||||
"sheet_number": "M2.1", "discipline": "Mechanical", "page_number": 7,
|
||||
"assertions": [{"attribute": "mounting", "value": "grade",
|
||||
"source_text": "AC-1 MOUNTED ON GRADE",
|
||||
"base64": "SHOULD-NOT-APPEAR"}],
|
||||
}],
|
||||
"clusters": [{"key": "roof-ac1", "location": "Roof / AC-1",
|
||||
"disciplines": ["Mechanical", "Architectural"],
|
||||
"assertions": [{"sheet_number": "M2.1", "attribute": "mounting",
|
||||
"value": "grade", "base64": "SHOULD-NOT-APPEAR"}]}],
|
||||
"decisions": [{"finding_refs": ["AGENT-0007"], "action": "kept",
|
||||
"reason": "supported", "kept_issue_id": "AGENT-0007"}],
|
||||
"suppressed": [],
|
||||
})
|
||||
return out_dir
|
||||
|
||||
|
||||
def test_item_scope_carries_the_reasoning_chain(tmp_path):
|
||||
context = build_context(_job_dir(tmp_path), "finding:AGENT-0007", _queue())
|
||||
assert context["scope"] == "item"
|
||||
assert context["finding"]["issue_id"] == "AGENT-0007"
|
||||
# The chain a "why does it think X" answer has to walk.
|
||||
assert context["originating_cluster"]["key"] == "roof-ac1"
|
||||
assert context["source_sheets"][0]["sheet_number"] == "M2.1"
|
||||
assert context["brain_decisions"][0]["action"] == "kept"
|
||||
assert context["finding"]["verification"]["status"] == "unverified"
|
||||
|
||||
|
||||
def test_context_never_leaks_base64(tmp_path):
|
||||
"""Page images blow up the prompt and are useless as quotable evidence."""
|
||||
context = build_context(_job_dir(tmp_path), "finding:AGENT-0007", _queue())
|
||||
assert "SHOULD-NOT-APPEAR" not in json.dumps(context)
|
||||
|
||||
|
||||
def test_run_scope_carries_coverage_material(tmp_path):
|
||||
"""The 'why didn't it pick up the Civil set' inputs are all present."""
|
||||
context = build_context(_job_dir(tmp_path), None, _queue(),
|
||||
question="why didn't it pick up the Civil set?")
|
||||
assert context["scope"] == "run"
|
||||
assert set(context["sheets_by_discipline"]) == {"Mechanical", "Architectural"}
|
||||
assert context["sheet_reconciliation"]["declared_not_in_set"] == ["C-001", "C-101"]
|
||||
assert "finding" not in context
|
||||
assert context["run"]["code_review_enabled"] in (True, False)
|
||||
|
||||
|
||||
def test_unknown_item_falls_back_to_run_scope(tmp_path):
|
||||
context = build_context(_job_dir(tmp_path), "finding:NOPE", _queue())
|
||||
assert context["scope"] == "run"
|
||||
|
||||
|
||||
def test_missing_artifacts_degrade_to_empty(tmp_path):
|
||||
context = build_context(str(tmp_path), None, [])
|
||||
assert context["scope"] == "run"
|
||||
assert context["artifacts_available"] == {
|
||||
"conflicts.json": False, "agent/memory.json": False, "job.log": False}
|
||||
|
||||
|
||||
def test_log_excerpt_matches_question_terms(tmp_path):
|
||||
out_dir = _job_dir(tmp_path)
|
||||
with open(os.path.join(out_dir, "job.log"), "w", encoding="utf-8") as f:
|
||||
f.write("[Extract] page 3 Civil sheet unreadable, skipped\n")
|
||||
f.write("[Brain] merged 2 findings\n")
|
||||
context = build_context(out_dir, None, [], question="why no Civil sheets?")
|
||||
assert any("Civil" in line for line in context["log_excerpt"])
|
||||
assert context["artifacts_available"]["job.log"] is True
|
||||
|
||||
|
||||
def test_reviewer_decision_so_far_is_included(tmp_path):
|
||||
decisions = {"finding:AGENT-0007": {"decision": "reject",
|
||||
"reason_code": "extraction_misread",
|
||||
"comment": "that is a power floor box"}}
|
||||
context = build_context(_job_dir(tmp_path), "finding:AGENT-0007", _queue(), decisions)
|
||||
assert context["reviewer_decision_so_far"]["reason_code"] == "extraction_misread"
|
||||
|
||||
|
||||
def test_clean_cluster_item_uses_cluster_scope(tmp_path):
|
||||
queue = [{"review_item_id": "clean_cluster:roof-ac1", "kind": "clean_cluster",
|
||||
"blocking": False, "reasons": ["audit_sample"],
|
||||
"payload": {"key": "roof-ac1", "location": "Roof / AC-1",
|
||||
"assertions": [{"sheet_number": "M2.1", "value": "grade"}]}}]
|
||||
context = build_context(_job_dir(tmp_path), "clean_cluster:roof-ac1", queue)
|
||||
assert context["scope"] == "item"
|
||||
assert context["cluster"]["key"] == "roof-ac1"
|
||||
assert "finding" not in context
|
||||
@@ -85,3 +85,34 @@ def test_write_label_appends_json_lines(tmp_path):
|
||||
with open(path, encoding="utf-8") as f:
|
||||
lines = [json.loads(line) for line in f if line.strip()]
|
||||
assert lines == [label1, label2]
|
||||
|
||||
|
||||
def test_decision_label_carries_reviewer_corrections():
|
||||
"""category/severity corrections reach the label instead of being dropped."""
|
||||
decision = {"decision": "reject", "reason_code": "extraction_misread",
|
||||
"category_correction": "power floor box",
|
||||
"severity_correction": "low"}
|
||||
label = decision_to_label(_queue_item(), decision, {"job_id": "abc123", "pipeline_mode": "agent",
|
||||
"report": {"summary": {}}})
|
||||
assert label["category_correction"] == "power floor box"
|
||||
assert label["severity_correction"] == "low"
|
||||
assert label["kind"] == "review_decision"
|
||||
|
||||
|
||||
def test_write_label_also_lands_in_the_cross_job_store(tmp_path):
|
||||
from backend import config
|
||||
from backend.review.feedback import read_shared_feedback
|
||||
label = decision_to_label(_queue_item(), {"decision": "reject",
|
||||
"reason_code": "extraction_misread"},
|
||||
{"job_id": "abc123", "pipeline_mode": "agent",
|
||||
"report": {"summary": {}}})
|
||||
write_label(str(tmp_path), label)
|
||||
assert os.path.isfile(os.path.join(config.REVIEW_FEEDBACK_DIR, "decisions.jsonl"))
|
||||
records = read_shared_feedback("review_decision")
|
||||
assert len(records) == 1
|
||||
assert records[0]["reason_code"] == "extraction_misread"
|
||||
|
||||
|
||||
def test_shared_feedback_read_is_empty_when_nothing_written():
|
||||
from backend.review.feedback import read_shared_feedback
|
||||
assert read_shared_feedback("review_decision") == []
|
||||
Reference in new issue
Block a user