Read-only Q&A on the review screen, per finding and per run, answered from
the job's own artifacts (evidence, cluster, extraction, verification, Brain
merge, sheet index, cover reconciliation, job.log). It never mutates findings,
decisions, or the report.
Turns are logged job-locally (review/chat_log.jsonl, transcript at
/jobs/{id}/review-chat/log) and to a cross-job feedback store
(REVIEW_FEEDBACK_DIR), which now also receives review decisions with their
category/severity corrections.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0115gGtrSxXE9DKvS9XPFSoT
316 lines
13 KiB
Python
316 lines
13 KiB
Python
"""Evidence bundles for the review chat.
|
|
|
|
The chat is an explainer, not an investigator: it may only answer from what the
|
|
run actually produced. This module assembles that material from the job's own
|
|
artifacts and hands the model a bounded, slimmed view.
|
|
|
|
Two shapes, matching the two kinds of question a reviewer asks:
|
|
|
|
- item scope ("why does it think the AC unit is on the ground?") - the finding,
|
|
its evidence, the cluster the finding came from, the sheets those assertions
|
|
were extracted from, any wave-5b verification verdict, the Brain's merge/drop
|
|
decision, and the reviewer's own saved decision.
|
|
- run scope ("why didn't it pick up the Civil set?") - the sheet index by
|
|
discipline, the deterministic cover-index reconciliation, per-stage counts,
|
|
what got suppressed and why, and matching job.log lines.
|
|
|
|
Everything here is read-only and degrades to empty on a missing or corrupt
|
|
artifact; a chat request must never be the thing that breaks a review screen.
|
|
"""
|
|
|
|
import json
|
|
import os
|
|
import re
|
|
from typing import Any, Dict, List, Optional
|
|
|
|
from backend import config
|
|
|
|
# Assertion/evidence text is quoted back verbatim so the reviewer can check the
|
|
# answer against the sheet, but a whole cluster of them would swamp the prompt.
|
|
_MAX_CLUSTER_ASSERTIONS = 40
|
|
_MAX_SHEET_ASSERTIONS = 25
|
|
_MAX_SOURCE_TEXT_CHARS = 400
|
|
_MAX_SHEETS_IN_ROSTER = 400
|
|
_MAX_SUPPRESSED = 25
|
|
_MAX_LOG_LINE_CHARS = 400
|
|
|
|
|
|
def _read_json(path: str, default):
|
|
try:
|
|
with open(path, encoding="utf-8") as f:
|
|
return json.load(f)
|
|
except (OSError, json.JSONDecodeError):
|
|
return default
|
|
|
|
|
|
def _truncate(value: Any, limit: int = _MAX_SOURCE_TEXT_CHARS) -> Any:
|
|
if not isinstance(value, str) or len(value) <= limit:
|
|
return value
|
|
return value[:limit] + "..."
|
|
|
|
|
|
def _slim_assertion(assertion: Dict) -> Dict:
|
|
"""Drop base64/bookkeeping; keep what explains where a value came from."""
|
|
out = {
|
|
"sheet_number": assertion.get("sheet_number"),
|
|
"discipline": assertion.get("discipline"),
|
|
"attribute": assertion.get("attribute"),
|
|
"value": assertion.get("value"),
|
|
"source_text": _truncate(assertion.get("source_text")),
|
|
"location_key": assertion.get("location_key"),
|
|
"normalized_value": assertion.get("normalized_value"),
|
|
"disputed": assertion.get("disputed"),
|
|
}
|
|
return {key: value for key, value in out.items() if value is not None}
|
|
|
|
|
|
def _slim_finding(finding: Dict) -> Dict:
|
|
"""The finding as the run recorded it, including how it was checked."""
|
|
out = {
|
|
"issue_id": finding.get("issue_id"),
|
|
"source_stage": finding.get("source_stage"),
|
|
"agent": finding.get("agent"),
|
|
"category": finding.get("category"),
|
|
"severity": finding.get("severity"),
|
|
"confidence": finding.get("confidence"),
|
|
"location": finding.get("location"),
|
|
"disciplines": finding.get("disciplines"),
|
|
"sheets": finding.get("sheets"),
|
|
"description": _truncate(finding.get("description"), 1200),
|
|
"recommended_resolution": finding.get("recommended_resolution"),
|
|
"code_reference": finding.get("code_reference"),
|
|
"risk_score": finding.get("risk_score"),
|
|
"recommended_priority": finding.get("recommended_priority"),
|
|
"scope_id": finding.get("scope_id"),
|
|
"evidence": [
|
|
{
|
|
"discipline": item.get("discipline"),
|
|
"sheet": item.get("sheet"),
|
|
"source_text": _truncate(item.get("source_text")),
|
|
"asserted_value": item.get("asserted_value"),
|
|
}
|
|
for item in (finding.get("evidence") or [])
|
|
if isinstance(item, dict)
|
|
],
|
|
# Wave 5b / Brain-clarify re-checked some findings against fresh sheet
|
|
# images + the text layer. When present this is the single best answer
|
|
# to "did it actually look again?", so it is never dropped.
|
|
"verification": finding.get("verification"),
|
|
"clarification_of": finding.get("clarification_of"),
|
|
}
|
|
return {key: value for key, value in out.items() if value is not None}
|
|
|
|
|
|
def _slim_sheet(sheet: Dict, limit: int = _MAX_SHEET_ASSERTIONS) -> Dict:
|
|
assertions = sheet.get("assertions") or []
|
|
out = {
|
|
"sheet_number": sheet.get("sheet_number"),
|
|
"sheet_title": sheet.get("sheet_title"),
|
|
"discipline": sheet.get("discipline"),
|
|
"level": sheet.get("level"),
|
|
"page_number": sheet.get("page_number"),
|
|
"assertion_count": len(assertions),
|
|
"assertions": [_slim_assertion(item) for item in assertions[:limit]],
|
|
}
|
|
if len(assertions) > limit:
|
|
out["assertions_omitted"] = len(assertions) - limit
|
|
return out
|
|
|
|
|
|
def _discipline_roster(sheet_index: Dict, sheets: List[Dict]) -> Dict[str, List[str]]:
|
|
"""Sheet numbers grouped by discipline - the 'is Civil in here?' answer.
|
|
|
|
Built from the classified sheet index when there is one, falling back to
|
|
raw extraction, so an empty/failed index stage does not read as "no sheets".
|
|
"""
|
|
entries = (sheet_index or {}).get("sheet_index") or []
|
|
if not entries:
|
|
entries = [
|
|
{"sheet_number": sheet.get("sheet_number"),
|
|
"discipline": sheet.get("discipline")}
|
|
for sheet in sheets or []
|
|
]
|
|
roster: Dict[str, List[str]] = {}
|
|
for entry in entries:
|
|
if not isinstance(entry, dict):
|
|
continue
|
|
discipline = str(entry.get("discipline") or "unknown")
|
|
number = entry.get("sheet_number") or entry.get("sheet_id") or "?"
|
|
bucket = roster.setdefault(discipline, [])
|
|
if len(bucket) < _MAX_SHEETS_IN_ROSTER and number not in bucket:
|
|
bucket.append(str(number))
|
|
return roster
|
|
|
|
|
|
def _log_excerpt(out_dir: str, terms: List[str], limit: int) -> List[str]:
|
|
"""job.log lines mentioning any search term, newest last.
|
|
|
|
The run log is where stage skips, retries, and coverage decisions are
|
|
recorded ("[Code] gated off", "[Extract] page 12 empty"), which is often
|
|
the literal answer to "why didn't it look at X".
|
|
"""
|
|
path = os.path.join(out_dir, "job.log")
|
|
needles = [term.lower() for term in terms if term and len(str(term)) >= 2]
|
|
if not needles or not os.path.isfile(path):
|
|
return []
|
|
hits: List[str] = []
|
|
try:
|
|
with open(path, encoding="utf-8", errors="replace") as f:
|
|
for line in f:
|
|
lowered = line.lower()
|
|
if any(needle in lowered for needle in needles):
|
|
hits.append(_truncate(line.rstrip("\n"), _MAX_LOG_LINE_CHARS))
|
|
except OSError:
|
|
return []
|
|
return hits[-limit:]
|
|
|
|
|
|
def _stage_terms(question: str) -> List[str]:
|
|
"""Search terms for the log: quoted sheet-ish tokens plus long words.
|
|
|
|
Deliberately crude - this only decides which log lines get shown, and an
|
|
over-broad match is bounded by REVIEW_CHAT_LOG_LINES anyway.
|
|
"""
|
|
tokens = re.findall(r"[A-Za-z][A-Za-z0-9.\-]{2,}", question or "")
|
|
stop = {"the", "why", "did", "not", "and", "for", "was", "were", "does",
|
|
"this", "that", "with", "from", "what", "how", "you", "its",
|
|
"it's", "there", "when", "have", "has", "any", "are", "but"}
|
|
return [token for token in tokens if token.lower() not in stop][:12]
|
|
|
|
|
|
def build_context(out_dir: str, review_item_id: Optional[str],
|
|
queue: Optional[List[Dict]] = None,
|
|
decisions: Optional[Dict[str, Dict]] = None,
|
|
question: str = "") -> Dict[str, Any]:
|
|
"""Assemble the evidence bundle for one chat turn.
|
|
|
|
``review_item_id`` selects item scope; None (or an id not in the queue)
|
|
gives run scope. Missing artifacts degrade to empty sections rather than
|
|
raising - the model is told what is missing via ``artifacts_available``.
|
|
"""
|
|
report = _read_json(os.path.join(out_dir, "conflicts.json"), {}) or {}
|
|
snapshot = _read_json(os.path.join(out_dir, "agent", "memory.json"), {}) or {}
|
|
summary = report.get("summary") or {}
|
|
sheets = snapshot.get("sheets") or []
|
|
sheet_index = report.get("sheet_index") or snapshot.get("sheet_index") or {}
|
|
|
|
context: Dict[str, Any] = {
|
|
"scope": "run",
|
|
"run": {
|
|
"source": report.get("source"),
|
|
"pipeline_mode": summary.get("pipeline_mode"),
|
|
"agent_status": summary.get("agent_status"),
|
|
"sheets_analyzed": summary.get("sheets_analyzed") or len(sheets),
|
|
"by_stage": summary.get("by_stage"),
|
|
"conflicts_found": summary.get("conflicts_found"),
|
|
"by_severity": summary.get("by_severity"),
|
|
"models_used": summary.get("models_used"),
|
|
# Stage gating is the answer to a whole class of "why didn't it
|
|
# check X" questions, so it is stated rather than left implied.
|
|
"code_review_enabled": config.ENABLE_CODE_REVIEW,
|
|
},
|
|
"sheets_by_discipline": _discipline_roster(sheet_index, sheets),
|
|
"sheet_reconciliation": report.get("sheet_reconciliation"),
|
|
"missing_expected_sheets": (sheet_index or {}).get("missing_expected_sheets"),
|
|
"suppressed_by_the_run": [
|
|
{
|
|
"issue_id": item.get("issue_id"),
|
|
"category": item.get("category"),
|
|
"description": _truncate(item.get("description"), 300),
|
|
"verification": item.get("verification"),
|
|
}
|
|
for item in (snapshot.get("suppressed") or [])[:_MAX_SUPPRESSED]
|
|
if isinstance(item, dict)
|
|
],
|
|
"artifacts_available": {
|
|
"conflicts.json": bool(report),
|
|
"agent/memory.json": bool(snapshot),
|
|
"job.log": os.path.isfile(os.path.join(out_dir, "job.log")),
|
|
},
|
|
}
|
|
|
|
item = None
|
|
for candidate in queue or []:
|
|
if candidate.get("review_item_id") == review_item_id:
|
|
item = candidate
|
|
break
|
|
if item is None:
|
|
context["log_excerpt"] = _log_excerpt(
|
|
out_dir, _stage_terms(question), config.REVIEW_CHAT_LOG_LINES)
|
|
return context
|
|
|
|
context["scope"] = "item"
|
|
payload = item.get("payload") or {}
|
|
context["review_item"] = {
|
|
"review_item_id": item.get("review_item_id"),
|
|
"kind": item.get("kind"),
|
|
"blocking": item.get("blocking"),
|
|
"review_triggers": item.get("reasons"),
|
|
}
|
|
saved = (decisions or {}).get(review_item_id) or {}
|
|
if saved:
|
|
context["reviewer_decision_so_far"] = {
|
|
"decision": saved.get("decision"),
|
|
"reason_code": saved.get("reason_code"),
|
|
"comment": _truncate(saved.get("comment")),
|
|
}
|
|
|
|
if item.get("kind") == "clean_cluster":
|
|
context["cluster"] = {
|
|
"key": payload.get("key"),
|
|
"location": payload.get("location"),
|
|
"disciplines": payload.get("disciplines"),
|
|
"kind": payload.get("kind"),
|
|
"assertions": [_slim_assertion(a)
|
|
for a in (payload.get("assertions") or [])[:_MAX_CLUSTER_ASSERTIONS]],
|
|
}
|
|
cited_sheets = [a.get("sheet_number") for a in payload.get("assertions") or []]
|
|
else:
|
|
context["finding"] = _slim_finding(payload)
|
|
cited_sheets = list(payload.get("sheets") or [])
|
|
cited_sheets += [e.get("sheet") for e in payload.get("evidence") or []
|
|
if isinstance(e, dict)]
|
|
scope_id = str(payload.get("scope_id") or "")
|
|
if scope_id.startswith("conflict:"):
|
|
cluster_key = scope_id.split(":", 1)[1]
|
|
cluster = next((c for c in snapshot.get("clusters") or []
|
|
if c.get("key") == cluster_key), None)
|
|
if cluster is not None:
|
|
assertions = cluster.get("assertions") or []
|
|
context["originating_cluster"] = {
|
|
"key": cluster.get("key"),
|
|
"location": cluster.get("location"),
|
|
"disciplines": cluster.get("disciplines"),
|
|
"kind": cluster.get("kind"),
|
|
"disputed_attributes": cluster.get("disputed_attributes"),
|
|
"assertion_count": len(assertions),
|
|
"assertions": [_slim_assertion(a)
|
|
for a in assertions[:_MAX_CLUSTER_ASSERTIONS]],
|
|
}
|
|
cited_sheets += [a.get("sheet_number") for a in assertions]
|
|
issue_id = payload.get("issue_id")
|
|
brain_decisions = [
|
|
decision for decision in snapshot.get("decisions") or []
|
|
if isinstance(decision, dict) and (
|
|
decision.get("kept_issue_id") == issue_id
|
|
or issue_id in (decision.get("finding_refs") or []))
|
|
]
|
|
if brain_decisions:
|
|
context["brain_decisions"] = brain_decisions[:10]
|
|
|
|
# The sheets the finding actually rests on, with their raw extraction -
|
|
# this is what lets the model say "it read 'MOUNTED ON GRADE' off M2.1".
|
|
wanted = {str(number) for number in cited_sheets if number}
|
|
if wanted:
|
|
context["source_sheets"] = [
|
|
_slim_sheet(sheet) for sheet in sheets
|
|
if str(sheet.get("sheet_number") or "") in wanted
|
|
]
|
|
|
|
context["log_excerpt"] = _log_excerpt(
|
|
out_dir,
|
|
_stage_terms(question) + sorted(wanted),
|
|
config.REVIEW_CHAT_LOG_LINES,
|
|
)
|
|
return context
|