feat: review chat — ask the run why it concluded a finding
Docker Release / build-and-push (push) Successful in 1m45s
Docker Release / release (push) Skipped

Read-only Q&A on the review screen, per finding and per run, answered from
the job's own artifacts (evidence, cluster, extraction, verification, Brain
merge, sheet index, cover reconciliation, job.log). It never mutates findings,
decisions, or the report.

Turns are logged job-locally (review/chat_log.jsonl, transcript at
/jobs/{id}/review-chat/log) and to a cross-job feedback store
(REVIEW_FEEDBACK_DIR), which now also receives review decisions with their
category/severity corrections.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0115gGtrSxXE9DKvS9XPFSoT
This commit is contained in:
woogiandClaude Opus 5 committed 2026-09-14 10:35:18 -05:00
1 parent 46db871152
commit 23e19f53b2
15 files changed
+1598 -9

No files matched your search

+315
View File
@@ -0,0 +1,315 @@
"""Evidence bundles for the review chat.
The chat is an explainer, not an investigator: it may only answer from what the
run actually produced. This module assembles that material from the job's own
artifacts and hands the model a bounded, slimmed view.
Two shapes, matching the two kinds of question a reviewer asks:
- item scope ("why does it think the AC unit is on the ground?") - the finding,
its evidence, the cluster the finding came from, the sheets those assertions
were extracted from, any wave-5b verification verdict, the Brain's merge/drop
decision, and the reviewer's own saved decision.
- run scope ("why didn't it pick up the Civil set?") - the sheet index by
discipline, the deterministic cover-index reconciliation, per-stage counts,
what got suppressed and why, and matching job.log lines.
Everything here is read-only and degrades to empty on a missing or corrupt
artifact; a chat request must never be the thing that breaks a review screen.
"""
import json
import os
import re
from typing import Any, Dict, List, Optional
from backend import config
# Assertion/evidence text is quoted back verbatim so the reviewer can check the
# answer against the sheet, but a whole cluster of them would swamp the prompt.
_MAX_CLUSTER_ASSERTIONS = 40
_MAX_SHEET_ASSERTIONS = 25
_MAX_SOURCE_TEXT_CHARS = 400
_MAX_SHEETS_IN_ROSTER = 400
_MAX_SUPPRESSED = 25
_MAX_LOG_LINE_CHARS = 400
def _read_json(path: str, default):
try:
with open(path, encoding="utf-8") as f:
return json.load(f)
except (OSError, json.JSONDecodeError):
return default
def _truncate(value: Any, limit: int = _MAX_SOURCE_TEXT_CHARS) -> Any:
if not isinstance(value, str) or len(value) <= limit:
return value
return value[:limit] + "..."
def _slim_assertion(assertion: Dict) -> Dict:
"""Drop base64/bookkeeping; keep what explains where a value came from."""
out = {
"sheet_number": assertion.get("sheet_number"),
"discipline": assertion.get("discipline"),
"attribute": assertion.get("attribute"),
"value": assertion.get("value"),
"source_text": _truncate(assertion.get("source_text")),
"location_key": assertion.get("location_key"),
"normalized_value": assertion.get("normalized_value"),
"disputed": assertion.get("disputed"),
}
return {key: value for key, value in out.items() if value is not None}
def _slim_finding(finding: Dict) -> Dict:
"""The finding as the run recorded it, including how it was checked."""
out = {
"issue_id": finding.get("issue_id"),
"source_stage": finding.get("source_stage"),
"agent": finding.get("agent"),
"category": finding.get("category"),
"severity": finding.get("severity"),
"confidence": finding.get("confidence"),
"location": finding.get("location"),
"disciplines": finding.get("disciplines"),
"sheets": finding.get("sheets"),
"description": _truncate(finding.get("description"), 1200),
"recommended_resolution": finding.get("recommended_resolution"),
"code_reference": finding.get("code_reference"),
"risk_score": finding.get("risk_score"),
"recommended_priority": finding.get("recommended_priority"),
"scope_id": finding.get("scope_id"),
"evidence": [
{
"discipline": item.get("discipline"),
"sheet": item.get("sheet"),
"source_text": _truncate(item.get("source_text")),
"asserted_value": item.get("asserted_value"),
}
for item in (finding.get("evidence") or [])
if isinstance(item, dict)
],
# Wave 5b / Brain-clarify re-checked some findings against fresh sheet
# images + the text layer. When present this is the single best answer
# to "did it actually look again?", so it is never dropped.
"verification": finding.get("verification"),
"clarification_of": finding.get("clarification_of"),
}
return {key: value for key, value in out.items() if value is not None}
def _slim_sheet(sheet: Dict, limit: int = _MAX_SHEET_ASSERTIONS) -> Dict:
assertions = sheet.get("assertions") or []
out = {
"sheet_number": sheet.get("sheet_number"),
"sheet_title": sheet.get("sheet_title"),
"discipline": sheet.get("discipline"),
"level": sheet.get("level"),
"page_number": sheet.get("page_number"),
"assertion_count": len(assertions),
"assertions": [_slim_assertion(item) for item in assertions[:limit]],
}
if len(assertions) > limit:
out["assertions_omitted"] = len(assertions) - limit
return out
def _discipline_roster(sheet_index: Dict, sheets: List[Dict]) -> Dict[str, List[str]]:
"""Sheet numbers grouped by discipline - the 'is Civil in here?' answer.
Built from the classified sheet index when there is one, falling back to
raw extraction, so an empty/failed index stage does not read as "no sheets".
"""
entries = (sheet_index or {}).get("sheet_index") or []
if not entries:
entries = [
{"sheet_number": sheet.get("sheet_number"),
"discipline": sheet.get("discipline")}
for sheet in sheets or []
]
roster: Dict[str, List[str]] = {}
for entry in entries:
if not isinstance(entry, dict):
continue
discipline = str(entry.get("discipline") or "unknown")
number = entry.get("sheet_number") or entry.get("sheet_id") or "?"
bucket = roster.setdefault(discipline, [])
if len(bucket) < _MAX_SHEETS_IN_ROSTER and number not in bucket:
bucket.append(str(number))
return roster
def _log_excerpt(out_dir: str, terms: List[str], limit: int) -> List[str]:
"""job.log lines mentioning any search term, newest last.
The run log is where stage skips, retries, and coverage decisions are
recorded ("[Code] gated off", "[Extract] page 12 empty"), which is often
the literal answer to "why didn't it look at X".
"""
path = os.path.join(out_dir, "job.log")
needles = [term.lower() for term in terms if term and len(str(term)) >= 2]
if not needles or not os.path.isfile(path):
return []
hits: List[str] = []
try:
with open(path, encoding="utf-8", errors="replace") as f:
for line in f:
lowered = line.lower()
if any(needle in lowered for needle in needles):
hits.append(_truncate(line.rstrip("\n"), _MAX_LOG_LINE_CHARS))
except OSError:
return []
return hits[-limit:]
def _stage_terms(question: str) -> List[str]:
"""Search terms for the log: quoted sheet-ish tokens plus long words.
Deliberately crude - this only decides which log lines get shown, and an
over-broad match is bounded by REVIEW_CHAT_LOG_LINES anyway.
"""
tokens = re.findall(r"[A-Za-z][A-Za-z0-9.\-]{2,}", question or "")
stop = {"the", "why", "did", "not", "and", "for", "was", "were", "does",
"this", "that", "with", "from", "what", "how", "you", "its",
"it's", "there", "when", "have", "has", "any", "are", "but"}
return [token for token in tokens if token.lower() not in stop][:12]
def build_context(out_dir: str, review_item_id: Optional[str],
queue: Optional[List[Dict]] = None,
decisions: Optional[Dict[str, Dict]] = None,
question: str = "") -> Dict[str, Any]:
"""Assemble the evidence bundle for one chat turn.
``review_item_id`` selects item scope; None (or an id not in the queue)
gives run scope. Missing artifacts degrade to empty sections rather than
raising - the model is told what is missing via ``artifacts_available``.
"""
report = _read_json(os.path.join(out_dir, "conflicts.json"), {}) or {}
snapshot = _read_json(os.path.join(out_dir, "agent", "memory.json"), {}) or {}
summary = report.get("summary") or {}
sheets = snapshot.get("sheets") or []
sheet_index = report.get("sheet_index") or snapshot.get("sheet_index") or {}
context: Dict[str, Any] = {
"scope": "run",
"run": {
"source": report.get("source"),
"pipeline_mode": summary.get("pipeline_mode"),
"agent_status": summary.get("agent_status"),
"sheets_analyzed": summary.get("sheets_analyzed") or len(sheets),
"by_stage": summary.get("by_stage"),
"conflicts_found": summary.get("conflicts_found"),
"by_severity": summary.get("by_severity"),
"models_used": summary.get("models_used"),
# Stage gating is the answer to a whole class of "why didn't it
# check X" questions, so it is stated rather than left implied.
"code_review_enabled": config.ENABLE_CODE_REVIEW,
},
"sheets_by_discipline": _discipline_roster(sheet_index, sheets),
"sheet_reconciliation": report.get("sheet_reconciliation"),
"missing_expected_sheets": (sheet_index or {}).get("missing_expected_sheets"),
"suppressed_by_the_run": [
{
"issue_id": item.get("issue_id"),
"category": item.get("category"),
"description": _truncate(item.get("description"), 300),
"verification": item.get("verification"),
}
for item in (snapshot.get("suppressed") or [])[:_MAX_SUPPRESSED]
if isinstance(item, dict)
],
"artifacts_available": {
"conflicts.json": bool(report),
"agent/memory.json": bool(snapshot),
"job.log": os.path.isfile(os.path.join(out_dir, "job.log")),
},
}
item = None
for candidate in queue or []:
if candidate.get("review_item_id") == review_item_id:
item = candidate
break
if item is None:
context["log_excerpt"] = _log_excerpt(
out_dir, _stage_terms(question), config.REVIEW_CHAT_LOG_LINES)
return context
context["scope"] = "item"
payload = item.get("payload") or {}
context["review_item"] = {
"review_item_id": item.get("review_item_id"),
"kind": item.get("kind"),
"blocking": item.get("blocking"),
"review_triggers": item.get("reasons"),
}
saved = (decisions or {}).get(review_item_id) or {}
if saved:
context["reviewer_decision_so_far"] = {
"decision": saved.get("decision"),
"reason_code": saved.get("reason_code"),
"comment": _truncate(saved.get("comment")),
}
if item.get("kind") == "clean_cluster":
context["cluster"] = {
"key": payload.get("key"),
"location": payload.get("location"),
"disciplines": payload.get("disciplines"),
"kind": payload.get("kind"),
"assertions": [_slim_assertion(a)
for a in (payload.get("assertions") or [])[:_MAX_CLUSTER_ASSERTIONS]],
}
cited_sheets = [a.get("sheet_number") for a in payload.get("assertions") or []]
else:
context["finding"] = _slim_finding(payload)
cited_sheets = list(payload.get("sheets") or [])
cited_sheets += [e.get("sheet") for e in payload.get("evidence") or []
if isinstance(e, dict)]
scope_id = str(payload.get("scope_id") or "")
if scope_id.startswith("conflict:"):
cluster_key = scope_id.split(":", 1)[1]
cluster = next((c for c in snapshot.get("clusters") or []
if c.get("key") == cluster_key), None)
if cluster is not None:
assertions = cluster.get("assertions") or []
context["originating_cluster"] = {
"key": cluster.get("key"),
"location": cluster.get("location"),
"disciplines": cluster.get("disciplines"),
"kind": cluster.get("kind"),
"disputed_attributes": cluster.get("disputed_attributes"),
"assertion_count": len(assertions),
"assertions": [_slim_assertion(a)
for a in assertions[:_MAX_CLUSTER_ASSERTIONS]],
}
cited_sheets += [a.get("sheet_number") for a in assertions]
issue_id = payload.get("issue_id")
brain_decisions = [
decision for decision in snapshot.get("decisions") or []
if isinstance(decision, dict) and (
decision.get("kept_issue_id") == issue_id
or issue_id in (decision.get("finding_refs") or []))
]
if brain_decisions:
context["brain_decisions"] = brain_decisions[:10]
# The sheets the finding actually rests on, with their raw extraction -
# this is what lets the model say "it read 'MOUNTED ON GRADE' off M2.1".
wanted = {str(number) for number in cited_sheets if number}
if wanted:
context["source_sheets"] = [
_slim_sheet(sheet) for sheet in sheets
if str(sheet.get("sheet_number") or "") in wanted
]
context["log_excerpt"] = _log_excerpt(
out_dir,
_stage_terms(question) + sorted(wanted),
config.REVIEW_CHAT_LOG_LINES,
)
return context