feat: review chat — ask the run why it concluded a finding
Read-only Q&A on the review screen, per finding and per run, answered from
the job's own artifacts (evidence, cluster, extraction, verification, Brain
merge, sheet index, cover reconciliation, job.log). It never mutates findings,
decisions, or the report.
Turns are logged job-locally (review/chat_log.jsonl, transcript at
/jobs/{id}/review-chat/log) and to a cross-job feedback store
(REVIEW_FEEDBACK_DIR), which now also receives review decisions with their
category/severity corrections.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0115gGtrSxXE9DKvS9XPFSoT
This commit is contained in:
1 parent
46db871152
commit
23e19f53b2
15 files changed
+1598
-9
No files matched your search
@@ -1,9 +1,19 @@
|
||||
"""Feedback labels: one label artifact per human-review decision, for metrics."""
|
||||
"""Feedback labels: one label artifact per human-review decision, for metrics.
|
||||
|
||||
Labels are written twice: job-locally under ``<out_dir>/review/`` (the
|
||||
auditable record for that run) and, via ``append_shared_feedback``, to the
|
||||
cross-job store at ``config.REVIEW_FEEDBACK_DIR``. The shared store is
|
||||
append-only and nothing reads it yet - it exists so that a later pass can prime
|
||||
a run with what reviewers corrected on previous sets without having to walk
|
||||
every job directory.
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
from datetime import datetime, timezone
|
||||
|
||||
from backend import config
|
||||
|
||||
|
||||
def _as_dict(value) -> dict:
|
||||
return value if isinstance(value, dict) else {}
|
||||
@@ -21,6 +31,7 @@ def decision_to_label(queue_item: dict, decision: dict, job: dict) -> dict:
|
||||
payload = _as_dict(queue_item.get("payload"))
|
||||
summary = _as_dict(_as_dict(job.get("report")).get("summary"))
|
||||
return {
|
||||
"kind": "review_decision",
|
||||
"review_item_id": queue_item.get("review_item_id"),
|
||||
"job_id": job.get("job_id"),
|
||||
"pipeline_mode": job.get("pipeline_mode"),
|
||||
@@ -30,6 +41,11 @@ def decision_to_label(queue_item: dict, decision: dict, job: dict) -> dict:
|
||||
"confidence": payload.get("confidence"),
|
||||
"decision": decision.get("decision"),
|
||||
"reason_code": decision.get("reason_code"),
|
||||
# The reviewer's structured corrections. Carried here (and into the
|
||||
# cross-job store) so "wrong category" survives as data rather than
|
||||
# only as free text on the suppressed issue.
|
||||
"category_correction": decision.get("category_correction"),
|
||||
"severity_correction": decision.get("severity_correction"),
|
||||
"location": payload.get("location"),
|
||||
"disciplines": payload.get("disciplines"),
|
||||
"sheets": payload.get("sheets"),
|
||||
@@ -40,7 +56,12 @@ def decision_to_label(queue_item: dict, decision: dict, job: dict) -> dict:
|
||||
|
||||
|
||||
def write_label(out_dir: str, label: dict) -> None:
|
||||
"""Append one label as a JSON line; never raises on I/O failure."""
|
||||
"""Append one label job-locally and to the cross-job store.
|
||||
|
||||
Never raises on I/O failure: a lost label must not fail the save that
|
||||
produced it.
|
||||
"""
|
||||
append_shared_feedback(label)
|
||||
try:
|
||||
review_dir = os.path.join(out_dir, "review")
|
||||
os.makedirs(review_dir, exist_ok=True)
|
||||
@@ -49,3 +70,49 @@ def write_label(out_dir: str, label: dict) -> None:
|
||||
f.write(json.dumps(label) + "\n")
|
||||
except OSError as e:
|
||||
print(f"[Review] feedback label write failed: {e}")
|
||||
|
||||
|
||||
def append_shared_feedback(record: dict) -> None:
|
||||
"""Append one record to the cross-job feedback store; never raises.
|
||||
|
||||
One JSONL file per record ``kind`` so a reader can pick up decisions and
|
||||
chat turns independently. Failure here is logged and swallowed: the
|
||||
cross-job roll-up is a convenience, and losing a line must never fail the
|
||||
review action that produced it.
|
||||
"""
|
||||
try:
|
||||
kind = str(record.get("kind") or "misc")
|
||||
os.makedirs(config.REVIEW_FEEDBACK_DIR, exist_ok=True)
|
||||
name = "chat_turns.jsonl" if kind == "review_chat_turn" else "decisions.jsonl"
|
||||
path = os.path.join(config.REVIEW_FEEDBACK_DIR, name)
|
||||
with open(path, "a", encoding="utf-8") as f:
|
||||
f.write(json.dumps(record) + "\n")
|
||||
except OSError as e:
|
||||
print(f"[Review] shared feedback write failed: {e}")
|
||||
|
||||
|
||||
def read_shared_feedback(kind: str = "review_decision") -> list:
|
||||
"""Read the cross-job store for one record kind, oldest first.
|
||||
|
||||
Corrupt lines are skipped so a partial write cannot hide the rest.
|
||||
"""
|
||||
name = "chat_turns.jsonl" if kind == "review_chat_turn" else "decisions.jsonl"
|
||||
path = os.path.join(config.REVIEW_FEEDBACK_DIR, name)
|
||||
if not os.path.isfile(path):
|
||||
return []
|
||||
records = []
|
||||
try:
|
||||
with open(path, encoding="utf-8") as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
try:
|
||||
value = json.loads(line)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
if isinstance(value, dict):
|
||||
records.append(value)
|
||||
except OSError:
|
||||
return []
|
||||
return records
|
||||
Reference in new issue
Block a user