256 lines
10 KiB
Python
256 lines
10 KiB
Python
"""ReviewFinalizer: apply human decisions, targeted reruns, RFIs, final artifacts.
|
|
|
|
All LLM-touching helpers degrade gracefully: a failed or empty targeted rerun
|
|
becomes a visible ``analysis_gap`` finding instead of raising, and RFI drafting
|
|
returns whatever was produced (possibly []). Finalization never crashes the job
|
|
on a single bad scope.
|
|
"""
|
|
|
|
import json
|
|
import os
|
|
from typing import Dict, List, Optional, Tuple
|
|
|
|
from backend import config
|
|
from backend.agents.base import AgentScope, AgentUsage
|
|
from backend.agents.conflict_critic import ConflictCriticAgent
|
|
from backend.agents.memory import ProjectMemory
|
|
from backend.agents.orchestrator import Orchestrator
|
|
from backend.agents.rfi_writer import RFIWriterAgent
|
|
# Private import, acceptable here: the runner's _finding_as_conflict is the
|
|
# canonical finding -> report["conflicts"] mapping; reusing it keeps the
|
|
# finalized report's conflicts in exactly the shape build_report produces.
|
|
from backend.agents.runner import _finding_as_conflict
|
|
from backend.pipeline.report import to_markdown
|
|
from backend.review.store import ReviewStore
|
|
|
|
|
|
def apply_decisions(prioritized: List[dict], decisions: Dict[str, dict]) -> Tuple[List[dict], List[dict]]:
|
|
kept: List[dict] = []
|
|
suppressed: List[dict] = []
|
|
for issue in prioritized:
|
|
review_id = f"finding:{issue.get('issue_id')}"
|
|
decision = decisions.get(review_id) or {}
|
|
action = decision.get("decision")
|
|
if action == "reject":
|
|
suppressed.append({
|
|
**issue,
|
|
"review_state": "rejected",
|
|
"reason_code": decision.get("reason_code"),
|
|
"review_comment": decision.get("comment") or "",
|
|
})
|
|
elif action == "unsure":
|
|
kept.append({**issue, "review_state": "unsure"})
|
|
else:
|
|
kept.append({**issue, "review_state": "confirmed" if action == "confirm" else "unreviewed"})
|
|
return kept, suppressed
|
|
|
|
|
|
def _gap_finding(index: int, scope_id: str, description: str) -> dict:
|
|
"""Same shape as the runner's gap_findings: low severity, high confidence."""
|
|
return {
|
|
"issue_id": f"AGENT-GAP-CLARIFY-{index + 1:03d}",
|
|
"source_stage": "qaqc",
|
|
"category": "analysis_gap",
|
|
"severity": "low",
|
|
"confidence": "high",
|
|
"location": scope_id.split(":", 2)[1] if ":" in scope_id else "",
|
|
"disciplines": [],
|
|
"sheets": [],
|
|
"description": description,
|
|
"evidence": [],
|
|
"recommended_resolution": "Review this scope manually or rerun the job.",
|
|
"code_reference": None,
|
|
"agent": "completeness",
|
|
"scope_id": scope_id,
|
|
}
|
|
|
|
|
|
def rerun_clarified_scopes(
|
|
memory_snapshot: dict,
|
|
decisions: Dict[str, dict],
|
|
prioritized: Optional[List[dict]] = None,
|
|
) -> List[dict]:
|
|
"""Bounded targeted reruns for ``needs_clarification`` decisions.
|
|
|
|
v1 reruns conflict scopes only: at most ONE ConflictCriticAgent scope per
|
|
clarified finding. The clarification answer is injected as a pseudo
|
|
"Reviewer" assertion prepended to the cluster's assertions so it reaches
|
|
the critic's evidence block (and survives front-truncation to
|
|
AGENT_CLUSTER_MAX_ASSERTIONS); ``page_to_b64`` is empty (cluster
|
|
assertions may carry their own
|
|
base64). Every per-scope failure degrades to an ``analysis_gap`` finding
|
|
and never raises. Non-conflict scopes are NOT rerun; they produce an
|
|
``analysis_gap`` noting the scope is not rerunnable in v1.
|
|
"""
|
|
findings_pool = list(prioritized or []) + list(memory_snapshot.get("findings") or [])
|
|
clusters = memory_snapshot.get("clusters") or []
|
|
out: List[dict] = []
|
|
for item_id, decision in (decisions or {}).items():
|
|
if (decision or {}).get("decision") != "needs_clarification":
|
|
continue
|
|
answer = str(decision.get("clarification_answer") or "").strip()
|
|
if not answer:
|
|
continue
|
|
issue_id = item_id.split(":", 1)[1] if item_id.startswith("finding:") else item_id
|
|
finding = next((f for f in findings_pool if f.get("issue_id") == issue_id), None)
|
|
scope_id = str((finding or {}).get("scope_id") or "")
|
|
if not scope_id.startswith("conflict:"):
|
|
out.append(_gap_finding(
|
|
len(out), scope_id or item_id,
|
|
f"Clarification rerun not supported in v1 for non-conflict scope "
|
|
f"{scope_id or item_id!r} (finding {issue_id}).",
|
|
))
|
|
continue
|
|
cluster_key = scope_id.split(":", 1)[1]
|
|
cluster = next((c for c in clusters if c.get("key") == cluster_key), None)
|
|
if cluster is None:
|
|
out.append(_gap_finding(
|
|
len(out), scope_id,
|
|
f"Clarification rerun failed: cluster {cluster_key!r} not found "
|
|
f"for scope {scope_id} (finding {issue_id}).",
|
|
))
|
|
continue
|
|
rerun_cluster = {
|
|
**cluster,
|
|
# Prepend: ConflictCriticAgent truncates assertions from the front
|
|
# (AGENT_CLUSTER_MAX_ASSERTIONS), so the clarification must come
|
|
# first or a full cluster would silently drop it.
|
|
"assertions": [{
|
|
"discipline": "Reviewer",
|
|
"sheet_number": "REVIEW",
|
|
"attribute": "clarification",
|
|
"value": answer,
|
|
"source_text": answer,
|
|
}] + list(cluster.get("assertions") or []),
|
|
}
|
|
scope = AgentScope(
|
|
scope_id=scope_id,
|
|
payload={"cluster": rerun_cluster, "page_to_b64": {}},
|
|
)
|
|
result = ConflictCriticAgent(AgentUsage()).run(scope)
|
|
if result.error or not result.artifacts:
|
|
out.append(_gap_finding(
|
|
len(out), scope_id,
|
|
f"Clarification rerun did not complete for scope {scope_id} "
|
|
f"(finding {issue_id}): {result.error or 'no findings produced'}.",
|
|
))
|
|
continue
|
|
for rerun_finding in result.artifacts:
|
|
rerun_finding["clarification_of"] = issue_id
|
|
out.append(rerun_finding)
|
|
return out
|
|
|
|
|
|
def _draft_rfis(kept: List[dict]) -> List[dict]:
|
|
"""Draft RFIs for kept issues only, mirroring the runner's wave 7."""
|
|
orchestrator = Orchestrator(ProjectMemory())
|
|
scopes = [
|
|
AgentScope(
|
|
scope_id=f"rfi:{finding.get('issue_id') or index + 1}",
|
|
payload={"finding": finding},
|
|
)
|
|
for index, finding in enumerate(kept)
|
|
]
|
|
try:
|
|
results = orchestrator.run_scopes(
|
|
RFIWriterAgent(AgentUsage()), scopes, config.AGENT_RFI_CONCURRENCY
|
|
)
|
|
except Exception:
|
|
return []
|
|
return [artifact for result in results for artifact in result.artifacts]
|
|
|
|
|
|
def _read_json(path: str, default):
|
|
try:
|
|
with open(path, encoding="utf-8") as f:
|
|
return json.load(f)
|
|
except (OSError, json.JSONDecodeError):
|
|
return default
|
|
|
|
|
|
def _dump(out_dir: str, name: str, value) -> None:
|
|
with open(os.path.join(out_dir, name), "w", encoding="utf-8") as f:
|
|
json.dump(value, f, indent=2)
|
|
|
|
|
|
def finalize_review(job_id: str, out_dir: str) -> dict:
|
|
"""Apply review decisions and write the final report artifacts.
|
|
|
|
Raises ValueError("incomplete review") if any blocking queue item lacks a
|
|
decision. Never raises for rerun/RFI degradation.
|
|
"""
|
|
store = ReviewStore(out_dir)
|
|
queue = store.read_queue()
|
|
decisions = store.read_decisions()
|
|
for item in queue:
|
|
if item.get("blocking") and item.get("review_item_id") not in decisions:
|
|
raise ValueError("incomplete review")
|
|
|
|
report = _read_json(os.path.join(out_dir, "conflicts.json"), {}) or {}
|
|
snapshot = _read_json(os.path.join(out_dir, "agent", "memory.json"), {}) or {}
|
|
prioritized = list(report.get("validated_issues") or [])
|
|
|
|
rerun_findings = rerun_clarified_scopes(snapshot, decisions, prioritized)
|
|
replacements: Dict[str, List[dict]] = {}
|
|
for finding in rerun_findings:
|
|
origin = finding.get("clarification_of")
|
|
if origin:
|
|
replacements.setdefault(origin, []).append(finding)
|
|
else:
|
|
prioritized.append(finding) # gap findings stay as additions
|
|
for origin, new_findings in replacements.items():
|
|
for index, issue in enumerate(prioritized):
|
|
if issue.get("issue_id") == origin:
|
|
prioritized[index:index + 1] = new_findings
|
|
break
|
|
|
|
kept, suppressed = apply_decisions(prioritized, decisions)
|
|
for issue in kept:
|
|
if issue.get("clarification_of"):
|
|
issue["review_state"] = "clarified"
|
|
else:
|
|
decision = decisions.get(f"finding:{issue.get('issue_id')}") or {}
|
|
if decision.get("decision") == "needs_clarification":
|
|
issue["review_state"] = "clarification_failed"
|
|
|
|
rfis = _draft_rfis(kept)
|
|
|
|
report["validated_issues"] = kept
|
|
report["suppressed_issues"] = (report.get("suppressed_issues") or []) + suppressed
|
|
report["rfis"] = rfis
|
|
summary = report.setdefault("summary", {})
|
|
summary["agent_status"] = "complete"
|
|
by_stage = summary.get("by_stage")
|
|
if isinstance(by_stage, dict):
|
|
if "validated" in by_stage:
|
|
by_stage["validated"] = len(kept)
|
|
if "rfis" in by_stage:
|
|
by_stage["rfis"] = len(rfis)
|
|
# Rebuild the conflicts view and headline counts from the KEPT
|
|
# conflict-stage findings so rejected findings no longer appear as
|
|
# conflicts in report.md / the UI (mirrors pipeline.report.build_report).
|
|
conflicts = [
|
|
_finding_as_conflict(finding)
|
|
for finding in kept
|
|
if finding.get("source_stage") == "conflict"
|
|
]
|
|
report["conflicts"] = conflicts
|
|
by_severity = {"high": 0, "medium": 0, "low": 0}
|
|
by_category: Dict[str, int] = {}
|
|
for conflict in conflicts:
|
|
by_severity[conflict["severity"]] = by_severity.get(conflict["severity"], 0) + 1
|
|
by_category[conflict["category"]] = by_category.get(conflict["category"], 0) + 1
|
|
summary["conflicts_found"] = len(conflicts)
|
|
summary["by_severity"] = by_severity
|
|
summary["by_category"] = by_category
|
|
summary["review"] = store.progress(queue)
|
|
|
|
os.makedirs(out_dir, exist_ok=True)
|
|
_dump(out_dir, "conflicts.json", report)
|
|
_dump(out_dir, "validated_issues.json", kept)
|
|
_dump(out_dir, "suppressed_issues.json", suppressed)
|
|
_dump(out_dir, "rfis.json", rfis)
|
|
with open(os.path.join(out_dir, "report.md"), "w", encoding="utf-8") as f:
|
|
f.write(to_markdown(report))
|
|
return report
|