Files
Conflict_Checker/backend/config.py
T
woogiandClaude Opus 5 23e19f53b2
Docker Release / build-and-push (push) Successful in 1m45s
Docker Release / release (push) Skipped
feat: review chat — ask the run why it concluded a finding
Read-only Q&A on the review screen, per finding and per run, answered from
the job's own artifacts (evidence, cluster, extraction, verification, Brain
merge, sheet index, cover reconciliation, job.log). It never mutates findings,
decisions, or the report.

Turns are logged job-locally (review/chat_log.jsonl, transcript at
/jobs/{id}/review-chat/log) and to a cross-job feedback store
(REVIEW_FEEDBACK_DIR), which now also receives review decisions with their
category/severity corrections.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0115gGtrSxXE9DKvS9XPFSoT
2026-09-14 10:35:18 -05:00

263 lines
16 KiB
Python

"""
config.py - Environment-driven configuration for the Conflict Checker.
All knobs come from backend/.env (see .env.example). Paths are derived from this
file's location so the app runs regardless of the current working directory.
"""
import os
from dotenv import load_dotenv
load_dotenv(os.path.join(os.path.dirname(os.path.abspath(__file__)), ".env"))
_BASE_DIR = os.path.dirname(os.path.abspath(__file__))
_TRUTHY = ("1", "true", "yes", "on")
def _flag(name: str, default: str) -> bool:
"""Parse a boolean env knob. Accepts 1/true/yes/on (case-insensitive) so a
knob set to "1" behaves the same as one set to "true" — mixing bare
`== "true"` comparisons with this set silently disabled features."""
return os.getenv(name, default).strip().lower() in _TRUTHY
# -- AI Backend (OpenRouter) ----------------------------------------
# One multimodal model does both extraction (Stage 1) and conflict
# reasoning (Stage 3). Override MODEL per-stage if you ever split them.
AI_BASE_URL = os.getenv("AI_BASE_URL", "https://openrouter.ai/api/v1")
AI_API_KEY = os.getenv("AI_API_KEY", "")
MODEL = os.getenv("MODEL", "google/gemini-2.5-pro")
# Text model when text stages run on OpenRouter (non-hybrid). Falls back to
# MODEL when unset.
TEXT_MODEL = os.getenv("TEXT_MODEL", "") or MODEL
# Agent-mode model overrides (OpenRouter IDs). Empty values inherit the
# matching general-purpose model so the skeleton requires no extra config.
AGENT_EXTRACT_MODEL = os.getenv("AGENT_EXTRACT_MODEL", "") or MODEL
AGENT_INDEX_MODEL = os.getenv("AGENT_INDEX_MODEL", "") or TEXT_MODEL
AGENT_JURISDICTION_MODEL = os.getenv("AGENT_JURISDICTION_MODEL", "") or TEXT_MODEL
AGENT_LINKER_MODEL = os.getenv("AGENT_LINKER_MODEL", "") or TEXT_MODEL
AGENT_CONFLICT_MODEL = os.getenv("AGENT_CONFLICT_MODEL", "") or MODEL
AGENT_CODE_MODEL = os.getenv("AGENT_CODE_MODEL", "") or TEXT_MODEL
AGENT_CONSTRUCT_MODEL = os.getenv("AGENT_CONSTRUCT_MODEL", "") or TEXT_MODEL
AGENT_COMPLETENESS_MODEL = os.getenv("AGENT_COMPLETENESS_MODEL", "") or TEXT_MODEL
AGENT_BRAIN_MODEL = os.getenv("AGENT_BRAIN_MODEL", "") or TEXT_MODEL
AGENT_RFI_MODEL = os.getenv("AGENT_RFI_MODEL", "") or TEXT_MODEL
# Agent-mode hard scope limits. These are intentionally independent of Classic
# batching so Agent workers can never grow into whole-set reasoning calls.
AGENT_LINK_MAX_ASSERTIONS = int(os.getenv("AGENT_LINK_MAX_ASSERTIONS", "60"))
AGENT_CLUSTER_MAX_ASSERTIONS = int(os.getenv("AGENT_CLUSTER_MAX_ASSERTIONS", "24"))
AGENT_CONFLICT_MAX_IMAGES = int(os.getenv("AGENT_CONFLICT_MAX_IMAGES", "6"))
AGENT_CODE_BATCH_SIZE = int(os.getenv("AGENT_CODE_BATCH_SIZE", "60"))
AGENT_BRAIN_MAX_TOKENS = int(os.getenv("AGENT_BRAIN_MAX_TOKENS", "16384"))
AGENT_LINK_CONCURRENCY = int(os.getenv("AGENT_LINK_CONCURRENCY", "4"))
AGENT_CONFLICT_CONCURRENCY = int(os.getenv("AGENT_CONFLICT_CONCURRENCY", "4"))
AGENT_SPECIALIST_CONCURRENCY = int(os.getenv("AGENT_SPECIALIST_CONCURRENCY", "4"))
AGENT_RFI_CONCURRENCY = int(os.getenv("AGENT_RFI_CONCURRENCY", "4"))
# -- Review focus toggles -------------------------------------------
# ENABLE_CODE_REVIEW gates the code/ADA/jurisdiction review path in BOTH
# pipelines. Default OFF: the product's focus is drawing-integrity and
# cross-discipline coordination, not code/accessibility compliance. When
# False the CodeAgent wave (agent) and the Code/ADA stage (classic) are
# skipped entirely, by_stage.code reports 0, and nothing in the ADA corpus
# or jurisdiction meta is deleted so the path can be re-enabled with one env
# flag. Set ENABLE_CODE_REVIEW=1 to restore code/ADA findings.
ENABLE_CODE_REVIEW = _flag("ENABLE_CODE_REVIEW", "false")
# Per-sheet Drawing Integrity QA wave (agent + classic). This is the
# drawing-focused pass: it reads ONE sheet's own objects + image + text layer
# and flags problems internal to that sheet -- dangling detail/callout/keynote
# references, schedule-vs-plan or legend disagreements on the same sheet,
# dimension strings that do not sum, missing title-block/scale/north-arrow,
# and notes that contradict each other. It complements (does not replace) the
# cross-sheet conflict critic. Default ON.
ENABLE_DRAWING_INTEGRITY = _flag("ENABLE_DRAWING_INTEGRITY", "true")
AGENT_INTEGRITY_MODEL = os.getenv("AGENT_INTEGRITY_MODEL", "") or MODEL
AGENT_INTEGRITY_CONCURRENCY = int(os.getenv("AGENT_INTEGRITY_CONCURRENCY", "4"))
AGENT_INTEGRITY_MAX_IMAGES = int(os.getenv("AGENT_INTEGRITY_MAX_IMAGES", "1"))
AGENT_INTEGRITY_MAX_ASSERTIONS = int(os.getenv("AGENT_INTEGRITY_MAX_ASSERTIONS", "80"))
INTEGRITY_MAX_TOKENS = int(os.getenv("INTEGRITY_MAX_TOKENS", "16384"))
# Skip sheets with fewer than this many extracted objects -- too sparse for a
# meaningful internal-consistency pass (avoids burning a call on near-empty pages).
INTEGRITY_MIN_ASSERTIONS = int(os.getenv("INTEGRITY_MIN_ASSERTIONS", "3"))
# Wave 5b evidence verification (vision fact-check of cited sheet text)
AGENT_VERIFY_MODEL = os.getenv("AGENT_VERIFY_MODEL", "") or MODEL
AGENT_VERIFY_CONCURRENCY = int(os.getenv("AGENT_VERIFY_CONCURRENCY", "4"))
AGENT_VERIFY_MAX_CHECKS = int(os.getenv("AGENT_VERIFY_MAX_CHECKS", "20"))
AGENT_VERIFY_SEVERITIES = {
s.strip().lower()
for s in os.getenv("AGENT_VERIFY_SEVERITIES", "critical,high").split(",")
if s.strip()
}
AGENT_VERIFY_REASONING_EFFORT = os.getenv("AGENT_VERIFY_REASONING_EFFORT", "low").strip()
VERIFY_MAX_TOKENS = int(os.getenv("VERIFY_MAX_TOKENS", "8192"))
# Wave 6.5 Brain-directed clarification. After the Brain merge, the Brain may
# name findings it is unsure about and emit typed clarification requests; v1
# executes verify_evidence requests by routing them back through the wave-5b
# EvidenceVerifierAgent (fresh page images + hi-DPI evidence crops + text-layer
# oracle). Bounded: one planning call, at most BRAIN_CLARIFY_MAX_REQUESTS
# verifications, a single iteration. Reuses AGENT_VERIFY_* / VERIFY_* knobs for
# the verification calls. Default ON.
ENABLE_BRAIN_CLARIFY = _flag("ENABLE_BRAIN_CLARIFY", "true")
BRAIN_CLARIFY_MAX_REQUESTS = int(os.getenv("BRAIN_CLARIFY_MAX_REQUESTS", "8"))
BRAIN_CLARIFY_MAX_TOKENS = int(os.getenv("BRAIN_CLARIFY_MAX_TOKENS", "4096"))
# -- Text-layer grounding (deterministic PDF text layer via PyMuPDF) ----
# The vector text layer is extracted once per job and grounds the extractor,
# rescues misquoted-but-real values in the grounding guard, and serves the
# wave-5b verifier as a text oracle plus high-DPI evidence crops.
TEXT_LAYER_ENABLED = os.getenv("TEXT_LAYER_ENABLED", "true").strip().lower() in ("1", "true", "yes")
TEXT_LAYER_MIN_CHARS = int(os.getenv("TEXT_LAYER_MIN_CHARS", "20")) # below this per page -> no text layer
TEXT_LAYER_MAX_CHARS = int(os.getenv("TEXT_LAYER_MAX_CHARS", "12000")) # cap per sheet in extractor prompt
VERIFY_TEXT_MAX_CHARS = int(os.getenv("VERIFY_TEXT_MAX_CHARS", "8000"))# cap of excerpt in verify scope
VERIFY_HI_DPI_CROPS = os.getenv("VERIFY_HI_DPI_CROPS", "true").strip().lower() in ("1", "true", "yes")
VERIFY_CROP_DPI = int(os.getenv("VERIFY_CROP_DPI", "300"))
VERIFY_CROP_MARGIN_PTS = int(os.getenv("VERIFY_CROP_MARGIN_PTS", "36"))# padding around evidence bbox (PDF points)
# Agent-mode human-review gate. When on (default), Agent runs stop after the
# Brain merge and wait for human decisions before RFIs/final report/email go
# out. AGENT_REVIEW_AUDIT_SAMPLE caps how many clean clusters get added to the
# queue as non-blocking spot-checks. REVIEW_AGGREGATE_INCLUDE_TEXT controls
# whether future cross-job review feedback aggregation may include verbatim
# source_text/images/comments (off by default = privacy-preserving).
AGENT_REQUIRE_REVIEW = os.getenv("AGENT_REQUIRE_REVIEW", "true").strip().lower() in ("1", "true", "yes")
AGENT_REVIEW_AUDIT_SAMPLE = int(os.getenv("AGENT_REVIEW_AUDIT_SAMPLE", "5"))
# NOTE: currently unwired - reserved for future cross-job aggregation tooling.
REVIEW_AGGREGATE_INCLUDE_TEXT = os.getenv("REVIEW_AGGREGATE_INCLUDE_TEXT", "false").strip().lower() in ("1", "true", "yes")
# -- Review chat (ask-the-run Q&A on the review screen) --------------
# A read-only explainer: it answers "why did the run decide X?" from the job's
# own artifacts and never mutates findings, decisions, or code. Every turn is
# appended to <out_dir>/review/chat_log.jsonl AND to the cross-job feedback
# store (REVIEW_FEEDBACK_DIR) so answers are available to future prompt priors.
# HISTORY_TURNS caps how much of a thread is replayed into the prompt;
# LOG_LINES caps how many job.log lines are searched into the context bundle.
ENABLE_REVIEW_CHAT = _flag("ENABLE_REVIEW_CHAT", "true")
REVIEW_CHAT_MODEL = os.getenv("REVIEW_CHAT_MODEL", "") or TEXT_MODEL
REVIEW_CHAT_MAX_TOKENS = int(os.getenv("REVIEW_CHAT_MAX_TOKENS", "4096"))
REVIEW_CHAT_HISTORY_TURNS = int(os.getenv("REVIEW_CHAT_HISTORY_TURNS", "6"))
REVIEW_CHAT_LOG_LINES = int(os.getenv("REVIEW_CHAT_LOG_LINES", "40"))
REVIEW_CHAT_MAX_QUESTION_CHARS = int(os.getenv("REVIEW_CHAT_MAX_QUESTION_CHARS", "2000"))
# Cross-job feedback store: where review decisions and chat turns accumulate so
# a future run can be primed with "what humans corrected last time". Job-local
# artifacts stay the source of truth; this is the append-only roll-up.
# (OUTPUT_DIR is defined further down; keep this in sync with it.)
REVIEW_FEEDBACK_DIR = os.getenv("REVIEW_FEEDBACK_DIR", "") or os.path.join(
_BASE_DIR, "outputs", "_feedback")
# -- Hybrid (local text LLM) ----------------------------------------
# Optional OpenAI-compatible local endpoint (e.g. a vLLM box) for the text-only
# QAQC stages. Vision stages ALWAYS use OpenRouter. The user picks hybrid per
# job in the UI; these say WHERE local is + the toggle's default state. Empty
# LOCAL_BASE_URL = hybrid disabled (falls back to OpenRouter).
LOCAL_BASE_URL = os.getenv("LOCAL_BASE_URL", "")
LOCAL_API_KEY = os.getenv("LOCAL_API_KEY", "local")
LOCAL_TEXT_MODEL = os.getenv("LOCAL_TEXT_MODEL", "")
HYBRID_DEFAULT = os.getenv("HYBRID_DEFAULT", "false").strip().lower() in ("1", "true", "yes")
# -- Pipeline -------------------------------------------------------
PDF_DPI = int(os.getenv("PDF_DPI", "100"))
MAX_PAGES = int(os.getenv("MAX_PAGES", "60"))
MAX_DIMENSION = int(os.getenv("MAX_DIMENSION", "2400")) # px cap on the long edge
LLM_TIMEOUT = int(os.getenv("LLM_TIMEOUT", "180")) # seconds per call
# Gemini 2.5 Pro counts thinking tokens against max_tokens, so the visible
# JSON budget is well under this number on dense sheets. 65536 is the model's
# output ceiling - give thinking all the room it wants so visible JSON never
# truncates; the thinking budget itself is capped separately below.
EXTRACT_MAX_TOKENS = int(os.getenv("EXTRACT_MAX_TOKENS", "65536"))
# Reasoning effort for the per-sheet extractor (OpenRouter reasoning knob).
# Extraction is perceptive, not deliberative - "low" keeps thinking tokens
# from eating the output budget. Empty string disables the parameter.
EXTRACT_REASONING_EFFORT = os.getenv("EXTRACT_REASONING_EFFORT", "low").strip()
# Hard thinking-token budget for the extractor (OpenRouter reasoning
# max_tokens -> Gemini thinking_budget). "low" effort alone still let Gemini
# burn ~25k thinking tokens per sheet (job 98194fa8d215); a hard cap forces
# the budget into visible output. 0 disables -> falls back to the effort knob.
# Mutually exclusive with effort when set (OpenRouter rejects both together).
EXTRACT_REASONING_MAX_TOKENS = int(os.getenv("EXTRACT_REASONING_MAX_TOKENS", "2048"))
# Coverage-driven extraction retry ladder. After the vision pass, the fraction
# of meaningful text-layer lines represented in extracted objects is measured;
# below EXTRACT_COVERAGE_FLOOR the page climbs the ladder: text-only
# structuring pass (rung 2), then deterministic text-layer fallback stubs
# (rung 3) so no text-bearing page goes dark.
EXTRACT_COVERAGE_FLOOR = float(os.getenv("EXTRACT_COVERAGE_FLOOR", "0.6"))
EXTRACT_TEXT_RETRY_ENABLED = _flag("EXTRACT_TEXT_RETRY_ENABLED", "true")
EXTRACT_FALLBACK_ENABLED = _flag("EXTRACT_FALLBACK_ENABLED", "true")
EXTRACT_FALLBACK_MAX_OBJECTS = int(os.getenv("EXTRACT_FALLBACK_MAX_OBJECTS", "200"))
REASON_MAX_TOKENS = int(os.getenv("REASON_MAX_TOKENS", "4096"))
# -- QAQC stage knobs (Stages 0-1, 3, 6-11) -------------------------
# Token caps per stage. Most are single whole-set calls, so they need more
# headroom than a per-cluster reason call.
JURISDICTION_MAX_TOKENS = int(os.getenv("JURISDICTION_MAX_TOKENS", "4096"))
SHEET_INDEX_MAX_TOKENS = int(os.getenv("SHEET_INDEX_MAX_TOKENS", "16384"))
NORMALIZE_MAX_TOKENS = int(os.getenv("NORMALIZE_MAX_TOKENS", "16384"))
QAQC_MAX_TOKENS = int(os.getenv("QAQC_MAX_TOKENS", "16384"))
CODE_MAX_TOKENS = int(os.getenv("CODE_MAX_TOKENS", "16384"))
CONSTRUCT_MAX_TOKENS = int(os.getenv("CONSTRUCT_MAX_TOKENS", "16384"))
VALIDATE_MAX_TOKENS = int(os.getenv("VALIDATE_MAX_TOKENS", "16384"))
RISK_MAX_TOKENS = int(os.getenv("RISK_MAX_TOKENS", "16384"))
RFI_MAX_TOKENS = int(os.getenv("RFI_MAX_TOKENS", "16384"))
# Stage 4 clustering engine: "llm" (semantic, fuzzy matches + catches more) or
# "deterministic" (clusterer.py, free/reproducible). Swap via env to A/B.
CLUSTERER = os.getenv("CLUSTERER", "llm").strip().lower()
CLUSTER_MAX_TOKENS = int(os.getenv("CLUSTER_MAX_TOKENS", "16384"))
CLUSTER_MAX = int(os.getenv("CLUSTER_MAX", "120"))
# Disk-backed LLM response cache for the testing loop. When on, identical calls
# (same model/prompt/images/params) replay the saved response at zero API cost,
# so re-running a set only pays for stages whose input actually changed. Off by
# default so production never serves stale results. Clear: rm -rf the dir.
LLM_CACHE = os.getenv("LLM_CACHE", "false").strip().lower() in ("1", "true", "yes")
LLM_CACHE_DIR = os.getenv("LLM_CACHE_DIR", os.path.join(_BASE_DIR, ".llm_cache"))
# Verbose LLM observability. Per call, one line lands in the job log (model,
# backend, prompt size, response size, parsed-item counts, per-call cost) and
# the full request/response is dumped to <out_dir>/llm_raw/ (base64 image
# payloads excluded; image count recorded instead) so missed or hallucinated
# items can be traced back to exactly what the model saw and returned.
LLM_VERBOSE = os.getenv("LLM_VERBOSE", "true").strip().lower() in ("1", "true", "yes")
LLM_RAW_DUMP = os.getenv("LLM_RAW_DUMP", "true").strip().lower() in ("1", "true", "yes")
# Parallelism (ThreadPoolExecutor workers)
EXTRACT_CONCURRENCY = int(os.getenv("EXTRACT_CONCURRENCY", "4"))
REASON_CONCURRENCY = int(os.getenv("REASON_CONCURRENCY", "4"))
# Batched stages (normalization, per-sheet code/constructability) reuse this.
NORMALIZE_CONCURRENCY = int(os.getenv("NORMALIZE_CONCURRENCY", "4"))
CODE_CONCURRENCY = int(os.getenv("CODE_CONCURRENCY", "4"))
# Max assertions per code-review call. Code review batches sheets to keep each
# call small (avoids the truncation that zeroed out a whole-set call).
CODE_BATCH_SIZE = int(os.getenv("CODE_BATCH_SIZE", "80"))
CONSTRUCT_CONCURRENCY = int(os.getenv("CONSTRUCT_CONCURRENCY", "4"))
# Assertions per batch for the normalization stage.
NORMALIZE_BATCH_SIZE = int(os.getenv("NORMALIZE_BATCH_SIZE", "60"))
# -- App ------------------------------------------------------------
UPLOAD_DIR = os.path.join(_BASE_DIR, "uploads")
OUTPUT_DIR = os.path.join(_BASE_DIR, "outputs")
APP_TITLE = "Conflict Checker"
APP_VERSION = "0.1.0"
# Public base URL used to build the "view results" link in notification
# emails. Set to whatever address users reach this server on (e.g. the
# Tailscale/LAN URL) so the link in the email actually resolves.
APP_BASE_URL = os.getenv("APP_BASE_URL", "https://conchecker.scoutitsystems.com")
# Build identifier baked into the Docker image by CI (sha-<short_sha>, matching
# the image tag). Shown in the site header and /health. "dev" for local runs.
APP_BUILD = os.getenv("APP_BUILD", "dev")
# -- Email / SMTP (optional notification on completion) -------------
# If unset, the app still works; it just logs "SMTP not configured" and
# skips the email. Mirrors IronBid's graceful behavior.
SMTP_HOST = os.getenv("SMTP_HOST", "")
SMTP_PORT = int(os.getenv("SMTP_PORT", "587"))
SMTP_USER = os.getenv("SMTP_USER", "")
SMTP_PASSWORD = os.getenv("SMTP_PASSWORD", "")
SMTP_FROM = os.getenv("SMTP_FROM", "")
SMTP_USE_TLS = _flag("SMTP_USE_TLS", "true")
SMTP_USE_SSL = _flag("SMTP_USE_SSL", "false")