Files
Conflict_Checker/backend/jobs.py
T
woogiandCursor 82a48d99cf Add scoped Agent-mode pipeline as experimental Classic fork.
Wire specialist waves, Brain consolidation, and Classic-compatible reports so Agent mode can run end-to-end via OpenRouter without changing the default Classic path.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-18 14:31:13 +00:00

182 lines
6.7 KiB
Python

"""
jobs.py - Lightweight async job registry for conflict checks.
A conflict check takes minutes, so the HTTP request must not block on it. Each
upload becomes a job that runs on a background thread; the client gets a job_id
immediately and can either poll GET /jobs/{id} or just close the page and wait
for the completion email.
State is in-memory (fine for a single-user tool); the report is also persisted
to outputs/<job_id>/ so results survive a restart even though live status does
not. No external queue/DB.
"""
import os
import time
import uuid
import shutil
import threading
from typing import Dict, Optional
from backend import config
from backend.agents.runner import run_agent_pipeline
from backend.pipeline.runner import run_pipeline
from backend.email_sender import send_conflict_report
_jobs: Dict[str, Dict] = {}
_lock = threading.Lock()
PIPELINE_MODES = {"classic", "agent"}
def _set(job_id: str, **fields) -> None:
with _lock:
_jobs[job_id].update(fields)
def create_job(pdf_path: str, source_filename: str, email: Optional[str] = None,
project_input: Optional[Dict] = None, text_local: bool = False,
pipeline_mode: str = "classic") -> str:
"""Register a job and kick off its background thread. Returns the job_id."""
pipeline_mode = pipeline_mode.strip().lower()
if pipeline_mode not in PIPELINE_MODES:
raise ValueError(f"Unsupported pipeline mode: {pipeline_mode!r}")
# Agent mode v1 is OpenRouter-only.
text_local = bool(text_local and pipeline_mode == "classic")
job_id = uuid.uuid4().hex[:12]
with _lock:
_jobs[job_id] = {
"job_id": job_id,
"status": "queued", # queued -> running -> done | error
"source": source_filename,
"email": email or None,
"project_input": project_input or {},
"text_local": text_local,
"pipeline_mode": pipeline_mode,
"stage": None,
"created_at": time.time(),
"finished_at": None,
"report": None,
"error": None,
}
threading.Thread(target=_run, args=(
job_id, pdf_path, project_input, text_local, pipeline_mode,
),
daemon=True).start()
return job_id
def _run(job_id: str, pdf_path: str, project_input: Optional[Dict] = None,
text_local: bool = False, pipeline_mode: str = "classic") -> None:
out_dir = os.path.join(config.OUTPUT_DIR, job_id)
try:
_set(job_id, status="running")
# Keep a copy of the source PDF so its sheets can be viewed later.
os.makedirs(out_dir, exist_ok=True)
shutil.copy2(pdf_path, os.path.join(out_dir, "source.pdf"))
runner = run_agent_pipeline if pipeline_mode == "agent" else run_pipeline
runner_kwargs = {
"out_dir": out_dir,
"on_stage": lambda name: _set(job_id, stage=name),
"project_input": project_input,
"source_name": _jobs[job_id].get("source"),
}
if pipeline_mode == "classic":
runner_kwargs["text_local"] = text_local
report = runner(pdf_path, **runner_kwargs)
report.setdefault("summary", {})["pipeline_mode"] = pipeline_mode
_set(job_id, status="done", report=report, finished_at=time.time(), stage=None)
_notify(job_id, report, out_dir)
except Exception as e:
print(f"[Jobs] Job {job_id} failed: {e}")
_set(job_id, status="error", error=str(e), finished_at=time.time())
_notify_error(job_id)
finally:
try:
os.remove(pdf_path)
except OSError:
pass
def _notify(job_id: str, report: Dict, out_dir: str) -> None:
email = _jobs[job_id].get("email")
if not email:
return
results_url = f"{config.APP_BASE_URL.rstrip('/')}/?job={job_id}"
attachments = [
os.path.join(out_dir, "report.md"),
os.path.join(out_dir, "conflicts.json"),
os.path.join(out_dir, "validated_issues.json"),
os.path.join(out_dir, "rfis.json"),
]
send_conflict_report(email, report, results_url=results_url, attachments=attachments)
def _notify_error(job_id: str) -> None:
job = _jobs[job_id]
email = job.get("email")
if not email:
return
# Reuse the report mailer with a minimal error-shaped payload.
err_report = {
"source": job.get("source", ""),
"summary": {"conflicts_found": 0, "by_severity": {}, "disciplines": []},
}
try:
from backend.email_sender import _smtp_ready, _send
from email.message import EmailMessage
if not _smtp_ready():
print(f"[Email] SMTP not configured - skipping error notice to {email}")
return
msg = EmailMessage()
msg["Subject"] = f"Conflict Checker - {job.get('source','')} - run FAILED"
msg["From"] = config.SMTP_FROM or config.SMTP_USER
msg["To"] = email
msg.set_content(
"Your conflict check did not complete.\n\n"
f"Drawing set: {job.get('source','')}\n"
f"Error: {job.get('error','unknown')}\n\n"
"Generated by Conflict Checker"
)
_send(msg)
except Exception as e:
print(f"[Email] Failed to send error notice: {e}")
def get_job(job_id: str) -> Optional[Dict]:
"""Public job view. Includes the full report only when done.
Falls back to the on-disk conflicts.json when the job isn't in the
in-memory registry (e.g. after a server restart).
"""
with _lock:
job = _jobs.get(job_id)
if job:
return dict(job)
# Try loading from disk
report_path = os.path.join(config.OUTPUT_DIR, job_id, "conflicts.json")
if not os.path.isfile(report_path):
return None
try:
import json
with open(report_path, encoding="utf-8") as f:
report = json.load(f)
source_pdf = os.path.join(config.OUTPUT_DIR, job_id, "source.pdf")
return {
"job_id": job_id,
"status": "done",
"source": report.get("source", os.path.basename(report_path)),
"email": None,
"project_input": report.get("project_input", {}),
"text_local": report.get("summary", {}).get("text_backend") == "local",
"pipeline_mode": report.get("summary", {}).get("pipeline_mode", "classic"),
"stage": None,
"created_at": os.path.getmtime(source_pdf) if os.path.isfile(source_pdf) else None,
"finished_at": os.path.getmtime(report_path),
"report": report,
"error": None,
}
except Exception as e:
print(f"[Jobs] Failed to load job {job_id} from disk: {e}")
return None