diff --git a/backend/text_coverage.py b/backend/text_coverage.py new file mode 100644 index 0000000..508b46e --- /dev/null +++ b/backend/text_coverage.py @@ -0,0 +1,135 @@ +"""text_coverage.py - deterministic extraction-coverage measurement. + +The coverage guarantee: for any page with a usable text layer, measure how +much of that layer ended up represented in extracted objects. Pages below +the floor route into the extraction retry ladder (agents/extractors.py and +pipeline/extractor.py). fallback_objects() is the last rung: stub objects +segmented straight from the text layer so no text-bearing page goes dark. +""" + +import re +from typing import Dict, List, Optional + +MIN_LINE_CHARS = 12 +_TICK_RE = re.compile(r"^[\d\s'\"/.,-]+$") +_WORD_RE = re.compile(r"[a-z0-9]+") + + +def _meaningful_lines(text: str) -> List[str]: + lines = [] + for raw in (text or "").splitlines(): + line = " ".join(raw.split()) + if len(line) < MIN_LINE_CHARS or _TICK_RE.match(line): + continue + lines.append(line) + return lines + + +def _norm(text: str) -> str: + return " ".join(_WORD_RE.findall((text or "").lower())) + + +def text_coverage(page_text: str, objects: List[Dict]) -> Dict: + """Fraction of meaningful text-layer lines whose normalized form appears + in the concatenated normalized source_text of extracted objects.""" + lines = _meaningful_lines(page_text) + if not lines: + return {"total_lines": 0, "covered_lines": 0, "ratio": 1.0} + haystack = " ".join( + _norm(str(o.get("source_text") or o.get("object_description") + or o.get("value") or "")) + for o in objects if isinstance(o, dict) + ) + covered = sum(1 for ln in lines if _norm(ln) and _norm(ln) in haystack) + return { + "total_lines": len(lines), + "covered_lines": covered, + "ratio": covered / len(lines) if lines else 1.0, + } + + +def segment_text_layer(text: str) -> List[str]: + """Segment a page text layer into note-sized blocks.""" + segments: List[str] = [] + buf: List[str] = [] + number_re = re.compile(r"^(\d{1,2}[.)]?|[A-Z]\d{0,2}[.)]?)\s*$") + + def flush(): + joined = " ".join(buf).strip() + if len(joined) >= MIN_LINE_CHARS: + segments.append(joined) + buf.clear() + + for raw in (text or "").splitlines(): + line = raw.strip() + if not line: + flush() + continue + if number_re.match(line): + flush() + buf.append(line.rstrip(".)")) + continue + buf.append(line) + if line.endswith(".") and len(" ".join(buf)) > 120: + flush() + flush() + return segments + + +def fallback_objects(page_text: str, page_number: int, + max_objects: int = 200) -> List[Dict]: + """Last-rung deterministic extraction: one stub object per text segment, + source_text verbatim from the text layer.""" + objs = [] + for idx, seg in enumerate(segment_text_layer(page_text)[:max_objects]): + objs.append({ + "object_id": f"p{page_number}-tl{idx}", + "object_type": "general_note", + "category": "general", + "tag": None, + "name": seg[:80], + "description": seg, + "attributes": {}, + "location_key": {}, + "source_text": seg, + "graphical_basis": None, + "review_uses": ["code_review", "constructability_review"], + "confidence": "low", + "grounding": "text_layer_fallback", + }) + return objs + + +def merge_objects(vision_objs: List[Dict], text_objs: List[Dict]) -> List[Dict]: + """Union of vision and text-structured objects. Vision results come first + and are never dropped. Text objects are appended unless their normalized + source_text is already represented.""" + merged = list(vision_objs or []) + seen = {_norm(str(o.get("source_text") or "")) + for o in merged if isinstance(o, dict)} + seen.discard("") + for obj in text_objs or []: + if not isinstance(obj, dict): + continue + key = _norm(str(obj.get("source_text") or "")) + if key and key in seen: + continue + seen.add(key) + merged.append(obj) + return merged + + +_SHEET_ID_RE = re.compile(r"\b([A-Z]{1,2}\d{2,3}(?:\.\d+)?)\b") + + +def recover_sheet_number(page_text: str) -> Optional[str]: + """Deterministic sheet id from the text layer: prefer candidates in the + last ~15% of the page (title block lives at the drawing edge).""" + text = page_text or "" + cands = _SHEET_ID_RE.findall(text) + if not cands: + return None + tail = text[int(len(text) * 0.85):] + for cand in reversed(_SHEET_ID_RE.findall(tail)): + return cand + return cands[0] diff --git a/tests/test_text_coverage.py b/tests/test_text_coverage.py new file mode 100644 index 0000000..0e10ab1 --- /dev/null +++ b/tests/test_text_coverage.py @@ -0,0 +1,76 @@ +from backend.text_coverage import (text_coverage, segment_text_layer, + fallback_objects, merge_objects, + recover_sheet_number) + + +def test_coverage_full(): + text = "NOTE 1\nALL LUMBER NO. 2 SOUTHERN PINE\nNOTE 2\nUSE 5/8\" PLYWOOD" + objects = [{"source_text": "ALL LUMBER NO. 2 SOUTHERN PINE"}, + {"source_text": "USE 5/8\" PLYWOOD"}] + cov = text_coverage(text, objects) + assert cov["covered_lines"] == 2 + assert cov["total_lines"] == 2 + assert cov["ratio"] == 1.0 + + +def test_coverage_zero_on_empty_objects(): + cov = text_coverage("LINE ALPHA CONTENT\nLINE BETA CONTENT\nLINE GAMMA CONTENT", []) + assert cov["ratio"] == 0.0 and cov["total_lines"] == 3 + + +def test_coverage_ignores_short_and_numeric_noise_lines(): + text = "15\"\n19\"\nA\nB\nREAL NOTE ABOUT FRAMING HERE" + cov = text_coverage(text, [{"source_text": "REAL NOTE ABOUT FRAMING HERE"}]) + assert cov["total_lines"] == 1 and cov["ratio"] == 1.0 + + +def test_segment_notes_and_rows(): + text = "WOOD CONSTRUCTION\n1. \nALL SAWN LUMBER TO BE SOUTHERN PINE.\n2. \nROOF SHEATHING 5/8\" PLYWOOD." + segs = segment_text_layer(text) + assert any("ALL SAWN LUMBER" in s for s in segs) + assert any("ROOF SHEATHING" in s for s in segs) + + +def test_fallback_objects_verbatim_and_stamped(): + objs = fallback_objects("1. \nALL SAWN LUMBER TO BE SOUTHERN PINE.", page_number=8) + assert len(objs) == 1 + assert objs[0]["source_text"] == "1 ALL SAWN LUMBER TO BE SOUTHERN PINE." + assert objs[0]["grounding"] == "text_layer_fallback" + assert objs[0]["confidence"] == "low" + + +def test_merge_objects_keeps_vision_and_unions_text(): + vision = [ + {"source_text": "2X6 WD STUD @ 16\" O.C.", "object_type": "wall"}, + {"source_text": None, "graphical_basis": "light fixture symbol, grid C-4", + "object_type": "lighting_fixture"}, + ] + text = [ + {"source_text": "2X6 WD STUD @ 16\" O.C.", "object_type": "wall"}, + {"source_text": "ALL LUMBER NO. 2 SOUTHERN PINE", "object_type": "general_note"}, + ] + merged = merge_objects(vision, text) + assert len(merged) == 3 + assert any(o.get("graphical_basis") for o in merged) + assert merged[0]["object_type"] == "wall" + + +def test_merge_objects_dedupes_by_normalized_text(): + a = [{"source_text": "RTU-1: 5 TON, 1600 CFM"}] + b = [{"source_text": "rtu 1 5 ton 1600 cfm"}] + assert len(merge_objects(a, b)) == 1 + + +def test_recover_sheet_number_from_title_block(): + text = ("WALL SECTIONS\n...\nSheet Information\nS301\n" + "Issue Date 05.29.26\nProject Number 25177") + assert recover_sheet_number(text) == "S301" + + +def test_recover_sheet_number_none_when_absent(): + assert recover_sheet_number("just some notes about lumber") is None + + +def test_recover_prefers_discipline_pattern_over_dates(): + text = "Issue Date 05.29.26\nProject Number 25177\nA102 REFLECTED CEILING PLAN" + assert recover_sheet_number(text) == "A102"