136 lines
4.5 KiB
Python
136 lines
4.5 KiB
Python
"""text_coverage.py - deterministic extraction-coverage measurement.
|
|
|
|
The coverage guarantee: for any page with a usable text layer, measure how
|
|
much of that layer ended up represented in extracted objects. Pages below
|
|
the floor route into the extraction retry ladder (agents/extractors.py and
|
|
pipeline/extractor.py). fallback_objects() is the last rung: stub objects
|
|
segmented straight from the text layer so no text-bearing page goes dark.
|
|
"""
|
|
|
|
import re
|
|
from typing import Dict, List, Optional
|
|
|
|
MIN_LINE_CHARS = 12
|
|
_TICK_RE = re.compile(r"^[\d\s'\"/.,-]+$")
|
|
_WORD_RE = re.compile(r"[a-z0-9]+")
|
|
|
|
|
|
def _meaningful_lines(text: str) -> List[str]:
|
|
lines = []
|
|
for raw in (text or "").splitlines():
|
|
line = " ".join(raw.split())
|
|
if len(line) < MIN_LINE_CHARS or _TICK_RE.match(line):
|
|
continue
|
|
lines.append(line)
|
|
return lines
|
|
|
|
|
|
def _norm(text: str) -> str:
|
|
return " ".join(_WORD_RE.findall((text or "").lower()))
|
|
|
|
|
|
def text_coverage(page_text: str, objects: List[Dict]) -> Dict:
|
|
"""Fraction of meaningful text-layer lines whose normalized form appears
|
|
in the concatenated normalized source_text of extracted objects."""
|
|
lines = _meaningful_lines(page_text)
|
|
if not lines:
|
|
return {"total_lines": 0, "covered_lines": 0, "ratio": 1.0}
|
|
haystack = " ".join(
|
|
_norm(str(o.get("source_text") or o.get("object_description")
|
|
or o.get("value") or ""))
|
|
for o in objects if isinstance(o, dict)
|
|
)
|
|
covered = sum(1 for ln in lines if _norm(ln) and _norm(ln) in haystack)
|
|
return {
|
|
"total_lines": len(lines),
|
|
"covered_lines": covered,
|
|
"ratio": covered / len(lines) if lines else 1.0,
|
|
}
|
|
|
|
|
|
def segment_text_layer(text: str) -> List[str]:
|
|
"""Segment a page text layer into note-sized blocks."""
|
|
segments: List[str] = []
|
|
buf: List[str] = []
|
|
number_re = re.compile(r"^(\d{1,2}[.)]?|[A-Z]\d{0,2}[.)]?)\s*$")
|
|
|
|
def flush():
|
|
joined = " ".join(buf).strip()
|
|
if len(joined) >= MIN_LINE_CHARS:
|
|
segments.append(joined)
|
|
buf.clear()
|
|
|
|
for raw in (text or "").splitlines():
|
|
line = raw.strip()
|
|
if not line:
|
|
flush()
|
|
continue
|
|
if number_re.match(line):
|
|
flush()
|
|
buf.append(line.rstrip(".)"))
|
|
continue
|
|
buf.append(line)
|
|
if line.endswith(".") and len(" ".join(buf)) > 120:
|
|
flush()
|
|
flush()
|
|
return segments
|
|
|
|
|
|
def fallback_objects(page_text: str, page_number: int,
|
|
max_objects: int = 200) -> List[Dict]:
|
|
"""Last-rung deterministic extraction: one stub object per text segment,
|
|
source_text verbatim from the text layer."""
|
|
objs = []
|
|
for idx, seg in enumerate(segment_text_layer(page_text)[:max_objects]):
|
|
objs.append({
|
|
"object_id": f"p{page_number}-tl{idx}",
|
|
"object_type": "general_note",
|
|
"category": "general",
|
|
"tag": None,
|
|
"name": seg[:80],
|
|
"description": seg,
|
|
"attributes": {},
|
|
"location_key": {},
|
|
"source_text": seg,
|
|
"graphical_basis": None,
|
|
"review_uses": ["code_review", "constructability_review"],
|
|
"confidence": "low",
|
|
"grounding": "text_layer_fallback",
|
|
})
|
|
return objs
|
|
|
|
|
|
def merge_objects(vision_objs: List[Dict], text_objs: List[Dict]) -> List[Dict]:
|
|
"""Union of vision and text-structured objects. Vision results come first
|
|
and are never dropped. Text objects are appended unless their normalized
|
|
source_text is already represented."""
|
|
merged = list(vision_objs or [])
|
|
seen = {_norm(str(o.get("source_text") or ""))
|
|
for o in merged if isinstance(o, dict)}
|
|
seen.discard("")
|
|
for obj in text_objs or []:
|
|
if not isinstance(obj, dict):
|
|
continue
|
|
key = _norm(str(obj.get("source_text") or ""))
|
|
if key and key in seen:
|
|
continue
|
|
seen.add(key)
|
|
merged.append(obj)
|
|
return merged
|
|
|
|
|
|
_SHEET_ID_RE = re.compile(r"\b([A-Z]{1,2}\d{2,3}(?:\.\d+)?)\b")
|
|
|
|
|
|
def recover_sheet_number(page_text: str) -> Optional[str]:
|
|
"""Deterministic sheet id from the text layer: prefer candidates in the
|
|
last ~15% of the page (title block lives at the drawing edge)."""
|
|
text = page_text or ""
|
|
cands = _SHEET_ID_RE.findall(text)
|
|
if not cands:
|
|
return None
|
|
tail = text[int(len(text) * 0.85):]
|
|
for cand in reversed(_SHEET_ID_RE.findall(tail)):
|
|
return cand
|
|
return cands[0]
|