From cf7209cd85933acb808ed5fe96976afe065c117d Mon Sep 17 00:00:00 2001 From: Chris Coutinho Date: Tue, 16 Jun 2026 20:06:35 +0200 Subject: [PATCH] fix(document-processors): escalate glyph-corrupt PDFs to the structured tier The fast (pypdfium2) extractor can leak raw glyph codes on subset fonts with a broken /ToUnicode CMap. The result scores high on the existing text-quality heuristic -- a uniform glyph/Caesar offset preserves whitespace and token lengths -- yet is unsearchable. The structured (pymupdf) tier extracts the same pages correctly. Add a language-agnostic C0-control-character-ratio signal to the tier-0 classifier that detects this corruption and routes the document to a new `structured` recommended_tier. Wire the fast->structured hop on the inline path and generalise it so a low-quality-but-non-empty layer also tries structured before OCR -- the inline and external ingest modes now follow the full fast->structured->ocr ladder identically. A scanned / no-text-layer document (total_chars == 0) still shortcuts straight to OCR, since a text extractor cannot recover a pure raster. New per-tenant tunable DOCUMENT_GLYPH_CORRUPTION_RATIO (default 0.02); escalation metrics gain a `corrupt_glyphs` reason label. Co-Authored-By: Claude Opus 4.8 (1M context) --- nextcloud_mcp_server/config.py | 12 ++ .../document_processors/classifier.py | 187 ++++++++++++++---- .../document_processors/escalation.py | 11 +- .../document_processors/registry.py | 77 +++++++- nextcloud_mcp_server/observability/metrics.py | 4 +- tests/unit/test_config.py | 24 +++ tests/unit/test_doc_classifier.py | 60 ++++++ tests/unit/test_registry_tiering.py | 91 ++++++++- 8 files changed, 412 insertions(+), 54 deletions(-) diff --git a/nextcloud_mcp_server/config.py b/nextcloud_mcp_server/config.py index 0265b04a..ad9531e4 100644 --- a/nextcloud_mcp_server/config.py +++ b/nextcloud_mcp_server/config.py @@ -171,6 +171,12 @@ _DEFAULTS: dict[str, Any] = { "document_ocr_page_fraction": 0.5, "document_ocr_min_page_chars": 16, "document_ocr_detect_scanned": True, + # Tier-0 glyph-corruption trigger. When the fast (pypdfium2) extraction's + # doc-level C0-control-char ratio exceeds this, the text layer is treated as + # glyph-corrupt (a broken /ToUnicode mapping leaking raw glyph codes) and the + # doc escalates fast->structured (pymupdf re-extracts it correctly -- no OCR). + # 0 disables. Clean docs sit ~0; affected PDFs measured ~1-11% in testing. + "document_glyph_corruption_ratio": 0.02, # OCR backend request timeout (seconds). Slow scanned newspapers can take # 20-60s; raise/lower per tenant. Configurable so a tenant isn't stuck with # the 180s default when its gateway has its own shorter ceiling. @@ -384,6 +390,7 @@ _dynaconf = Dynaconf( Validator("DOCUMENT_OCR_MIN_TEXT_QUALITY", gte=0, lte=1), Validator("DOCUMENT_OCR_PAGE_FRACTION", gte=0, lte=1), Validator("DOCUMENT_OCR_MIN_PAGE_CHARS", gte=0), + Validator("DOCUMENT_GLYPH_CORRUPTION_RATIO", gte=0, lte=1), # Non-negative Validator("DOCUMENT_CHUNK_OVERLAP", gte=0), # Non-empty strings @@ -896,6 +903,10 @@ class Settings: document_ocr_page_fraction: float = 0.5 document_ocr_min_page_chars: int = 16 document_ocr_detect_scanned: bool = True + # Tier-0 glyph-corruption trigger: doc-level C0-control-char ratio above which + # the fast (pypdfium2) text layer is treated as glyph-corrupt and escalated + # fast->structured (pymupdf). 0 disables. See classifier._control_char_ratio. + document_glyph_corruption_ratio: float = 0.02 # Observability settings metrics_enabled: bool = True @@ -1532,6 +1543,7 @@ def get_settings() -> Settings: "document_ocr_page_fraction": "DOCUMENT_OCR_PAGE_FRACTION", "document_ocr_min_page_chars": "DOCUMENT_OCR_MIN_PAGE_CHARS", "document_ocr_detect_scanned": "DOCUMENT_OCR_DETECT_SCANNED", + "document_glyph_corruption_ratio": "DOCUMENT_GLYPH_CORRUPTION_RATIO", # Observability settings "metrics_enabled": "METRICS_ENABLED", "metrics_port": "METRICS_PORT", diff --git a/nextcloud_mcp_server/document_processors/classifier.py b/nextcloud_mcp_server/document_processors/classifier.py index 720f679c..54d99414 100644 --- a/nextcloud_mcp_server/document_processors/classifier.py +++ b/nextcloud_mcp_server/document_processors/classifier.py @@ -27,9 +27,10 @@ Two entry points: Recommended tier: * ``ocr`` -- scanned / no-usable-text-layer (route to tier 3, when enabled) + * ``structured`` -- a text layer that is present but glyph-corrupt (the fast + extractor leaked raw glyph codes; high C0-control-char ratio). A different + in-cluster extractor (the pymupdf ``structured`` tier) recovers it -- no OCR. * ``fast`` -- a usable digital text layer (stay on tier 1) - -``structured`` (tier 2 / docling) is a separate service, not produced here. """ import logging @@ -56,9 +57,20 @@ MIN_TEXT_QUALITY = 0.5 OCR_PAGE_FRACTION = 0.5 # A page with fewer extracted chars than this has effectively no text layer. MIN_PAGE_CHARS = 16 +# Doc-level control-character ratio above which the text layer is treated as +# glyph-corrupt and routed to the ``structured`` (pymupdf) tier, which re-extracts +# such PDFs correctly. Kept in sync with the DOCUMENT_GLYPH_CORRUPTION_RATIO +# setting default (the registry passes the per-tenant value). See +# ``_control_char_ratio``. +GLYPH_CORRUPTION_RATIO = 0.02 _WORD_RE = re.compile(r"\S+") +# Whitespace control characters that legitimately appear in extracted text +# (tab / newline / carriage-return / form-feed / vertical-tab). Every OTHER C0 +# control char is a corruption signal -- see ``_control_char_ratio``. +_TEXT_WHITESPACE_CONTROLS = frozenset("\t\n\r\f\v") + @dataclass class PageSignals: @@ -67,6 +79,7 @@ class PageSignals: image_coverage: float # 0..1 of page area covered by images text_quality: float # 0..1; low = mashed/space-less/garbage layer needs_ocr: bool # scanned or unusable text layer + control_ratio: float = 0.0 # 0..1; high = corrupt/glyph-leak text layer @dataclass @@ -76,10 +89,11 @@ class DocClassification: total_chars: int mean_text_quality: float ocr_page_fraction: float # fraction of sampled pages flagged needs_ocr - recommended_tier: str # "fast" | "ocr" + recommended_tier: str # "fast" | "structured" | "ocr" + mean_control_ratio: float = 0.0 # doc-level C0-control-char ratio (glyph-leak) flags: set[str] = field( default_factory=set - ) # scanned | bad_text_layer | image_heavy + ) # scanned | bad_text_layer | image_heavy | corrupt_glyphs pages: list[PageSignals] = field(default_factory=list) @@ -116,6 +130,25 @@ def _text_quality(text: str) -> float: return round(ws_score * len_score * overlong_score * merge_score, 3) +def _control_char_ratio(text: str) -> float: + """Fraction of C0 control characters (excluding whitespace controls) in ``text``. + + Near 0 for clean text in ANY script; elevated when the extractor leaked raw + glyph codes instead of Unicode -- the broken-/ToUnicode failure mode where a + subset font's character codes are returned uniformly offset (e.g. "WKH" for + "THE"). This is the language-agnostic counterpart to :func:`_text_quality`: a + uniform glyph/Caesar offset preserves whitespace and token lengths (so every + ``_text_quality`` factor scores it ~1.0), but it litters the text with C0 + controls -- digits/punctuation map to bytes below 0x20 -- which clean prose + never contains. Unlike a dictionary or stop-word probe it makes no assumption + about the document's language. + """ + if not text: + return 0.0 + bad = sum(1 for c in text if ord(c) < 0x20 and c not in _TEXT_WHITESPACE_CONTROLS) + return bad / len(text) + + def _sample_indices(page_count: int) -> list[int]: if page_count <= MAX_SAMPLED_PAGES: return list(range(page_count)) @@ -143,6 +176,56 @@ def _page_image_coverage(page: Any) -> float: return min(img_area / page_area, 1.0) +def _route_from_signals( + *, + total_chars: int, + ocr_frac: float, + mean_quality: float, + control_ratio: float, + image_heavy: bool, + page_fraction: float, + min_text_quality: float, + glyph_corruption_ratio: float, +) -> tuple[set[str], str]: + """Shared flag-set + recommended-tier decision for both classifier paths. + + Routing precedence (cheapest correct fix first): + 1. scanned / no text layer (``ocr_frac >= fraction`` AND ``total_chars == 0``) + -> ``"ocr"`` + 2. glyph-corrupt text layer (``control_ratio > glyph_corruption_ratio``) + -> ``"structured"``. pypdfium2 leaked glyph codes; the pymupdf + ``structured`` tier re-extracts these correctly, so no paid OCR is + needed. The registry re-classifies the structured output, so a doc that + is ALSO partly scanned can still escalate to OCR from there. + 3. junk/mashed text layer (``ocr_frac >= fraction``) -> ``"ocr"`` + 4. otherwise -> ``"fast"`` + + Flags are diagnostic and independent of the verdict (e.g. ``image_heavy`` + fires on ANY image-heavy page; the OCR route needs a page FRACTION). + """ + glyph_corrupt = total_chars > 0 and control_ratio > glyph_corruption_ratio + + flags: set[str] = set() + if ocr_frac >= page_fraction and total_chars == 0: + flags.add("scanned") + elif ocr_frac >= page_fraction and mean_quality < min_text_quality: + flags.add("bad_text_layer") + if glyph_corrupt: + flags.add("corrupt_glyphs") + if image_heavy: + flags.add("image_heavy") + + if ocr_frac >= page_fraction and total_chars == 0: + recommended = "ocr" + elif glyph_corrupt: + recommended = "structured" + elif ocr_frac >= page_fraction: + recommended = "ocr" + else: + recommended = "fast" + return flags, recommended + + def classify_pdf(content: bytes) -> DocClassification: """Classify a PDF from its bytes. @@ -171,7 +254,14 @@ def classify_pdf(content: bytes) -> DocClassification: # raises the diagnostic image_heavy flag below. needs_ocr = quality < MIN_TEXT_QUALITY or len(text.strip()) < MIN_PAGE_CHARS pages.append( - PageSignals(n, len(text), round(coverage, 3), quality, needs_ocr) + PageSignals( + n, + len(text), + round(coverage, 3), + quality, + needs_ocr, + round(_control_char_ratio(text), 4), + ) ) sampled = len(pages) @@ -180,25 +270,24 @@ def classify_pdf(content: bytes) -> DocClassification: round(sum(p.text_quality for p in pages) / sampled, 3) if sampled else 0.0 ) ocr_frac = (sum(p.needs_ocr for p in pages) / sampled) if sampled else 0.0 + # Char-weighted doc-level control-char ratio (p.control_ratio * char_count is + # the per-page bad-char count). The glyph-leak signal -- see _control_char_ratio. + control_ratio = ( + sum(p.control_ratio * p.char_count for p in pages) / total_chars + if total_chars + else 0.0 + ) - # Flags are diagnostic signals, intentionally independent of the routing - # verdict: image_heavy fires if ANY page is image-heavy, while the OCR route - # needs a FRACTION of pages (OCR_PAGE_FRACTION). So a mostly-digital doc with - # one full-page photo is flagged image_heavy yet still routes "fast" -- the - # flag_total{image_heavy} count is expected to exceed classified{ocr}. - flags: set[str] = set() - if any(p.image_coverage >= IMAGE_HEAVY_THRESHOLD for p in pages): - flags.add("image_heavy") - if ( - ocr_frac >= OCR_PAGE_FRACTION - and total_chars - and mean_quality < MIN_TEXT_QUALITY - ): - flags.add("bad_text_layer") - if ocr_frac >= OCR_PAGE_FRACTION and total_chars == 0: - flags.add("scanned") - - recommended = "ocr" if ocr_frac >= OCR_PAGE_FRACTION else "fast" + flags, recommended = _route_from_signals( + total_chars=total_chars, + ocr_frac=ocr_frac, + mean_quality=mean_quality, + control_ratio=control_ratio, + image_heavy=any(p.image_coverage >= IMAGE_HEAVY_THRESHOLD for p in pages), + page_fraction=OCR_PAGE_FRACTION, + min_text_quality=MIN_TEXT_QUALITY, + glyph_corruption_ratio=GLYPH_CORRUPTION_RATIO, + ) return DocClassification( page_count=page_count, @@ -207,6 +296,7 @@ def classify_pdf(content: bytes) -> DocClassification: mean_text_quality=mean_quality, ocr_page_fraction=round(ocr_frac, 3), recommended_tier=recommended, + mean_control_ratio=round(control_ratio, 4), flags=flags, pages=pages, ) @@ -239,6 +329,7 @@ def classify_from_text( min_text_quality: float = MIN_TEXT_QUALITY, min_page_chars: int = MIN_PAGE_CHARS, page_fraction: float = OCR_PAGE_FRACTION, + glyph_corruption_ratio: float = GLYPH_CORRUPTION_RATIO, image_coverage: list[float] | None = None, ) -> DocClassification: """Classify from text already extracted by tier-1 -- no PDF re-open by default. @@ -246,9 +337,13 @@ def classify_from_text( The hot-path classifier. A page is OCR-worthy when its text is near-empty (``< min_page_chars``) or its text-quality is junk (``< min_text_quality`` -- the word-merging signal). The doc recommends ``ocr`` once - ``ocr_frac >= page_fraction``. Thresholds are passed in by the registry from - per-tenant settings. ``image_coverage`` (when supplied) only feeds the - ``image_heavy`` diagnostic flag -- it does NOT route (see module docstring). + ``ocr_frac >= page_fraction``. A doc whose text layer is present but + glyph-corrupt (doc-level C0-control-char ratio ``> glyph_corruption_ratio``, + the broken-/ToUnicode failure mode) instead recommends ``structured`` -- the + pymupdf tier re-extracts it correctly, no OCR needed. Thresholds are passed in + by the registry from per-tenant settings. ``image_coverage`` (when supplied) + only feeds the ``image_heavy`` diagnostic flag -- it does NOT route (see + module docstring). ``page_boundaries`` are ``{page, start_offset, end_offset}`` indexing into ``full_text``; ``image_coverage[i]`` (if given) aligns with the i-th boundary. @@ -293,7 +388,14 @@ def classify_from_text( # ``cov`` still feeds the diagnostic ``image_heavy`` flag below. needs_ocr = len(seg.strip()) < min_page_chars or quality < min_text_quality pages.append( - PageSignals(b["page"], len(seg), round(cov, 3), quality, needs_ocr) + PageSignals( + b["page"], + len(seg), + round(cov, 3), + quality, + needs_ocr, + round(_control_char_ratio(seg), 4), + ) ) sampled = len(pages) @@ -305,23 +407,21 @@ def classify_from_text( # page_count guard also skips escalation; defaulting to 0.0 keeps the # recorded classification metric accurate rather than a misleading "ocr"). ocr_frac = (sum(p.needs_ocr for p in pages) / sampled) if sampled else 0.0 + # Doc-level (char-weighted) control-char ratio -- the glyph-leak signal that + # routes to the structured tier. Computed over full_text so it is robust to + # boundary edge cases. + control_ratio = _control_char_ratio(full_text) - # Flags gated on ocr_frac >= page_fraction (matching classify_pdf): a doc that - # routes "fast" must not carry a junk-layer flag just because a few isolated - # pages are bad -- otherwise the metric diverges from classify_pdf. - flags: set[str] = set() - if sampled and ocr_frac >= page_fraction: - if total_chars == 0: - # "scanned" (not "no_text_layer"): same name + meaning as classify_pdf - # so astrolabe_document_classifier_flag_total isn't split across two - # labels for the empty-text-layer case. - flags.add("scanned") - elif mean_quality < min_text_quality: - flags.add("bad_text_layer") - if any(p.image_coverage >= IMAGE_HEAVY_THRESHOLD for p in pages): - flags.add("image_heavy") - - recommended = "ocr" if ocr_frac >= page_fraction else "fast" + flags, recommended = _route_from_signals( + total_chars=total_chars, + ocr_frac=ocr_frac, + mean_quality=mean_quality, + control_ratio=control_ratio, + image_heavy=any(p.image_coverage >= IMAGE_HEAVY_THRESHOLD for p in pages), + page_fraction=page_fraction, + min_text_quality=min_text_quality, + glyph_corruption_ratio=glyph_corruption_ratio, + ) return DocClassification( page_count=len(page_boundaries), @@ -330,6 +430,7 @@ def classify_from_text( mean_text_quality=mean_quality, ocr_page_fraction=round(ocr_frac, 3), recommended_tier=recommended, + mean_control_ratio=round(control_ratio, 4), flags=flags, pages=pages, ) diff --git a/nextcloud_mcp_server/document_processors/escalation.py b/nextcloud_mcp_server/document_processors/escalation.py index 8bfddf50..d9c3126e 100644 --- a/nextcloud_mcp_server/document_processors/escalation.py +++ b/nextcloud_mcp_server/document_processors/escalation.py @@ -49,7 +49,7 @@ class EscalationDecision: kind: Literal["hop", "suppressed"] to_tier: str - reason: Literal["empty_text", "low_confidence"] + reason: Literal["empty_text", "low_confidence", "corrupt_glyphs"] def next_tier(current: str) -> str | None: @@ -80,10 +80,11 @@ class EscalateError(Exception): the junk text is never indexed, and it must never be swallowed by a broad ``except Exception`` on the indexing path. - ``reason`` uses the existing escalation label vocabulary. This PR raises - ``empty_text`` (scanned / no text layer) and ``low_confidence`` (junk text - layer); ``unsupported`` and ``forced`` are reserved for future callers and - not raised yet. + ``reason`` uses the existing escalation label vocabulary: ``empty_text`` + (scanned / no text layer), ``low_confidence`` (junk text layer), and + ``corrupt_glyphs`` (a usable-looking layer whose extractor leaked raw glyph + codes -- the broken-/ToUnicode case -- recovered by a different in-cluster + extractor); ``unsupported`` and ``forced`` are reserved for future callers. """ def __init__(self, *, from_tier: str, to_tier: str, reason: str) -> None: diff --git a/nextcloud_mcp_server/document_processors/registry.py b/nextcloud_mcp_server/document_processors/registry.py index 33174409..c39fa619 100644 --- a/nextcloud_mcp_server/document_processors/registry.py +++ b/nextcloud_mcp_server/document_processors/registry.py @@ -247,6 +247,63 @@ class ProcessorRegistry: result, content, settings, record=True, filename=filename ) + # Escalate a poor fast extraction up the ladder (fast -> structured -> ocr), + # mirroring the external per-tier path so both modes behave identically. A + # glyph-corrupt layer (the extractor leaked raw glyph codes -- the + # broken-/ToUnicode case) OR a low-quality-but-non-empty layer first tries + # the structured (pymupdf) tier: free, in-cluster, and able to recover both. + # Only a scanned / no-text-layer doc (total_chars == 0) skips structured -- + # a text extractor cannot conjure text from a pure raster -- and drops + # straight to OCR via the gate below. Structured is therefore NOT gated on + # document_ocr_enabled. Its output is re-classified (record=False -- the doc + # was already counted at the fast tier) so a doc that is ALSO partly scanned + # still reaches the OCR gate. + if ( + classification is not None + and classification.page_count > 0 + and ( + classification.recommended_tier == "structured" + or ( + classification.recommended_tier == "ocr" + and classification.total_chars > 0 + ) + ) + ): + structured = self._pdf_processor_for_tier("structured") + if structured is not None: + reason = ( + "corrupt_glyphs" + if classification.recommended_tier == "structured" + else "low_confidence" + ) + record_document_escalation("fast", "structured", reason) + logger.info( + "Escalating %s fast->structured (reason=%s)", + filename or "", + reason, + ) + structured_result = await self._run_processor( + structured, + content, + content_type, + filename, + options, + progress_callback, + escalated=True, + ) + if structured_result.success: + result = structured_result + classification = self._classify_result( + result, content, settings, record=False, filename=filename + ) + else: + logger.warning( + "structured escalation did not succeed for %s (%s); keeping " + "the tier-1 result", + filename or "", + structured_result.metadata.get("parse_failed_reason", "error"), + ) + # NOTE: the suppressed-escalation metric (document_escalation_suppressed_total, # the "what-if OCR" signal; Deck #324) is intentionally NOT emitted on this # inline/memory path -- it is instrumented only on the per-tier external @@ -379,6 +436,7 @@ class ProcessorRegistry: min_text_quality=settings.document_ocr_min_text_quality, min_page_chars=settings.document_ocr_min_page_chars, page_fraction=settings.document_ocr_page_fraction, + glyph_corruption_ratio=settings.document_glyph_corruption_ratio, image_coverage=image_coverage, ) except Exception: @@ -520,6 +578,9 @@ class ProcessorRegistry: - ``total_chars == 0`` (scanned / no text layer) -> target the ``ocr`` tier directly. Text-extractor tiers (``structured``) cannot conjure text from a pure raster scan, so a structured hop would just be wasted. + - glyph-corrupt text layer (``recommended_tier == "structured"``) -> target + the ``structured`` tier; pymupdf re-extracts a broken-/ToUnicode layer + correctly, so OCR is never the target for this case. - low-confidence but non-empty layer -> escalate to the next rung, so a different in-cluster extractor can try before paying for OCR. @@ -538,13 +599,23 @@ class ProcessorRegistry: record=(current_tier == TIER_LADDER[0]), filename=filename, ) - if classification is None or classification.recommended_tier != "ocr": + if classification is None or classification.recommended_tier not in ( + "structured", + "ocr", + ): return None # A zero-page (empty/corrupt) PDF gains nothing from any tier. if classification.page_count <= 0: return None - if classification.total_chars == 0: - minimum: str | None = "ocr" + minimum: str | None + if classification.recommended_tier == "structured": + # Glyph-corrupt text layer (the extractor leaked glyph codes): a + # different in-cluster extractor (the structured/pymupdf tier) recovers + # it -- never pay for OCR here. Target the structured rung specifically. + minimum = "structured" + reason = "corrupt_glyphs" + elif classification.total_chars == 0: + minimum = "ocr" reason = "empty_text" else: minimum = None diff --git a/nextcloud_mcp_server/observability/metrics.py b/nextcloud_mcp_server/observability/metrics.py index 92e87871..c1e64a12 100644 --- a/nextcloud_mcp_server/observability/metrics.py +++ b/nextcloud_mcp_server/observability/metrics.py @@ -272,7 +272,7 @@ document_bytes_processed_total = Counter( document_escalation_total = Counter( "astrolabe_document_escalation_total", "Total document parse escalations between tiers", - # reason: low_confidence | empty_text | unsupported | error | forced + # reason: low_confidence | empty_text | corrupt_glyphs | unsupported | error | forced ["from_tier", "to_tier", "reason"], ) @@ -754,7 +754,7 @@ def record_document_escalation(from_tier: str, to_tier: str, reason: str) -> Non Args: from_tier: Tier that could not satisfactorily parse the document to_tier: Tier the document was escalated to - reason: low_confidence | empty_text | unsupported | error | forced + reason: low_confidence | empty_text | corrupt_glyphs | unsupported | error | forced """ document_escalation_total.labels( from_tier=from_tier, to_tier=to_tier, reason=reason diff --git a/tests/unit/test_config.py b/tests/unit/test_config.py index 4a15d77c..dc23bf8e 100644 --- a/tests/unit/test_config.py +++ b/tests/unit/test_config.py @@ -266,6 +266,30 @@ class TestChunkConfigValidation: _reload_config() assert get_settings().document_max_pdf_size_mb == pytest.approx(12.5) + def test_glyph_corruption_ratio_default_and_env_override(self): + """document_glyph_corruption_ratio defaults to 0.02 and reads its env var. + + Guards the _DEFAULTS-key-must-match-env-var footgun. + """ + assert Settings().document_glyph_corruption_ratio == pytest.approx(0.02) + with patch.dict( + os.environ, {"DOCUMENT_GLYPH_CORRUPTION_RATIO": "0.05"}, clear=True + ): + _reload_config() + assert get_settings().document_glyph_corruption_ratio == pytest.approx(0.05) + + @patch.dict( + os.environ, + {"DOCUMENT_GLYPH_CORRUPTION_RATIO": "1.5"}, + clear=True, + ) + def test_glyph_corruption_ratio_out_of_range_raises_error(self): + """The ratio must be within [0, 1].""" + from dynaconf import ValidationError + + with pytest.raises(ValidationError, match="DOCUMENT_GLYPH_CORRUPTION_RATIO"): + _reload_config() + def test_valid_chunk_settings(self): """Test valid chunk size and overlap configuration.""" settings = Settings( diff --git a/tests/unit/test_doc_classifier.py b/tests/unit/test_doc_classifier.py index b80f7b52..84429c3c 100644 --- a/tests/unit/test_doc_classifier.py +++ b/tests/unit/test_doc_classifier.py @@ -320,3 +320,63 @@ def test_scan_coverage_shorter_than_pages_aligns_without_crash(): assert all(p.needs_ocr is False for p in c.pages) # coverage no longer routes assert "image_heavy" in c.flags # but page 0 still flags image_heavy assert c.recommended_tier == "fast" + + +# --- glyph-corruption signal (broken /ToUnicode -> structured escalation) ----- + +# pypdfium2-style leak: a uniform glyph/Caesar offset turns clean prose into +# alphabetic-but-wrong tokens (normal spacing + token length => HIGH text_quality) +# while digits/punctuation map to C0 control bytes. The control-char ratio is the +# only signal that catches this; _text_quality scores it ~1.0. +_GLYPH_CORRUPT = "WKH \x0f TXLFN \x10 EURZQ \x11 IRA MXPSV \x0f RYHU \x10 GRJ " * 6 + + +def test_control_char_ratio_clean_is_zero(): + assert clf._control_char_ratio("the quick brown fox") == 0.0 + # legitimate whitespace controls (tab/newline/CR/form-feed/vtab) don't count + assert clf._control_char_ratio("a\tb\nc\r\nd\f\ve") == 0.0 + + +def test_control_char_ratio_detects_glyph_leak(): + assert clf._control_char_ratio(_GLYPH_CORRUPT) > clf.GLYPH_CORRUPTION_RATIO + + +def test_clean_text_not_flagged_corrupt(): + txt = "the quick brown fox jumps over the lazy dog " * 3 + c = clf.classify_from_text( + txt, [{"page": 1, "start_offset": 0, "end_offset": len(txt)}] + ) + assert "corrupt_glyphs" not in c.flags + assert c.mean_control_ratio == pytest.approx(0.0) + assert c.recommended_tier == "fast" + + +def test_glyph_corrupt_routes_structured_not_ocr(): + full = _GLYPH_CORRUPT + c = clf.classify_from_text( + full, [{"page": 1, "start_offset": 0, "end_offset": len(full)}] + ) + assert c.recommended_tier == "structured" + assert "corrupt_glyphs" in c.flags + # The point: it is NOT a low-quality signal -- the cipher scores high, so only + # the control-char ratio diverts it (to structured, the free pymupdf re-parse). + assert c.mean_text_quality >= clf.MIN_TEXT_QUALITY + assert c.mean_control_ratio > clf.GLYPH_CORRUPTION_RATIO + + +def test_glyph_corruption_ratio_override_disables_trigger(): + full = _GLYPH_CORRUPT + bounds = [{"page": 1, "start_offset": 0, "end_offset": len(full)}] + # A threshold of 1.0 can never be exceeded => not treated as corrupt => the + # other (high-quality) signals win => fast. + c = clf.classify_from_text(full, bounds, glyph_corruption_ratio=1.0) + assert c.recommended_tier == "fast" + assert "corrupt_glyphs" not in c.flags + + +def test_empty_doc_routes_ocr_not_structured(): + # Precedence: a scanned/empty doc (no text layer) has no control chars to leak, + # so it must stay an OCR case, never structured. + c = clf.classify_from_text("", [{"page": 1, "start_offset": 0, "end_offset": 0}]) + assert c.recommended_tier == "ocr" + assert "corrupt_glyphs" not in c.flags diff --git a/tests/unit/test_registry_tiering.py b/tests/unit/test_registry_tiering.py index 65d08768..c654912d 100644 --- a/tests/unit/test_registry_tiering.py +++ b/tests/unit/test_registry_tiering.py @@ -24,7 +24,9 @@ class _Fake(DocumentProcessor): self, name: str, tier: str, - text: str = "clean text here", + # >= MIN_PAGE_CHARS of clean, whitespace-separated prose so the default + # classifies "fast" (a shorter string trips the near-empty OCR signal). + text: str = "this is clean readable prose text", success=True, pages: int = 1, ): @@ -75,6 +77,7 @@ class _Settings: page_fraction=0.5, min_page_chars=16, detect_scanned=False, + glyph_corruption_ratio=0.02, # Guard off by default so existing tiering tests are unaffected; tests # that exercise the size guard pass an explicit cap. max_pdf_size_mb=0.0, @@ -86,6 +89,7 @@ class _Settings: self.document_ocr_page_fraction = page_fraction self.document_ocr_min_page_chars = min_page_chars self.document_ocr_detect_scanned = detect_scanned + self.document_glyph_corruption_ratio = glyph_corruption_ratio self.document_max_pdf_size_mb = max_pdf_size_mb @@ -244,6 +248,91 @@ async def test_no_ocr_escalation_when_disabled(monkeypatch): assert res.processor == "fast" +# --- glyph-corruption escalation + full-ladder parity ------------------------ + +# A fast-tier text layer that looks like words (normal spacing/token lengths -> +# HIGH text_quality) but leaks C0 control chars: the broken-/ToUnicode signature +# the control-char ratio catches. Decodes to a pangram under a -3 shift. +_GLYPH = "WKH \x0f TXLFN \x10 EURZQ \x11 IRA MXPSV \x0f RYHU \x10 GRJ " * 6 + + +async def test_glyph_corrupt_escalates_fast_to_structured(monkeypatch): + # Not gated on OCR: structured is free + in-cluster, so a glyph-corrupt layer + # escalates fast->structured even with OCR disabled. + monkeypatch.setattr(reg_mod, "get_settings", lambda: _Settings(ocr=False)) + esc = MagicMock() + monkeypatch.setattr(reg_mod, "record_document_escalation", esc) + r = _registry( + (_Fake("fast", "fast", text=_GLYPH), 20), + (_Fake("structured", "structured", text="clean recovered prose text"), 10), + ) + res = await r.process(b"%PDF-1.7", "application/pdf") + assert res.processor == "structured" + esc.assert_called_once_with("fast", "structured", "corrupt_glyphs") + + +async def test_glyph_corrupt_no_structured_stays_fast(monkeypatch): + # No structured processor registered -> nothing to escalate to; keep fast. + monkeypatch.setattr(reg_mod, "get_settings", lambda: _Settings(ocr=False)) + r = _registry((_Fake("fast", "fast", text=_GLYPH), 20)) + res = await r.process(b"%PDF-1.7", "application/pdf") + assert res.processor == "fast" + + +async def test_inline_lowconf_tries_structured_before_ocr(monkeypatch): + # Full-ladder parity with the external path: a junk-but-non-empty fast layer + # tries structured (fast->structured) BEFORE any OCR, even with OCR enabled. + monkeypatch.setattr(reg_mod, "get_settings", lambda: _Settings(ocr=True)) + esc = MagicMock() + monkeypatch.setattr(reg_mod, "record_document_escalation", esc) + r = _registry( + (_Fake("fast", "fast", text="x" * 40), 20), # one long token -> quality ~0 + (_Fake("structured", "structured", text="clean recovered prose text here"), 10), + (_Fake("ocr", "ocr", text="ocr text"), 5), + ) + res = await r.process(b"%PDF-1.7", "application/pdf") + assert res.processor == "structured" + esc.assert_called_once_with("fast", "structured", "low_confidence") + + +async def test_inline_empty_skips_structured_straight_to_ocr(monkeypatch): + # The one intended shortcut: a scanned/no-text-layer doc (total_chars == 0) + # skips structured (it cannot extract text from a raster) and goes to OCR. + monkeypatch.setattr(reg_mod, "get_settings", lambda: _Settings(ocr=True)) + esc = MagicMock() + monkeypatch.setattr(reg_mod, "record_document_escalation", esc) + r = _registry( + (_Fake("fast", "fast", text=""), 20), + (_Fake("structured", "structured", text="should not run"), 10), + (_Fake("ocr", "ocr", text="ocr text"), 5), + ) + res = await r.process(b"%PDF-1.7", "application/pdf") + assert res.processor == "ocr" + esc.assert_called_once_with("fast", "ocr", "empty_text") + + +def test_evaluate_escalation_glyph_corrupt_goes_structured(monkeypatch): + # External path mirrors the inline path: glyph-corrupt -> structured, never OCR. + monkeypatch.setattr(reg_mod, "record_document_classification", MagicMock()) + r = _registry( + (_Fake("fast", "fast"), 20), + (_Fake("structured", "structured"), 10), + (_Fake("ocr", "ocr"), 5), + ) + res = ProcessingResult( + text=_GLYPH, + metadata={ + "page_count": 1, + "page_boundaries": [ + {"page": 1, "start_offset": 0, "end_offset": len(_GLYPH)} + ], + }, + processor="fast", + ) + decision = r.evaluate_escalation(res, b"%PDF", "fast", _Settings(ocr=True)) + assert decision == EscalationDecision("hop", "structured", "corrupt_glyphs") + + # --- Per-tier external path (Deck #323) -------------------------------------