diff --git a/nextcloud_mcp_server/document_processors/escalation.py b/nextcloud_mcp_server/document_processors/escalation.py index 3eba0986..84c37dcf 100644 --- a/nextcloud_mcp_server/document_processors/escalation.py +++ b/nextcloud_mcp_server/document_processors/escalation.py @@ -49,7 +49,7 @@ class EscalationDecision: kind: Literal["hop", "suppressed"] to_tier: str - reason: str # empty_text | low_confidence + reason: Literal["empty_text", "low_confidence"] def next_tier(current: str) -> str | None: diff --git a/nextcloud_mcp_server/vector/processor.py b/nextcloud_mcp_server/vector/processor.py index be570ad8..64b968e1 100644 --- a/nextcloud_mcp_server/vector/processor.py +++ b/nextcloud_mcp_server/vector/processor.py @@ -157,33 +157,35 @@ async def _parse_pdf_tier( decision = registry.evaluate_escalation( result, content, tier, settings, filename=filename ) - if decision is not None and decision.kind == "suppressed": - # The ideal next tier (e.g. ocr) is disabled, so we do NOT hop: index - # this tier's output as terminal and record the would-be escalation - # so operators see the latent demand ("what-if OCR enabled"; #324). - record_document_escalation_suppressed( - tier, decision.to_tier, decision.reason - ) - logger.info( - "Escalation suppressed for %s: %s->%s disabled (reason=%s), " - "indexing at current tier", - filename or "", - tier, - decision.to_tier, - decision.reason, - ) - elif decision is not None: - record_document_escalation(tier, decision.to_tier, decision.reason) - logger.info( - "Escalating %s %s->%s (reason=%s)", - filename or "", - tier, - decision.to_tier, - decision.reason, - ) - raise EscalateError( - from_tier=tier, to_tier=decision.to_tier, reason=decision.reason - ) + if decision is not None: + if decision.kind == "suppressed": + # The ideal next tier (e.g. ocr) is disabled, so we do NOT hop: + # index this tier's output as terminal and record the would-be + # escalation so operators see the latent demand ("what-if OCR + # enabled"; #324). + record_document_escalation_suppressed( + tier, decision.to_tier, decision.reason + ) + logger.info( + "Escalation suppressed for %s: %s->%s disabled (reason=%s), " + "indexing at current tier", + filename or "", + tier, + decision.to_tier, + decision.reason, + ) + else: # "hop" — the Literal kind makes this branch exhaustive. + record_document_escalation(tier, decision.to_tier, decision.reason) + logger.info( + "Escalating %s %s->%s (reason=%s)", + filename or "", + tier, + decision.to_tier, + decision.reason, + ) + raise EscalateError( + from_tier=tier, to_tier=decision.to_tier, reason=decision.reason + ) return result diff --git a/tests/unit/test_registry_tiering.py b/tests/unit/test_registry_tiering.py index 0fa38a82..828dfa73 100644 --- a/tests/unit/test_registry_tiering.py +++ b/tests/unit/test_registry_tiering.py @@ -482,3 +482,19 @@ def test_evaluate_escalation_structured_hop_not_suppressed_when_ocr_off(monkeypa ) decision = r.evaluate_escalation(res, b"%PDF", "fast", _Settings(ocr=False)) assert decision == EscalationDecision("hop", "structured", "low_confidence") + + +def test_evaluate_escalation_terminal_when_ocr_unregistered_and_off(monkeypatch): + """No OCR processor registered at all (not merely disabled) → genuinely + terminal: returns None, NOT a suppressed decision. 'Absent' != 'disabled'.""" + monkeypatch.setattr(reg_mod, "record_document_classification", MagicMock()) + r = _registry((_Fake("fast", "fast"), 20)) # only fast; no ocr processor + res = ProcessingResult( + text="", + metadata={ + "page_count": 1, + "page_boundaries": [{"page": 1, "start_offset": 0, "end_offset": 0}], + }, + processor="fast", + ) + assert r.evaluate_escalation(res, b"%PDF", "fast", _Settings(ocr=False)) is None