fix(review): lock OCR backend init, warn on rollback fallthrough, zero-page metric
Address PR #858 review round 2: - OcrProcessor backend resolution is now guarded by an anyio.Lock (lazy-init, double-checked) so a burst of concurrent first-OCR calls resolves the backend once instead of each fetching its own gateway M2M token. - The document_tier1_engine=pymupdf rollback now logs a warning when it falls back to the fast processor (no 'structured' registered) instead of silently using the very engine the operator opted out of. - classify_from_text defaults ocr_frac to 0.0 (not 1.0) for a zero-page PDF, so the recorded classification metric is "fast" (no OCR evidence) rather than a misleading "ocr"; the no_text_layer/bad_text_layer flags are gated on having sampled at least one page. New tests: zero-page classify routes fast, rollback-fallback warning. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.8
parent
1634e8adc2
commit
f1272dfe84
@@ -208,13 +208,17 @@ def classify_from_text(
|
||||
mean_quality = (
|
||||
round(sum(p.text_quality for p in pages) / sampled, 3) if sampled else 0.0
|
||||
)
|
||||
ocr_frac = (sum(p.needs_ocr for p in pages) / sampled) if sampled else 1.0
|
||||
# No pages (empty/corrupt PDF) => no OCR evidence => "fast" (the registry's
|
||||
# page_count guard also skips escalation; defaulting to 0.0 keeps the
|
||||
# recorded classification metric accurate rather than a misleading "ocr").
|
||||
ocr_frac = (sum(p.needs_ocr for p in pages) / sampled) if sampled else 0.0
|
||||
|
||||
flags: set[str] = set()
|
||||
if total_chars == 0:
|
||||
flags.add("no_text_layer")
|
||||
elif mean_quality < MIN_TEXT_QUALITY:
|
||||
flags.add("bad_text_layer")
|
||||
if sampled:
|
||||
if total_chars == 0:
|
||||
flags.add("no_text_layer")
|
||||
elif mean_quality < MIN_TEXT_QUALITY:
|
||||
flags.add("bad_text_layer")
|
||||
|
||||
recommended = "ocr" if ocr_frac >= OCR_PAGE_FRACTION else "fast"
|
||||
|
||||
|
||||
@@ -22,6 +22,7 @@ from abc import ABC, abstractmethod
|
||||
from collections.abc import Awaitable, Callable
|
||||
from typing import Any
|
||||
|
||||
import anyio
|
||||
import httpx
|
||||
|
||||
from nextcloud_mcp_server.config import Settings, get_settings
|
||||
@@ -184,6 +185,10 @@ class OcrProcessor(DocumentProcessor):
|
||||
# lifetime is safe.
|
||||
self._backend_resolved = False
|
||||
self._backend: _OcrBackend | None = None
|
||||
# Serialise first-call resolution so a burst of concurrent OCR requests
|
||||
# doesn't each build a backend (and fetch its own M2M token). Lazy-init:
|
||||
# anyio primitives must not be created at import time.
|
||||
self._backend_lock: anyio.Lock | None = None
|
||||
|
||||
@property
|
||||
def name(self) -> str:
|
||||
@@ -209,8 +214,12 @@ class OcrProcessor(DocumentProcessor):
|
||||
) -> ProcessingResult:
|
||||
settings = get_settings()
|
||||
if not self._backend_resolved:
|
||||
self._backend = build_ocr_backend(settings)
|
||||
self._backend_resolved = True
|
||||
if self._backend_lock is None:
|
||||
self._backend_lock = anyio.Lock()
|
||||
async with self._backend_lock:
|
||||
if not self._backend_resolved: # double-checked
|
||||
self._backend = build_ocr_backend(settings)
|
||||
self._backend_resolved = True
|
||||
backend = self._backend
|
||||
if backend is None:
|
||||
logger.warning(
|
||||
|
||||
@@ -203,11 +203,19 @@ class ProcessorRegistry:
|
||||
settings = get_settings()
|
||||
|
||||
if settings.document_tier1_engine == "pymupdf":
|
||||
processor = self._pdf_processor_for_tier(
|
||||
"structured"
|
||||
) or self.find_processor(content_type)
|
||||
processor = self._pdf_processor_for_tier("structured")
|
||||
if processor is None:
|
||||
raise ProcessorError("No PDF processor registered")
|
||||
# The rollback was set to opt OUT of pypdfium2, so falling back
|
||||
# to it (the highest-priority PDF processor) silently would
|
||||
# defeat that intent -- warn loudly.
|
||||
processor = self.find_processor(content_type)
|
||||
if processor is None:
|
||||
raise ProcessorError("No PDF processor registered")
|
||||
logger.warning(
|
||||
"document_tier1_engine=pymupdf but no 'structured' processor "
|
||||
"is registered; falling back to '%s'",
|
||||
processor.name,
|
||||
)
|
||||
return await self._run_processor(
|
||||
processor, content, content_type, filename, options, progress_callback
|
||||
)
|
||||
|
||||
@@ -176,3 +176,12 @@ def test_classify_from_text_empty_routes_ocr():
|
||||
assert c.recommended_tier == "ocr"
|
||||
assert "no_text_layer" in c.flags
|
||||
assert c.total_chars == 0
|
||||
|
||||
|
||||
def test_classify_from_text_no_pages_routes_fast():
|
||||
# An empty/corrupt PDF (no page boundaries) is not OCR evidence -> "fast",
|
||||
# so the recorded classification metric isn't a misleading "ocr".
|
||||
c = clf.classify_from_text("", [])
|
||||
assert c.recommended_tier == "fast"
|
||||
assert c.ocr_page_fraction == pytest.approx(0.0)
|
||||
assert c.flags == set()
|
||||
|
||||
@@ -92,6 +92,20 @@ async def test_engine_rollback_uses_structured(monkeypatch):
|
||||
assert res.processor == "structured"
|
||||
|
||||
|
||||
async def test_engine_rollback_warns_when_no_structured(monkeypatch, caplog):
|
||||
# pymupdf rollback with no structured processor registered: it falls back to
|
||||
# the fast processor but must warn (it silently used what the user opted out
|
||||
# of otherwise).
|
||||
monkeypatch.setattr(reg_mod, "get_settings", lambda: _Settings(engine="pymupdf"))
|
||||
r = _registry((_Fake("fast", "fast"), 20))
|
||||
with caplog.at_level(
|
||||
"WARNING", logger="nextcloud_mcp_server.document_processors.registry"
|
||||
):
|
||||
res = await r.process(b"%PDF-1.7", "application/pdf")
|
||||
assert res.processor == "fast"
|
||||
assert any("no 'structured' processor" in rec.message for rec in caplog.records)
|
||||
|
||||
|
||||
async def test_records_classification(monkeypatch):
|
||||
monkeypatch.setattr(reg_mod, "get_settings", lambda: _Settings())
|
||||
rec = MagicMock()
|
||||
|
||||
Reference in New Issue
Block a user