Merge pull request #903 from cbcoutinho/feat/classifier-image-coverage-gate
fix(classifier): make image coverage diagnostic-only, not an OCR routing trigger
This commit is contained in:
@@ -4,17 +4,24 @@ Decides which extraction tier a PDF should escalate to, from cheap signals:
|
|||||||
* text_quality -- is the text layer usable, or mashed/space-less junk? (the
|
* text_quality -- is the text layer usable, or mashed/space-less junk? (the
|
||||||
"Student 147" lesson: a text layer can exist yet be unusable, e.g.
|
"Student 147" lesson: a text layer can exist yet be unusable, e.g.
|
||||||
"01322234567mobile")
|
"01322234567mobile")
|
||||||
* image_coverage -- a page that is mostly a raster image is a scan/photo whose
|
|
||||||
content isn't fully in any text layer.
|
|
||||||
* no text layer -- the strongest OCR signal available from text alone.
|
* no text layer -- the strongest OCR signal available from text alone.
|
||||||
|
|
||||||
|
Routing is on the TEXT signals only: a page escalates to OCR when its text is
|
||||||
|
near-empty or junk-quality. ``image_coverage`` is computed but is a DIAGNOSTIC
|
||||||
|
signal (the ``image_heavy`` flag), NOT a routing trigger: a mostly-raster page
|
||||||
|
whose embedded text is already usable (a scan carrying a clean OCR layer, or a
|
||||||
|
digital page dominated by a figure) gains nothing from re-OCR, so escalating it
|
||||||
|
to the paid OCR tier was wasteful -- on OHR-Bench the coverage trigger drove
|
||||||
|
~45% of escalations. The trade-off: image-only content on an otherwise-clean
|
||||||
|
page (handwriting, stamps, figure text) is no longer force-routed to OCR.
|
||||||
|
|
||||||
Two entry points:
|
Two entry points:
|
||||||
* ``classify_from_text(text, page_boundaries, ...)`` -- the HOT PATH. Routes on
|
* ``classify_from_text(text, page_boundaries, ...)`` -- the HOT PATH. Routes on
|
||||||
text-quality + near-empty pages derived from the tier-1 extraction (~no
|
text-quality + near-empty pages derived from the tier-1 extraction (~no
|
||||||
cost). When OCR + scan-detection are enabled the registry also passes
|
cost). When OCR + scan-detection are enabled the registry also passes
|
||||||
per-page ``image_coverage`` (from ``image_coverage_per_page``) so scans are
|
per-page ``image_coverage`` (from ``image_coverage_per_page``) for the
|
||||||
caught too; that image pass is the only added cost and only OCR-opted-in
|
``image_heavy`` diagnostic flag; that image pass is the only added cost and
|
||||||
tenants pay it. Thresholds come from per-tenant settings.
|
only OCR-opted-in tenants pay it. Thresholds come from per-tenant settings.
|
||||||
* ``classify_pdf(content)`` -- a standalone/diagnostic pass that re-opens the
|
* ``classify_pdf(content)`` -- a standalone/diagnostic pass that re-opens the
|
||||||
PDF and does image-coverage analysis inline. Off the hot path.
|
PDF and does image-coverage analysis inline. Off the hot path.
|
||||||
|
|
||||||
@@ -36,8 +43,10 @@ logger = logging.getLogger(__name__)
|
|||||||
# so the pass stays bounded regardless of page count.
|
# so the pass stays bounded regardless of page count.
|
||||||
MAX_SAMPLED_PAGES = 24
|
MAX_SAMPLED_PAGES = 24
|
||||||
|
|
||||||
# A page counts as "scanned-like" when a raster image covers most of it.
|
# Raster-image coverage above which a page raises the DIAGNOSTIC ``image_heavy``
|
||||||
IMAGE_COVERAGE_SCANNED = 0.80
|
# flag. This is observability only -- it does NOT route to OCR (see module
|
||||||
|
# docstring); routing is on the text signals alone.
|
||||||
|
IMAGE_HEAVY_THRESHOLD = 0.80
|
||||||
# Text-quality score below which the layer is treated as junk (mashed tokens).
|
# Text-quality score below which the layer is treated as junk (mashed tokens).
|
||||||
# Kept in sync with the DOCUMENT_OCR_MIN_TEXT_QUALITY setting default so the
|
# Kept in sync with the DOCUMENT_OCR_MIN_TEXT_QUALITY setting default so the
|
||||||
# module/diagnostic default matches production (the registry always passes the
|
# module/diagnostic default matches production (the registry always passes the
|
||||||
@@ -152,15 +161,15 @@ def classify_pdf(content: bytes) -> DocClassification:
|
|||||||
text = page.get_text("text")
|
text = page.get_text("text")
|
||||||
quality = _text_quality(text)
|
quality = _text_quality(text)
|
||||||
coverage = _page_image_coverage(page)
|
coverage = _page_image_coverage(page)
|
||||||
# OCR-worthy on the same three signals as classify_from_text (kept in
|
# OCR-worthy on TEXT signals only (kept in sync with
|
||||||
# sync so an operator reproducing routing offline gets the pipeline's
|
# classify_from_text): a junk/low-quality text layer (the
|
||||||
# answer): a mostly-raster scan, a junk/low-quality text layer (the
|
# word-merging case) or an effectively empty one. Image coverage is
|
||||||
# word-merging case), or an effectively empty text layer.
|
# deliberately NOT a routing trigger -- a mostly-raster page whose
|
||||||
needs_ocr = (
|
# embedded text is already usable (a scan with a clean OCR layer, or
|
||||||
coverage >= IMAGE_COVERAGE_SCANNED
|
# a digital page dominated by a figure) gains nothing from re-OCR, so
|
||||||
or quality < MIN_TEXT_QUALITY
|
# routing it to the paid OCR tier was wasteful. High coverage still
|
||||||
or len(text.strip()) < MIN_PAGE_CHARS
|
# raises the diagnostic image_heavy flag below.
|
||||||
)
|
needs_ocr = quality < MIN_TEXT_QUALITY or len(text.strip()) < MIN_PAGE_CHARS
|
||||||
pages.append(
|
pages.append(
|
||||||
PageSignals(n, len(text), round(coverage, 3), quality, needs_ocr)
|
PageSignals(n, len(text), round(coverage, 3), quality, needs_ocr)
|
||||||
)
|
)
|
||||||
@@ -178,7 +187,7 @@ def classify_pdf(content: bytes) -> DocClassification:
|
|||||||
# one full-page photo is flagged image_heavy yet still routes "fast" -- the
|
# one full-page photo is flagged image_heavy yet still routes "fast" -- the
|
||||||
# flag_total{image_heavy} count is expected to exceed classified{ocr}.
|
# flag_total{image_heavy} count is expected to exceed classified{ocr}.
|
||||||
flags: set[str] = set()
|
flags: set[str] = set()
|
||||||
if any(p.image_coverage >= IMAGE_COVERAGE_SCANNED for p in pages):
|
if any(p.image_coverage >= IMAGE_HEAVY_THRESHOLD for p in pages):
|
||||||
flags.add("image_heavy")
|
flags.add("image_heavy")
|
||||||
if (
|
if (
|
||||||
ocr_frac >= OCR_PAGE_FRACTION
|
ocr_frac >= OCR_PAGE_FRACTION
|
||||||
@@ -206,16 +215,13 @@ def classify_pdf(content: bytes) -> DocClassification:
|
|||||||
def image_coverage_per_page(content: bytes) -> list[float]:
|
def image_coverage_per_page(content: bytes) -> list[float]:
|
||||||
"""Raster-image coverage in ``[0, 1]`` for every page (document order).
|
"""Raster-image coverage in ``[0, 1]`` for every page (document order).
|
||||||
|
|
||||||
Lets the hot path flag scanned pages whose embedded text layer is junk but
|
Feeds the ``image_heavy`` DIAGNOSTIC flag only (coverage no longer routes --
|
||||||
statistically clean-looking. Re-opens the PDF, so the registry calls it only
|
see module docstring). Re-opens the PDF, so the registry calls it only when
|
||||||
when OCR + scan detection are enabled (the cost is borne by OCR-opted-in
|
OCR + scan detection are enabled (the cost is borne by OCR-opted-in tenants).
|
||||||
tenants). Returned list is aligned by index with the leading page boundaries.
|
Returned list is aligned by index with the leading page boundaries.
|
||||||
|
|
||||||
Bounded to the first ``MAX_SAMPLED_PAGES`` pages -- the image pass is the
|
Bounded to the first ``MAX_SAMPLED_PAGES`` pages -- the image pass is the
|
||||||
costly part, so a 200-page scan isn't fully rasterised on the hot path. Pages
|
costly part, so a 200-page scan isn't fully rasterised on the hot path.
|
||||||
beyond the cap fall back to the text-quality signal in ``classify_from_text``
|
|
||||||
(a scanned tail has junk text too), and ``page_fraction`` still gates over
|
|
||||||
every page.
|
|
||||||
"""
|
"""
|
||||||
import pymupdf # noqa: PLC0415 -- keep the heavy import lazy
|
import pymupdf # noqa: PLC0415 -- keep the heavy import lazy
|
||||||
|
|
||||||
@@ -238,18 +244,18 @@ def classify_from_text(
|
|||||||
"""Classify from text already extracted by tier-1 -- no PDF re-open by default.
|
"""Classify from text already extracted by tier-1 -- no PDF re-open by default.
|
||||||
|
|
||||||
The hot-path classifier. A page is OCR-worthy when its text is near-empty
|
The hot-path classifier. A page is OCR-worthy when its text is near-empty
|
||||||
(``< min_page_chars``), its text-quality is junk (``< min_text_quality`` --
|
(``< min_page_chars``) or its text-quality is junk (``< min_text_quality`` --
|
||||||
the word-merging signal), OR (when ``image_coverage`` is supplied, i.e. OCR +
|
the word-merging signal). The doc recommends ``ocr`` once
|
||||||
scan-detection are on) the page is mostly a raster image. The doc recommends
|
``ocr_frac >= page_fraction``. Thresholds are passed in by the registry from
|
||||||
``ocr`` once ``ocr_frac >= page_fraction``. Thresholds are passed in by the
|
per-tenant settings. ``image_coverage`` (when supplied) only feeds the
|
||||||
registry from per-tenant settings.
|
``image_heavy`` diagnostic flag -- it does NOT route (see module docstring).
|
||||||
|
|
||||||
``page_boundaries`` are ``{page, start_offset, end_offset}`` indexing into
|
``page_boundaries`` are ``{page, start_offset, end_offset}`` indexing into
|
||||||
``full_text``; ``image_coverage[i]`` (if given) aligns with the i-th boundary.
|
``full_text``; ``image_coverage[i]`` (if given) aligns with the i-th boundary.
|
||||||
|
|
||||||
Note: the ``image_heavy`` flag (and the image-coverage trigger) are only set
|
Note: the ``image_heavy`` flag is only set when ``image_coverage`` is
|
||||||
when ``image_coverage`` is supplied, so for tenants with scan detection off
|
supplied, so for tenants with scan detection off that flag is always zero.
|
||||||
that flag is always zero -- the text-quality/empty signals still route.
|
Routing is unaffected either way -- it is on the text signals alone.
|
||||||
"""
|
"""
|
||||||
# image_coverage is expected to be one entry per page, capped at
|
# image_coverage is expected to be one entry per page, capped at
|
||||||
# MAX_SAMPLED_PAGES (see image_coverage_per_page). Any other length means the
|
# MAX_SAMPLED_PAGES (see image_coverage_per_page). Any other length means the
|
||||||
@@ -278,11 +284,14 @@ def classify_from_text(
|
|||||||
if image_coverage is not None and idx < len(image_coverage)
|
if image_coverage is not None and idx < len(image_coverage)
|
||||||
else 0.0
|
else 0.0
|
||||||
)
|
)
|
||||||
needs_ocr = (
|
# Routing is on TEXT signals only: a near-empty layer or a junk/
|
||||||
len(seg.strip()) < min_page_chars
|
# space-mangled one. Image coverage (``cov``) is intentionally not a
|
||||||
or quality < min_text_quality
|
# routing trigger -- a mostly-raster page with an already-usable text
|
||||||
or cov >= IMAGE_COVERAGE_SCANNED
|
# layer does not benefit from re-OCR, so escalating it to the paid OCR
|
||||||
)
|
# tier was wasteful (on OHR-Bench this drove ~45% of escalations: clean
|
||||||
|
# digital figure-pages and scans that already carry a good OCR layer).
|
||||||
|
# ``cov`` still feeds the diagnostic ``image_heavy`` flag below.
|
||||||
|
needs_ocr = len(seg.strip()) < min_page_chars or quality < min_text_quality
|
||||||
pages.append(
|
pages.append(
|
||||||
PageSignals(b["page"], len(seg), round(cov, 3), quality, needs_ocr)
|
PageSignals(b["page"], len(seg), round(cov, 3), quality, needs_ocr)
|
||||||
)
|
)
|
||||||
@@ -309,7 +318,7 @@ def classify_from_text(
|
|||||||
flags.add("scanned")
|
flags.add("scanned")
|
||||||
elif mean_quality < min_text_quality:
|
elif mean_quality < min_text_quality:
|
||||||
flags.add("bad_text_layer")
|
flags.add("bad_text_layer")
|
||||||
if any(p.image_coverage >= IMAGE_COVERAGE_SCANNED for p in pages):
|
if any(p.image_coverage >= IMAGE_HEAVY_THRESHOLD for p in pages):
|
||||||
flags.add("image_heavy")
|
flags.add("image_heavy")
|
||||||
|
|
||||||
recommended = "ocr" if ocr_frac >= page_fraction else "fast"
|
recommended = "ocr" if ocr_frac >= page_fraction else "fast"
|
||||||
|
|||||||
@@ -3,8 +3,10 @@
|
|||||||
Pins the routing decisions and the text-quality heuristic that drive which
|
Pins the routing decisions and the text-quality heuristic that drive which
|
||||||
extraction tier a PDF starts in:
|
extraction tier a PDF starts in:
|
||||||
* a clean born-digital PDF (text, no full-page images) -> ``fast`` (tier 1);
|
* a clean born-digital PDF (text, no full-page images) -> ``fast`` (tier 1);
|
||||||
* a full-page-image scan -> ``ocr`` (tier 3), since handwriting/stamps aren't
|
* a full-page-image scan with no usable text layer -> ``ocr`` (tier 3);
|
||||||
in any text layer;
|
* routing is on TEXT signals only -- image coverage feeds the ``image_heavy``
|
||||||
|
diagnostic flag but does not route (a mostly-raster page with a clean text
|
||||||
|
layer stays ``fast``);
|
||||||
* the text-quality score distinguishes clean prose from mashed/space-less junk.
|
* the text-quality score distinguishes clean prose from mashed/space-less junk.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
@@ -116,6 +118,23 @@ def _image_with_mashed_text_pdf(pages: int = 2) -> bytes:
|
|||||||
return data
|
return data
|
||||||
|
|
||||||
|
|
||||||
|
def _image_with_clean_text_pdf(pages: int = 2) -> bytes:
|
||||||
|
# Full-page image with a CLEAN embedded text layer -- a scan carrying a good
|
||||||
|
# OCR layer, or a figure-heavy digital page. Image-heavy but usable text.
|
||||||
|
doc = pymupdf.open()
|
||||||
|
pix = pymupdf.Pixmap(pymupdf.csRGB, pymupdf.IRect(0, 0, 600, 850))
|
||||||
|
pix.clear_with(255)
|
||||||
|
img = pix.tobytes("png")
|
||||||
|
del pix # Pixmap holds native memory; release it before the loop
|
||||||
|
for _ in range(pages):
|
||||||
|
page = doc.new_page(width=595, height=842)
|
||||||
|
page.insert_image(page.rect, stream=img)
|
||||||
|
page.insert_text((50, 60), "Hello world this is clean text. " * 8)
|
||||||
|
data: bytes = doc.tobytes()
|
||||||
|
doc.close()
|
||||||
|
return data
|
||||||
|
|
||||||
|
|
||||||
def test_scanned_flag_when_no_text_layer():
|
def test_scanned_flag_when_no_text_layer():
|
||||||
c = clf.classify_pdf(_full_page_image_pdf())
|
c = clf.classify_pdf(_full_page_image_pdf())
|
||||||
assert c.total_chars == 0
|
assert c.total_chars == 0
|
||||||
@@ -236,12 +255,26 @@ def test_quality_floor_override_disables_trigger():
|
|||||||
assert c.recommended_tier == "fast"
|
assert c.recommended_tier == "fast"
|
||||||
|
|
||||||
|
|
||||||
def test_scan_signal_routes_ocr_even_with_clean_text():
|
def test_image_heavy_clean_text_stays_fast():
|
||||||
# clean text but every page is a raster scan -> OCR (the Student-147 case)
|
# Image coverage is diagnostic, not routing: fully-raster pages whose text
|
||||||
|
# layer is already clean (a scan carrying a good OCR layer, or a figure-heavy
|
||||||
|
# digital page) carry the image_heavy flag but stay on the fast tier --
|
||||||
|
# re-OCR adds nothing. This was the ~45% over-escalation on OHR-Bench.
|
||||||
full, bounds = _two_page(_CLEAN, _CLEAN)
|
full, bounds = _two_page(_CLEAN, _CLEAN)
|
||||||
c = clf.classify_from_text(full, bounds, image_coverage=[1.0, 1.0])
|
c = clf.classify_from_text(full, bounds, image_coverage=[1.0, 1.0])
|
||||||
assert c.recommended_tier == "ocr"
|
assert c.recommended_tier == "fast"
|
||||||
assert "image_heavy" in c.flags
|
assert "image_heavy" in c.flags
|
||||||
|
assert all(p.needs_ocr is False for p in c.pages)
|
||||||
|
|
||||||
|
|
||||||
|
def test_classify_pdf_image_heavy_clean_text_stays_fast():
|
||||||
|
# classify_pdf symmetry with test_image_heavy_clean_text_stays_fast: full-page
|
||||||
|
# raster images WITH a clean embedded text layer are image_heavy but route
|
||||||
|
# fast -- coverage is diagnostic, not routing, on the classify_pdf path too.
|
||||||
|
c = clf.classify_pdf(_image_with_clean_text_pdf())
|
||||||
|
assert c.recommended_tier == "fast"
|
||||||
|
assert "image_heavy" in c.flags
|
||||||
|
assert c.mean_text_quality >= clf.MIN_TEXT_QUALITY
|
||||||
|
|
||||||
|
|
||||||
def test_scan_signal_ignored_when_coverage_low():
|
def test_scan_signal_ignored_when_coverage_low():
|
||||||
@@ -270,9 +303,10 @@ def test_image_coverage_per_page():
|
|||||||
assert len(digital) == 2 and all(c < 0.1 for c in digital)
|
assert len(digital) == 2 and all(c < 0.1 for c in digital)
|
||||||
|
|
||||||
|
|
||||||
def test_scan_coverage_shorter_than_pages_falls_back_to_text():
|
def test_scan_coverage_shorter_than_pages_aligns_without_crash():
|
||||||
# image_coverage shorter than the boundaries (the MAX_SAMPLED_PAGES cap):
|
# image_coverage shorter than the boundaries (the MAX_SAMPLED_PAGES cap):
|
||||||
# page 0 is flagged scanned; later pages fall back to the text-quality signal.
|
# the single entry aligns to page 0; later pages get no coverage entry. With
|
||||||
|
# clean text everywhere, routing is on text only, so nothing escalates.
|
||||||
n = len(_CLEAN)
|
n = len(_CLEAN)
|
||||||
full = _CLEAN * 3
|
full = _CLEAN * 3
|
||||||
bounds = [
|
bounds = [
|
||||||
@@ -281,6 +315,8 @@ def test_scan_coverage_shorter_than_pages_falls_back_to_text():
|
|||||||
{"page": 3, "start_offset": 2 * n, "end_offset": 3 * n},
|
{"page": 3, "start_offset": 2 * n, "end_offset": 3 * n},
|
||||||
]
|
]
|
||||||
c = clf.classify_from_text(full, bounds, image_coverage=[1.0])
|
c = clf.classify_from_text(full, bounds, image_coverage=[1.0])
|
||||||
assert c.pages[0].needs_ocr is True # scanned (coverage)
|
assert c.pages[0].image_coverage == pytest.approx(1.0) # entry aligned
|
||||||
assert c.pages[1].needs_ocr is False # clean text, no coverage entry
|
assert c.pages[1].image_coverage == pytest.approx(0.0) # no entry -> 0
|
||||||
assert c.recommended_tier == "fast" # only 1/3 pages bad
|
assert all(p.needs_ocr is False for p in c.pages) # coverage no longer routes
|
||||||
|
assert "image_heavy" in c.flags # but page 0 still flags image_heavy
|
||||||
|
assert c.recommended_tier == "fast"
|
||||||
|
|||||||
Reference in New Issue
Block a user