refactor(vector): address PR #775 review round 3 — fix unused var, harden boundary lookup, rename trace span

- pdf_highlighter.compute_chunk_bboxes_batch: drop unused chunk_text
  destructure (SonarQube finding), and replace positional
  page_boundaries[page_num - 1] with a key-based next() match so
  reordered or non-1-indexed boundaries can't silently shift the bbox.
  Convert touched f-string log to lazy %s formatting.
- vector/processor: rename the trace_operation span from
  "vector_sync.generate_highlights" to "vector_sync.compute_chunk_bboxes"
  to match what the function actually does.
- Add test_compute_chunk_bboxes_handles_unordered_page_boundaries —
  reverses the boundaries list and asserts identical results to the
  in-order case, guarding the boundary-lookup regression class.
- Pin pre-push-review skill to sonnet model.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
Chris Coutinho
2026-05-08 23:05:31 +02:00
co-authored by Claude Opus 4.7
parent 8bc87ed37d
commit 0b004f54bd
4 changed files with 59 additions and 4 deletions
+1
View File
@@ -7,6 +7,7 @@ description: |
in this repo's automated PR reviews. Use when the user is about to push, says "ready
to push", "review my work", "check before PR", or invokes /pre-push-review.
Report-only — does not modify code.
model: sonnet
allowed-tools:
- Bash
- Read
+12 -3
View File
@@ -739,17 +739,26 @@ class PDFHighlighter:
start_offset,
end_offset,
_,
chunk_text,
_,
) in chunks:
chunk_page_info = PDFHighlighter.find_chunk_page(
start_offset, end_offset, page_boundaries
)
if not chunk_page_info:
logger.debug(f"Chunk {chunk_index}: not found on any page")
logger.debug("Chunk %s: not found on any page", chunk_index)
continue
page_num = chunk_page_info["page_num"]
page_boundary = page_boundaries[page_num - 1]
page_boundary = next(
(b for b in page_boundaries if b["page"] == page_num), None
)
if page_boundary is None:
logger.debug(
"Chunk %s: page %s not found in boundaries",
chunk_index,
page_num,
)
continue
page_text_length = (
page_boundary["end_offset"] - page_boundary["start_offset"]
)
+1 -1
View File
@@ -582,7 +582,7 @@ async def _index_document(
assert content_bytes is not None
with trace_operation(
"vector_sync.generate_highlights",
"vector_sync.compute_chunk_bboxes",
attributes={
"vector_sync.chunk_count": len(chunks),
"vector_sync.pdf_size": len(content_bytes),
@@ -185,3 +185,48 @@ def test_compute_chunk_bboxes_assigns_correct_page(page_index: int):
assert results, "expected a bbox for the chunk"
_, page_num = results[0]
assert page_num == page_index + 1
@pytest.mark.unit
def test_compute_chunk_bboxes_handles_unordered_page_boundaries():
"""Page lookup must match by ``page`` key, not by list position.
Regression guard: an earlier implementation indexed
``page_boundaries[page_num - 1]``, which silently produces a wrong
bbox if boundaries are passed out of order. Reverse the boundaries
and assert the result is identical to the in-order case.
"""
pages = [
"Page one talks about apples and oranges in detail.",
"Page two discusses bananas and grapes thoroughly.",
]
pdf_bytes = _make_pdf(pages)
boundaries, full_text = _page_boundaries(pages)
chunks = [
(0, 0, len(pages[0]), 1, "apples and oranges"),
(
1,
len(pages[0]),
len(pages[0]) + len(pages[1]),
2,
"bananas and grapes",
),
]
in_order = PDFHighlighter.compute_chunk_bboxes_batch(
pdf_bytes=pdf_bytes,
chunks=chunks,
page_boundaries=boundaries,
full_text=full_text,
)
reversed_order = PDFHighlighter.compute_chunk_bboxes_batch(
pdf_bytes=pdf_bytes,
chunks=chunks,
page_boundaries=list(reversed(boundaries)),
full_text=full_text,
)
assert in_order == reversed_order
assert reversed_order[0][1] == 1
assert reversed_order[1][1] == 2