fix(chunk-context): address PR #767 review — extract bbox helper, fix page_number overwrite

Resolves both 🟡 important issues from the latest review:

1. `page_number` was unconditionally overwritten in `viz_routes.py:696` even
   when Qdrant's payload lacked the field, clobbering the value resolved
   from `chunk_context.page_number`. The new helper returns each field
   independently and both call sites only overwrite via `is not None`
   guards, matching the existing logic in `visualization.py`.

2. The ~60-line `if chunk_index is not None: ... else: ...` Qdrant scroll
   block was duplicated between `api/visualization.py` and
   `auth/viz_routes.py`. Extracted into `get_chunk_bbox_and_page_from_qdrant`
   in `search/context.py` alongside the existing private `_get_chunk_*_from_qdrant`
   helpers; both routes now share ~12 lines of caller code.

New unit tests at `tests/unit/test_chunk_bbox_helper.py` cover the indexed
and offset paths, the `(bbox, None)` regression case, and graceful
degradation on Qdrant strict-mode 400 (which also closes nit #4).

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
Chris Coutinho
2026-05-09 13:29:22 +02:00
co-authored by Claude Opus 4.7
parent c780f96d2b
commit 7ef8760d27
4 changed files with 342 additions and 152 deletions
+17 -76
View File
@@ -14,7 +14,6 @@ import logging
from typing import Any
import pymupdf
from qdrant_client.models import FieldCondition, Filter, MatchValue
from starlette.requests import Request
from starlette.responses import JSONResponse
@@ -31,13 +30,14 @@ from nextcloud_mcp_server.search import (
BM25HybridSearchAlgorithm,
SemanticSearchAlgorithm,
)
from nextcloud_mcp_server.search.context import get_chunk_with_context
from nextcloud_mcp_server.search.context import (
get_chunk_bbox_and_page_from_qdrant,
get_chunk_with_context,
)
from nextcloud_mcp_server.vector.oauth_sync import (
NotProvisionedError,
get_user_client_basic_auth,
)
from nextcloud_mcp_server.vector.placeholder import get_placeholder_filter
from nextcloud_mcp_server.vector.qdrant_client import get_qdrant_client
from nextcloud_mcp_server.vector.visualization import compute_pca_coordinates
logger = logging.getLogger(__name__)
@@ -567,82 +567,23 @@ async def get_chunk_context(request: Request) -> JSONResponse:
# For PDF files, also fetch the chunk's bounding box from Qdrant if
# available so the client can overlay a highlight on top of a
# render-on-demand page image (Deck #76).
# render-on-demand page image (Deck #76). Qdrant's page_number is
# trusted over the context-expansion fallback when present.
chunk_bbox = None
page_number = chunk_context.page_number
if doc_type == "file":
try:
settings = get_settings()
qdrant_client = await get_qdrant_client()
# Prefer chunk_index for the chunk-bbox lookup (always indexed);
# fall back to (chunk_start_offset, chunk_end_offset) when not provided.
if chunk_index is not None:
points_response = await qdrant_client.scroll(
collection_name=settings.get_collection_name(),
scroll_filter=Filter(
must=[
get_placeholder_filter(),
FieldCondition(
key="doc_id", match=MatchValue(value=doc_id_val)
),
FieldCondition(
key="user_id", match=MatchValue(value=user_id)
),
FieldCondition(
key="chunk_index",
match=MatchValue(value=chunk_index),
),
]
),
limit=1,
with_vectors=False,
with_payload=["chunk_bbox", "page_number"],
)
else:
# Legacy fallback for clients that don't send chunk_index
# (pre-cbcoutinho/astrolabe#75). chunk_start/end_offset
# aren't indexed in Qdrant Cloud strict mode, so this
# call may fail with HTTP 400 there; the outer except
# logs a warning and the response degrades gracefully
# (no chunk_bbox).
points_response = await qdrant_client.scroll(
collection_name=settings.get_collection_name(),
scroll_filter=Filter(
must=[
get_placeholder_filter(),
FieldCondition(
key="doc_id", match=MatchValue(value=doc_id_val)
),
FieldCondition(
key="user_id", match=MatchValue(value=user_id)
),
FieldCondition(
key="chunk_start_offset",
match=MatchValue(value=start),
),
FieldCondition(
key="chunk_end_offset",
match=MatchValue(value=end),
),
]
),
limit=1,
with_vectors=False,
with_payload=["chunk_bbox", "page_number"],
)
if points_response[0]:
payload = points_response[0][0].payload
if payload:
chunk_bbox = payload.get("chunk_bbox")
# Trust Qdrant page number if available (might be more accurate than context expansion logic)
if payload.get("page_number") is not None:
page_number = payload.get("page_number")
except Exception as e:
logger.warning(f"Failed to fetch chunk bbox: {e}")
qdrant_bbox, qdrant_page = await get_chunk_bbox_and_page_from_qdrant(
user_id=user_id,
doc_id=doc_id_val,
chunk_index=chunk_index,
chunk_start=start,
chunk_end=end,
)
if qdrant_bbox is not None:
chunk_bbox = qdrant_bbox
if qdrant_page is not None:
page_number = qdrant_page
# Build response
response_data = {
+17 -76
View File
@@ -18,7 +18,6 @@ from pathlib import Path
import anyio
import numpy as np
from jinja2 import Environment, FileSystemLoader
from qdrant_client.models import FieldCondition, Filter, MatchValue
from starlette.authentication import requires
from starlette.requests import Request
from starlette.responses import HTMLResponse, JSONResponse
@@ -34,13 +33,15 @@ from nextcloud_mcp_server.search import (
BM25HybridSearchAlgorithm,
SemanticSearchAlgorithm,
)
from nextcloud_mcp_server.search.context import get_chunk_with_context
from nextcloud_mcp_server.search.context import (
get_chunk_bbox_and_page_from_qdrant,
get_chunk_with_context,
)
from nextcloud_mcp_server.vector.oauth_sync import (
NotProvisionedError,
get_user_client_basic_auth,
)
from nextcloud_mcp_server.vector.pca import PCA
from nextcloud_mcp_server.vector.placeholder import get_placeholder_filter
from nextcloud_mcp_server.vector.qdrant_client import get_qdrant_client
logger = logging.getLogger(__name__)
@@ -625,82 +626,22 @@ async def chunk_context_endpoint(request: Request) -> JSONResponse:
# For PDF files, also fetch the chunk bbox from Qdrant so the client
# can overlay a highlight on top of a render-on-demand page image
# (Deck #76).
# (Deck #76). Qdrant's page_number is trusted over the
# context-expansion fallback when present.
chunk_bbox = None
page_number = chunk_context.page_number
if doc_type == "file":
try:
settings = get_settings()
qdrant_client = await get_qdrant_client()
# Prefer chunk_index for the chunk-bbox lookup (always indexed);
# fall back to (chunk_start_offset, chunk_end_offset) when not provided.
if chunk_index is not None:
points_response = await qdrant_client.scroll(
collection_name=settings.get_collection_name(),
scroll_filter=Filter(
must=[
get_placeholder_filter(),
FieldCondition(
key="doc_id", match=MatchValue(value=doc_id_int)
),
FieldCondition(
key="user_id", match=MatchValue(value=user_id)
),
FieldCondition(
key="chunk_index",
match=MatchValue(value=chunk_index),
),
]
),
limit=1,
with_vectors=False,
with_payload=["chunk_bbox", "page_number"],
)
else:
# Legacy fallback for clients that don't send chunk_index
# (pre-cbcoutinho/astrolabe#75). chunk_start/end_offset
# aren't indexed in Qdrant Cloud strict mode, so this
# call may fail with HTTP 400 there; the outer except
# logs a warning and the response degrades gracefully
# (no chunk_bbox).
points_response = await qdrant_client.scroll(
collection_name=settings.get_collection_name(),
scroll_filter=Filter(
must=[
get_placeholder_filter(),
FieldCondition(
key="doc_id", match=MatchValue(value=doc_id_int)
),
FieldCondition(
key="user_id", match=MatchValue(value=user_id)
),
FieldCondition(
key="chunk_start_offset",
match=MatchValue(value=start),
),
FieldCondition(
key="chunk_end_offset",
match=MatchValue(value=end),
),
]
),
limit=1,
with_vectors=False,
with_payload=["chunk_bbox", "page_number"],
)
points = points_response[0]
if points and points[0].payload:
chunk_bbox = points[0].payload.get("chunk_bbox")
page_number = points[0].payload.get("page_number")
if chunk_bbox:
logger.info(
f"Found chunk bbox: page={page_number}, "
f"rects={len(chunk_bbox)}"
)
except Exception as e:
logger.warning(f"Failed to fetch chunk bbox: {e}")
qdrant_bbox, qdrant_page = await get_chunk_bbox_and_page_from_qdrant(
user_id=user_id,
doc_id=doc_id_int,
chunk_index=chunk_index,
chunk_start=start,
chunk_end=end,
)
if qdrant_bbox is not None:
chunk_bbox = qdrant_bbox
if qdrant_page is not None:
page_number = qdrant_page
# Return response compatible with frontend expectations
response_data: dict = {
+93
View File
@@ -12,6 +12,7 @@ from qdrant_client.models import FieldCondition, Filter, MatchValue
from nextcloud_mcp_server.client import NextcloudClient
from nextcloud_mcp_server.config import get_settings
from nextcloud_mcp_server.vector.html_processor import html_to_markdown
from nextcloud_mcp_server.vector.placeholder import get_placeholder_filter
from nextcloud_mcp_server.vector.qdrant_client import get_qdrant_client
logger = logging.getLogger(__name__)
@@ -195,6 +196,98 @@ async def _get_deck_metadata_from_qdrant(
return None
async def get_chunk_bbox_and_page_from_qdrant(
user_id: str,
doc_id: int | str,
chunk_index: int | None,
chunk_start: int,
chunk_end: int,
) -> tuple[list | None, int | None]:
"""Fetch chunk_bbox and page_number for a chunk from Qdrant payload.
Prefers chunk_index for the lookup (always indexed); falls back to
(chunk_start_offset, chunk_end_offset) when chunk_index is not provided
— this is the legacy path for clients pre-cbcoutinho/astrolabe#75. The
fallback may 400 in Qdrant Cloud strict mode because those offset fields
aren't indexed there; that's logged as a warning and (None, None) is
returned so callers degrade gracefully.
Args:
user_id: User ID who owns the document
doc_id: Document ID (int for file/note, str for some doc types)
chunk_index: Zero-based chunk index, or None to use offset fallback
chunk_start: Character offset where chunk starts (used when
chunk_index is None)
chunk_end: Character offset where chunk ends (used when chunk_index
is None)
Returns:
Tuple of (chunk_bbox, page_number); either field may be None
independently if absent from the payload, or both may be None on
miss/error.
"""
try:
settings = get_settings()
qdrant_client = await get_qdrant_client()
if chunk_index is not None:
points_response = await qdrant_client.scroll(
collection_name=settings.get_collection_name(),
scroll_filter=Filter(
must=[
get_placeholder_filter(),
FieldCondition(key="doc_id", match=MatchValue(value=doc_id)),
FieldCondition(key="user_id", match=MatchValue(value=user_id)),
FieldCondition(
key="chunk_index", match=MatchValue(value=chunk_index)
),
]
),
limit=1,
with_vectors=False,
with_payload=["chunk_bbox", "page_number"],
)
else:
points_response = await qdrant_client.scroll(
collection_name=settings.get_collection_name(),
scroll_filter=Filter(
must=[
get_placeholder_filter(),
FieldCondition(key="doc_id", match=MatchValue(value=doc_id)),
FieldCondition(key="user_id", match=MatchValue(value=user_id)),
FieldCondition(
key="chunk_start_offset",
match=MatchValue(value=chunk_start),
),
FieldCondition(
key="chunk_end_offset",
match=MatchValue(value=chunk_end),
),
]
),
limit=1,
with_vectors=False,
with_payload=["chunk_bbox", "page_number"],
)
points = points_response[0]
if not points or not points[0].payload:
return None, None
payload = points[0].payload
chunk_bbox = payload.get("chunk_bbox")
page_number = payload.get("page_number")
if chunk_bbox:
logger.info(
"Found chunk bbox: page=%s, rects=%d", page_number, len(chunk_bbox)
)
return chunk_bbox, page_number
except Exception as e:
logger.warning("Failed to fetch chunk bbox: %s", e)
return None, None
@dataclass
class ChunkContext:
"""Expanded chunk with surrounding context and position markers.