fix(vector): address PR review — wait=True backfill, batched writes, search helper

Addresses reviewer feedback on PR #773:

- Backfill set_payload now uses wait=True to avoid a race where
  _ensure_keyword_payload_indexes builds the KEYWORD index before
  fire-and-forget writes have committed, leaving int payloads
  invisible to filters.
- Batch points sharing the same int doc_id into a single set_payload
  call (one document → many chunks → one round-trip instead of N).
- Drop _has_int_doc_id_sample short-circuit. The sample's false-negative
  window (clean first 256 results, ints further in) is gone; full scroll
  is the dominant cost on first run anyway.
- Simplify _ensure_keyword_payload_indexes: the "already exists" 400
  branch was dead code (Qdrant returns 200 on identical re-create); any
  400 now logs a warning and continues.
- search/context.py: comment the broadened file-type guard. Add explicit
  not doc_id.isdigit() checks at the top of note/news_item/deck_card
  branches in _fetch_document_text so malformed payloads surface as
  warnings instead of being swallowed by the broad except.

Also extracts build_search_result_from_point into search/algorithms.py
to deduplicate the 71-line payload-extraction loop shared by
SemanticSearchAlgorithm and BM25HybridSearchAlgorithm. This fixes
SonarQube's quality-gate failure (4.0% new-code duplication, max 3%).

Test coverage:
- 7 new unit tests for build_search_result_from_point covering missing
  payload, note/file/deck_card metadata, int doc_id coercion, and
  metadata_extras merging.
- Replace _has_int_doc_id_sample tests with clean-collection no-op and
  per-batch grouping tests.
- Update set_payload assertions from wait=False to wait=True.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
Chris Coutinho
2026-05-08 21:14:28 +02:00
co-authored by Claude Opus 4.7
parent 719b3b5034
commit 6aba589a6e
7 changed files with 385 additions and 252 deletions
+65 -1
View File
@@ -5,7 +5,7 @@ from abc import ABC, abstractmethod
from dataclasses import dataclass
from typing import Any, Protocol, runtime_checkable
from qdrant_client.models import FieldCondition, Filter, MatchValue
from qdrant_client.models import FieldCondition, Filter, MatchValue, ScoredPoint
from nextcloud_mcp_server.config import get_settings
from nextcloud_mcp_server.vector.placeholder import get_placeholder_filter
@@ -181,6 +181,70 @@ class SearchResult:
raise ValueError(f"Score must be non-negative, got {self.score}")
def build_search_result_from_point(
point: ScoredPoint,
*,
metadata_extras: dict[str, Any] | None = None,
) -> SearchResult | None:
"""Construct a SearchResult from a Qdrant ScoredPoint payload.
Returns ``None`` when the payload is missing — callers should skip the
point. The defensive ``str()`` coercion on ``doc_id`` covers legacy int
payloads until the startup backfill has run everywhere (see
``vector/qdrant_client.py:_backfill_doc_id_to_string``).
Args:
point: A Qdrant ``ScoredPoint`` from a search response.
metadata_extras: Algorithm-specific metadata merged into the result's
``metadata`` dict (e.g., ``{"search_method": "bm25_hybrid_rrf"}``).
Returns:
A populated ``SearchResult``, or ``None`` if ``point.payload`` is
missing.
"""
if point.payload is None:
return None
doc_id = str(point.payload["doc_id"])
doc_type = point.payload.get("doc_type", "note")
metadata: dict[str, Any] = {
"chunk_index": point.payload.get("chunk_index"),
"total_chunks": point.payload.get("total_chunks"),
}
if metadata_extras:
metadata.update(metadata_extras)
# File-specific metadata for PDF viewer
if doc_type == "file" and (path := point.payload.get("file_path")):
metadata["path"] = path
# Deck-card metadata for frontend URL construction and verify-on-read
# (ADR-019) — both board_id and stack_id are required to call
# deck.get_card without an O(boards × stacks) iteration fallback.
if doc_type == "deck_card":
if board_id := point.payload.get("board_id"):
metadata["board_id"] = board_id
if stack_id := point.payload.get("stack_id"):
metadata["stack_id"] = stack_id
return SearchResult(
id=doc_id,
doc_type=doc_type,
title=point.payload.get("title", "Untitled"),
excerpt=point.payload.get("excerpt", ""),
score=point.score,
metadata=metadata,
chunk_start_offset=point.payload.get("chunk_start_offset"),
chunk_end_offset=point.payload.get("chunk_end_offset"),
page_number=point.payload.get("page_number"),
page_count=point.payload.get("page_count"),
chunk_index=point.payload.get("chunk_index", 0),
total_chunks=point.payload.get("total_chunks", 1),
point_id=str(point.id),
)
class SearchAlgorithm(ABC):
"""Abstract base class for search algorithms.
+22 -54
View File
@@ -10,7 +10,11 @@ from nextcloud_mcp_server.config import get_settings
from nextcloud_mcp_server.embedding import get_bm25_service, get_embedding_service
from nextcloud_mcp_server.observability.metrics import record_qdrant_operation
from nextcloud_mcp_server.observability.tracing import trace_operation
from nextcloud_mcp_server.search.algorithms import SearchAlgorithm, SearchResult
from nextcloud_mcp_server.search.algorithms import (
SearchAlgorithm,
SearchResult,
build_search_result_from_point,
)
from nextcloud_mcp_server.vector.placeholder import get_placeholder_filter
from nextcloud_mcp_server.vector.qdrant_client import get_qdrant_client
@@ -202,66 +206,30 @@ class BM25HybridSearchAlgorithm(SearchAlgorithm):
"search.deduplicate",
attributes={"dedupe.num_points": len(search_response.points)},
):
seen_chunks = set()
results = []
seen_chunks: set[tuple[str, str, Any, Any]] = set()
results: list[SearchResult] = []
metadata_extras = {
"search_method": f"bm25_hybrid_{self.fusion_name}",
}
for result in search_response.points:
if result.payload is None:
for point in search_response.points:
sr = build_search_result_from_point(
point, metadata_extras=metadata_extras
)
if sr is None:
continue
# doc_id is always str post-normalization, but defensively coerce
# legacy int payloads on read until the backfill has run everywhere.
doc_id = str(result.payload["doc_id"])
doc_type = result.payload.get("doc_type", "note")
chunk_start = result.payload.get("chunk_start_offset")
chunk_end = result.payload.get("chunk_end_offset")
chunk_key = (doc_id, doc_type, chunk_start, chunk_end)
# Skip if we've already seen this exact chunk
chunk_key = (
sr.id,
sr.doc_type,
sr.chunk_start_offset,
sr.chunk_end_offset,
)
if chunk_key in seen_chunks:
continue
seen_chunks.add(chunk_key)
# Build metadata dict with common fields
metadata = {
"chunk_index": result.payload.get("chunk_index"),
"total_chunks": result.payload.get("total_chunks"),
"search_method": f"bm25_hybrid_{self.fusion_name}",
}
# Add file-specific metadata for PDF viewer
if doc_type == "file" and (path := result.payload.get("file_path")):
metadata["path"] = path
# Add deck_card-specific metadata for frontend URL construction
# and verify-on-read (ADR-019) — both board_id and stack_id are
# required to call deck.get_card without an O(boards × stacks)
# iteration fallback.
if doc_type == "deck_card":
if board_id := result.payload.get("board_id"):
metadata["board_id"] = board_id
if stack_id := result.payload.get("stack_id"):
metadata["stack_id"] = stack_id
# Return unverified results (verification happens at output stage)
results.append(
SearchResult(
id=doc_id,
doc_type=doc_type,
title=result.payload.get("title", "Untitled"),
excerpt=result.payload.get("excerpt", ""),
score=result.score, # Fusion score (RRF or DBSF)
metadata=metadata,
chunk_start_offset=result.payload.get("chunk_start_offset"),
chunk_end_offset=result.payload.get("chunk_end_offset"),
page_number=result.payload.get("page_number"),
page_count=result.payload.get("page_count"),
chunk_index=result.payload.get("chunk_index", 0),
total_chunks=result.payload.get("total_chunks", 1),
point_id=str(result.id), # Qdrant point ID for batch retrieval
)
)
results.append(sr)
if len(results) >= limit:
break
+36 -2
View File
@@ -415,8 +415,13 @@ async def get_chunk_with_context(
f"(Qdrant cache miss, possibly legacy data)"
)
# For files, the doc_id is the numeric file ID (as a string) — resolve it
# to a WebDAV path so _fetch_document_text can retrieve the binary content.
# For files, doc_id is always the stringified numeric file ID after
# producer normalization — resolve it to a WebDAV path so
# _fetch_document_text can retrieve the binary content. The previous
# `isinstance(doc_id, int)` guard is no longer needed: file producers
# write str(file_id) and the startup backfill rewrites legacy int
# payloads. If lookup fails (e.g. truly malformed legacy data), the
# caller logs and returns None below — a re-index is the recovery path.
resolved_doc_id = doc_id
if doc_type == "file":
file_path = await _get_file_path_from_qdrant(
@@ -506,6 +511,15 @@ async def _fetch_document_text(
"""
try:
if doc_type == "note":
# Note IDs are integers in the Nextcloud API; reject non-numeric
# doc_ids explicitly so a malformed payload surfaces in logs
# rather than getting silently swallowed by `except Exception`.
if not doc_id.isdigit():
logger.warning(
"Expected numeric note doc_id, got %r — skipping document fetch",
doc_id,
)
return None
# Fetch note by ID
note = await nc_client.notes.get_note(note_id=int(doc_id))
# Reconstruct full content as indexed: title + "\n\n" + content
@@ -562,6 +576,15 @@ async def _fetch_document_text(
)
return None
elif doc_type == "news_item":
# News item IDs are integers in the Nextcloud News API; reject
# non-numeric doc_ids explicitly so malformed payloads surface
# rather than getting swallowed by the broad except below.
if not doc_id.isdigit():
logger.warning(
"Expected numeric news_item doc_id, got %r — skipping document fetch",
doc_id,
)
return None
# Fetch news item by ID
item = await nc_client.news.get_item(int(doc_id))
# Reconstruct full content as indexed: title + source + URL + body
@@ -580,6 +603,17 @@ async def _fetch_document_text(
content_parts.append(body_markdown)
return "\n".join(content_parts)
elif doc_type == "deck_card":
# Deck card IDs are integers in the Nextcloud Deck API; reject
# non-numeric doc_ids explicitly so malformed payloads surface
# rather than getting swallowed by the broad except below. The
# numeric check covers both the metadata-fast-path (line ~600)
# and the iteration fallback (line ~635).
if not doc_id.isdigit():
logger.warning(
"Expected numeric deck_card doc_id, got %r — skipping document fetch",
doc_id,
)
return None
# Fetch card from Deck API
# Try to get board_id/stack_id from Qdrant metadata (O(1) lookup)
# Otherwise fall back to iteration (legacy data)
+12 -53
View File
@@ -8,7 +8,11 @@ from qdrant_client.models import FieldCondition, Filter, MatchValue
from nextcloud_mcp_server.config import get_settings
from nextcloud_mcp_server.embedding import get_embedding_service
from nextcloud_mcp_server.observability.metrics import record_qdrant_operation
from nextcloud_mcp_server.search.algorithms import SearchAlgorithm, SearchResult
from nextcloud_mcp_server.search.algorithms import (
SearchAlgorithm,
SearchResult,
build_search_result_from_point,
)
from nextcloud_mcp_server.vector.placeholder import get_placeholder_filter
from nextcloud_mcp_server.vector.qdrant_client import get_qdrant_client
@@ -134,65 +138,20 @@ class SemanticSearchAlgorithm(SearchAlgorithm):
# Deduplicate by (doc_id, doc_type, chunk_start, chunk_end)
# This allows multiple chunks from same doc, but removes duplicate chunks
seen_chunks = set()
results = []
seen_chunks: set[tuple[str, str, Any, Any]] = set()
results: list[SearchResult] = []
for result in search_response.points:
if result.payload is None:
for point in search_response.points:
sr = build_search_result_from_point(point)
if sr is None:
continue
# doc_id is always str post-normalization, but defensively coerce
# legacy int payloads on read until the backfill has run everywhere.
doc_id = str(result.payload["doc_id"])
doc_type = result.payload.get("doc_type", "note")
chunk_start = result.payload.get("chunk_start_offset")
chunk_end = result.payload.get("chunk_end_offset")
chunk_key = (doc_id, doc_type, chunk_start, chunk_end)
# Skip if we've already seen this exact chunk
chunk_key = (sr.id, sr.doc_type, sr.chunk_start_offset, sr.chunk_end_offset)
if chunk_key in seen_chunks:
continue
seen_chunks.add(chunk_key)
# Build metadata dict with common fields
metadata = {
"chunk_index": result.payload.get("chunk_index"),
"total_chunks": result.payload.get("total_chunks"),
}
# Add file-specific metadata for PDF viewer
if doc_type == "file" and (path := result.payload.get("file_path")):
metadata["path"] = path
# Add deck_card-specific metadata for frontend URL construction
# and verify-on-read (ADR-019) — both board_id and stack_id are
# required to call deck.get_card without an O(boards × stacks)
# iteration fallback.
if doc_type == "deck_card":
if board_id := result.payload.get("board_id"):
metadata["board_id"] = board_id
if stack_id := result.payload.get("stack_id"):
metadata["stack_id"] = stack_id
# Return unverified results (verification happens at output stage)
results.append(
SearchResult(
id=doc_id,
doc_type=doc_type,
title=result.payload.get("title", "Untitled"),
excerpt=result.payload.get("excerpt", ""),
score=result.score,
metadata=metadata,
chunk_start_offset=result.payload.get("chunk_start_offset"),
chunk_end_offset=result.payload.get("chunk_end_offset"),
page_number=result.payload.get("page_number"),
page_count=result.payload.get("page_count"),
chunk_index=result.payload.get("chunk_index", 0),
total_chunks=result.payload.get("total_chunks", 1),
point_id=str(result.id), # Qdrant point ID for batch retrieval
)
)
results.append(sr)
if len(results) >= limit:
break
+30 -46
View File
@@ -1,6 +1,7 @@
"""Qdrant client wrapper."""
import logging
from typing import Any
from qdrant_client import AsyncQdrantClient, models
from qdrant_client.http.exceptions import UnexpectedResponse
@@ -28,8 +29,10 @@ async def _ensure_keyword_payload_indexes(
) -> None:
"""Create KEYWORD payload indexes for fields used in exact-match filters.
Idempotent: tolerates 'already exists' errors so it can run on every
startup against existing collections.
Idempotent at the Qdrant layer: re-creating an identical index returns
200, so this can run on every startup. Schema conflicts (a pre-existing
index with a different type) surface as a 400 — log loudly so operators
can intervene, but keep going so the remaining fields still get indexed.
"""
for field in _KEYWORD_PAYLOAD_FIELDS:
try:
@@ -41,39 +44,11 @@ async def _ensure_keyword_payload_indexes(
)
logger.info("Created KEYWORD payload index on '%s'", field)
except UnexpectedResponse as e:
# Qdrant returns 400 if the index already exists with a different
# schema, or simply succeeds if it already matches. Treat
# already-exists as benign; surface schema conflicts loudly.
body = getattr(e, "content", b"") or b""
body_text = body.decode("utf-8", errors="replace")
if "already exists" in body_text.lower():
logger.debug("Payload index on '%s' already exists", field)
else:
logger.warning(
"Failed to create payload index on '%s': %s", field, body_text
)
async def _has_int_doc_id_sample(
client: AsyncQdrantClient, collection_name: str, sample_size: int = 256
) -> bool:
"""Quick sample to decide whether the full backfill scroll is needed.
Reading the first batch is cheap; if all sampled doc_ids are already str
(the steady-state on healthy collections), we skip the full pass.
"""
points, _ = await client.scroll(
collection_name=collection_name,
limit=sample_size,
with_payload=["doc_id"],
with_vectors=False,
)
for point in points:
payload = point.payload or {}
value = payload.get("doc_id")
if value is not None and not isinstance(value, str):
return True
return False
logger.warning(
"Failed to create payload index on '%s': %s", field, body_text
)
async def _backfill_doc_id_to_string(
@@ -84,18 +59,15 @@ async def _backfill_doc_id_to_string(
Producers now uniformly write str(doc_id), but historical points may carry
int values from before normalization. A KEYWORD index does not match int
payloads, so any leftover int doc_ids would be silently invisible to
filters. Scroll all points and convert in-place. Idempotent.
filters. Scrolls all points once and converts in-place; idempotent (a
second pass over the same collection performs zero writes).
Skipped when the first sample batch already contains only str doc_ids.
Within each scroll batch, points sharing the same int doc_id are batched
into a single ``set_payload`` call to minimize Qdrant round-trips.
"""
if not await _has_int_doc_id_sample(client, collection_name):
logger.debug(
"doc_id backfill: sample shows no legacy int payloads; skipping full scan"
)
return
logger.info(
"Running doc_id backfill on '%s' (this may take a moment for large collections)",
"Scanning '%s' for legacy int doc_id payloads (this is a one-time "
"migration on first start after upgrade)",
collection_name,
)
@@ -117,6 +89,11 @@ async def _backfill_doc_id_to_string(
if not points:
break
# Group by stringified value so points sharing a doc_id (one document
# → many chunks) collapse into a single set_payload call. Point IDs
# can be int/str/UUID, so widen the value type to satisfy the qdrant
# client's PointsSelector signature without re-spelling the union.
by_value: dict[str, list[Any]] = {}
for point in points:
scanned += 1
# Qdrant client typing allows None payload even when with_payload
@@ -125,13 +102,20 @@ async def _backfill_doc_id_to_string(
value = payload.get("doc_id")
if value is None or isinstance(value, str):
continue
by_value.setdefault(str(value), []).append(point.id)
for str_val, point_ids in by_value.items():
# wait=True is required: _ensure_keyword_payload_indexes runs
# immediately after this function and only indexes committed
# data — fire-and-forget writes would leave int payloads
# invisible to KEYWORD filters.
await client.set_payload(
collection_name=collection_name,
payload={"doc_id": str(value)},
points=[point.id],
wait=False,
payload={"doc_id": str_val},
points=point_ids,
wait=True,
)
rewritten += 1
rewritten += len(point_ids)
if next_offset is None:
break