refactor(search): address PR #750 review feedback

- Cap all_results to limit*2 after sort in the per-doc_types branch of
  nc_semantic_search to bound over-verification (was unbounded N-types).
- Switch BatchVerifier from (client, doc_ids, user_id) to (client, results,
  semaphore). Verifiers now read file paths and deck board/stack ids from
  SearchResult.metadata instead of doing fresh Qdrant scrolls — eliminates
  one duplicate round-trip per file/deck-card verification.
- Bound per-id verification concurrency with a shared anyio.Semaphore
  (default 20, matching server/semantic.py context-expansion convention).
  Prevents httpx pool exhaustion / rate limiting on large search pages.
- Propagate stack_id from Qdrant payload to SearchResult.metadata in both
  bm25_hybrid.py and semantic.py (board_id was already propagated).
- Drop now-unused _resolve_file_path / _resolve_deck_metadata helpers.
- Drop redundant int(d) in requested predicate from _verify_news_items.
- Rewrite eviction comment to be honest about inline (not background)
  execution and the resulting latency coupling.
- ADR-019 status: Proposed -> Accepted.
- Add news property to NextcloudClientProtocol.
- Widen SearchResult.id and SemanticSearchResult.id to int | str to match
  BatchVerifier signature and document support for future string-id types.
- Flip openWorldHint to True on nc_semantic_search_answer (it calls into
  Nextcloud via nc_semantic_search).

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
Chris Coutinho
2026-05-01 08:22:28 +02:00
co-authored by Claude Opus 4.7
parent d90e793d19
commit 7784ec02d7
8 changed files with 380 additions and 270 deletions
@@ -1,6 +1,6 @@
# ADR-019: Verify-on-Read for Semantic Search Results # ADR-019: Verify-on-Read for Semantic Search Results
**Status**: Proposed **Status**: Accepted
**Date**: 2026-05-01 **Date**: 2026-05-01
**Depends On**: ADR-007 (Background Vector Sync), ADR-010 (Webhook-Based Vector Sync) **Depends On**: ADR-007 (Background Vector Sync), ADR-010 (Webhook-Based Vector Sync)
+7 -1
View File
@@ -10,7 +10,13 @@ from .base import BaseResponse
class SemanticSearchResult(BaseModel): class SemanticSearchResult(BaseModel):
"""Model for semantic search results with additional metadata.""" """Model for semantic search results with additional metadata."""
id: int = Field(description="Document ID (int for all document types)") id: int | str = Field(
description=(
"Document ID. Numeric for all currently indexed types (notes, files, "
"deck cards, news items); typed as int|str to allow future doc types "
"that use string identifiers."
)
)
doc_type: str = Field( doc_type: str = Field(
description="Document type (note, calendar_event, deck_card, etc.)" description="Document type (note, calendar_event, deck_card, etc.)"
) )
+9 -2
View File
@@ -67,6 +67,11 @@ class NextcloudClientProtocol(Protocol):
"""Tables client for accessing table row documents.""" """Tables client for accessing table row documents."""
... ...
@property
def news(self) -> Any:
"""News client for accessing news item documents."""
...
async def get_indexed_doc_types(user_id: str) -> set[str]: async def get_indexed_doc_types(user_id: str) -> set[str]:
"""Query Qdrant to get actually-indexed document types for a user. """Query Qdrant to get actually-indexed document types for a user.
@@ -127,7 +132,9 @@ class SearchResult:
"""A single search result with metadata and score. """A single search result with metadata and score.
Attributes: Attributes:
id: Document ID (int for all document types) id: Document ID. Numeric for indexed types today (notes, files,
deck cards, news items), but typed as ``int | str`` to allow
future doc types that use string identifiers (e.g., file paths).
doc_type: Document type (note, file, calendar, contact, etc.) doc_type: Document type (note, file, calendar, contact, etc.)
title: Document title title: Document title
excerpt: Content excerpt showing match context excerpt: Content excerpt showing match context
@@ -144,7 +151,7 @@ class SearchResult:
point_id: Qdrant point ID for batch vector retrieval (None if not from Qdrant) point_id: Qdrant point ID for batch vector retrieval (None if not from Qdrant)
""" """
id: int id: int | str
doc_type: str doc_type: str
title: str title: str
excerpt: str excerpt: str
@@ -233,9 +233,14 @@ class BM25HybridSearchAlgorithm(SearchAlgorithm):
metadata["path"] = path metadata["path"] = path
# Add deck_card-specific metadata for frontend URL construction # Add deck_card-specific metadata for frontend URL construction
# and verify-on-read (ADR-019) — both board_id and stack_id are
# required to call deck.get_card without an O(boards × stacks)
# iteration fallback.
if doc_type == "deck_card": if doc_type == "deck_card":
if board_id := result.payload.get("board_id"): if board_id := result.payload.get("board_id"):
metadata["board_id"] = board_id metadata["board_id"] = board_id
if stack_id := result.payload.get("stack_id"):
metadata["stack_id"] = stack_id
# Return unverified results (verification happens at output stage) # Return unverified results (verification happens at output stage)
results.append( results.append(
+5
View File
@@ -164,9 +164,14 @@ class SemanticSearchAlgorithm(SearchAlgorithm):
metadata["path"] = path metadata["path"] = path
# Add deck_card-specific metadata for frontend URL construction # Add deck_card-specific metadata for frontend URL construction
# and verify-on-read (ADR-019) — both board_id and stack_id are
# required to call deck.get_card without an O(boards × stacks)
# iteration fallback.
if doc_type == "deck_card": if doc_type == "deck_card":
if board_id := result.payload.get("board_id"): if board_id := result.payload.get("board_id"):
metadata["board_id"] = board_id metadata["board_id"] = board_id
if stack_id := result.payload.get("stack_id"):
metadata["stack_id"] = stack_id
# Return unverified results (verification happens at output stage) # Return unverified results (verification happens at output stage)
results.append( results.append(
+187 -219
View File
@@ -6,10 +6,18 @@ against Nextcloud at query time, dropping any that the user can no longer
access (deleted, unshared, etc.) and lazily evicting them from the index. access (deleted, unshared, etc.) and lazily evicting them from the index.
Per-doc_type verifiers are registered in ``_VERIFIERS``. Each takes the Per-doc_type verifiers are registered in ``_VERIFIERS``. Each takes the
authenticated client, a list of doc_ids, and the user_id, and returns the authenticated client, the (deduplicated) list of ``SearchResult``s for that
subset of doc_ids that are currently accessible. The dispatch deliberately doc_type, and a shared concurrency semaphore. They return the subset of
groups by doc_type so doc-types with cheap batch endpoints (news_item) can ``doc_id`` values that are currently accessible. Verifiers read whatever
do a single fetch rather than one round-trip per result. metadata they need (file path, deck card board/stack ids) directly from the
SearchResult — these fields are populated at index-time and propagated by
the algorithm layer (see ``search/bm25_hybrid.py`` and ``search/semantic.py``)
so verification adds zero extra Qdrant round-trips.
Concurrency is bounded by a shared semaphore (default 20) so a large search
result page (or a multi-doc_type query) cannot exhaust the httpx connection
pool or trigger Nextcloud rate limiting. The 20-slot default matches the
context-expansion convention in ``server/semantic.py``.
Failure policy: Failure policy:
@@ -28,18 +36,22 @@ from typing import Any
import anyio import anyio
from httpx import HTTPStatusError from httpx import HTTPStatusError
from qdrant_client.models import FieldCondition, Filter, MatchValue
from nextcloud_mcp_server.config import get_settings
from nextcloud_mcp_server.search.algorithms import SearchResult from nextcloud_mcp_server.search.algorithms import SearchResult
from nextcloud_mcp_server.vector.eviction import delete_document_points from nextcloud_mcp_server.vector.eviction import delete_document_points
from nextcloud_mcp_server.vector.qdrant_client import get_qdrant_client
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
BatchVerifier = Callable[[Any, list[int | str], str], Awaitable[set[int | str]]] # Default cap on concurrent verification round-trips against Nextcloud. Matches
"""(client, doc_ids, user_id) -> set of accessible doc_ids.""" # the convention in ``server/semantic.py`` for context-expansion fan-out.
DEFAULT_VERIFICATION_CONCURRENCY = 20
BatchVerifier = Callable[
[Any, list[SearchResult], anyio.Semaphore], Awaitable[set[int | str]]
]
"""(client, results, semaphore) -> set of doc_ids accessible to the user."""
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
@@ -55,185 +67,200 @@ def _is_definitive_404_or_403(exc: BaseException) -> bool:
async def _verify_notes( async def _verify_notes(
client: Any, doc_ids: list[int | str], user_id: str client: Any, results: list[SearchResult], semaphore: anyio.Semaphore
) -> set[int | str]: ) -> set[int | str]:
accessible: set[int | str] = set() accessible: set[int | str] = set()
async def check(doc_id: int | str) -> None: async def check(result: SearchResult) -> None:
try: async with semaphore:
await client.notes.get_note(int(doc_id)) doc_id = result.id
accessible.add(doc_id) try:
except HTTPStatusError as e: await client.notes.get_note(int(doc_id))
if _is_definitive_404_or_403(e): accessible.add(doc_id)
return except HTTPStatusError as e:
logger.warning( if _is_definitive_404_or_403(e):
"Transient error verifying note %s: %s %s; keeping result", return
doc_id, logger.warning(
e.response.status_code, "Transient error verifying note %s: %s %s; keeping result",
e, doc_id,
) e.response.status_code,
accessible.add(doc_id) e,
except Exception as e: )
logger.warning( accessible.add(doc_id)
"Unexpected error verifying note %s: %s; keeping result", except Exception as e:
doc_id, logger.warning(
e, "Unexpected error verifying note %s: %s; keeping result",
) doc_id,
accessible.add(doc_id) e,
)
accessible.add(doc_id)
async with anyio.create_task_group() as tg: async with anyio.create_task_group() as tg:
for doc_id in doc_ids: for r in results:
tg.start_soon(check, doc_id) tg.start_soon(check, r)
return accessible return accessible
async def _verify_files( async def _verify_files(
client: Any, doc_ids: list[int | str], user_id: str client: Any, results: list[SearchResult], semaphore: anyio.Semaphore
) -> set[int | str]: ) -> set[int | str]:
accessible: set[int | str] = set() accessible: set[int | str] = set()
async def check(doc_id: int | str) -> None: async def check(result: SearchResult) -> None:
# Resolve file_id → file_path from Qdrant payload doc_id = result.id
file_path = await _resolve_file_path(user_id, doc_id) # file_path is propagated from the Qdrant payload by the algorithm
if file_path is None: # layer (bm25_hybrid.py / semantic.py). No extra Qdrant round-trip.
file_path = (result.metadata or {}).get("path")
if not file_path:
# Cannot verify without a path; treat as accessible to avoid # Cannot verify without a path; treat as accessible to avoid
# silently dropping legitimate results when payload is missing # silently dropping legitimate results when payload is missing
# (legacy data, or a future doc_type that doesn't propagate path).
logger.warning( logger.warning(
"No file_path in Qdrant for file_id %s; keeping result " "No file path in metadata for file_id %s; keeping result "
"(verification skipped)", "(verification skipped)",
doc_id, doc_id,
) )
accessible.add(doc_id) accessible.add(doc_id)
return return
try: async with semaphore:
info = await client.webdav.get_file_info(file_path) try:
if info is None: info = await client.webdav.get_file_info(file_path)
# get_file_info returns None on definitive 404 if info is None:
return # get_file_info returns None on definitive 404
accessible.add(doc_id) return
except HTTPStatusError as e: accessible.add(doc_id)
if _is_definitive_404_or_403(e): except HTTPStatusError as e:
return if _is_definitive_404_or_403(e):
logger.warning( return
"Transient error verifying file %s (%s): %s %s; keeping result", logger.warning(
doc_id, "Transient error verifying file %s (%s): %s %s; keeping result",
file_path, doc_id,
e.response.status_code, file_path,
e, e.response.status_code,
) e,
accessible.add(doc_id) )
except Exception as e: accessible.add(doc_id)
logger.warning( except Exception as e:
"Unexpected error verifying file %s (%s): %s; keeping result", logger.warning(
doc_id, "Unexpected error verifying file %s (%s): %s; keeping result",
file_path, doc_id,
e, file_path,
) e,
accessible.add(doc_id) )
accessible.add(doc_id)
async with anyio.create_task_group() as tg: async with anyio.create_task_group() as tg:
for doc_id in doc_ids: for r in results:
tg.start_soon(check, doc_id) tg.start_soon(check, r)
return accessible return accessible
async def _verify_deck_cards( async def _verify_deck_cards(
client: Any, doc_ids: list[int | str], user_id: str client: Any, results: list[SearchResult], semaphore: anyio.Semaphore
) -> set[int | str]: ) -> set[int | str]:
accessible: set[int | str] = set() accessible: set[int | str] = set()
async def check(doc_id: int | str) -> None: async def check(result: SearchResult) -> None:
# Resolve card_id → (board_id, stack_id) from Qdrant payload doc_id = result.id
meta = await _resolve_deck_metadata(user_id, int(doc_id)) # board_id and stack_id are propagated from the Qdrant payload by the
if meta is None: # algorithm layer. No extra Qdrant round-trip.
meta = result.metadata or {}
board_id = meta.get("board_id")
stack_id = meta.get("stack_id")
if board_id is None or stack_id is None:
# Without metadata we cannot run the cheap fast-path. Per ADR-019 # Without metadata we cannot run the cheap fast-path. Per ADR-019
# we deliberately do NOT fall back to O(boards × stacks) iteration # we deliberately do NOT fall back to O(boards × stacks) iteration
# in the search hot path; treat as accessible. # in the search hot path; treat as accessible.
logger.warning( logger.warning(
"No deck metadata in Qdrant for card %s; keeping result " "Incomplete deck metadata for card %s (board_id=%s, stack_id=%s); "
"(verification skipped, legacy data without board_id/stack_id)", "keeping result (verification skipped, legacy data)",
doc_id, doc_id,
board_id,
stack_id,
) )
accessible.add(doc_id) accessible.add(doc_id)
return return
try: async with semaphore:
await client.deck.get_card( try:
board_id=meta["board_id"], await client.deck.get_card(
stack_id=meta["stack_id"], board_id=int(board_id),
card_id=int(doc_id), stack_id=int(stack_id),
) card_id=int(doc_id),
accessible.add(doc_id) )
except HTTPStatusError as e: accessible.add(doc_id)
if _is_definitive_404_or_403(e): except HTTPStatusError as e:
return if _is_definitive_404_or_403(e):
logger.warning( return
"Transient error verifying deck card %s: %s %s; keeping result", logger.warning(
doc_id, "Transient error verifying deck card %s: %s %s; keeping result",
e.response.status_code, doc_id,
e, e.response.status_code,
) e,
accessible.add(doc_id) )
except Exception as e: accessible.add(doc_id)
logger.warning( except Exception as e:
"Unexpected error verifying deck card %s: %s; keeping result", logger.warning(
doc_id, "Unexpected error verifying deck card %s: %s; keeping result",
e, doc_id,
) e,
accessible.add(doc_id) )
accessible.add(doc_id)
async with anyio.create_task_group() as tg: async with anyio.create_task_group() as tg:
for doc_id in doc_ids: for r in results:
tg.start_soon(check, doc_id) tg.start_soon(check, r)
return accessible return accessible
async def _verify_news_items( async def _verify_news_items(
client: Any, doc_ids: list[int | str], user_id: str client: Any, results: list[SearchResult], semaphore: anyio.Semaphore
) -> set[int | str]: ) -> set[int | str]:
"""Batch-verify news items with a single fetch. """Batch-verify news items with a single fetch.
The Nextcloud News API has no per-item endpoint, so ``news.get_item`` is The Nextcloud News API has no per-item endpoint, so ``news.get_item`` is
implemented as a fetch-all + filter — which would be O(N × all_items) if implemented as a fetch-all + filter — which would be O(N × all_items) if
called per id. Instead we fetch once and intersect. called per id. Instead we fetch once and intersect. The semaphore is
accepted for signature symmetry but not heavily used (one round-trip total).
""" """
requested = {int(d) for d in doc_ids} doc_ids = [r.id for r in results]
try: async with semaphore:
items = await client.news.get_items(batch_size=-1, get_read=True) try:
except HTTPStatusError as e: items = await client.news.get_items(batch_size=-1, get_read=True)
# If the News API itself is gone (app disabled, user lost access), except HTTPStatusError as e:
# treat *all* requested items as inaccessible. Eviction will reclaim. # If the News API itself is gone (app disabled, user lost access),
if _is_definitive_404_or_403(e): # treat *all* requested items as inaccessible. Eviction will reclaim.
logger.info( if _is_definitive_404_or_403(e):
"News API returned %s for user %s; treating all %d news_items as inaccessible", logger.info(
"News API returned %s for user %s; treating all %d news_items as inaccessible",
e.response.status_code,
client.username,
len(doc_ids),
)
return set()
logger.warning(
"Transient error fetching news items for verification: %s %s; keeping all results",
e.response.status_code, e.response.status_code,
user_id, e,
len(requested),
) )
return set() return set(doc_ids)
logger.warning( except Exception as e:
"Transient error fetching news items for verification: %s %s; keeping all results", logger.warning(
e.response.status_code, "Unexpected error fetching news items for verification: %s; keeping all results",
e, e,
) )
return set(doc_ids) return set(doc_ids)
except Exception as e:
logger.warning(
"Unexpected error fetching news items for verification: %s; keeping all results",
e,
)
return set(doc_ids)
present_ids = {int(item.get("id")) for item in items if item.get("id") is not None} present_ids = {int(item.get("id")) for item in items if item.get("id") is not None}
# Map back to the original doc_id types (the caller may pass ints or strs) # Map back to the original doc_id types (caller may pass ints or strs).
accessible: set[int | str] = set() accessible: set[int | str] = set()
for d in doc_ids: for d in doc_ids:
if int(d) in present_ids and int(d) in requested: if int(d) in present_ids:
accessible.add(d) accessible.add(d)
return accessible return accessible
@@ -255,77 +282,6 @@ def get_supported_doc_types() -> set[str]:
return set(_VERIFIERS.keys()) return set(_VERIFIERS.keys())
# ---------------------------------------------------------------------------
# Qdrant payload lookup helpers
# ---------------------------------------------------------------------------
async def _resolve_file_path(user_id: str, doc_id: int | str) -> str | None:
"""Look up file_path for a file_id from any chunk's Qdrant payload."""
try:
qdrant_client = await get_qdrant_client()
settings = get_settings()
scroll_result = await qdrant_client.scroll(
collection_name=settings.get_collection_name(),
scroll_filter=Filter(
must=[
FieldCondition(key="user_id", match=MatchValue(value=user_id)),
FieldCondition(key="doc_id", match=MatchValue(value=doc_id)),
FieldCondition(key="doc_type", match=MatchValue(value="file")),
]
),
limit=1,
with_payload=["file_path"],
with_vectors=False,
)
if scroll_result[0]:
point = scroll_result[0][0]
file_path = point.payload.get("file_path") if point.payload else None
if file_path:
return str(file_path)
return None
except Exception as e:
logger.debug("Error resolving file_path for file_id %s: %s", doc_id, e)
return None
async def _resolve_deck_metadata(user_id: str, card_id: int) -> dict[str, int] | None:
"""Look up (board_id, stack_id) for a deck card from any chunk's payload."""
try:
qdrant_client = await get_qdrant_client()
settings = get_settings()
scroll_result = await qdrant_client.scroll(
collection_name=settings.get_collection_name(),
scroll_filter=Filter(
must=[
FieldCondition(key="user_id", match=MatchValue(value=user_id)),
FieldCondition(key="doc_id", match=MatchValue(value=card_id)),
FieldCondition(key="doc_type", match=MatchValue(value="deck_card")),
]
),
limit=1,
with_payload=["board_id", "stack_id"],
with_vectors=False,
)
if scroll_result[0]:
point = scroll_result[0][0]
payload = point.payload or {}
board_id = payload.get("board_id")
stack_id = payload.get("stack_id")
if board_id is not None and stack_id is not None:
return {"board_id": int(board_id), "stack_id": int(stack_id)}
return None
except Exception as e:
logger.debug("Error resolving deck metadata for card %s: %s", card_id, e)
return None
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
# Public entry point # Public entry point
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
@@ -336,13 +292,14 @@ async def verify_search_results(
results: list[SearchResult], results: list[SearchResult],
*, *,
evict_on_missing: bool = True, evict_on_missing: bool = True,
max_concurrent: int = DEFAULT_VERIFICATION_CONCURRENCY,
) -> list[SearchResult]: ) -> list[SearchResult]:
"""Filter search results to those the user can currently access. """Filter search results to those the user can currently access.
Deduplicates by ``(doc_id, doc_type)`` before verifying, so multiple Deduplicates by ``(doc_id, doc_type)`` before verifying, so multiple
chunks from the same document cost a single check. Verifiers run chunks from the same document cost a single check. Verifiers run
concurrently per doc_type (and within each doc_type, per id where that concurrently per doc_type and concurrently per id within each verifier,
is cheaper than batching). bounded by a shared semaphore (``max_concurrent``).
When ``evict_on_missing=True``, points for documents that fail When ``evict_on_missing=True``, points for documents that fail
verification are deleted from Qdrant in-line. Eviction failures are verification are deleted from Qdrant in-line. Eviction failures are
@@ -353,6 +310,8 @@ async def verify_search_results(
results: SearchResult list from the algorithm layer (may include results: SearchResult list from the algorithm layer (may include
multiple chunks per document). multiple chunks per document).
evict_on_missing: Schedule lazy eviction for inaccessible docs. evict_on_missing: Schedule lazy eviction for inaccessible docs.
max_concurrent: Cap on concurrent verification round-trips against
Nextcloud. Defaults to ``DEFAULT_VERIFICATION_CONCURRENCY``.
Returns: Returns:
Filtered list preserving the original order. Filtered list preserving the original order.
@@ -363,28 +322,33 @@ async def verify_search_results(
user_id: str = client.username user_id: str = client.username
# Group unique (doc_id, doc_type) by doc_type so each verifier sees a # Group unique (doc_id, doc_type) by doc_type so each verifier sees a
# deduplicated batch. # deduplicated batch. We pick one SearchResult per (id, doc_type) to carry
by_type: dict[str, set[int | str]] = {} # metadata (path, board_id/stack_id) into the verifier — chunks of the
# same document share these fields, so any chunk works.
by_type: dict[str, dict[int | str, SearchResult]] = {}
for r in results: for r in results:
by_type.setdefault(r.doc_type, set()).add(r.id) by_type.setdefault(r.doc_type, {}).setdefault(r.id, r)
# Shared semaphore bounds total Nextcloud round-trips across all
# per-id verifiers. Without it, a 50-result mostly-notes page could fan
# out 50 concurrent get_note calls and exhaust the connection pool.
semaphore = anyio.Semaphore(max_concurrent)
# Run all type verifiers concurrently. Per-id failures are absorbed
# inside each verifier; this outer task group only fans out per type.
accessible_by_type: dict[str, set[int | str]] = {} accessible_by_type: dict[str, set[int | str]] = {}
async def run_verifier(doc_type: str, doc_ids: set[int | str]) -> None: async def run_verifier(doc_type: str, unique_results: list[SearchResult]) -> None:
verifier = _VERIFIERS.get(doc_type) verifier = _VERIFIERS.get(doc_type)
if verifier is None: if verifier is None:
logger.warning( logger.warning(
"No verifier registered for doc_type=%r; keeping %d result(s) unverified", "No verifier registered for doc_type=%r; keeping %d result(s) unverified",
doc_type, doc_type,
len(doc_ids), len(unique_results),
) )
accessible_by_type[doc_type] = doc_ids accessible_by_type[doc_type] = {r.id for r in unique_results}
return return
try: try:
accessible_by_type[doc_type] = await verifier( accessible_by_type[doc_type] = await verifier(
client, list(doc_ids), user_id client, unique_results, semaphore
) )
except Exception as e: except Exception as e:
# Verifier itself blew up (not per-id) — fail open. # Verifier itself blew up (not per-id) — fail open.
@@ -392,20 +356,20 @@ async def verify_search_results(
"Verifier for doc_type=%s raised: %s; keeping all %d result(s) unverified", "Verifier for doc_type=%s raised: %s; keeping all %d result(s) unverified",
doc_type, doc_type,
e, e,
len(doc_ids), len(unique_results),
exc_info=True, exc_info=True,
) )
accessible_by_type[doc_type] = doc_ids accessible_by_type[doc_type] = {r.id for r in unique_results}
async with anyio.create_task_group() as tg: async with anyio.create_task_group() as tg:
for doc_type, doc_ids in by_type.items(): for doc_type, id_to_result in by_type.items():
tg.start_soon(run_verifier, doc_type, doc_ids) tg.start_soon(run_verifier, doc_type, list(id_to_result.values()))
# Compute (doc_id, doc_type) pairs that failed verification # Compute (doc_id, doc_type) pairs that failed verification
inaccessible: set[tuple[int | str, str]] = set() inaccessible: set[tuple[int | str, str]] = set()
for doc_type, doc_ids in by_type.items(): for doc_type, id_to_result in by_type.items():
accessible = accessible_by_type.get(doc_type, doc_ids) accessible = accessible_by_type.get(doc_type, set(id_to_result.keys()))
for doc_id in doc_ids: for doc_id in id_to_result.keys():
if doc_id not in accessible: if doc_id not in accessible:
inaccessible.add((doc_id, doc_type)) inaccessible.add((doc_id, doc_type))
@@ -416,11 +380,15 @@ async def verify_search_results(
sorted((str(d), t) for d, t in inaccessible), sorted((str(d), t) for d, t in inaccessible),
) )
# Filter results in-place-style, preserving order # Filter results, preserving order. All chunks of an inaccessible document
# are dropped together (dedup happened before verification, but the result
# list still contains all chunks).
kept = [r for r in results if (r.id, r.doc_type) not in inaccessible] kept = [r for r in results if (r.id, r.doc_type) not in inaccessible]
# Lazy eviction — fire and forget, but bounded inline so we don't lose # Lazy eviction. Runs inline before returning — slow Qdrant will delay
# the user_id binding by escaping the task group. # the search response. Background eviction would need a task registered
# on the lifespan context; the inline approach is acceptable given
# typical Qdrant delete latency, but callers should be aware.
if evict_on_missing and inaccessible: if evict_on_missing and inaccessible:
async def evict(doc_id: int | str, doc_type: str) -> None: async def evict(doc_id: int | str, doc_type: str) -> None:
+8 -3
View File
@@ -142,8 +142,13 @@ def configure_semantic_tools(mcp: FastMCP):
) )
all_results.extend(unverified_results) all_results.extend(unverified_results)
# Sort combined results by score # Sort combined results by score, then cap to `limit * 2` to
# match the cross-app branch's over-fetch budget. Without this
# cap, N requested doc_types × `limit * 2` results would all
# flow into verification, multiplying the Nextcloud round-trip
# cost by N.
all_results.sort(key=lambda r: r.score, reverse=True) all_results.sort(key=lambda r: r.score, reverse=True)
all_results = all_results[: limit * 2]
# ADR-019: Verify-on-read. The vector index is a recall layer; # ADR-019: Verify-on-read. The vector index is a recall layer;
# Nextcloud is the source of truth for access. Filter out ghost # Nextcloud is the source of truth for access. Filter out ghost
@@ -300,7 +305,7 @@ def configure_semantic_tools(mcp: FastMCP):
title="Search with AI-Generated Answer", title="Search with AI-Generated Answer",
annotations=ToolAnnotations( annotations=ToolAnnotations(
readOnlyHint=True, # Search doesn't modify data readOnlyHint=True, # Search doesn't modify data
openWorldHint=False, # Searches only indexed Nextcloud data openWorldHint=True, # Calls into Nextcloud via nc_semantic_search
), ),
) )
@require_scopes("semantic.read") @require_scopes("semantic.read")
@@ -432,7 +437,7 @@ def configure_semantic_tools(mcp: FastMCP):
async with semaphore: async with semaphore:
if result.doc_type == "note": if result.doc_type == "note":
try: try:
note = await client.notes.get_note(result.id) note = await client.notes.get_note(int(result.id))
content = note.get("content", "") content = note.get("content", "")
accessible_results[index] = result accessible_results[index] = result
full_contents[index] = content full_contents[index] = content
+158 -44
View File
@@ -2,6 +2,7 @@
from types import SimpleNamespace from types import SimpleNamespace
import anyio
import httpx import httpx
import pytest import pytest
from httpx import HTTPStatusError from httpx import HTTPStatusError
@@ -22,11 +23,16 @@ from nextcloud_mcp_server.search.verification import (
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
def _sem(slots: int = 20) -> anyio.Semaphore:
return anyio.Semaphore(slots)
def _make_result( def _make_result(
doc_id: int, doc_id: int | str,
doc_type: str = "note", doc_type: str = "note",
chunk_index: int = 0, chunk_index: int = 0,
score: float = 0.9, score: float = 0.9,
metadata: dict | None = None,
) -> SearchResult: ) -> SearchResult:
return SearchResult( return SearchResult(
id=doc_id, id=doc_id,
@@ -35,6 +41,7 @@ def _make_result(
excerpt="...", excerpt="...",
score=score, score=score,
chunk_index=chunk_index, chunk_index=chunk_index,
metadata=metadata,
) )
@@ -72,7 +79,9 @@ async def test_verify_notes_200_keeps_all(mocker):
) )
client = SimpleNamespace(notes=notes_client, username="alice") client = SimpleNamespace(notes=notes_client, username="alice")
result = await _verify_notes(client, [1, 2, 3], "alice") result = await _verify_notes(
client, [_make_result(1), _make_result(2), _make_result(3)], _sem()
)
assert result == {1, 2, 3} assert result == {1, 2, 3}
assert notes_client.get_note.await_count == 3 assert notes_client.get_note.await_count == 3
@@ -85,7 +94,7 @@ async def test_verify_notes_404_drops(mocker):
) )
client = SimpleNamespace(notes=notes_client, username="alice") client = SimpleNamespace(notes=notes_client, username="alice")
result = await _verify_notes(client, [42], "alice") result = await _verify_notes(client, [_make_result(42)], _sem())
assert result == set() assert result == set()
@@ -97,7 +106,7 @@ async def test_verify_notes_403_drops(mocker):
) )
client = SimpleNamespace(notes=notes_client, username="alice") client = SimpleNamespace(notes=notes_client, username="alice")
result = await _verify_notes(client, [42], "alice") result = await _verify_notes(client, [_make_result(42)], _sem())
assert result == set() assert result == set()
@@ -110,7 +119,7 @@ async def test_verify_notes_transient_5xx_keeps(mocker):
) )
client = SimpleNamespace(notes=notes_client, username="alice") client = SimpleNamespace(notes=notes_client, username="alice")
result = await _verify_notes(client, [42], "alice") result = await _verify_notes(client, [_make_result(42)], _sem())
assert result == {42} assert result == {42}
@@ -122,7 +131,7 @@ async def test_verify_notes_unexpected_exception_keeps(mocker):
) )
client = SimpleNamespace(notes=notes_client, username="alice") client = SimpleNamespace(notes=notes_client, username="alice")
result = await _verify_notes(client, [7], "alice") result = await _verify_notes(client, [_make_result(7)], _sem())
assert result == {7} assert result == {7}
@@ -143,7 +152,9 @@ async def test_verify_notes_mixed_outcomes(mocker):
notes_client = SimpleNamespace(get_note=mocker.AsyncMock(side_effect=side_effect)) notes_client = SimpleNamespace(get_note=mocker.AsyncMock(side_effect=side_effect))
client = SimpleNamespace(notes=notes_client, username="alice") client = SimpleNamespace(notes=notes_client, username="alice")
result = await _verify_notes(client, [1, 2, 3], "alice") result = await _verify_notes(
client, [_make_result(1), _make_result(2), _make_result(3)], _sem()
)
assert result == {1, 3} assert result == {1, 3}
@@ -161,7 +172,15 @@ async def test_verify_news_items_intersects_with_fetched_set(mocker):
) )
client = SimpleNamespace(news=news_client, username="alice") client = SimpleNamespace(news=news_client, username="alice")
result = await _verify_news_items(client, [10, 20, 99], "alice") result = await _verify_news_items(
client,
[
_make_result(10, doc_type="news_item"),
_make_result(20, doc_type="news_item"),
_make_result(99, doc_type="news_item"),
],
_sem(),
)
assert result == {10, 20} assert result == {10, 20}
assert news_client.get_items.await_count == 1 assert news_client.get_items.await_count == 1
@@ -174,7 +193,15 @@ async def test_verify_news_items_api_404_drops_all(mocker):
) )
client = SimpleNamespace(news=news_client, username="alice") client = SimpleNamespace(news=news_client, username="alice")
result = await _verify_news_items(client, [1, 2, 3], "alice") result = await _verify_news_items(
client,
[
_make_result(1, doc_type="news_item"),
_make_result(2, doc_type="news_item"),
_make_result(3, doc_type="news_item"),
],
_sem(),
)
assert result == set() assert result == set()
@@ -186,7 +213,15 @@ async def test_verify_news_items_transient_keeps_all(mocker):
) )
client = SimpleNamespace(news=news_client, username="alice") client = SimpleNamespace(news=news_client, username="alice")
result = await _verify_news_items(client, [1, 2, 3], "alice") result = await _verify_news_items(
client,
[
_make_result(1, doc_type="news_item"),
_make_result(2, doc_type="news_item"),
_make_result(3, doc_type="news_item"),
],
_sem(),
)
assert result == {1, 2, 3} assert result == {1, 2, 3}
@@ -197,16 +232,18 @@ async def test_verify_news_items_transient_keeps_all(mocker):
@pytest.mark.unit @pytest.mark.unit
async def test_verify_files_uses_propfind_when_path_resolves(mocker): async def test_verify_files_uses_path_from_metadata(mocker):
mocker.patch.object( """File verifier reads path from SearchResult.metadata, no Qdrant round-trip."""
verification, "_resolve_file_path", return_value="Documents/foo.txt"
)
webdav_client = SimpleNamespace( webdav_client = SimpleNamespace(
get_file_info=mocker.AsyncMock(return_value={"id": 100}) get_file_info=mocker.AsyncMock(return_value={"id": 100})
) )
client = SimpleNamespace(webdav=webdav_client, username="alice") client = SimpleNamespace(webdav=webdav_client, username="alice")
result = await _verify_files(client, [100], "alice") result = await _verify_files(
client,
[_make_result(100, doc_type="file", metadata={"path": "Documents/foo.txt"})],
_sem(),
)
assert result == {100} assert result == {100}
webdav_client.get_file_info.assert_awaited_once_with("Documents/foo.txt") webdav_client.get_file_info.assert_awaited_once_with("Documents/foo.txt")
@@ -215,27 +252,54 @@ async def test_verify_files_uses_propfind_when_path_resolves(mocker):
@pytest.mark.unit @pytest.mark.unit
async def test_verify_files_404_via_get_file_info_drops(mocker): async def test_verify_files_404_via_get_file_info_drops(mocker):
"""get_file_info returns None on 404 — that's a definitive drop.""" """get_file_info returns None on 404 — that's a definitive drop."""
mocker.patch.object(verification, "_resolve_file_path", return_value="gone.txt")
webdav_client = SimpleNamespace(get_file_info=mocker.AsyncMock(return_value=None)) webdav_client = SimpleNamespace(get_file_info=mocker.AsyncMock(return_value=None))
client = SimpleNamespace(webdav=webdav_client, username="alice") client = SimpleNamespace(webdav=webdav_client, username="alice")
result = await _verify_files(client, [123], "alice") result = await _verify_files(
client,
[_make_result(123, doc_type="file", metadata={"path": "gone.txt"})],
_sem(),
)
assert result == set() assert result == set()
@pytest.mark.unit @pytest.mark.unit
async def test_verify_files_missing_payload_keeps_unverified(mocker): async def test_verify_files_missing_path_metadata_keeps_unverified(mocker):
"""Without a file_path we cannot verify — fail open, don't drop.""" """Without a path in metadata we cannot verify — fail open, don't drop."""
mocker.patch.object(verification, "_resolve_file_path", return_value=None) webdav_client = SimpleNamespace(
webdav_client = SimpleNamespace(get_file_info=mocker.AsyncMock(return_value=None)) get_file_info=mocker.AsyncMock(side_effect=AssertionError("must not be called"))
)
client = SimpleNamespace(webdav=webdav_client, username="alice") client = SimpleNamespace(webdav=webdav_client, username="alice")
result = await _verify_files(client, [555], "alice") # No metadata at all
result = await _verify_files(client, [_make_result(555, doc_type="file")], _sem())
assert result == {555} assert result == {555}
webdav_client.get_file_info.assert_not_awaited() webdav_client.get_file_info.assert_not_awaited()
# Metadata present but no "path" key
result = await _verify_files(
client, [_make_result(556, doc_type="file", metadata={})], _sem()
)
assert result == {556}
webdav_client.get_file_info.assert_not_awaited()
@pytest.mark.unit
async def test_verify_files_transient_5xx_keeps(mocker):
webdav_client = SimpleNamespace(
get_file_info=mocker.AsyncMock(side_effect=_http_error(503))
)
client = SimpleNamespace(webdav=webdav_client, username="alice")
result = await _verify_files(
client,
[_make_result(7, doc_type="file", metadata={"path": "x.txt"})],
_sem(),
)
assert result == {7}
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
# Deck card verifier # Deck card verifier
@@ -244,15 +308,21 @@ async def test_verify_files_missing_payload_keeps_unverified(mocker):
@pytest.mark.unit @pytest.mark.unit
async def test_verify_deck_cards_uses_metadata_fast_path(mocker): async def test_verify_deck_cards_uses_metadata_fast_path(mocker):
mocker.patch.object( """Deck verifier reads board_id+stack_id from metadata, no Qdrant round-trip."""
verification,
"_resolve_deck_metadata",
return_value={"board_id": 1, "stack_id": 2},
)
deck_client = SimpleNamespace(get_card=mocker.AsyncMock(return_value=object())) deck_client = SimpleNamespace(get_card=mocker.AsyncMock(return_value=object()))
client = SimpleNamespace(deck=deck_client, username="alice") client = SimpleNamespace(deck=deck_client, username="alice")
result = await _verify_deck_cards(client, [42], "alice") result = await _verify_deck_cards(
client,
[
_make_result(
42,
doc_type="deck_card",
metadata={"board_id": 1, "stack_id": 2},
)
],
_sem(),
)
assert result == {42} assert result == {42}
deck_client.get_card.assert_awaited_once_with(board_id=1, stack_id=2, card_id=42) deck_client.get_card.assert_awaited_once_with(board_id=1, stack_id=2, card_id=42)
@@ -261,33 +331,56 @@ async def test_verify_deck_cards_uses_metadata_fast_path(mocker):
@pytest.mark.unit @pytest.mark.unit
async def test_verify_deck_cards_403_drops(mocker): async def test_verify_deck_cards_403_drops(mocker):
"""Board unshared with user → 403 from get_card → drop.""" """Board unshared with user → 403 from get_card → drop."""
mocker.patch.object(
verification,
"_resolve_deck_metadata",
return_value={"board_id": 1, "stack_id": 2},
)
deck_client = SimpleNamespace( deck_client = SimpleNamespace(
get_card=mocker.AsyncMock(side_effect=_http_error(403)) get_card=mocker.AsyncMock(side_effect=_http_error(403))
) )
client = SimpleNamespace(deck=deck_client, username="alice") client = SimpleNamespace(deck=deck_client, username="alice")
result = await _verify_deck_cards(client, [42], "alice") result = await _verify_deck_cards(
client,
[
_make_result(
42,
doc_type="deck_card",
metadata={"board_id": 1, "stack_id": 2},
)
],
_sem(),
)
assert result == set() assert result == set()
@pytest.mark.unit @pytest.mark.unit
async def test_verify_deck_cards_no_metadata_skips_verification(mocker): async def test_verify_deck_cards_missing_metadata_keeps_unverified(mocker):
"""Legacy data without board_id/stack_id payload → keep, do NOT iterate.""" """Legacy data without board_id/stack_id → keep, do NOT iterate or call API."""
mocker.patch.object(verification, "_resolve_deck_metadata", return_value=None)
deck_client = SimpleNamespace( deck_client = SimpleNamespace(
get_card=mocker.AsyncMock(side_effect=AssertionError("must not be called")) get_card=mocker.AsyncMock(side_effect=AssertionError("must not be called"))
) )
client = SimpleNamespace(deck=deck_client, username="alice") client = SimpleNamespace(deck=deck_client, username="alice")
result = await _verify_deck_cards(client, [42], "alice") # No metadata at all
result = await _verify_deck_cards(
client, [_make_result(42, doc_type="deck_card")], _sem()
)
assert result == {42} assert result == {42}
# Only board_id (stack_id missing)
result = await _verify_deck_cards(
client,
[_make_result(43, doc_type="deck_card", metadata={"board_id": 1})],
_sem(),
)
assert result == {43}
# Only stack_id (board_id missing)
result = await _verify_deck_cards(
client,
[_make_result(44, doc_type="deck_card", metadata={"stack_id": 2})],
_sem(),
)
assert result == {44}
deck_client.get_card.assert_not_awaited() deck_client.get_card.assert_not_awaited()
@@ -320,9 +413,12 @@ async def test_verify_search_results_dedupes_chunks_per_document(mocker):
assert len(kept) == 3 # all kept, all reference the same accessible doc assert len(kept) == 3 # all kept, all reference the same accessible doc
spy.assert_awaited_once() spy.assert_awaited_once()
# Verifier received the single deduplicated id, not three copies # Verifier received exactly one SearchResult (the deduplicated representative)
args, _kwargs = spy.call_args args, _kwargs = spy.call_args
assert args[1] == [1] assert len(args[1]) == 1
assert args[1][0].id == 1
# And a semaphore as the third arg
assert isinstance(args[2], anyio.Semaphore)
@pytest.mark.unit @pytest.mark.unit
@@ -454,8 +550,8 @@ async def test_verify_search_results_dispatches_per_doc_type_concurrently(mocker
results = [ results = [
_make_result(1, doc_type="note"), _make_result(1, doc_type="note"),
_make_result(500, doc_type="file"), _make_result(500, doc_type="file", metadata={"path": "a.txt"}),
_make_result(999, doc_type="file"), # to be dropped _make_result(999, doc_type="file", metadata={"path": "b.txt"}), # to be dropped
] ]
client = SimpleNamespace(username="alice") client = SimpleNamespace(username="alice")
@@ -464,3 +560,21 @@ async def test_verify_search_results_dispatches_per_doc_type_concurrently(mocker
assert {(r.id, r.doc_type) for r in kept} == {(1, "note"), (500, "file")} assert {(r.id, r.doc_type) for r in kept} == {(1, "note"), (500, "file")}
note_verifier.assert_awaited_once() note_verifier.assert_awaited_once()
file_verifier.assert_awaited_once() file_verifier.assert_awaited_once()
@pytest.mark.unit
async def test_verify_search_results_passes_semaphore_to_verifier(mocker):
"""The dispatcher must construct a Semaphore and pass it to verifiers."""
captured: dict[str, anyio.Semaphore] = {}
async def verifier(client, results, semaphore):
captured["sem"] = semaphore
return {r.id for r in results}
mocker.patch.dict(verification._VERIFIERS, {"note": verifier}, clear=False)
mocker.patch.object(verification, "delete_document_points", mocker.AsyncMock())
client = SimpleNamespace(username="alice")
await verify_search_results(client, [_make_result(1)], max_concurrent=5)
assert isinstance(captured["sem"], anyio.Semaphore)