refactor(search): address PR #750 review feedback

- Cap all_results to limit*2 after sort in the per-doc_types branch of
  nc_semantic_search to bound over-verification (was unbounded N-types).
- Switch BatchVerifier from (client, doc_ids, user_id) to (client, results,
  semaphore). Verifiers now read file paths and deck board/stack ids from
  SearchResult.metadata instead of doing fresh Qdrant scrolls — eliminates
  one duplicate round-trip per file/deck-card verification.
- Bound per-id verification concurrency with a shared anyio.Semaphore
  (default 20, matching server/semantic.py context-expansion convention).
  Prevents httpx pool exhaustion / rate limiting on large search pages.
- Propagate stack_id from Qdrant payload to SearchResult.metadata in both
  bm25_hybrid.py and semantic.py (board_id was already propagated).
- Drop now-unused _resolve_file_path / _resolve_deck_metadata helpers.
- Drop redundant int(d) in requested predicate from _verify_news_items.
- Rewrite eviction comment to be honest about inline (not background)
  execution and the resulting latency coupling.
- ADR-019 status: Proposed -> Accepted.
- Add news property to NextcloudClientProtocol.
- Widen SearchResult.id and SemanticSearchResult.id to int | str to match
  BatchVerifier signature and document support for future string-id types.
- Flip openWorldHint to True on nc_semantic_search_answer (it calls into
  Nextcloud via nc_semantic_search).

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
Chris Coutinho
2026-05-01 08:22:28 +02:00
co-authored by Claude Opus 4.7
parent d90e793d19
commit 7784ec02d7
8 changed files with 380 additions and 270 deletions
@@ -1,6 +1,6 @@
# ADR-019: Verify-on-Read for Semantic Search Results
**Status**: Proposed
**Status**: Accepted
**Date**: 2026-05-01
**Depends On**: ADR-007 (Background Vector Sync), ADR-010 (Webhook-Based Vector Sync)
+7 -1
View File
@@ -10,7 +10,13 @@ from .base import BaseResponse
class SemanticSearchResult(BaseModel):
"""Model for semantic search results with additional metadata."""
id: int = Field(description="Document ID (int for all document types)")
id: int | str = Field(
description=(
"Document ID. Numeric for all currently indexed types (notes, files, "
"deck cards, news items); typed as int|str to allow future doc types "
"that use string identifiers."
)
)
doc_type: str = Field(
description="Document type (note, calendar_event, deck_card, etc.)"
)
+9 -2
View File
@@ -67,6 +67,11 @@ class NextcloudClientProtocol(Protocol):
"""Tables client for accessing table row documents."""
...
@property
def news(self) -> Any:
"""News client for accessing news item documents."""
...
async def get_indexed_doc_types(user_id: str) -> set[str]:
"""Query Qdrant to get actually-indexed document types for a user.
@@ -127,7 +132,9 @@ class SearchResult:
"""A single search result with metadata and score.
Attributes:
id: Document ID (int for all document types)
id: Document ID. Numeric for indexed types today (notes, files,
deck cards, news items), but typed as ``int | str`` to allow
future doc types that use string identifiers (e.g., file paths).
doc_type: Document type (note, file, calendar, contact, etc.)
title: Document title
excerpt: Content excerpt showing match context
@@ -144,7 +151,7 @@ class SearchResult:
point_id: Qdrant point ID for batch vector retrieval (None if not from Qdrant)
"""
id: int
id: int | str
doc_type: str
title: str
excerpt: str
@@ -233,9 +233,14 @@ class BM25HybridSearchAlgorithm(SearchAlgorithm):
metadata["path"] = path
# Add deck_card-specific metadata for frontend URL construction
# and verify-on-read (ADR-019) — both board_id and stack_id are
# required to call deck.get_card without an O(boards × stacks)
# iteration fallback.
if doc_type == "deck_card":
if board_id := result.payload.get("board_id"):
metadata["board_id"] = board_id
if stack_id := result.payload.get("stack_id"):
metadata["stack_id"] = stack_id
# Return unverified results (verification happens at output stage)
results.append(
+5
View File
@@ -164,9 +164,14 @@ class SemanticSearchAlgorithm(SearchAlgorithm):
metadata["path"] = path
# Add deck_card-specific metadata for frontend URL construction
# and verify-on-read (ADR-019) — both board_id and stack_id are
# required to call deck.get_card without an O(boards × stacks)
# iteration fallback.
if doc_type == "deck_card":
if board_id := result.payload.get("board_id"):
metadata["board_id"] = board_id
if stack_id := result.payload.get("stack_id"):
metadata["stack_id"] = stack_id
# Return unverified results (verification happens at output stage)
results.append(
+187 -219
View File
@@ -6,10 +6,18 @@ against Nextcloud at query time, dropping any that the user can no longer
access (deleted, unshared, etc.) and lazily evicting them from the index.
Per-doc_type verifiers are registered in ``_VERIFIERS``. Each takes the
authenticated client, a list of doc_ids, and the user_id, and returns the
subset of doc_ids that are currently accessible. The dispatch deliberately
groups by doc_type so doc-types with cheap batch endpoints (news_item) can
do a single fetch rather than one round-trip per result.
authenticated client, the (deduplicated) list of ``SearchResult``s for that
doc_type, and a shared concurrency semaphore. They return the subset of
``doc_id`` values that are currently accessible. Verifiers read whatever
metadata they need (file path, deck card board/stack ids) directly from the
SearchResult — these fields are populated at index-time and propagated by
the algorithm layer (see ``search/bm25_hybrid.py`` and ``search/semantic.py``)
so verification adds zero extra Qdrant round-trips.
Concurrency is bounded by a shared semaphore (default 20) so a large search
result page (or a multi-doc_type query) cannot exhaust the httpx connection
pool or trigger Nextcloud rate limiting. The 20-slot default matches the
context-expansion convention in ``server/semantic.py``.
Failure policy:
@@ -28,18 +36,22 @@ from typing import Any
import anyio
from httpx import HTTPStatusError
from qdrant_client.models import FieldCondition, Filter, MatchValue
from nextcloud_mcp_server.config import get_settings
from nextcloud_mcp_server.search.algorithms import SearchResult
from nextcloud_mcp_server.vector.eviction import delete_document_points
from nextcloud_mcp_server.vector.qdrant_client import get_qdrant_client
logger = logging.getLogger(__name__)
BatchVerifier = Callable[[Any, list[int | str], str], Awaitable[set[int | str]]]
"""(client, doc_ids, user_id) -> set of accessible doc_ids."""
# Default cap on concurrent verification round-trips against Nextcloud. Matches
# the convention in ``server/semantic.py`` for context-expansion fan-out.
DEFAULT_VERIFICATION_CONCURRENCY = 20
BatchVerifier = Callable[
[Any, list[SearchResult], anyio.Semaphore], Awaitable[set[int | str]]
]
"""(client, results, semaphore) -> set of doc_ids accessible to the user."""
# ---------------------------------------------------------------------------
@@ -55,185 +67,200 @@ def _is_definitive_404_or_403(exc: BaseException) -> bool:
async def _verify_notes(
client: Any, doc_ids: list[int | str], user_id: str
client: Any, results: list[SearchResult], semaphore: anyio.Semaphore
) -> set[int | str]:
accessible: set[int | str] = set()
async def check(doc_id: int | str) -> None:
try:
await client.notes.get_note(int(doc_id))
accessible.add(doc_id)
except HTTPStatusError as e:
if _is_definitive_404_or_403(e):
return
logger.warning(
"Transient error verifying note %s: %s %s; keeping result",
doc_id,
e.response.status_code,
e,
)
accessible.add(doc_id)
except Exception as e:
logger.warning(
"Unexpected error verifying note %s: %s; keeping result",
doc_id,
e,
)
accessible.add(doc_id)
async def check(result: SearchResult) -> None:
async with semaphore:
doc_id = result.id
try:
await client.notes.get_note(int(doc_id))
accessible.add(doc_id)
except HTTPStatusError as e:
if _is_definitive_404_or_403(e):
return
logger.warning(
"Transient error verifying note %s: %s %s; keeping result",
doc_id,
e.response.status_code,
e,
)
accessible.add(doc_id)
except Exception as e:
logger.warning(
"Unexpected error verifying note %s: %s; keeping result",
doc_id,
e,
)
accessible.add(doc_id)
async with anyio.create_task_group() as tg:
for doc_id in doc_ids:
tg.start_soon(check, doc_id)
for r in results:
tg.start_soon(check, r)
return accessible
async def _verify_files(
client: Any, doc_ids: list[int | str], user_id: str
client: Any, results: list[SearchResult], semaphore: anyio.Semaphore
) -> set[int | str]:
accessible: set[int | str] = set()
async def check(doc_id: int | str) -> None:
# Resolve file_id → file_path from Qdrant payload
file_path = await _resolve_file_path(user_id, doc_id)
if file_path is None:
async def check(result: SearchResult) -> None:
doc_id = result.id
# file_path is propagated from the Qdrant payload by the algorithm
# layer (bm25_hybrid.py / semantic.py). No extra Qdrant round-trip.
file_path = (result.metadata or {}).get("path")
if not file_path:
# Cannot verify without a path; treat as accessible to avoid
# silently dropping legitimate results when payload is missing
# (legacy data, or a future doc_type that doesn't propagate path).
logger.warning(
"No file_path in Qdrant for file_id %s; keeping result "
"No file path in metadata for file_id %s; keeping result "
"(verification skipped)",
doc_id,
)
accessible.add(doc_id)
return
try:
info = await client.webdav.get_file_info(file_path)
if info is None:
# get_file_info returns None on definitive 404
return
accessible.add(doc_id)
except HTTPStatusError as e:
if _is_definitive_404_or_403(e):
return
logger.warning(
"Transient error verifying file %s (%s): %s %s; keeping result",
doc_id,
file_path,
e.response.status_code,
e,
)
accessible.add(doc_id)
except Exception as e:
logger.warning(
"Unexpected error verifying file %s (%s): %s; keeping result",
doc_id,
file_path,
e,
)
accessible.add(doc_id)
async with semaphore:
try:
info = await client.webdav.get_file_info(file_path)
if info is None:
# get_file_info returns None on definitive 404
return
accessible.add(doc_id)
except HTTPStatusError as e:
if _is_definitive_404_or_403(e):
return
logger.warning(
"Transient error verifying file %s (%s): %s %s; keeping result",
doc_id,
file_path,
e.response.status_code,
e,
)
accessible.add(doc_id)
except Exception as e:
logger.warning(
"Unexpected error verifying file %s (%s): %s; keeping result",
doc_id,
file_path,
e,
)
accessible.add(doc_id)
async with anyio.create_task_group() as tg:
for doc_id in doc_ids:
tg.start_soon(check, doc_id)
for r in results:
tg.start_soon(check, r)
return accessible
async def _verify_deck_cards(
client: Any, doc_ids: list[int | str], user_id: str
client: Any, results: list[SearchResult], semaphore: anyio.Semaphore
) -> set[int | str]:
accessible: set[int | str] = set()
async def check(doc_id: int | str) -> None:
# Resolve card_id → (board_id, stack_id) from Qdrant payload
meta = await _resolve_deck_metadata(user_id, int(doc_id))
if meta is None:
async def check(result: SearchResult) -> None:
doc_id = result.id
# board_id and stack_id are propagated from the Qdrant payload by the
# algorithm layer. No extra Qdrant round-trip.
meta = result.metadata or {}
board_id = meta.get("board_id")
stack_id = meta.get("stack_id")
if board_id is None or stack_id is None:
# Without metadata we cannot run the cheap fast-path. Per ADR-019
# we deliberately do NOT fall back to O(boards × stacks) iteration
# in the search hot path; treat as accessible.
logger.warning(
"No deck metadata in Qdrant for card %s; keeping result "
"(verification skipped, legacy data without board_id/stack_id)",
"Incomplete deck metadata for card %s (board_id=%s, stack_id=%s); "
"keeping result (verification skipped, legacy data)",
doc_id,
board_id,
stack_id,
)
accessible.add(doc_id)
return
try:
await client.deck.get_card(
board_id=meta["board_id"],
stack_id=meta["stack_id"],
card_id=int(doc_id),
)
accessible.add(doc_id)
except HTTPStatusError as e:
if _is_definitive_404_or_403(e):
return
logger.warning(
"Transient error verifying deck card %s: %s %s; keeping result",
doc_id,
e.response.status_code,
e,
)
accessible.add(doc_id)
except Exception as e:
logger.warning(
"Unexpected error verifying deck card %s: %s; keeping result",
doc_id,
e,
)
accessible.add(doc_id)
async with semaphore:
try:
await client.deck.get_card(
board_id=int(board_id),
stack_id=int(stack_id),
card_id=int(doc_id),
)
accessible.add(doc_id)
except HTTPStatusError as e:
if _is_definitive_404_or_403(e):
return
logger.warning(
"Transient error verifying deck card %s: %s %s; keeping result",
doc_id,
e.response.status_code,
e,
)
accessible.add(doc_id)
except Exception as e:
logger.warning(
"Unexpected error verifying deck card %s: %s; keeping result",
doc_id,
e,
)
accessible.add(doc_id)
async with anyio.create_task_group() as tg:
for doc_id in doc_ids:
tg.start_soon(check, doc_id)
for r in results:
tg.start_soon(check, r)
return accessible
async def _verify_news_items(
client: Any, doc_ids: list[int | str], user_id: str
client: Any, results: list[SearchResult], semaphore: anyio.Semaphore
) -> set[int | str]:
"""Batch-verify news items with a single fetch.
The Nextcloud News API has no per-item endpoint, so ``news.get_item`` is
implemented as a fetch-all + filter — which would be O(N × all_items) if
called per id. Instead we fetch once and intersect.
called per id. Instead we fetch once and intersect. The semaphore is
accepted for signature symmetry but not heavily used (one round-trip total).
"""
requested = {int(d) for d in doc_ids}
doc_ids = [r.id for r in results]
try:
items = await client.news.get_items(batch_size=-1, get_read=True)
except HTTPStatusError as e:
# If the News API itself is gone (app disabled, user lost access),
# treat *all* requested items as inaccessible. Eviction will reclaim.
if _is_definitive_404_or_403(e):
logger.info(
"News API returned %s for user %s; treating all %d news_items as inaccessible",
async with semaphore:
try:
items = await client.news.get_items(batch_size=-1, get_read=True)
except HTTPStatusError as e:
# If the News API itself is gone (app disabled, user lost access),
# treat *all* requested items as inaccessible. Eviction will reclaim.
if _is_definitive_404_or_403(e):
logger.info(
"News API returned %s for user %s; treating all %d news_items as inaccessible",
e.response.status_code,
client.username,
len(doc_ids),
)
return set()
logger.warning(
"Transient error fetching news items for verification: %s %s; keeping all results",
e.response.status_code,
user_id,
len(requested),
e,
)
return set()
logger.warning(
"Transient error fetching news items for verification: %s %s; keeping all results",
e.response.status_code,
e,
)
return set(doc_ids)
except Exception as e:
logger.warning(
"Unexpected error fetching news items for verification: %s; keeping all results",
e,
)
return set(doc_ids)
return set(doc_ids)
except Exception as e:
logger.warning(
"Unexpected error fetching news items for verification: %s; keeping all results",
e,
)
return set(doc_ids)
present_ids = {int(item.get("id")) for item in items if item.get("id") is not None}
# Map back to the original doc_id types (the caller may pass ints or strs)
# Map back to the original doc_id types (caller may pass ints or strs).
accessible: set[int | str] = set()
for d in doc_ids:
if int(d) in present_ids and int(d) in requested:
if int(d) in present_ids:
accessible.add(d)
return accessible
@@ -255,77 +282,6 @@ def get_supported_doc_types() -> set[str]:
return set(_VERIFIERS.keys())
# ---------------------------------------------------------------------------
# Qdrant payload lookup helpers
# ---------------------------------------------------------------------------
async def _resolve_file_path(user_id: str, doc_id: int | str) -> str | None:
"""Look up file_path for a file_id from any chunk's Qdrant payload."""
try:
qdrant_client = await get_qdrant_client()
settings = get_settings()
scroll_result = await qdrant_client.scroll(
collection_name=settings.get_collection_name(),
scroll_filter=Filter(
must=[
FieldCondition(key="user_id", match=MatchValue(value=user_id)),
FieldCondition(key="doc_id", match=MatchValue(value=doc_id)),
FieldCondition(key="doc_type", match=MatchValue(value="file")),
]
),
limit=1,
with_payload=["file_path"],
with_vectors=False,
)
if scroll_result[0]:
point = scroll_result[0][0]
file_path = point.payload.get("file_path") if point.payload else None
if file_path:
return str(file_path)
return None
except Exception as e:
logger.debug("Error resolving file_path for file_id %s: %s", doc_id, e)
return None
async def _resolve_deck_metadata(user_id: str, card_id: int) -> dict[str, int] | None:
"""Look up (board_id, stack_id) for a deck card from any chunk's payload."""
try:
qdrant_client = await get_qdrant_client()
settings = get_settings()
scroll_result = await qdrant_client.scroll(
collection_name=settings.get_collection_name(),
scroll_filter=Filter(
must=[
FieldCondition(key="user_id", match=MatchValue(value=user_id)),
FieldCondition(key="doc_id", match=MatchValue(value=card_id)),
FieldCondition(key="doc_type", match=MatchValue(value="deck_card")),
]
),
limit=1,
with_payload=["board_id", "stack_id"],
with_vectors=False,
)
if scroll_result[0]:
point = scroll_result[0][0]
payload = point.payload or {}
board_id = payload.get("board_id")
stack_id = payload.get("stack_id")
if board_id is not None and stack_id is not None:
return {"board_id": int(board_id), "stack_id": int(stack_id)}
return None
except Exception as e:
logger.debug("Error resolving deck metadata for card %s: %s", card_id, e)
return None
# ---------------------------------------------------------------------------
# Public entry point
# ---------------------------------------------------------------------------
@@ -336,13 +292,14 @@ async def verify_search_results(
results: list[SearchResult],
*,
evict_on_missing: bool = True,
max_concurrent: int = DEFAULT_VERIFICATION_CONCURRENCY,
) -> list[SearchResult]:
"""Filter search results to those the user can currently access.
Deduplicates by ``(doc_id, doc_type)`` before verifying, so multiple
chunks from the same document cost a single check. Verifiers run
concurrently per doc_type (and within each doc_type, per id where that
is cheaper than batching).
concurrently per doc_type and concurrently per id within each verifier,
bounded by a shared semaphore (``max_concurrent``).
When ``evict_on_missing=True``, points for documents that fail
verification are deleted from Qdrant in-line. Eviction failures are
@@ -353,6 +310,8 @@ async def verify_search_results(
results: SearchResult list from the algorithm layer (may include
multiple chunks per document).
evict_on_missing: Schedule lazy eviction for inaccessible docs.
max_concurrent: Cap on concurrent verification round-trips against
Nextcloud. Defaults to ``DEFAULT_VERIFICATION_CONCURRENCY``.
Returns:
Filtered list preserving the original order.
@@ -363,28 +322,33 @@ async def verify_search_results(
user_id: str = client.username
# Group unique (doc_id, doc_type) by doc_type so each verifier sees a
# deduplicated batch.
by_type: dict[str, set[int | str]] = {}
# deduplicated batch. We pick one SearchResult per (id, doc_type) to carry
# metadata (path, board_id/stack_id) into the verifier — chunks of the
# same document share these fields, so any chunk works.
by_type: dict[str, dict[int | str, SearchResult]] = {}
for r in results:
by_type.setdefault(r.doc_type, set()).add(r.id)
by_type.setdefault(r.doc_type, {}).setdefault(r.id, r)
# Shared semaphore bounds total Nextcloud round-trips across all
# per-id verifiers. Without it, a 50-result mostly-notes page could fan
# out 50 concurrent get_note calls and exhaust the connection pool.
semaphore = anyio.Semaphore(max_concurrent)
# Run all type verifiers concurrently. Per-id failures are absorbed
# inside each verifier; this outer task group only fans out per type.
accessible_by_type: dict[str, set[int | str]] = {}
async def run_verifier(doc_type: str, doc_ids: set[int | str]) -> None:
async def run_verifier(doc_type: str, unique_results: list[SearchResult]) -> None:
verifier = _VERIFIERS.get(doc_type)
if verifier is None:
logger.warning(
"No verifier registered for doc_type=%r; keeping %d result(s) unverified",
doc_type,
len(doc_ids),
len(unique_results),
)
accessible_by_type[doc_type] = doc_ids
accessible_by_type[doc_type] = {r.id for r in unique_results}
return
try:
accessible_by_type[doc_type] = await verifier(
client, list(doc_ids), user_id
client, unique_results, semaphore
)
except Exception as e:
# Verifier itself blew up (not per-id) — fail open.
@@ -392,20 +356,20 @@ async def verify_search_results(
"Verifier for doc_type=%s raised: %s; keeping all %d result(s) unverified",
doc_type,
e,
len(doc_ids),
len(unique_results),
exc_info=True,
)
accessible_by_type[doc_type] = doc_ids
accessible_by_type[doc_type] = {r.id for r in unique_results}
async with anyio.create_task_group() as tg:
for doc_type, doc_ids in by_type.items():
tg.start_soon(run_verifier, doc_type, doc_ids)
for doc_type, id_to_result in by_type.items():
tg.start_soon(run_verifier, doc_type, list(id_to_result.values()))
# Compute (doc_id, doc_type) pairs that failed verification
inaccessible: set[tuple[int | str, str]] = set()
for doc_type, doc_ids in by_type.items():
accessible = accessible_by_type.get(doc_type, doc_ids)
for doc_id in doc_ids:
for doc_type, id_to_result in by_type.items():
accessible = accessible_by_type.get(doc_type, set(id_to_result.keys()))
for doc_id in id_to_result.keys():
if doc_id not in accessible:
inaccessible.add((doc_id, doc_type))
@@ -416,11 +380,15 @@ async def verify_search_results(
sorted((str(d), t) for d, t in inaccessible),
)
# Filter results in-place-style, preserving order
# Filter results, preserving order. All chunks of an inaccessible document
# are dropped together (dedup happened before verification, but the result
# list still contains all chunks).
kept = [r for r in results if (r.id, r.doc_type) not in inaccessible]
# Lazy eviction — fire and forget, but bounded inline so we don't lose
# the user_id binding by escaping the task group.
# Lazy eviction. Runs inline before returning — slow Qdrant will delay
# the search response. Background eviction would need a task registered
# on the lifespan context; the inline approach is acceptable given
# typical Qdrant delete latency, but callers should be aware.
if evict_on_missing and inaccessible:
async def evict(doc_id: int | str, doc_type: str) -> None:
+8 -3
View File
@@ -142,8 +142,13 @@ def configure_semantic_tools(mcp: FastMCP):
)
all_results.extend(unverified_results)
# Sort combined results by score
# Sort combined results by score, then cap to `limit * 2` to
# match the cross-app branch's over-fetch budget. Without this
# cap, N requested doc_types × `limit * 2` results would all
# flow into verification, multiplying the Nextcloud round-trip
# cost by N.
all_results.sort(key=lambda r: r.score, reverse=True)
all_results = all_results[: limit * 2]
# ADR-019: Verify-on-read. The vector index is a recall layer;
# Nextcloud is the source of truth for access. Filter out ghost
@@ -300,7 +305,7 @@ def configure_semantic_tools(mcp: FastMCP):
title="Search with AI-Generated Answer",
annotations=ToolAnnotations(
readOnlyHint=True, # Search doesn't modify data
openWorldHint=False, # Searches only indexed Nextcloud data
openWorldHint=True, # Calls into Nextcloud via nc_semantic_search
),
)
@require_scopes("semantic.read")
@@ -432,7 +437,7 @@ def configure_semantic_tools(mcp: FastMCP):
async with semaphore:
if result.doc_type == "note":
try:
note = await client.notes.get_note(result.id)
note = await client.notes.get_note(int(result.id))
content = note.get("content", "")
accessible_results[index] = result
full_contents[index] = content
+158 -44
View File
@@ -2,6 +2,7 @@
from types import SimpleNamespace
import anyio
import httpx
import pytest
from httpx import HTTPStatusError
@@ -22,11 +23,16 @@ from nextcloud_mcp_server.search.verification import (
# ---------------------------------------------------------------------------
def _sem(slots: int = 20) -> anyio.Semaphore:
return anyio.Semaphore(slots)
def _make_result(
doc_id: int,
doc_id: int | str,
doc_type: str = "note",
chunk_index: int = 0,
score: float = 0.9,
metadata: dict | None = None,
) -> SearchResult:
return SearchResult(
id=doc_id,
@@ -35,6 +41,7 @@ def _make_result(
excerpt="...",
score=score,
chunk_index=chunk_index,
metadata=metadata,
)
@@ -72,7 +79,9 @@ async def test_verify_notes_200_keeps_all(mocker):
)
client = SimpleNamespace(notes=notes_client, username="alice")
result = await _verify_notes(client, [1, 2, 3], "alice")
result = await _verify_notes(
client, [_make_result(1), _make_result(2), _make_result(3)], _sem()
)
assert result == {1, 2, 3}
assert notes_client.get_note.await_count == 3
@@ -85,7 +94,7 @@ async def test_verify_notes_404_drops(mocker):
)
client = SimpleNamespace(notes=notes_client, username="alice")
result = await _verify_notes(client, [42], "alice")
result = await _verify_notes(client, [_make_result(42)], _sem())
assert result == set()
@@ -97,7 +106,7 @@ async def test_verify_notes_403_drops(mocker):
)
client = SimpleNamespace(notes=notes_client, username="alice")
result = await _verify_notes(client, [42], "alice")
result = await _verify_notes(client, [_make_result(42)], _sem())
assert result == set()
@@ -110,7 +119,7 @@ async def test_verify_notes_transient_5xx_keeps(mocker):
)
client = SimpleNamespace(notes=notes_client, username="alice")
result = await _verify_notes(client, [42], "alice")
result = await _verify_notes(client, [_make_result(42)], _sem())
assert result == {42}
@@ -122,7 +131,7 @@ async def test_verify_notes_unexpected_exception_keeps(mocker):
)
client = SimpleNamespace(notes=notes_client, username="alice")
result = await _verify_notes(client, [7], "alice")
result = await _verify_notes(client, [_make_result(7)], _sem())
assert result == {7}
@@ -143,7 +152,9 @@ async def test_verify_notes_mixed_outcomes(mocker):
notes_client = SimpleNamespace(get_note=mocker.AsyncMock(side_effect=side_effect))
client = SimpleNamespace(notes=notes_client, username="alice")
result = await _verify_notes(client, [1, 2, 3], "alice")
result = await _verify_notes(
client, [_make_result(1), _make_result(2), _make_result(3)], _sem()
)
assert result == {1, 3}
@@ -161,7 +172,15 @@ async def test_verify_news_items_intersects_with_fetched_set(mocker):
)
client = SimpleNamespace(news=news_client, username="alice")
result = await _verify_news_items(client, [10, 20, 99], "alice")
result = await _verify_news_items(
client,
[
_make_result(10, doc_type="news_item"),
_make_result(20, doc_type="news_item"),
_make_result(99, doc_type="news_item"),
],
_sem(),
)
assert result == {10, 20}
assert news_client.get_items.await_count == 1
@@ -174,7 +193,15 @@ async def test_verify_news_items_api_404_drops_all(mocker):
)
client = SimpleNamespace(news=news_client, username="alice")
result = await _verify_news_items(client, [1, 2, 3], "alice")
result = await _verify_news_items(
client,
[
_make_result(1, doc_type="news_item"),
_make_result(2, doc_type="news_item"),
_make_result(3, doc_type="news_item"),
],
_sem(),
)
assert result == set()
@@ -186,7 +213,15 @@ async def test_verify_news_items_transient_keeps_all(mocker):
)
client = SimpleNamespace(news=news_client, username="alice")
result = await _verify_news_items(client, [1, 2, 3], "alice")
result = await _verify_news_items(
client,
[
_make_result(1, doc_type="news_item"),
_make_result(2, doc_type="news_item"),
_make_result(3, doc_type="news_item"),
],
_sem(),
)
assert result == {1, 2, 3}
@@ -197,16 +232,18 @@ async def test_verify_news_items_transient_keeps_all(mocker):
@pytest.mark.unit
async def test_verify_files_uses_propfind_when_path_resolves(mocker):
mocker.patch.object(
verification, "_resolve_file_path", return_value="Documents/foo.txt"
)
async def test_verify_files_uses_path_from_metadata(mocker):
"""File verifier reads path from SearchResult.metadata, no Qdrant round-trip."""
webdav_client = SimpleNamespace(
get_file_info=mocker.AsyncMock(return_value={"id": 100})
)
client = SimpleNamespace(webdav=webdav_client, username="alice")
result = await _verify_files(client, [100], "alice")
result = await _verify_files(
client,
[_make_result(100, doc_type="file", metadata={"path": "Documents/foo.txt"})],
_sem(),
)
assert result == {100}
webdav_client.get_file_info.assert_awaited_once_with("Documents/foo.txt")
@@ -215,27 +252,54 @@ async def test_verify_files_uses_propfind_when_path_resolves(mocker):
@pytest.mark.unit
async def test_verify_files_404_via_get_file_info_drops(mocker):
"""get_file_info returns None on 404 — that's a definitive drop."""
mocker.patch.object(verification, "_resolve_file_path", return_value="gone.txt")
webdav_client = SimpleNamespace(get_file_info=mocker.AsyncMock(return_value=None))
client = SimpleNamespace(webdav=webdav_client, username="alice")
result = await _verify_files(client, [123], "alice")
result = await _verify_files(
client,
[_make_result(123, doc_type="file", metadata={"path": "gone.txt"})],
_sem(),
)
assert result == set()
@pytest.mark.unit
async def test_verify_files_missing_payload_keeps_unverified(mocker):
"""Without a file_path we cannot verify — fail open, don't drop."""
mocker.patch.object(verification, "_resolve_file_path", return_value=None)
webdav_client = SimpleNamespace(get_file_info=mocker.AsyncMock(return_value=None))
async def test_verify_files_missing_path_metadata_keeps_unverified(mocker):
"""Without a path in metadata we cannot verify — fail open, don't drop."""
webdav_client = SimpleNamespace(
get_file_info=mocker.AsyncMock(side_effect=AssertionError("must not be called"))
)
client = SimpleNamespace(webdav=webdav_client, username="alice")
result = await _verify_files(client, [555], "alice")
# No metadata at all
result = await _verify_files(client, [_make_result(555, doc_type="file")], _sem())
assert result == {555}
webdav_client.get_file_info.assert_not_awaited()
# Metadata present but no "path" key
result = await _verify_files(
client, [_make_result(556, doc_type="file", metadata={})], _sem()
)
assert result == {556}
webdav_client.get_file_info.assert_not_awaited()
@pytest.mark.unit
async def test_verify_files_transient_5xx_keeps(mocker):
webdav_client = SimpleNamespace(
get_file_info=mocker.AsyncMock(side_effect=_http_error(503))
)
client = SimpleNamespace(webdav=webdav_client, username="alice")
result = await _verify_files(
client,
[_make_result(7, doc_type="file", metadata={"path": "x.txt"})],
_sem(),
)
assert result == {7}
# ---------------------------------------------------------------------------
# Deck card verifier
@@ -244,15 +308,21 @@ async def test_verify_files_missing_payload_keeps_unverified(mocker):
@pytest.mark.unit
async def test_verify_deck_cards_uses_metadata_fast_path(mocker):
mocker.patch.object(
verification,
"_resolve_deck_metadata",
return_value={"board_id": 1, "stack_id": 2},
)
"""Deck verifier reads board_id+stack_id from metadata, no Qdrant round-trip."""
deck_client = SimpleNamespace(get_card=mocker.AsyncMock(return_value=object()))
client = SimpleNamespace(deck=deck_client, username="alice")
result = await _verify_deck_cards(client, [42], "alice")
result = await _verify_deck_cards(
client,
[
_make_result(
42,
doc_type="deck_card",
metadata={"board_id": 1, "stack_id": 2},
)
],
_sem(),
)
assert result == {42}
deck_client.get_card.assert_awaited_once_with(board_id=1, stack_id=2, card_id=42)
@@ -261,33 +331,56 @@ async def test_verify_deck_cards_uses_metadata_fast_path(mocker):
@pytest.mark.unit
async def test_verify_deck_cards_403_drops(mocker):
"""Board unshared with user → 403 from get_card → drop."""
mocker.patch.object(
verification,
"_resolve_deck_metadata",
return_value={"board_id": 1, "stack_id": 2},
)
deck_client = SimpleNamespace(
get_card=mocker.AsyncMock(side_effect=_http_error(403))
)
client = SimpleNamespace(deck=deck_client, username="alice")
result = await _verify_deck_cards(client, [42], "alice")
result = await _verify_deck_cards(
client,
[
_make_result(
42,
doc_type="deck_card",
metadata={"board_id": 1, "stack_id": 2},
)
],
_sem(),
)
assert result == set()
@pytest.mark.unit
async def test_verify_deck_cards_no_metadata_skips_verification(mocker):
"""Legacy data without board_id/stack_id payload → keep, do NOT iterate."""
mocker.patch.object(verification, "_resolve_deck_metadata", return_value=None)
async def test_verify_deck_cards_missing_metadata_keeps_unverified(mocker):
"""Legacy data without board_id/stack_id → keep, do NOT iterate or call API."""
deck_client = SimpleNamespace(
get_card=mocker.AsyncMock(side_effect=AssertionError("must not be called"))
)
client = SimpleNamespace(deck=deck_client, username="alice")
result = await _verify_deck_cards(client, [42], "alice")
# No metadata at all
result = await _verify_deck_cards(
client, [_make_result(42, doc_type="deck_card")], _sem()
)
assert result == {42}
# Only board_id (stack_id missing)
result = await _verify_deck_cards(
client,
[_make_result(43, doc_type="deck_card", metadata={"board_id": 1})],
_sem(),
)
assert result == {43}
# Only stack_id (board_id missing)
result = await _verify_deck_cards(
client,
[_make_result(44, doc_type="deck_card", metadata={"stack_id": 2})],
_sem(),
)
assert result == {44}
deck_client.get_card.assert_not_awaited()
@@ -320,9 +413,12 @@ async def test_verify_search_results_dedupes_chunks_per_document(mocker):
assert len(kept) == 3 # all kept, all reference the same accessible doc
spy.assert_awaited_once()
# Verifier received the single deduplicated id, not three copies
# Verifier received exactly one SearchResult (the deduplicated representative)
args, _kwargs = spy.call_args
assert args[1] == [1]
assert len(args[1]) == 1
assert args[1][0].id == 1
# And a semaphore as the third arg
assert isinstance(args[2], anyio.Semaphore)
@pytest.mark.unit
@@ -454,8 +550,8 @@ async def test_verify_search_results_dispatches_per_doc_type_concurrently(mocker
results = [
_make_result(1, doc_type="note"),
_make_result(500, doc_type="file"),
_make_result(999, doc_type="file"), # to be dropped
_make_result(500, doc_type="file", metadata={"path": "a.txt"}),
_make_result(999, doc_type="file", metadata={"path": "b.txt"}), # to be dropped
]
client = SimpleNamespace(username="alice")
@@ -464,3 +560,21 @@ async def test_verify_search_results_dispatches_per_doc_type_concurrently(mocker
assert {(r.id, r.doc_type) for r in kept} == {(1, "note"), (500, "file")}
note_verifier.assert_awaited_once()
file_verifier.assert_awaited_once()
@pytest.mark.unit
async def test_verify_search_results_passes_semaphore_to_verifier(mocker):
"""The dispatcher must construct a Semaphore and pass it to verifiers."""
captured: dict[str, anyio.Semaphore] = {}
async def verifier(client, results, semaphore):
captured["sem"] = semaphore
return {r.id for r in results}
mocker.patch.dict(verification._VERIFIERS, {"note": verifier}, clear=False)
mocker.patch.object(verification, "delete_document_points", mocker.AsyncMock())
client = SimpleNamespace(username="alice")
await verify_search_results(client, [_make_result(1)], max_concurrent=5)
assert isinstance(captured["sem"], anyio.Semaphore)