fix: address PR #813 review round 4 (log leak, cross-user chunk ctx, algo, overlap)
1. Don't log unverified result titles: both search algorithms logged top-5 titles at DEBUG before verify-on-read; with owner-level share expansion the unverified set can contain other users' docs. Algorithms now log a count only; the verifying callers (server/semantic, viz_routes, api/visualization) log verified titles after verify-on-read. 2. Cross-user FILE chunk context: get_chunk_with_context + the Qdrant chunk helpers now take accessible_owners and use build_ownership_filter. For files the expanded scope is honoured only after a per-file file_accessible_by_id check (accessible_owners is owner-level, so the gate prevents a one-file share recipient from reading any of the owner's cached chunks). note/deck/ news stay self-only (per-user APIs) — a documented gap. Both chunk endpoints pass accessible_owners. 3. Algorithm usage: SemanticSearchAlgorithm is not dead (it backs the dense-only option on the viz/API surfaces); added a clarifying comment in server/ semantic.py. Additionally wired accessible_owners + verify-on-read into the /api/v1 search routes (unified_search, vector_search) so the astrolabe surface is ACL-aware too — degrading gracefully to self-only/unverified for non-provisioned callers instead of 401. 4. Overlapping conditions: build_ownership_filter no longer lists self in the owner_id MatchAny branch (self is already covered by the user_id branch); the owner_id branch carries only the OTHER owners. Tests: build_ownership_filter dedup + chunk-bbox filter-shape updates; new ACL-aware get_indexed_doc_types, cached-chunk lookup, and end-to-end cross-user file chunk-context (recipient gets the chunk, non-recipient denied) tests. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.8
parent
ac041d5c97
commit
8deb48e6fa
@@ -30,10 +30,12 @@ from nextcloud_mcp_server.search import (
|
|||||||
BM25HybridSearchAlgorithm,
|
BM25HybridSearchAlgorithm,
|
||||||
SemanticSearchAlgorithm,
|
SemanticSearchAlgorithm,
|
||||||
)
|
)
|
||||||
|
from nextcloud_mcp_server.search.access_filter import list_accessible_owners
|
||||||
from nextcloud_mcp_server.search.context import (
|
from nextcloud_mcp_server.search.context import (
|
||||||
get_chunk_bbox_and_page_from_qdrant,
|
get_chunk_bbox_and_page_from_qdrant,
|
||||||
get_chunk_with_context,
|
get_chunk_with_context,
|
||||||
)
|
)
|
||||||
|
from nextcloud_mcp_server.search.verification import verify_search_results
|
||||||
from nextcloud_mcp_server.utils.validation import is_valid_nextcloud_doc_id
|
from nextcloud_mcp_server.utils.validation import is_valid_nextcloud_doc_id
|
||||||
from nextcloud_mcp_server.vector.oauth_sync import (
|
from nextcloud_mcp_server.vector.oauth_sync import (
|
||||||
NotProvisionedError,
|
NotProvisionedError,
|
||||||
@@ -164,24 +166,75 @@ async def unified_search(request: Request) -> JSONResponse:
|
|||||||
# Request extra results to handle offset
|
# Request extra results to handle offset
|
||||||
search_limit = limit + offset
|
search_limit = limit + offset
|
||||||
|
|
||||||
# Execute search
|
async def _execute(owners: list[str] | None) -> list:
|
||||||
all_results = []
|
"""Run the search across requested doc_types with the given owner
|
||||||
if doc_types and isinstance(doc_types, list):
|
scope (None ⇒ self-only)."""
|
||||||
for doc_type in doc_types:
|
results: list = []
|
||||||
if doc_type:
|
if doc_types and isinstance(doc_types, list):
|
||||||
results = await search_algo.search(
|
for doc_type in doc_types:
|
||||||
query=query,
|
if doc_type:
|
||||||
user_id=user_id,
|
results.extend(
|
||||||
limit=search_limit,
|
await search_algo.search(
|
||||||
doc_type=doc_type,
|
query=query,
|
||||||
)
|
user_id=user_id,
|
||||||
all_results.extend(results)
|
limit=search_limit,
|
||||||
all_results.sort(key=lambda r: r.score, reverse=True)
|
doc_type=doc_type,
|
||||||
|
accessible_owners=owners,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
results.sort(key=lambda r: r.score, reverse=True)
|
||||||
|
else:
|
||||||
|
results = await search_algo.search(
|
||||||
|
query=query,
|
||||||
|
user_id=user_id,
|
||||||
|
limit=search_limit,
|
||||||
|
accessible_owners=owners,
|
||||||
|
)
|
||||||
|
return results
|
||||||
|
|
||||||
|
# Resolve a Nextcloud client so search is ACL-aware and verify-on-read
|
||||||
|
# can confirm access. The OAuth bearer only authenticates Astrolabe →
|
||||||
|
# MCP Server; MCP Server → Nextcloud uses the provisioned app password.
|
||||||
|
# If the caller never provisioned background sync there is no client to
|
||||||
|
# expand shares or verify with — fall back to self-only, unverified
|
||||||
|
# search (the pre-ACL behaviour) rather than 401, so unified search keeps
|
||||||
|
# working for users who haven't opted into background indexing.
|
||||||
|
oauth_ctx = request.app.state.oauth_context
|
||||||
|
nextcloud_host = oauth_ctx.get("config", {}).get("nextcloud_host", "")
|
||||||
|
if not nextcloud_host:
|
||||||
|
raise ValueError("Nextcloud host not configured")
|
||||||
|
try:
|
||||||
|
nc_client = await get_user_client_basic_auth(user_id, nextcloud_host)
|
||||||
|
except NotProvisionedError:
|
||||||
|
logger.debug(
|
||||||
|
"User %s not provisioned; self-only unverified search", user_id
|
||||||
|
)
|
||||||
|
all_results = await _execute(None)
|
||||||
else:
|
else:
|
||||||
all_results = await search_algo.search(
|
async with nc_client:
|
||||||
query=query,
|
# Expand to owners who shared content with the caller (same as
|
||||||
user_id=user_id,
|
# the MCP tool path) so shared documents are searchable.
|
||||||
limit=search_limit,
|
accessible_owners = await list_accessible_owners(
|
||||||
|
nc_client.sharing, user_id
|
||||||
|
)
|
||||||
|
all_results = await _execute(accessible_owners)
|
||||||
|
# Verify-on-read (ADR-019): drop documents the caller can no
|
||||||
|
# longer access (e.g. a revoked share) before formatting.
|
||||||
|
# Eviction runs inline — this Starlette route has no FastMCP
|
||||||
|
# lifespan task group.
|
||||||
|
all_results, _dropped = await verify_search_results(
|
||||||
|
nc_client, all_results
|
||||||
|
)
|
||||||
|
|
||||||
|
# Safe to log titles now: provisioned callers passed verify-on-read;
|
||||||
|
# non-provisioned ran self-only (unverified titles are never logged).
|
||||||
|
if all_results:
|
||||||
|
logger.debug(
|
||||||
|
"Top verified results: %s",
|
||||||
|
", ".join(
|
||||||
|
f"{r.doc_type}_{r.id} (score={r.score:.3f}, title='{r.title}')"
|
||||||
|
for r in all_results[:5]
|
||||||
|
),
|
||||||
)
|
)
|
||||||
|
|
||||||
# Sort results by score (no deduplication - show all chunks)
|
# Sort results by score (no deduplication - show all chunks)
|
||||||
@@ -357,28 +410,76 @@ async def vector_search(request: Request) -> JSONResponse:
|
|||||||
score_threshold=score_threshold, fusion=fusion
|
score_threshold=score_threshold, fusion=fusion
|
||||||
)
|
)
|
||||||
|
|
||||||
# Execute search for each doc_type if specified, otherwise search all
|
async def _execute(owners: list[str] | None) -> list:
|
||||||
all_results = []
|
"""Run the search across requested doc_types with the given owner
|
||||||
if doc_types and isinstance(doc_types, list):
|
scope (None ⇒ self-only)."""
|
||||||
# Search each doc_type separately and merge results
|
results: list = []
|
||||||
for doc_type in doc_types:
|
if doc_types and isinstance(doc_types, list):
|
||||||
if doc_type: # Skip empty strings
|
# Search each doc_type separately and merge results
|
||||||
results = await search_algo.search(
|
for doc_type in doc_types:
|
||||||
query=query,
|
if doc_type: # Skip empty strings
|
||||||
user_id=user_id,
|
results.extend(
|
||||||
limit=limit,
|
await search_algo.search(
|
||||||
doc_type=doc_type,
|
query=query,
|
||||||
)
|
user_id=user_id,
|
||||||
all_results.extend(results)
|
limit=limit,
|
||||||
# Sort merged results by score and limit
|
doc_type=doc_type,
|
||||||
all_results.sort(key=lambda r: r.score, reverse=True)
|
accessible_owners=owners,
|
||||||
all_results = all_results[:limit]
|
)
|
||||||
|
)
|
||||||
|
# Sort merged results by score and limit
|
||||||
|
results.sort(key=lambda r: r.score, reverse=True)
|
||||||
|
results = results[:limit]
|
||||||
|
else:
|
||||||
|
# Search all document types
|
||||||
|
results = await search_algo.search(
|
||||||
|
query=query,
|
||||||
|
user_id=user_id,
|
||||||
|
limit=limit,
|
||||||
|
accessible_owners=owners,
|
||||||
|
)
|
||||||
|
return results
|
||||||
|
|
||||||
|
# Resolve a Nextcloud client so search is ACL-aware and verify-on-read
|
||||||
|
# can confirm access (same pattern as /api/v1/search). If the caller
|
||||||
|
# never provisioned background sync there is no client to expand shares
|
||||||
|
# or verify with — fall back to self-only, unverified search (pre-ACL
|
||||||
|
# behaviour) rather than 401.
|
||||||
|
oauth_ctx = request.app.state.oauth_context
|
||||||
|
nextcloud_host = oauth_ctx.get("config", {}).get("nextcloud_host", "")
|
||||||
|
if not nextcloud_host:
|
||||||
|
raise ValueError("Nextcloud host not configured")
|
||||||
|
try:
|
||||||
|
nc_client = await get_user_client_basic_auth(user_id, nextcloud_host)
|
||||||
|
except NotProvisionedError:
|
||||||
|
logger.debug(
|
||||||
|
"User %s not provisioned; self-only unverified search", user_id
|
||||||
|
)
|
||||||
|
all_results = await _execute(None)
|
||||||
else:
|
else:
|
||||||
# Search all document types
|
async with nc_client:
|
||||||
all_results = await search_algo.search(
|
# Expand to owners who shared content with the caller (same as
|
||||||
query=query,
|
# the MCP tool path) so shared documents are searchable.
|
||||||
user_id=user_id,
|
accessible_owners = await list_accessible_owners(
|
||||||
limit=limit,
|
nc_client.sharing, user_id
|
||||||
|
)
|
||||||
|
all_results = await _execute(accessible_owners)
|
||||||
|
# Verify-on-read (ADR-019): drop now-inaccessible docs before
|
||||||
|
# formatting (inline eviction — no FastMCP lifespan task group
|
||||||
|
# on this Starlette route).
|
||||||
|
all_results, _dropped = await verify_search_results(
|
||||||
|
nc_client, all_results
|
||||||
|
)
|
||||||
|
|
||||||
|
# Safe to log titles now: provisioned callers passed verify-on-read;
|
||||||
|
# non-provisioned ran self-only (unverified titles are never logged).
|
||||||
|
if all_results:
|
||||||
|
logger.debug(
|
||||||
|
"Top verified results: %s",
|
||||||
|
", ".join(
|
||||||
|
f"{r.doc_type}_{r.id} (score={r.score:.3f}, title='{r.title}')"
|
||||||
|
for r in all_results[:5]
|
||||||
|
),
|
||||||
)
|
)
|
||||||
|
|
||||||
# Format results for PHP client
|
# Format results for PHP client
|
||||||
@@ -569,6 +670,10 @@ async def get_chunk_context(request: Request) -> JSONResponse:
|
|||||||
)
|
)
|
||||||
|
|
||||||
async with nc_client:
|
async with nc_client:
|
||||||
|
# Expand to owners who shared content with the caller so the cached
|
||||||
|
# chunk lookup can resolve cross-user SHARED FILES (gated per-file
|
||||||
|
# inside get_chunk_with_context). Same expansion as the search path.
|
||||||
|
accessible_owners = await list_accessible_owners(nc_client.sharing, user_id)
|
||||||
chunk_context = await get_chunk_with_context(
|
chunk_context = await get_chunk_with_context(
|
||||||
nc_client=nc_client,
|
nc_client=nc_client,
|
||||||
user_id=user_id,
|
user_id=user_id,
|
||||||
@@ -579,6 +684,7 @@ async def get_chunk_context(request: Request) -> JSONResponse:
|
|||||||
chunk_index=chunk_index,
|
chunk_index=chunk_index,
|
||||||
total_chunks=total_chunks,
|
total_chunks=total_chunks,
|
||||||
context_chars=context_chars,
|
context_chars=context_chars,
|
||||||
|
accessible_owners=accessible_owners,
|
||||||
)
|
)
|
||||||
|
|
||||||
if chunk_context is None:
|
if chunk_context is None:
|
||||||
@@ -598,12 +704,16 @@ async def get_chunk_context(request: Request) -> JSONResponse:
|
|||||||
page_number = chunk_context.page_number
|
page_number = chunk_context.page_number
|
||||||
|
|
||||||
if doc_type == "file":
|
if doc_type == "file":
|
||||||
|
# Reaching here means the file chunk context resolved, so access was
|
||||||
|
# already confirmed (get_chunk_with_context gates files by id);
|
||||||
|
# the bbox/page lookup uses the same owner scope for cross-user files.
|
||||||
qdrant_bbox, qdrant_page = await get_chunk_bbox_and_page_from_qdrant(
|
qdrant_bbox, qdrant_page = await get_chunk_bbox_and_page_from_qdrant(
|
||||||
user_id=user_id,
|
user_id=user_id,
|
||||||
doc_id=doc_id,
|
doc_id=doc_id,
|
||||||
chunk_index=chunk_index,
|
chunk_index=chunk_index,
|
||||||
chunk_start=start,
|
chunk_start=start,
|
||||||
chunk_end=end,
|
chunk_end=end,
|
||||||
|
accessible_owners=accessible_owners,
|
||||||
)
|
)
|
||||||
if qdrant_bbox is not None:
|
if qdrant_bbox is not None:
|
||||||
chunk_bbox = qdrant_bbox
|
chunk_bbox = qdrant_bbox
|
||||||
|
|||||||
@@ -244,6 +244,16 @@ async def vector_visualization_search(request: Request) -> JSONResponse:
|
|||||||
verified_results, _dropped = await verify_search_results(
|
verified_results, _dropped = await verify_search_results(
|
||||||
nc_client, all_results
|
nc_client, all_results
|
||||||
)
|
)
|
||||||
|
# Safe to log titles now: these passed verify-on-read (unverified
|
||||||
|
# titles are never logged — see the search algorithms).
|
||||||
|
if verified_results:
|
||||||
|
logger.debug(
|
||||||
|
"Top verified results: %s",
|
||||||
|
", ".join(
|
||||||
|
f"{r.doc_type}_{r.id} (score={r.score:.3f}, title='{r.title}')"
|
||||||
|
for r in verified_results[:5]
|
||||||
|
),
|
||||||
|
)
|
||||||
search_results = verified_results[:limit]
|
search_results = verified_results[:limit]
|
||||||
search_duration = time.perf_counter() - search_start
|
search_duration = time.perf_counter() - search_start
|
||||||
|
|
||||||
@@ -661,6 +671,10 @@ async def chunk_context_endpoint(request: Request) -> JSONResponse:
|
|||||||
)
|
)
|
||||||
|
|
||||||
async with nc_client:
|
async with nc_client:
|
||||||
|
# Expand to owners who shared content with the caller so the cached
|
||||||
|
# chunk lookup can resolve cross-user SHARED FILES (gated per-file
|
||||||
|
# inside get_chunk_with_context). Same expansion as the search path.
|
||||||
|
accessible_owners = await list_accessible_owners(nc_client.sharing, user_id)
|
||||||
chunk_context = await get_chunk_with_context(
|
chunk_context = await get_chunk_with_context(
|
||||||
nc_client=nc_client,
|
nc_client=nc_client,
|
||||||
user_id=user_id,
|
user_id=user_id,
|
||||||
@@ -671,6 +685,7 @@ async def chunk_context_endpoint(request: Request) -> JSONResponse:
|
|||||||
chunk_index=chunk_index,
|
chunk_index=chunk_index,
|
||||||
total_chunks=total_chunks,
|
total_chunks=total_chunks,
|
||||||
context_chars=context_chars,
|
context_chars=context_chars,
|
||||||
|
accessible_owners=accessible_owners,
|
||||||
)
|
)
|
||||||
|
|
||||||
# Check if context expansion succeeded
|
# Check if context expansion succeeded
|
||||||
@@ -699,12 +714,16 @@ async def chunk_context_endpoint(request: Request) -> JSONResponse:
|
|||||||
chunk_bbox = None
|
chunk_bbox = None
|
||||||
page_number = chunk_context.page_number
|
page_number = chunk_context.page_number
|
||||||
if doc_type == "file":
|
if doc_type == "file":
|
||||||
|
# Reaching here means the file chunk context resolved, so access was
|
||||||
|
# already confirmed (get_chunk_with_context gates files by id);
|
||||||
|
# the bbox/page lookup uses the same owner scope for cross-user files.
|
||||||
qdrant_bbox, qdrant_page = await get_chunk_bbox_and_page_from_qdrant(
|
qdrant_bbox, qdrant_page = await get_chunk_bbox_and_page_from_qdrant(
|
||||||
user_id=user_id,
|
user_id=user_id,
|
||||||
doc_id=doc_id,
|
doc_id=doc_id,
|
||||||
chunk_index=chunk_index,
|
chunk_index=chunk_index,
|
||||||
chunk_start=start,
|
chunk_start=start,
|
||||||
chunk_end=end,
|
chunk_end=end,
|
||||||
|
accessible_owners=accessible_owners,
|
||||||
)
|
)
|
||||||
if qdrant_bbox is not None:
|
if qdrant_bbox is not None:
|
||||||
chunk_bbox = qdrant_bbox
|
chunk_bbox = qdrant_bbox
|
||||||
|
|||||||
@@ -136,13 +136,15 @@ def build_ownership_filter(
|
|||||||
) -> Filter:
|
) -> Filter:
|
||||||
"""Build the Qdrant ``Filter`` constraining a search to readable points.
|
"""Build the Qdrant ``Filter`` constraining a search to readable points.
|
||||||
|
|
||||||
Matches points whose ``owner_id`` is in ``accessible_owners`` OR whose
|
Matches points whose ``owner_id`` is in ``accessible_owners`` (excluding
|
||||||
legacy ``user_id`` equals ``user_id``. The legacy branch keeps points
|
self) OR whose ``user_id`` equals ``user_id``. The ``user_id`` branch covers
|
||||||
indexed before this change reachable until they're re-indexed.
|
*all* of the caller's own content — both new points (where
|
||||||
|
``owner_id == user_id``) and legacy points indexed before ``owner_id``
|
||||||
|
existed — so self is intentionally NOT repeated in the ``owner_id`` branch.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
user_id: Querying user (used for the legacy ``user_id`` fallback
|
user_id: Querying user (matched by the ``user_id`` branch, which is the
|
||||||
and as the only-self default when ``accessible_owners`` is None).
|
self-only default when ``accessible_owners`` is None).
|
||||||
accessible_owners: Pre-computed list of owner UIDs the user has
|
accessible_owners: Pre-computed list of owner UIDs the user has
|
||||||
access to. When None, defaults to ``[user_id]`` (no shares
|
access to. When None, defaults to ``[user_id]`` (no shares
|
||||||
expansion — used by callers that genuinely want self-only
|
expansion — used by callers that genuinely want self-only
|
||||||
@@ -152,15 +154,18 @@ def build_ownership_filter(
|
|||||||
A Qdrant ``Filter`` ready to be nested under a parent ``must`` clause.
|
A Qdrant ``Filter`` ready to be nested under a parent ``must`` clause.
|
||||||
"""
|
"""
|
||||||
owners = accessible_owners if accessible_owners is not None else [user_id]
|
owners = accessible_owners if accessible_owners is not None else [user_id]
|
||||||
# The legacy ``user_id`` branch is always present (self-owned content,
|
# The ``user_id`` branch is always present and already covers self-owned
|
||||||
# incl. pre-migration points). The ``owner_id`` branch is appended only for
|
# content (new + legacy). The ``owner_id`` branch is added only for OTHER
|
||||||
# a non-empty owner set: an empty list is handled explicitly here rather
|
# owners (share senders) — listing self there too would overlap the
|
||||||
# than relying on ``MatchAny(any=[])`` matching nothing, which is not a
|
# ``user_id`` branch for no benefit. When there are no other owners the
|
||||||
# documented Qdrant guarantee and could change across versions. Self always
|
# ``owner_id`` branch is omitted entirely, so we never depend on
|
||||||
# matches via the ``user_id`` branch, so an empty owner set is safe.
|
# ``MatchAny(any=[])`` matching nothing (not a documented Qdrant guarantee).
|
||||||
|
other_owners = [owner for owner in owners if owner != user_id]
|
||||||
conditions: list[Condition] = [
|
conditions: list[Condition] = [
|
||||||
FieldCondition(key="user_id", match=MatchValue(value=user_id)),
|
FieldCondition(key="user_id", match=MatchValue(value=user_id)),
|
||||||
]
|
]
|
||||||
if owners:
|
if other_owners:
|
||||||
conditions.insert(0, FieldCondition(key="owner_id", match=MatchAny(any=owners)))
|
conditions.insert(
|
||||||
|
0, FieldCondition(key="owner_id", match=MatchAny(any=other_owners))
|
||||||
|
)
|
||||||
return Filter(should=conditions)
|
return Filter(should=conditions)
|
||||||
|
|||||||
@@ -241,12 +241,10 @@ class BM25HybridSearchAlgorithm(SearchAlgorithm):
|
|||||||
if len(results) >= limit:
|
if len(results) >= limit:
|
||||||
break
|
break
|
||||||
|
|
||||||
|
# Log the count only — NOT titles. These results are unverified: with
|
||||||
|
# owner-level share expansion the candidate set can include other users'
|
||||||
|
# documents that verify-on-read will drop, so titles must not be logged
|
||||||
|
# until after verification (the verifying callers log verified titles).
|
||||||
logger.info("Returning %s unverified results after deduplication", len(results))
|
logger.info("Returning %s unverified results after deduplication", len(results))
|
||||||
if results:
|
|
||||||
result_details = [
|
|
||||||
f"{r.doc_type}_{r.id} (score={r.score:.3f}, title='{r.title}')"
|
|
||||||
for r in results[:5] # Show top 5
|
|
||||||
]
|
|
||||||
logger.debug("Top results: %s", ", ".join(result_details))
|
|
||||||
|
|
||||||
return results
|
return results
|
||||||
|
|||||||
@@ -7,10 +7,12 @@ position markers for better visualization and understanding of search results.
|
|||||||
import logging
|
import logging
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
|
|
||||||
|
from httpx import HTTPStatusError
|
||||||
from qdrant_client.models import FieldCondition, Filter, MatchValue
|
from qdrant_client.models import FieldCondition, Filter, MatchValue
|
||||||
|
|
||||||
from nextcloud_mcp_server.client import NextcloudClient
|
from nextcloud_mcp_server.client import NextcloudClient
|
||||||
from nextcloud_mcp_server.config import get_settings
|
from nextcloud_mcp_server.config import get_settings
|
||||||
|
from nextcloud_mcp_server.search.access_filter import build_ownership_filter
|
||||||
from nextcloud_mcp_server.utils.validation import is_valid_nextcloud_doc_id
|
from nextcloud_mcp_server.utils.validation import is_valid_nextcloud_doc_id
|
||||||
from nextcloud_mcp_server.vector.html_processor import html_to_markdown
|
from nextcloud_mcp_server.vector.html_processor import html_to_markdown
|
||||||
from nextcloud_mcp_server.vector.placeholder import get_placeholder_filter
|
from nextcloud_mcp_server.vector.placeholder import get_placeholder_filter
|
||||||
@@ -20,7 +22,12 @@ logger = logging.getLogger(__name__)
|
|||||||
|
|
||||||
|
|
||||||
async def _get_chunk_from_qdrant(
|
async def _get_chunk_from_qdrant(
|
||||||
user_id: str, doc_id: str, doc_type: str, chunk_start: int, chunk_end: int
|
user_id: str,
|
||||||
|
doc_id: str,
|
||||||
|
doc_type: str,
|
||||||
|
chunk_start: int,
|
||||||
|
chunk_end: int,
|
||||||
|
accessible_owners: list[str] | None = None,
|
||||||
) -> str | None:
|
) -> str | None:
|
||||||
"""Retrieve full chunk text from Qdrant payload.
|
"""Retrieve full chunk text from Qdrant payload.
|
||||||
|
|
||||||
@@ -28,11 +35,15 @@ async def _get_chunk_from_qdrant(
|
|||||||
chunk content already stored in Qdrant.
|
chunk content already stored in Qdrant.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
user_id: User ID who owns the document
|
user_id: Querying user.
|
||||||
doc_id: Document ID
|
doc_id: Document ID
|
||||||
doc_type: Document type (e.g., "note", "file")
|
doc_type: Document type (e.g., "note", "file")
|
||||||
chunk_start: Character offset where chunk starts
|
chunk_start: Character offset where chunk starts
|
||||||
chunk_end: Character offset where chunk ends
|
chunk_end: Character offset where chunk ends
|
||||||
|
accessible_owners: Owner UIDs the caller may read (self + share senders).
|
||||||
|
When None, the lookup is self-only. Callers must only pass an
|
||||||
|
expanded set after confirming the caller can access the document
|
||||||
|
(see ``get_chunk_with_context``) — the filter is owner-level.
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
Full chunk text from Qdrant excerpt field, or None if not found
|
Full chunk text from Qdrant excerpt field, or None if not found
|
||||||
@@ -46,7 +57,7 @@ async def _get_chunk_from_qdrant(
|
|||||||
collection_name=settings.get_collection_name(),
|
collection_name=settings.get_collection_name(),
|
||||||
scroll_filter=Filter(
|
scroll_filter=Filter(
|
||||||
must=[
|
must=[
|
||||||
FieldCondition(key="user_id", match=MatchValue(value=user_id)),
|
build_ownership_filter(user_id, accessible_owners),
|
||||||
FieldCondition(key="doc_id", match=MatchValue(value=doc_id)),
|
FieldCondition(key="doc_id", match=MatchValue(value=doc_id)),
|
||||||
FieldCondition(key="doc_type", match=MatchValue(value=doc_type)),
|
FieldCondition(key="doc_type", match=MatchValue(value=doc_type)),
|
||||||
FieldCondition(
|
FieldCondition(
|
||||||
@@ -93,17 +104,24 @@ async def _get_chunk_from_qdrant(
|
|||||||
|
|
||||||
|
|
||||||
async def _get_chunk_by_index_from_qdrant(
|
async def _get_chunk_by_index_from_qdrant(
|
||||||
user_id: str, doc_id: str, doc_type: str, chunk_index: int
|
user_id: str,
|
||||||
|
doc_id: str,
|
||||||
|
doc_type: str,
|
||||||
|
chunk_index: int,
|
||||||
|
accessible_owners: list[str] | None = None,
|
||||||
) -> str | None:
|
) -> str | None:
|
||||||
"""Retrieve chunk text by chunk_index from Qdrant payload.
|
"""Retrieve chunk text by chunk_index from Qdrant payload.
|
||||||
|
|
||||||
Used to fetch adjacent chunks for context expansion.
|
Used to fetch adjacent chunks for context expansion.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
user_id: User ID who owns the document
|
user_id: Querying user.
|
||||||
doc_id: Document ID
|
doc_id: Document ID
|
||||||
doc_type: Document type (e.g., "note", "file")
|
doc_type: Document type (e.g., "note", "file")
|
||||||
chunk_index: Zero-based chunk index in document
|
chunk_index: Zero-based chunk index in document
|
||||||
|
accessible_owners: Owner UIDs the caller may read; None ⇒ self-only.
|
||||||
|
Only pass an expanded set after a per-document access check (see
|
||||||
|
``get_chunk_with_context``).
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
Full chunk text from Qdrant excerpt field, or None if not found
|
Full chunk text from Qdrant excerpt field, or None if not found
|
||||||
@@ -117,7 +135,7 @@ async def _get_chunk_by_index_from_qdrant(
|
|||||||
collection_name=settings.get_collection_name(),
|
collection_name=settings.get_collection_name(),
|
||||||
scroll_filter=Filter(
|
scroll_filter=Filter(
|
||||||
must=[
|
must=[
|
||||||
FieldCondition(key="user_id", match=MatchValue(value=user_id)),
|
build_ownership_filter(user_id, accessible_owners),
|
||||||
FieldCondition(key="doc_id", match=MatchValue(value=doc_id)),
|
FieldCondition(key="doc_id", match=MatchValue(value=doc_id)),
|
||||||
FieldCondition(key="doc_type", match=MatchValue(value=doc_type)),
|
FieldCondition(key="doc_type", match=MatchValue(value=doc_type)),
|
||||||
FieldCondition(
|
FieldCondition(
|
||||||
@@ -217,6 +235,7 @@ async def get_chunk_bbox_and_page_from_qdrant(
|
|||||||
chunk_index: int | None,
|
chunk_index: int | None,
|
||||||
chunk_start: int,
|
chunk_start: int,
|
||||||
chunk_end: int,
|
chunk_end: int,
|
||||||
|
accessible_owners: list[str] | None = None,
|
||||||
) -> tuple[list | None, int | None]:
|
) -> tuple[list | None, int | None]:
|
||||||
"""Fetch chunk_bbox and page_number for a chunk from Qdrant payload.
|
"""Fetch chunk_bbox and page_number for a chunk from Qdrant payload.
|
||||||
|
|
||||||
@@ -256,7 +275,7 @@ async def get_chunk_bbox_and_page_from_qdrant(
|
|||||||
must=[
|
must=[
|
||||||
get_placeholder_filter(),
|
get_placeholder_filter(),
|
||||||
FieldCondition(key="doc_id", match=MatchValue(value=doc_id)),
|
FieldCondition(key="doc_id", match=MatchValue(value=doc_id)),
|
||||||
FieldCondition(key="user_id", match=MatchValue(value=user_id)),
|
build_ownership_filter(user_id, accessible_owners),
|
||||||
FieldCondition(
|
FieldCondition(
|
||||||
key="chunk_index", match=MatchValue(value=chunk_index)
|
key="chunk_index", match=MatchValue(value=chunk_index)
|
||||||
),
|
),
|
||||||
@@ -273,7 +292,7 @@ async def get_chunk_bbox_and_page_from_qdrant(
|
|||||||
must=[
|
must=[
|
||||||
get_placeholder_filter(),
|
get_placeholder_filter(),
|
||||||
FieldCondition(key="doc_id", match=MatchValue(value=doc_id)),
|
FieldCondition(key="doc_id", match=MatchValue(value=doc_id)),
|
||||||
FieldCondition(key="user_id", match=MatchValue(value=user_id)),
|
build_ownership_filter(user_id, accessible_owners),
|
||||||
FieldCondition(
|
FieldCondition(
|
||||||
key="chunk_start_offset",
|
key="chunk_start_offset",
|
||||||
match=MatchValue(value=chunk_start),
|
match=MatchValue(value=chunk_start),
|
||||||
@@ -352,6 +371,7 @@ async def get_chunk_with_context(
|
|||||||
chunk_index: int | None = None,
|
chunk_index: int | None = None,
|
||||||
total_chunks: int = 1,
|
total_chunks: int = 1,
|
||||||
context_chars: int = 300,
|
context_chars: int = 300,
|
||||||
|
accessible_owners: list[str] | None = None,
|
||||||
) -> ChunkContext | None:
|
) -> ChunkContext | None:
|
||||||
"""Fetch chunk with surrounding context.
|
"""Fetch chunk with surrounding context.
|
||||||
|
|
||||||
@@ -361,7 +381,7 @@ async def get_chunk_with_context(
|
|||||||
|
|
||||||
Args:
|
Args:
|
||||||
nc_client: Authenticated Nextcloud client
|
nc_client: Authenticated Nextcloud client
|
||||||
user_id: User ID who owns the document
|
user_id: Querying user.
|
||||||
doc_id: Document ID (str — keyword-indexed in Qdrant payload)
|
doc_id: Document ID (str — keyword-indexed in Qdrant payload)
|
||||||
doc_type: Type of document ("note", "file", etc.)
|
doc_type: Type of document ("note", "file", etc.)
|
||||||
chunk_start: Character offset where chunk starts
|
chunk_start: Character offset where chunk starts
|
||||||
@@ -372,6 +392,10 @@ async def get_chunk_with_context(
|
|||||||
field). When None, falls back to the (chunk_start, chunk_end) lookup.
|
field). When None, falls back to the (chunk_start, chunk_end) lookup.
|
||||||
total_chunks: Total number of chunks in document
|
total_chunks: Total number of chunks in document
|
||||||
context_chars: Number of characters to include before/after chunk
|
context_chars: Number of characters to include before/after chunk
|
||||||
|
accessible_owners: Owner UIDs the caller may read (self + share senders).
|
||||||
|
Used to support cross-user context for SHARED FILES only, and only
|
||||||
|
after a per-file access check (see ``lookup_owners`` below). For
|
||||||
|
non-file types the lookup stays self-only.
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
ChunkContext with expanded context and markers, or None if document
|
ChunkContext with expanded context and markers, or None if document
|
||||||
@@ -380,13 +404,53 @@ async def get_chunk_with_context(
|
|||||||
# doc_id is keyword-indexed in Qdrant as str — pass through verbatim
|
# doc_id is keyword-indexed in Qdrant as str — pass through verbatim
|
||||||
# (no int coercion; producers always stringify on write).
|
# (no int coercion; producers always stringify on write).
|
||||||
|
|
||||||
|
# Determine the ownership scope for the Qdrant cached-chunk lookups.
|
||||||
|
#
|
||||||
|
# ``accessible_owners`` is OWNER-level (every owner who shared anything with
|
||||||
|
# the caller), so widening the lookup to it unconditionally would let a
|
||||||
|
# recipient of a single shared file read ANY of that owner's cached chunks
|
||||||
|
# by guessing doc_ids. We therefore honour it only for FILES, and only after
|
||||||
|
# confirming the caller can access THIS file by id (``file_accessible_by_id``
|
||||||
|
# is cross-user-safe: a WebDAV SEARCH over the caller's whole tree incl.
|
||||||
|
# mounted shares). For per-user types (note/deck/news) there is no
|
||||||
|
# share-mounted by-id access via the caller's credentials, so the lookup
|
||||||
|
# stays self-only — cross-user context for those types is a known gap.
|
||||||
|
lookup_owners: list[str] | None = None # None ⇒ self-only
|
||||||
|
if doc_type == "file" and accessible_owners:
|
||||||
|
try:
|
||||||
|
if await nc_client.webdav.file_accessible_by_id(int(doc_id)):
|
||||||
|
lookup_owners = accessible_owners
|
||||||
|
else:
|
||||||
|
# Not owned and not shared with the caller → no access. Return
|
||||||
|
# early rather than falling back to a self-only lookup that
|
||||||
|
# would also miss (and so the result is the same None, but this
|
||||||
|
# is explicit and skips a pointless Qdrant round-trip).
|
||||||
|
logger.debug(
|
||||||
|
"File %s not accessible to %s; no cross-user chunk context",
|
||||||
|
doc_id,
|
||||||
|
user_id,
|
||||||
|
)
|
||||||
|
return None
|
||||||
|
except (ValueError, TypeError):
|
||||||
|
# Non-numeric doc_id: shouldn't happen (endpoints validate), but
|
||||||
|
# degrade to self-only rather than raising.
|
||||||
|
logger.warning("Non-numeric file doc_id %r; using self-only scope", doc_id)
|
||||||
|
except HTTPStatusError as exc:
|
||||||
|
# Transient transport/server error — treat as inconclusive and fall
|
||||||
|
# back to self-only so the caller's own files still resolve.
|
||||||
|
logger.warning(
|
||||||
|
"file_accessible_by_id(%s) failed (%s); using self-only scope",
|
||||||
|
doc_id,
|
||||||
|
exc,
|
||||||
|
)
|
||||||
|
|
||||||
# Try to get chunk from Qdrant (fast path).
|
# Try to get chunk from Qdrant (fast path).
|
||||||
# Prefer chunk_index lookup (always-indexed field) when caller supplied it;
|
# Prefer chunk_index lookup (always-indexed field) when caller supplied it;
|
||||||
# fall back to (chunk_start, chunk_end) lookup otherwise.
|
# fall back to (chunk_start, chunk_end) lookup otherwise.
|
||||||
chunk_text: str | None = None
|
chunk_text: str | None = None
|
||||||
if chunk_index is not None:
|
if chunk_index is not None:
|
||||||
chunk_text = await _get_chunk_by_index_from_qdrant(
|
chunk_text = await _get_chunk_by_index_from_qdrant(
|
||||||
user_id, doc_id, doc_type, chunk_index
|
user_id, doc_id, doc_type, chunk_index, accessible_owners=lookup_owners
|
||||||
)
|
)
|
||||||
# When chunk_index is supplied, the indexed lookup is canonical: both the
|
# When chunk_index is supplied, the indexed lookup is canonical: both the
|
||||||
# index path and the offset path query the same Qdrant collection, so an
|
# index path and the offset path query the same Qdrant collection, so an
|
||||||
@@ -398,7 +462,12 @@ async def get_chunk_with_context(
|
|||||||
skip_offset_lookup = chunk_index is not None
|
skip_offset_lookup = chunk_index is not None
|
||||||
if chunk_text is None and not skip_offset_lookup:
|
if chunk_text is None and not skip_offset_lookup:
|
||||||
chunk_text = await _get_chunk_from_qdrant(
|
chunk_text = await _get_chunk_from_qdrant(
|
||||||
user_id, doc_id, doc_type, chunk_start, chunk_end
|
user_id,
|
||||||
|
doc_id,
|
||||||
|
doc_type,
|
||||||
|
chunk_start,
|
||||||
|
chunk_end,
|
||||||
|
accessible_owners=lookup_owners,
|
||||||
)
|
)
|
||||||
|
|
||||||
if chunk_text:
|
if chunk_text:
|
||||||
@@ -422,7 +491,11 @@ async def get_chunk_with_context(
|
|||||||
# Fetch previous chunk if not first chunk
|
# Fetch previous chunk if not first chunk
|
||||||
if chunk_index > 0:
|
if chunk_index > 0:
|
||||||
before_chunk = await _get_chunk_by_index_from_qdrant(
|
before_chunk = await _get_chunk_by_index_from_qdrant(
|
||||||
user_id, doc_id, doc_type, chunk_index - 1
|
user_id,
|
||||||
|
doc_id,
|
||||||
|
doc_type,
|
||||||
|
chunk_index - 1,
|
||||||
|
accessible_owners=lookup_owners,
|
||||||
)
|
)
|
||||||
if before_chunk:
|
if before_chunk:
|
||||||
# Remove overlap: the last chunk_overlap chars of previous chunk
|
# Remove overlap: the last chunk_overlap chars of previous chunk
|
||||||
@@ -443,7 +516,11 @@ async def get_chunk_with_context(
|
|||||||
# Fetch next chunk if not last chunk
|
# Fetch next chunk if not last chunk
|
||||||
if chunk_index < total_chunks - 1:
|
if chunk_index < total_chunks - 1:
|
||||||
after_chunk = await _get_chunk_by_index_from_qdrant(
|
after_chunk = await _get_chunk_by_index_from_qdrant(
|
||||||
user_id, doc_id, doc_type, chunk_index + 1
|
user_id,
|
||||||
|
doc_id,
|
||||||
|
doc_type,
|
||||||
|
chunk_index + 1,
|
||||||
|
accessible_owners=lookup_owners,
|
||||||
)
|
)
|
||||||
if after_chunk:
|
if after_chunk:
|
||||||
# Remove overlap: the first chunk_overlap chars of next chunk
|
# Remove overlap: the first chunk_overlap chars of next chunk
|
||||||
|
|||||||
@@ -163,12 +163,10 @@ class SemanticSearchAlgorithm(SearchAlgorithm):
|
|||||||
if len(results) >= limit:
|
if len(results) >= limit:
|
||||||
break
|
break
|
||||||
|
|
||||||
|
# Log the count only — NOT titles. These results are unverified: with
|
||||||
|
# owner-level share expansion the candidate set can include other users'
|
||||||
|
# documents that verify-on-read will drop, so titles must not be logged
|
||||||
|
# until after verification (the verifying callers log verified titles).
|
||||||
logger.info("Returning %s unverified results after deduplication", len(results))
|
logger.info("Returning %s unverified results after deduplication", len(results))
|
||||||
if results:
|
|
||||||
result_details = [
|
|
||||||
f"{r.doc_type}_{r.id} (score={r.score:.3f}, title='{r.title}')"
|
|
||||||
for r in results[:5] # Show top 5
|
|
||||||
]
|
|
||||||
logger.debug("Top results: %s", ", ".join(result_details))
|
|
||||||
|
|
||||||
return results
|
return results
|
||||||
|
|||||||
@@ -129,7 +129,12 @@ def configure_semantic_tools(mcp: FastMCP):
|
|||||||
accessible_owners = await list_accessible_owners(client.sharing, username)
|
accessible_owners = await list_accessible_owners(client.sharing, username)
|
||||||
|
|
||||||
try:
|
try:
|
||||||
# Create BM25 hybrid search algorithm with specified fusion
|
# The nc_semantic_search tool deliberately uses BM25-hybrid (dense +
|
||||||
|
# sparse with RRF/DBSF fusion) as the single tool-layer algorithm.
|
||||||
|
# SemanticSearchAlgorithm is not dead code — it backs the dense-only
|
||||||
|
# option that the visualization/API surfaces expose explicitly
|
||||||
|
# (auth/viz_routes.py and api/visualization.py). Both algorithms take
|
||||||
|
# accessible_owners, so ACL-aware search works on every surface.
|
||||||
search_algo = BM25HybridSearchAlgorithm(
|
search_algo = BM25HybridSearchAlgorithm(
|
||||||
score_threshold=score_threshold, fusion=fusion
|
score_threshold=score_threshold, fusion=fusion
|
||||||
)
|
)
|
||||||
@@ -230,6 +235,17 @@ def configure_semantic_tools(mcp: FastMCP):
|
|||||||
verified_chunk_count,
|
verified_chunk_count,
|
||||||
dropped_count,
|
dropped_count,
|
||||||
)
|
)
|
||||||
|
# Safe to log titles now: these results passed verify-on-read, so the
|
||||||
|
# caller is confirmed to have access (unverified titles were never
|
||||||
|
# logged — see the search algorithms).
|
||||||
|
if verified_results:
|
||||||
|
logger.debug(
|
||||||
|
"Top verified results: %s",
|
||||||
|
", ".join(
|
||||||
|
f"{r.doc_type}_{r.id} (score={r.score:.3f}, title='{r.title}')"
|
||||||
|
for r in verified_results[:5]
|
||||||
|
),
|
||||||
|
)
|
||||||
search_results = verified_results[:limit]
|
search_results = verified_results[:limit]
|
||||||
|
|
||||||
# Convert SearchResult objects to SemanticSearchResult for response.
|
# Convert SearchResult objects to SemanticSearchResult for response.
|
||||||
|
|||||||
@@ -24,6 +24,7 @@ from qdrant_client.models import Distance, PointStruct, VectorParams
|
|||||||
from nextcloud_mcp_server.config import get_settings
|
from nextcloud_mcp_server.config import get_settings
|
||||||
from nextcloud_mcp_server.embedding import SimpleEmbeddingProvider
|
from nextcloud_mcp_server.embedding import SimpleEmbeddingProvider
|
||||||
from nextcloud_mcp_server.search.algorithms import get_indexed_doc_types
|
from nextcloud_mcp_server.search.algorithms import get_indexed_doc_types
|
||||||
|
from nextcloud_mcp_server.search.context import _get_chunk_by_index_from_qdrant
|
||||||
from nextcloud_mcp_server.search.semantic import SemanticSearchAlgorithm
|
from nextcloud_mcp_server.search.semantic import SemanticSearchAlgorithm
|
||||||
|
|
||||||
pytestmark = pytest.mark.integration
|
pytestmark = pytest.mark.integration
|
||||||
@@ -98,6 +99,11 @@ async def seeded_collection(monkeypatch):
|
|||||||
"nextcloud_mcp_server.search.algorithms.get_qdrant_client",
|
"nextcloud_mcp_server.search.algorithms.get_qdrant_client",
|
||||||
AsyncMock(return_value=client),
|
AsyncMock(return_value=client),
|
||||||
)
|
)
|
||||||
|
# The cached-chunk lookups read the client from the context module.
|
||||||
|
monkeypatch.setattr(
|
||||||
|
"nextcloud_mcp_server.search.context.get_qdrant_client",
|
||||||
|
AsyncMock(return_value=client),
|
||||||
|
)
|
||||||
|
|
||||||
yield provider
|
yield provider
|
||||||
|
|
||||||
@@ -189,3 +195,18 @@ async def test_get_indexed_doc_types_is_acl_aware(seeded_collection):
|
|||||||
}
|
}
|
||||||
# Self-only (default): Bob owns nothing here → discovers nothing.
|
# Self-only (default): Bob owns nothing here → discovers nothing.
|
||||||
assert await get_indexed_doc_types("bob") == set()
|
assert await get_indexed_doc_types("bob") == set()
|
||||||
|
|
||||||
|
|
||||||
|
async def test_cached_chunk_lookup_is_acl_aware(seeded_collection):
|
||||||
|
"""The cached-chunk Qdrant lookup honours accessible_owners: Bob retrieves
|
||||||
|
the excerpt of Alice's file point (owner_id=alice, chunk_index=0) when alice
|
||||||
|
is in his accessible owners, but not when scoped self-only. This is the
|
||||||
|
Qdrant-layer half of cross-user file chunk context (the per-file access
|
||||||
|
gate lives in get_chunk_with_context / file_accessible_by_id)."""
|
||||||
|
# Alice's seeded file point (_ALICE_FILE) carries excerpt=_DOC_TEXT at chunk 0.
|
||||||
|
text = await _get_chunk_by_index_from_qdrant(
|
||||||
|
"bob", "101", "file", 0, accessible_owners=["bob", "alice"]
|
||||||
|
)
|
||||||
|
assert text == _DOC_TEXT
|
||||||
|
# Self-only Bob cannot reach Alice's cached chunk.
|
||||||
|
assert await _get_chunk_by_index_from_qdrant("bob", "101", "file", 0) is None
|
||||||
|
|||||||
@@ -37,6 +37,7 @@ from nextcloud_mcp_server.search.access_filter import (
|
|||||||
clear_accessible_owners_cache,
|
clear_accessible_owners_cache,
|
||||||
list_accessible_owners,
|
list_accessible_owners,
|
||||||
)
|
)
|
||||||
|
from nextcloud_mcp_server.search.context import get_chunk_with_context
|
||||||
from nextcloud_mcp_server.search.semantic import SemanticSearchAlgorithm
|
from nextcloud_mcp_server.search.semantic import SemanticSearchAlgorithm
|
||||||
from nextcloud_mcp_server.search.verification import verify_search_results
|
from nextcloud_mcp_server.search.verification import verify_search_results
|
||||||
|
|
||||||
@@ -149,6 +150,12 @@ async def seeded_semantic(monkeypatch, shared_file):
|
|||||||
"nextcloud_mcp_server.search.semantic.get_embedding_service",
|
"nextcloud_mcp_server.search.semantic.get_embedding_service",
|
||||||
lambda: provider,
|
lambda: provider,
|
||||||
)
|
)
|
||||||
|
# The cached-chunk lookups (get_chunk_with_context) read the client from the
|
||||||
|
# context module — point it at the same in-memory Qdrant.
|
||||||
|
monkeypatch.setattr(
|
||||||
|
"nextcloud_mcp_server.search.context.get_qdrant_client",
|
||||||
|
AsyncMock(return_value=client),
|
||||||
|
)
|
||||||
yield file_id
|
yield file_id
|
||||||
await client.close()
|
await client.close()
|
||||||
|
|
||||||
@@ -215,3 +222,50 @@ async def test_file_accessible_by_id_resolves_shares(acl_users, shared_file):
|
|||||||
assert await acl_users["bob"].webdav.file_accessible_by_id(fid) is True
|
assert await acl_users["bob"].webdav.file_accessible_by_id(fid) is True
|
||||||
# ...the non-recipient cannot.
|
# ...the non-recipient cannot.
|
||||||
assert await acl_users["diana"].webdav.file_accessible_by_id(fid) is False
|
assert await acl_users["diana"].webdav.file_accessible_by_id(fid) is False
|
||||||
|
|
||||||
|
|
||||||
|
async def test_cross_user_file_chunk_context(acl_users, seeded_semantic):
|
||||||
|
"""End-to-end cross-user FILE chunk context: Bob (a share recipient) gets
|
||||||
|
Alice's cached chunk text, Diana (no share) gets None.
|
||||||
|
|
||||||
|
Exercises the full secure path: the ACL-aware Qdrant cached-chunk lookup
|
||||||
|
(owner_id=alice surfaces for Bob) gated by a real per-file
|
||||||
|
``file_accessible_by_id`` check against live Nextcloud. Diana fails the gate
|
||||||
|
and is denied even though the chunk is cached. Per-user types are covered by
|
||||||
|
the self-only behaviour elsewhere — this is the file path the feature adds.
|
||||||
|
"""
|
||||||
|
file_id = seeded_semantic
|
||||||
|
bob = acl_users["bob"]
|
||||||
|
diana = acl_users["diana"]
|
||||||
|
|
||||||
|
bob_owners = await list_accessible_owners(bob.sharing, "bob")
|
||||||
|
assert "alice" in bob_owners
|
||||||
|
|
||||||
|
ctx = await get_chunk_with_context(
|
||||||
|
nc_client=bob,
|
||||||
|
user_id="bob",
|
||||||
|
doc_id=str(file_id),
|
||||||
|
doc_type="file",
|
||||||
|
chunk_start=0,
|
||||||
|
chunk_end=len(_DOC_TEXT),
|
||||||
|
chunk_index=0,
|
||||||
|
total_chunks=1,
|
||||||
|
accessible_owners=bob_owners,
|
||||||
|
)
|
||||||
|
assert ctx is not None, "Bob (share recipient) must get Alice's cached chunk"
|
||||||
|
assert ctx.chunk_text == _DOC_TEXT
|
||||||
|
|
||||||
|
# Diana has no share → per-file gate denies even though the chunk is cached.
|
||||||
|
diana_owners = await list_accessible_owners(diana.sharing, "diana")
|
||||||
|
denied = await get_chunk_with_context(
|
||||||
|
nc_client=diana,
|
||||||
|
user_id="diana",
|
||||||
|
doc_id=str(file_id),
|
||||||
|
doc_type="file",
|
||||||
|
chunk_start=0,
|
||||||
|
chunk_end=len(_DOC_TEXT),
|
||||||
|
chunk_index=0,
|
||||||
|
total_chunks=1,
|
||||||
|
accessible_owners=diana_owners,
|
||||||
|
)
|
||||||
|
assert denied is None, "Diana (no share) must not get cross-user chunk context"
|
||||||
|
|||||||
@@ -156,11 +156,11 @@ class TestBuildOwnershipFilter:
|
|||||||
def test_defaults_to_self_only_when_owners_omitted(self) -> None:
|
def test_defaults_to_self_only_when_owners_omitted(self) -> None:
|
||||||
flt = build_ownership_filter("alice")
|
flt = build_ownership_filter("alice")
|
||||||
|
|
||||||
|
# Self-only: just the user_id branch. Self is NOT duplicated into an
|
||||||
|
# owner_id branch (the user_id branch already covers self-owned content).
|
||||||
assert flt.should is not None
|
assert flt.should is not None
|
||||||
assert len(flt.should) == 2 # owner_id branch + legacy user_id branch
|
assert len(flt.should) == 1
|
||||||
owner_branch, user_branch = flt.should
|
(user_branch,) = flt.should
|
||||||
assert owner_branch.key == "owner_id"
|
|
||||||
assert owner_branch.match.any == ["alice"]
|
|
||||||
assert user_branch.key == "user_id"
|
assert user_branch.key == "user_id"
|
||||||
assert user_branch.match.value == "alice"
|
assert user_branch.match.value == "alice"
|
||||||
|
|
||||||
@@ -168,10 +168,10 @@ class TestBuildOwnershipFilter:
|
|||||||
flt = build_ownership_filter("alice", ["alice", "bob", "carol"])
|
flt = build_ownership_filter("alice", ["alice", "bob", "carol"])
|
||||||
|
|
||||||
owner_branch, user_branch = flt.should
|
owner_branch, user_branch = flt.should
|
||||||
# Owner branch reflects the expanded set.
|
# Owner branch holds only the OTHER owners — self ("alice") is excluded
|
||||||
assert set(owner_branch.match.any) == {"alice", "bob", "carol"}
|
# because the user_id branch already matches self-owned content.
|
||||||
# Legacy user_id branch keeps the original user — that's the only
|
assert set(owner_branch.match.any) == {"bob", "carol"}
|
||||||
# legacy match path, so it must NOT widen to other owners.
|
assert user_branch.key == "user_id"
|
||||||
assert user_branch.match.value == "alice"
|
assert user_branch.match.value == "alice"
|
||||||
|
|
||||||
def test_explicit_empty_list_omits_owner_branch_keeps_legacy(self) -> None:
|
def test_explicit_empty_list_omits_owner_branch_keeps_legacy(self) -> None:
|
||||||
|
|||||||
@@ -68,7 +68,11 @@ class TestIndexedPath:
|
|||||||
# One scroll call, and the filter must include chunk_index (not offsets)
|
# One scroll call, and the filter must include chunk_index (not offsets)
|
||||||
qdrant_client.scroll.assert_awaited_once()
|
qdrant_client.scroll.assert_awaited_once()
|
||||||
scroll_kwargs = qdrant_client.scroll.await_args.kwargs
|
scroll_kwargs = qdrant_client.scroll.await_args.kwargs
|
||||||
filter_keys = [c.key for c in scroll_kwargs["scroll_filter"].must]
|
# Skip nested Filters (the ACL ownership sub-filter) — only field
|
||||||
|
# conditions carry a `.key`.
|
||||||
|
filter_keys = [
|
||||||
|
c.key for c in scroll_kwargs["scroll_filter"].must if hasattr(c, "key")
|
||||||
|
]
|
||||||
assert "chunk_index" in filter_keys
|
assert "chunk_index" in filter_keys
|
||||||
assert "chunk_start_offset" not in filter_keys
|
assert "chunk_start_offset" not in filter_keys
|
||||||
assert "chunk_end_offset" not in filter_keys
|
assert "chunk_end_offset" not in filter_keys
|
||||||
@@ -92,7 +96,11 @@ class TestOffsetFallbackPath:
|
|||||||
|
|
||||||
assert result == (bbox, 2)
|
assert result == (bbox, 2)
|
||||||
scroll_kwargs = qdrant_client.scroll.await_args.kwargs
|
scroll_kwargs = qdrant_client.scroll.await_args.kwargs
|
||||||
filter_keys = [c.key for c in scroll_kwargs["scroll_filter"].must]
|
# Skip nested Filters (the ACL ownership sub-filter) — only field
|
||||||
|
# conditions carry a `.key`.
|
||||||
|
filter_keys = [
|
||||||
|
c.key for c in scroll_kwargs["scroll_filter"].must if hasattr(c, "key")
|
||||||
|
]
|
||||||
assert "chunk_start_offset" in filter_keys
|
assert "chunk_start_offset" in filter_keys
|
||||||
assert "chunk_end_offset" in filter_keys
|
assert "chunk_end_offset" in filter_keys
|
||||||
assert "chunk_index" not in filter_keys
|
assert "chunk_index" not in filter_keys
|
||||||
|
|||||||
Reference in New Issue
Block a user