feat: dedup shared-file parsing/embedding across users in vector sync
A file shared across many users — directly, or via a group folder shared to a group — was parsed and embedded once per user. Chunk point IDs are user-agnostic (uuid5(tenant_id, doc_id=fileid, chunk_index)), but the per-user freshness gate filtered Qdrant by user_id, so two readers ping-ponged: each overwrote the other's points and each kept seeing "not indexed for me", reprocessing every scan. Production telemetry (note 386945, finding #5) measured identical docs re-processed every few hours at 7-13s each, with PDF parse ~62% of per-doc cost. Layer 1 — tenant-wide dedup: - Thread the scanner's tag-REPORT etag into the file DocumentTask and the chunk payload; index `etag` as a KEYWORD field. - vector/sharing_state.find_indexed_content scrolls tenant-wide (no user_id filter) for a non-placeholder point matching (doc_id, doc_type, etag), gated on embedding_identity in Python so a model switch correctly forces a re-embed. - Scanner skips enqueue and the processor skips fetch/parse/embed when a match exists (cross-worker race-guard before WebDAV read). Dedup is fail-safe: a Qdrant error degrades to "process normally". Layer 2 — observed-access ACL (no admin / GroupFolders API needed): - Each point carries `acl_principals` = the set of user:<uid> whose scanner has observed (hence can read) the file. The per-user tag REPORT is the access oracle; group membership/GroupFolders enumeration is admin-only and unavailable in multi-user modes. - build_ownership_filter ORs MatchAny(acl_principals, ["user:<me>"]) so a deduplicated shared/group-folder point surfaces to every reader; verify-on-read (_verify_files) remains the precise ACL gate. - Deletion/eviction become "release one user": drop the principal and delete the points only when the set empties, so one user untagging a shared file doesn't evict it for the others. Legacy points without the field keep the original per-user delete. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.8
parent
0919513f21
commit
1c93e7286d
@@ -11,7 +11,7 @@ from typing import Any, cast
|
||||
import anyio
|
||||
from anyio.abc import TaskStatus
|
||||
from anyio.streams.memory import MemoryObjectReceiveStream
|
||||
from qdrant_client.models import FieldCondition, Filter, MatchValue, PointStruct
|
||||
from qdrant_client.models import PointStruct
|
||||
|
||||
from nextcloud_mcp_server.acl_hash import compute_acl_hash
|
||||
from nextcloud_mcp_server.client import NextcloudClient
|
||||
@@ -33,6 +33,11 @@ from nextcloud_mcp_server.vector.html_processor import html_to_markdown
|
||||
from nextcloud_mcp_server.vector.placeholder import delete_placeholder_point
|
||||
from nextcloud_mcp_server.vector.qdrant_client import get_qdrant_client
|
||||
from nextcloud_mcp_server.vector.scanner import DocumentTask
|
||||
from nextcloud_mcp_server.vector.sharing_state import (
|
||||
claim_existing_index,
|
||||
existing_principals,
|
||||
release_document_for_user,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -192,28 +197,15 @@ async def process_document(
|
||||
):
|
||||
try:
|
||||
qdrant_client = await get_qdrant_client()
|
||||
settings = get_settings()
|
||||
|
||||
# Handle deletion
|
||||
if doc_task.operation == "delete":
|
||||
await qdrant_client.delete(
|
||||
collection_name=settings.get_collection_name(),
|
||||
points_selector=Filter(
|
||||
must=[
|
||||
FieldCondition(
|
||||
key="user_id",
|
||||
match=MatchValue(value=doc_task.user_id),
|
||||
),
|
||||
FieldCondition(
|
||||
key="doc_id",
|
||||
match=MatchValue(value=doc_task.doc_id),
|
||||
),
|
||||
FieldCondition(
|
||||
key="doc_type",
|
||||
match=MatchValue(value=doc_task.doc_type),
|
||||
),
|
||||
]
|
||||
),
|
||||
# Release this user rather than blind-delete: a file shared across
|
||||
# users has one user-agnostic point set referenced by multiple
|
||||
# principals, so the points are removed only once the last reader
|
||||
# is gone (see vector/sharing_state.release_document_for_user).
|
||||
await release_document_for_user(
|
||||
doc_task.doc_id, doc_task.doc_type, doc_task.user_id
|
||||
)
|
||||
logger.info(
|
||||
"Deleted %s_%s for %s",
|
||||
@@ -479,6 +471,28 @@ async def _index_document(
|
||||
)
|
||||
file_path = doc_task.file_path
|
||||
|
||||
# Cross-worker dedup race-guard: two users' tasks for the same shared
|
||||
# file can be enqueued before either finishes. If another worker has
|
||||
# already indexed this exact content (fileid + etag + embedding model)
|
||||
# in the tenant, claim it for this user (observed-access ACL) and skip
|
||||
# the expensive fetch/parse/embed entirely.
|
||||
if doc_task.etag and await claim_existing_index(
|
||||
doc_task.doc_id, "file", doc_task.etag, doc_task.user_id
|
||||
):
|
||||
await delete_placeholder_point(
|
||||
doc_id=doc_task.doc_id,
|
||||
doc_type="file",
|
||||
user_id=doc_task.user_id,
|
||||
)
|
||||
logger.info(
|
||||
"Dedup hit for file %s (etag=%s); claimed for user %s "
|
||||
"without reprocessing",
|
||||
doc_task.doc_id,
|
||||
doc_task.etag,
|
||||
doc_task.user_id,
|
||||
)
|
||||
return
|
||||
|
||||
# Read file content via WebDAV
|
||||
content_bytes, content_type = await nc_client.webdav.read_file(file_path)
|
||||
else:
|
||||
@@ -510,7 +524,10 @@ async def _index_document(
|
||||
content = result.text
|
||||
file_metadata = result.metadata
|
||||
title = file_metadata.get("title") or file_path.split("/")[-1]
|
||||
etag = "" # WebDAV read_file doesn't return etag
|
||||
# etag comes from the scanner's tag REPORT (threaded via the
|
||||
# DocumentTask); read_file itself returns no etag. It is the
|
||||
# tenant-wide content-dedup key, so it must be persisted.
|
||||
etag = doc_task.etag or ""
|
||||
|
||||
# Diagnostic: Log page boundary information if available
|
||||
if "page_boundaries" in file_metadata:
|
||||
@@ -763,6 +780,19 @@ async def _index_document(
|
||||
_embedding_identity = settings.get_embedding_model_name()
|
||||
_acl_hash = compute_acl_hash([("user", doc_task.user_id)])
|
||||
|
||||
# Observed-access ACL principals (computed once per document, not per chunk).
|
||||
# Seed with the indexer (and owner, if distinct) unioned with any principals
|
||||
# already recorded — so re-indexing after a content change preserves
|
||||
# visibility for readers who had previously claimed the file rather than
|
||||
# resetting it to just the indexer.
|
||||
_acl_principals = sorted(
|
||||
set(await existing_principals(doc_task.doc_id, doc_task.doc_type))
|
||||
| {
|
||||
f"user:{doc_task.user_id}",
|
||||
f"user:{doc_task.owner_id or doc_task.user_id}",
|
||||
}
|
||||
)
|
||||
|
||||
# Surface deck card data quality issues at indexing time rather than
|
||||
# only at verification time (where _verify_deck_cards falls through to
|
||||
# legacy-data pass-through when board_id/stack_id are missing). This is
|
||||
@@ -807,6 +837,13 @@ async def _index_document(
|
||||
# crawl shared-with-them content can set owner_id to the
|
||||
# true owner without losing the "who indexed this" trail.
|
||||
"owner_id": doc_task.owner_id or doc_task.user_id,
|
||||
# Observed-access ACL set: every user whose scanner has seen
|
||||
# (hence can read) this document. Seeded with the indexer (and
|
||||
# owner, if distinct); grown lazily as other readers' scanners
|
||||
# hit the tenant-wide dedup path. Search ORs a
|
||||
# MatchAny(acl_principals, ["user:<me>"]) branch so a
|
||||
# deduplicated shared file stays findable by every reader.
|
||||
"acl_principals": _acl_principals,
|
||||
"doc_id": doc_task.doc_id,
|
||||
"doc_type": doc_task.doc_type,
|
||||
"is_placeholder": False, # Real indexed document (not placeholder)
|
||||
|
||||
Reference in New Issue
Block a user