feat(vector): replace inline page-image payloads with chunk_bbox (Deck #76)
Per-chunk PDF page renders (~150–700 KB base64 PNG each) were the dominant disk consumer in production, repeatedly tripping `No space left on device: WAL buffer size exceeds available disk space` on welcomed-malamute Qdrant. Replace the inline highlighted_page_image / highlighted_page_number / highlight_count fields with a small `chunk_bbox` field: list[(x0, y0, x1, y1)] of normalized [0, 1] floats, ~32 bytes per chunk. Astrolabe (the only known consumer) renders the highlight client-side as a percentage-positioned overlay on top of the existing /api/v1/pdf-preview render-on-demand path (cbcoutinho/astrolabe#76). - pdf_highlighter: new compute_chunk_bboxes_batch() that reuses the existing _find_chunk_bbox text-search path, skipping all pixmap/PIL/PNG work. - processor: store chunk_bbox + chunk_bbox_page in the Qdrant payload, drop highlighted_page_image + friends, drop the base64 import. - visualization /api/v1/chunk-context and auth/viz_routes: read chunk_bbox instead of highlighted_page_image. - vector/__init__: stop eagerly re-exporting `processor`/`scanner` — fixes a pre-existing circular import (search.algorithms -> vector.placeholder -> vector/__init__ -> processor -> scanner -> server.semantic -> search.bm25_hybrid -> search.algorithms partial). Test suite that was broken on master (test_bm25_hybrid.py et al.) now collects and passes. - scripts/purge_page_images.py: ad-hoc, idempotent migration that delete_payload's the legacy keys from existing points. No reindex required; legacy chunks render the page with no overlay. Pairs with cbcoutinho/astrolabe#76. Frontend handles missing chunk_bbox gracefully, so this can land in either order. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.7
parent
2e1ab99a88
commit
ee402ea00e
Executable
+136
@@ -0,0 +1,136 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Purge legacy `highlighted_page_image` payloads from Qdrant (Deck #76).
|
||||
|
||||
Iterates all points in the configured Qdrant collection and deletes the
|
||||
legacy payload keys `highlighted_page_image`, `highlighted_page_number`,
|
||||
and `highlight_count`. This relieves disk pressure caused by inline
|
||||
base64 PNGs that the new code path no longer writes.
|
||||
|
||||
Idempotent: deleting non-existent keys is a no-op, so re-runs are safe.
|
||||
|
||||
Usage:
|
||||
uv run python scripts/purge_page_images.py [--dry-run] [--batch-size 256]
|
||||
|
||||
Connection settings (Qdrant URL/API key, collection name) are read from
|
||||
the same `Settings` object the server uses.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
|
||||
from qdrant_client import AsyncQdrantClient
|
||||
|
||||
from nextcloud_mcp_server.config import get_settings
|
||||
|
||||
logger = logging.getLogger("purge_page_images")
|
||||
|
||||
LEGACY_FIELDS = [
|
||||
"highlighted_page_image",
|
||||
"highlighted_page_number",
|
||||
"highlight_count",
|
||||
]
|
||||
|
||||
|
||||
def make_client() -> AsyncQdrantClient:
|
||||
settings = get_settings()
|
||||
if not settings.qdrant_url:
|
||||
raise SystemExit(
|
||||
"qdrant_url is not configured. Set QDRANT_URL (and QDRANT_API_KEY "
|
||||
"if required) before running this script."
|
||||
)
|
||||
return AsyncQdrantClient(
|
||||
url=settings.qdrant_url,
|
||||
api_key=settings.qdrant_api_key,
|
||||
timeout=60,
|
||||
)
|
||||
|
||||
|
||||
async def purge(dry_run: bool, batch_size: int) -> None:
|
||||
settings = get_settings()
|
||||
collection = settings.get_collection_name()
|
||||
client = make_client()
|
||||
|
||||
next_offset = None
|
||||
total_seen = 0
|
||||
total_updated = 0
|
||||
|
||||
logger.info(
|
||||
"Scanning collection %s; will delete keys %s%s",
|
||||
collection,
|
||||
LEGACY_FIELDS,
|
||||
" (dry run)" if dry_run else "",
|
||||
)
|
||||
|
||||
while True:
|
||||
points, next_offset = await client.scroll(
|
||||
collection_name=collection,
|
||||
limit=batch_size,
|
||||
offset=next_offset,
|
||||
with_payload=False,
|
||||
with_vectors=False,
|
||||
)
|
||||
if not points:
|
||||
break
|
||||
|
||||
ids = [p.id for p in points]
|
||||
total_seen += len(ids)
|
||||
|
||||
if not dry_run:
|
||||
await client.delete_payload(
|
||||
collection_name=collection,
|
||||
keys=LEGACY_FIELDS,
|
||||
points=ids,
|
||||
)
|
||||
total_updated += len(ids)
|
||||
|
||||
logger.info(
|
||||
"Batch: ids=%d total_seen=%d total_updated=%d",
|
||||
len(ids),
|
||||
total_seen,
|
||||
total_updated,
|
||||
)
|
||||
|
||||
if next_offset is None:
|
||||
break
|
||||
|
||||
logger.info(
|
||||
"Done. total_seen=%d total_updated=%d%s",
|
||||
total_seen,
|
||||
total_updated,
|
||||
" (dry run, no writes)" if dry_run else "",
|
||||
)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument(
|
||||
"--dry-run",
|
||||
action="store_true",
|
||||
help="Scan only; do not write any changes.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--batch-size",
|
||||
type=int,
|
||||
default=256,
|
||||
help="Points per scroll/update batch (default: 256).",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-v", "--verbose", action="store_true", help="Enable debug logging."
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
logging.basicConfig(
|
||||
level=logging.DEBUG if args.verbose else logging.INFO,
|
||||
format="%(asctime)s %(levelname)s %(name)s: %(message)s",
|
||||
)
|
||||
|
||||
asyncio.run(purge(dry_run=args.dry_run, batch_size=args.batch_size))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user