fix(vector): don't inflate qdrant-error metric on embed drops (#893 r3)

Round-3 review on PR #893:
- record_qdrant_operation("upsert","error") now fires only when the exhausted
  retry was actually a Qdrant failure (reason=="qdrant"); an embed/connection
  failure exhausts retries before Qdrant is called, so attributing it to
  mcp_qdrant_operations_total{error} inflated that signal. The cause is still
  captured by record_ingest_dropped.
- Add test_mistral_embed_retries_on_5xx: exercises the full Mistral retry path
  (5xx SDKError then success), not just the predicate.
- Add test_generate_does_not_retry_on_bad_request: generate() fast-fails on a
  permanent 4xx.
- Move astrolabe_vector_ingest_dropped_total's definition into the astrolabe_
  pipeline-metrics block (was in the mcp_ section).

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
Chris Coutinho
2026-06-11 06:17:59 +02:00
co-authored by Claude Opus 4.8
parent 8f7a8432f5
commit c4b6d4a017
4 changed files with 66 additions and 20 deletions
+12 -12
View File
@@ -161,18 +161,6 @@ vector_sync_processing_duration_seconds = Histogram(
buckets=(0.1, 0.5, 1.0, 2.5, 5.0, 10.0, 30.0, 60.0),
)
# Documents dropped after exhausting in-process indexing retries (the scanner
# re-picks them on a later full scan, so this is "dropped for this cycle", not
# "lost forever"). Labelled by classified cause so the embed-drop rate from a
# transient backend-pod rollover (connection/timeout) is alertable distinctly
# from a persistent fault (card 309). astrolabe_ prefix: pipeline metric.
vector_ingest_dropped_total = Counter(
"astrolabe_vector_ingest_dropped_total",
"Documents dropped after exhausting indexing retries, by cause",
# reason: connection | timeout | rate_limit | server | qdrant | other
["reason"],
)
vector_sync_queue_size = Gauge(
"mcp_vector_sync_queue_size",
"Current number of documents in processing queue",
@@ -284,6 +272,18 @@ document_parse_failed_total = Counter(
["reason"], # reason: timeout | oom | error
)
# Documents dropped after exhausting in-process indexing retries (the scanner
# re-picks them on a later full scan, so this is "dropped for this cycle", not
# "lost forever"). Labelled by classified cause so the embed-drop rate from a
# transient backend-pod rollover (connection/timeout) is alertable distinctly
# from a persistent fault (card 309).
vector_ingest_dropped_total = Counter(
"astrolabe_vector_ingest_dropped_total",
"Documents dropped after exhausting indexing retries, by cause",
# reason: connection | timeout | rate_limit | server | qdrant | other
["reason"],
)
# --- Tier-0 classifier (shadow mode) -----------------------------------------
#
# The classifier runs a cheap pre-pass per PDF and recommends a starting tier.
+11 -8
View File
@@ -528,14 +528,17 @@ async def process_document(
"drop_reason": reason,
},
)
# Record the failed Qdrant upsert. The processing-error
# metric is recorded once by the outer handler below, so
# exhausted-retry failures aren't double-counted. The
# drop counter is labelled by cause so a transient
# rollover (connection/timeout) is alertable distinctly.
# The document is NOT marked failed, so the next scan
# re-picks it (re-queue via the scan loop, card 309).
record_qdrant_operation("upsert", "error")
# Count a failed Qdrant upsert ONLY when Qdrant was the
# failing component; an embed/connection failure exhausts
# retries before Qdrant is ever called, so attributing it
# to mcp_qdrant_operations_total{error} would inflate that
# signal. The cause is captured by record_ingest_dropped
# instead, and the processing-error metric is recorded
# once by the outer handler below (no double-count). The
# document is NOT marked failed, so the next scan re-picks
# it (re-queue via the scan loop, card 309).
if reason == "qdrant":
record_qdrant_operation("upsert", "error")
record_ingest_dropped(reason)
raise
+21
View File
@@ -332,3 +332,24 @@ def test_mistral_is_transient_predicate():
assert _is_transient(err_400) is False # permanent client error
# ValueError has no status_code attr → getattr returns None → False.
assert _is_transient(ValueError()) is False
@pytest.mark.unit
async def test_mistral_embed_retries_on_5xx(mock_mistral_client, monkeypatch):
"""A 5xx SDKError is retried end-to-end (not just classified by the predicate)."""
from nextcloud_mcp_server.providers import _retry
monkeypatch.setattr(_retry.anyio, "sleep", AsyncMock(return_value=None))
# Real SDKError instance (so `except SDKError` catches it) with a 5xx status.
err = SDKError.__new__(SDKError)
err.status_code = 500
mock_mistral_client.embeddings.create_async = AsyncMock(
side_effect=[err, _make_response([[0.1, 0.2, 0.3]])]
)
provider = MistralProvider(api_key="test-key", embedding_model="mistral-embed")
embedding = await provider.embed("hello")
assert embedding == [0.1, 0.2, 0.3]
assert mock_mistral_client.embeddings.create_async.await_count == 2
+22
View File
@@ -457,3 +457,25 @@ async def test_generate_retries_on_connection_error(mock_openai_client, monkeypa
text = await provider.generate("prompt")
assert text == "Generated response"
assert create.await_count == 2
@pytest.mark.unit
async def test_generate_does_not_retry_on_bad_request(mock_openai_client, monkeypatch):
"""generate() fast-fails (no retry) on a permanent 4xx."""
import httpx
from openai import BadRequestError
from nextcloud_mcp_server.providers import _retry
monkeypatch.setattr(_retry.anyio, "sleep", AsyncMock(return_value=None))
err = BadRequestError(
"bad", response=httpx.Response(400, request=_req()), body=None
)
create = AsyncMock(side_effect=err)
mock_openai_client.chat.completions.create = create
provider = OpenAIProvider(api_key="test-key", generation_model="gpt-4o-mini")
with pytest.raises(BadRequestError):
await provider.generate("prompt")
assert create.await_count == 1 # no retry on a permanent 4xx