fix(vector): retry transient embed errors so a pod rollover drops 0 docs
From card 309 (OHR-Bench smoke-test triage): during a backend-pod rollover the
embedding endpoint was briefly unreachable, and openai.APIConnectionError /
ConnectError propagated unretried (the provider only retried 429). Documents
exhausted the 3 in-process retries and were dropped for that scan cycle.
Broaden the provider-level retry to the transient set -- APIConnectionError,
APITimeoutError, 429, and 5xx -- on the existing exponential backoff (2s->60s,
5 attempts), so a few seconds of retry rides through the rollover. Permanent
4xx (auth, bad request) still re-raise immediately. Generalize the shared
_retry helper (retry_on_rate_limit -> retry_on_transient, predicate renamed to
should_retry, accurate log label) with a back-compat alias; Mistral gets 429+5xx
for parity. The production gateway path inherits this via GatewayProvider, which
delegates to the decorated OpenAIProvider methods.
Add astrolabe_vector_ingest_dropped_total{reason}, incremented when a document
exhausts retries, classified (connection|timeout|rate_limit|server|qdrant|other)
by _drop_reason so the embed-drop rate is alertable per cause. Dropped docs are
NOT marked failed, so the next full scan re-picks them (re-queue via scan loop).
Refs: Deck board 12 card 309 (AC #1 no permanently-dropped docs; embed-drop
metric for AC #5).
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.8
parent
457c115ef4
commit
258ee96f4c
@@ -330,3 +330,85 @@ async def test_openai_close(mock_openai_client):
|
||||
|
||||
await provider.close()
|
||||
mock_openai_client.close.assert_called_once()
|
||||
|
||||
|
||||
# --- transient-error retry (card 309) ----------------------------------------
|
||||
|
||||
|
||||
def _req():
|
||||
import httpx
|
||||
|
||||
return httpx.Request("POST", "http://gw/v1/embeddings")
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_is_transient_classifies_retryable_errors():
|
||||
"""Connection / timeout / 429 / 5xx are transient; 4xx and others are not."""
|
||||
import httpx
|
||||
from openai import (
|
||||
APIConnectionError,
|
||||
APITimeoutError,
|
||||
BadRequestError,
|
||||
InternalServerError,
|
||||
RateLimitError,
|
||||
)
|
||||
|
||||
from nextcloud_mcp_server.providers.openai import _is_transient
|
||||
|
||||
req = _req()
|
||||
assert _is_transient(APIConnectionError(request=req)) is True
|
||||
assert _is_transient(APITimeoutError(request=req)) is True
|
||||
assert (
|
||||
_is_transient(
|
||||
RateLimitError("rl", response=httpx.Response(429, request=req), body=None)
|
||||
)
|
||||
is True
|
||||
)
|
||||
assert (
|
||||
_is_transient(
|
||||
InternalServerError(
|
||||
"boom", response=httpx.Response(500, request=req), body=None
|
||||
)
|
||||
)
|
||||
is True
|
||||
)
|
||||
# Permanent client errors must NOT be retried.
|
||||
assert (
|
||||
_is_transient(
|
||||
BadRequestError("bad", response=httpx.Response(400, request=req), body=None)
|
||||
)
|
||||
is False
|
||||
)
|
||||
assert _is_transient(ValueError("unrelated")) is False
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
async def test_embed_retries_on_connection_error(mock_openai_client, monkeypatch):
|
||||
"""A transient APIConnectionError (pod rollover) is retried, not dropped."""
|
||||
from openai import APIConnectionError
|
||||
|
||||
from nextcloud_mcp_server.providers import _retry
|
||||
|
||||
monkeypatch.setattr(_retry.anyio, "sleep", AsyncMock(return_value=None))
|
||||
|
||||
mock_embedding_data = MagicMock()
|
||||
mock_embedding_data.embedding = [0.1, 0.2, 0.3]
|
||||
mock_response = MagicMock()
|
||||
mock_response.data = [mock_embedding_data]
|
||||
|
||||
calls = {"n": 0}
|
||||
|
||||
async def _flaky(*args, **kwargs):
|
||||
calls["n"] += 1
|
||||
if calls["n"] == 1:
|
||||
raise APIConnectionError(request=_req())
|
||||
return mock_response
|
||||
|
||||
mock_openai_client.embeddings.create = _flaky
|
||||
provider = OpenAIProvider(
|
||||
api_key="test-key", embedding_model="text-embedding-3-small"
|
||||
)
|
||||
|
||||
result = await provider.embed("hello")
|
||||
assert result == [0.1, 0.2, 0.3]
|
||||
assert calls["n"] == 2 # one failure, one success
|
||||
|
||||
Reference in New Issue
Block a user