fix(vector): retry transient embed errors so a pod rollover drops 0 docs
From card 309 (OHR-Bench smoke-test triage): during a backend-pod rollover the
embedding endpoint was briefly unreachable, and openai.APIConnectionError /
ConnectError propagated unretried (the provider only retried 429). Documents
exhausted the 3 in-process retries and were dropped for that scan cycle.
Broaden the provider-level retry to the transient set -- APIConnectionError,
APITimeoutError, 429, and 5xx -- on the existing exponential backoff (2s->60s,
5 attempts), so a few seconds of retry rides through the rollover. Permanent
4xx (auth, bad request) still re-raise immediately. Generalize the shared
_retry helper (retry_on_rate_limit -> retry_on_transient, predicate renamed to
should_retry, accurate log label) with a back-compat alias; Mistral gets 429+5xx
for parity. The production gateway path inherits this via GatewayProvider, which
delegates to the decorated OpenAIProvider methods.
Add astrolabe_vector_ingest_dropped_total{reason}, incremented when a document
exhausts retries, classified (connection|timeout|rate_limit|server|qdrant|other)
by _drop_reason so the embed-drop rate is alertable per cause. Dropped docs are
NOT marked failed, so the next full scan re-picks them (re-queue via scan loop).
Refs: Deck board 12 card 309 (AC #1 no permanently-dropped docs; embed-drop
metric for AC #5).
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.8
parent
457c115ef4
commit
258ee96f4c
@@ -1,4 +1,4 @@
|
||||
"""Unit tests for the shared rate-limit retry decorator."""
|
||||
"""Unit tests for the shared transient-error retry decorator."""
|
||||
|
||||
from unittest.mock import AsyncMock
|
||||
|
||||
@@ -26,9 +26,7 @@ async def test_retry_succeeds_after_429():
|
||||
"""A 429 followed by success returns the success value."""
|
||||
calls = {"n": 0}
|
||||
|
||||
@_retry.retry_on_rate_limit(
|
||||
_FakeError, is_rate_limit=lambda e: e.status_code == 429
|
||||
)
|
||||
@_retry.retry_on_transient(_FakeError, should_retry=lambda e: e.status_code == 429)
|
||||
async def flaky():
|
||||
calls["n"] += 1
|
||||
if calls["n"] < 3:
|
||||
@@ -45,9 +43,7 @@ async def test_retry_reraises_non_rate_limit_immediately():
|
||||
"""A non-rate-limit error of the same class is re-raised on first hit."""
|
||||
calls = {"n": 0}
|
||||
|
||||
@_retry.retry_on_rate_limit(
|
||||
_FakeError, is_rate_limit=lambda e: e.status_code == 429
|
||||
)
|
||||
@_retry.retry_on_transient(_FakeError, should_retry=lambda e: e.status_code == 429)
|
||||
async def boom():
|
||||
calls["n"] += 1
|
||||
raise _FakeError(500)
|
||||
@@ -62,9 +58,7 @@ async def test_retry_gives_up_after_max_retries():
|
||||
"""After MAX_RETRIES failed attempts the last error is re-raised."""
|
||||
calls = {"n": 0}
|
||||
|
||||
@_retry.retry_on_rate_limit(
|
||||
_FakeError, is_rate_limit=lambda e: e.status_code == 429
|
||||
)
|
||||
@_retry.retry_on_transient(_FakeError, should_retry=lambda e: e.status_code == 429)
|
||||
async def always_429():
|
||||
calls["n"] += 1
|
||||
raise _FakeError(429)
|
||||
@@ -79,7 +73,7 @@ async def test_retry_default_predicate_treats_all_as_rate_limit():
|
||||
"""Default predicate (`lambda _: True`) retries every caught exception."""
|
||||
calls = {"n": 0}
|
||||
|
||||
@_retry.retry_on_rate_limit(_FakeError)
|
||||
@_retry.retry_on_transient(_FakeError)
|
||||
async def fail_once():
|
||||
calls["n"] += 1
|
||||
if calls["n"] < 2:
|
||||
@@ -95,9 +89,37 @@ async def test_retry_default_predicate_treats_all_as_rate_limit():
|
||||
async def test_retry_does_not_catch_unrelated_exceptions():
|
||||
"""Exceptions of a different class bypass the decorator entirely."""
|
||||
|
||||
@_retry.retry_on_rate_limit(_FakeError)
|
||||
@_retry.retry_on_transient(_FakeError)
|
||||
async def value_error():
|
||||
raise ValueError("nope")
|
||||
|
||||
with pytest.raises(ValueError, match="nope"):
|
||||
await value_error()
|
||||
|
||||
|
||||
class _ConnError(Exception):
|
||||
pass
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
async def test_retry_accepts_tuple_of_exception_types():
|
||||
"""A tuple of exception classes is caught (the OpenAI transient set shape)."""
|
||||
calls = {"n": 0}
|
||||
|
||||
@_retry.retry_on_transient((_FakeError, _ConnError))
|
||||
async def flaky():
|
||||
calls["n"] += 1
|
||||
if calls["n"] == 1:
|
||||
raise _ConnError("dropped")
|
||||
if calls["n"] == 2:
|
||||
raise _FakeError(503)
|
||||
return "ok"
|
||||
|
||||
assert await flaky() == "ok"
|
||||
assert calls["n"] == 3
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
async def test_retry_on_rate_limit_is_backcompat_alias():
|
||||
"""The old name still resolves to the generalized helper."""
|
||||
assert _retry.retry_on_rate_limit is _retry.retry_on_transient
|
||||
|
||||
Reference in New Issue
Block a user