test(integration): address round-7 review — keep RAG assertion live, fix races
- test_no_results_for_unrelated_query: replace pytest.skip with `unrelated = ... or 0.0` and fall through. The physics query almost always returns nothing on this corpus, so the skip meant the comparison (and the manual-is-indexed check) never ran. Treating no-results as score 0.0 keeps the test live and vacuously satisfies `0.0 <= relevant`. - test_sampling: the three limit/threshold/max-tokens tests now gate on a representative created note being searchable (search_term + note_id) instead of a bare idle signal that can fire before the new notes are enqueued. - _get_with_retry: only sleep between attempts, not before giving up. - _search_helpers: log the id/doc_type schema-drift mismatch at WARNING (CI runs --log-cli-level=WARN) so it surfaces instead of hiding behind a timeout. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.8
parent
4c7c627e51
commit
ca313e7271
@@ -49,9 +49,10 @@ async def document_is_searchable(
|
|||||||
if r.get("id") == note_id:
|
if r.get("id") == note_id:
|
||||||
if r.get("doc_type") == "note":
|
if r.get("doc_type") == "note":
|
||||||
return True
|
return True
|
||||||
# id matched but not a note — surface possible schema drift
|
# id matched but not a note — surface possible schema drift at
|
||||||
# rather than silently timing out.
|
# WARNING (CI runs --log-cli-level=WARN) instead of letting the
|
||||||
logger.debug(
|
# caller time out with a generic message.
|
||||||
|
logger.warning(
|
||||||
"search hit id=%s has doc_type=%s (expected note)",
|
"search hit id=%s has doc_type=%s (expected note)",
|
||||||
note_id,
|
note_id,
|
||||||
r.get("doc_type"),
|
r.get("doc_type"),
|
||||||
|
|||||||
@@ -56,7 +56,8 @@ async def _get_with_retry(
|
|||||||
logger.warning(
|
logger.warning(
|
||||||
"GET %s failed (attempt %s/%s): %s", url, attempt, max_attempts, e
|
"GET %s failed (attempt %s/%s): %s", url, attempt, max_attempts, e
|
||||||
)
|
)
|
||||||
await anyio.sleep(2)
|
if attempt < max_attempts:
|
||||||
|
await anyio.sleep(2) # no point sleeping before we give up
|
||||||
assert last_exc is not None # loop ran at least once, so this is set
|
assert last_exc is not None # loop ran at least once, so this is set
|
||||||
raise last_exc
|
raise last_exc
|
||||||
|
|
||||||
|
|||||||
@@ -432,13 +432,16 @@ async def test_no_results_for_unrelated_query(nc_mcp_client, indexed_manual_pdf)
|
|||||||
query happened to retrieve any chunk at all). Comparing against a relevant
|
query happened to retrieve any chunk at all). Comparing against a relevant
|
||||||
query on the same corpus is self-calibrating and stable.
|
query on the same corpus is self-calibrating and stable.
|
||||||
"""
|
"""
|
||||||
unrelated = await _top_score(
|
# No results for the nonsense query is the ideal outcome — treat as score
|
||||||
nc_mcp_client, "quantum entanglement hadron collider particle physics"
|
# 0.0 and fall through, so the comparison (and the manual-is-indexed check
|
||||||
|
# below) still runs instead of the test silently skipping every time the
|
||||||
|
# physics query finds nothing.
|
||||||
|
unrelated = (
|
||||||
|
await _top_score(
|
||||||
|
nc_mcp_client, "quantum entanglement hadron collider particle physics"
|
||||||
|
)
|
||||||
|
or 0.0
|
||||||
)
|
)
|
||||||
if unrelated is None:
|
|
||||||
# No results for the nonsense query is the ideal outcome; skip (rather
|
|
||||||
# than a silent pass) so the test report shows the path was taken.
|
|
||||||
pytest.skip("No results for nonsense physics query — the ideal outcome")
|
|
||||||
|
|
||||||
relevant = await _top_score(
|
relevant = await _top_score(
|
||||||
nc_mcp_client, "how do I enable two-factor authentication"
|
nc_mcp_client, "how do I enable two-factor authentication"
|
||||||
|
|||||||
@@ -274,8 +274,12 @@ async def test_semantic_search_answer_with_limit(nc_mcp_client, temporary_note_f
|
|||||||
category="Development",
|
category="Development",
|
||||||
)
|
)
|
||||||
|
|
||||||
# Wait for vector indexing to complete
|
# Wait until the batch is indexed — gate on the last note being searchable
|
||||||
await wait_for_vector_sync(nc_mcp_client)
|
# rather than a bare idle signal, which can fire before the new notes are
|
||||||
|
# even enqueued.
|
||||||
|
await wait_for_vector_sync(
|
||||||
|
nc_mcp_client, search_term="async context managers", note_id=_note3["id"]
|
||||||
|
)
|
||||||
|
|
||||||
call_result = await nc_mcp_client.call_tool(
|
call_result = await nc_mcp_client.call_tool(
|
||||||
"nc_semantic_search_answer",
|
"nc_semantic_search_answer",
|
||||||
@@ -315,8 +319,10 @@ async def test_semantic_search_answer_score_threshold(
|
|||||||
category="Test",
|
category="Test",
|
||||||
)
|
)
|
||||||
|
|
||||||
# Wait for vector indexing to complete
|
# Gate on the new note being searchable (not a bare idle signal).
|
||||||
await wait_for_vector_sync(nc_mcp_client)
|
await wait_for_vector_sync(
|
||||||
|
nc_mcp_client, search_term="widget manufacturing", note_id=_note["id"]
|
||||||
|
)
|
||||||
|
|
||||||
# Query with exact match
|
# Query with exact match
|
||||||
call_result = await nc_mcp_client.call_tool(
|
call_result = await nc_mcp_client.call_tool(
|
||||||
@@ -362,8 +368,10 @@ async def test_semantic_search_answer_max_tokens(nc_mcp_client, temporary_note_f
|
|||||||
category="Test",
|
category="Test",
|
||||||
)
|
)
|
||||||
|
|
||||||
# Wait for vector indexing to complete
|
# Gate on the new note being searchable (not a bare idle signal).
|
||||||
await wait_for_vector_sync(nc_mcp_client)
|
await wait_for_vector_sync(
|
||||||
|
nc_mcp_client, search_term="Long Document content", note_id=_note["id"]
|
||||||
|
)
|
||||||
|
|
||||||
call_result = await nc_mcp_client.call_tool(
|
call_result = await nc_mcp_client.call_tool(
|
||||||
"nc_semantic_search_answer",
|
"nc_semantic_search_answer",
|
||||||
|
|||||||
Reference in New Issue
Block a user