Round-2 claude-review findings: - 🟡 Base-class recursion invariant: documented on embed_with_usage / embed_batch_with_usage that a provider overriding embed()/embed_batch() to delegate to the *_with_usage variant MUST also override that variant, or the two recurse. (No recursion today; the shipped providers pair the overrides.) - 🟡 Processor metering had no unit test: extracted the two-event recording into a module-level record_indexing_usage() helper and added tests/unit/test_processor_metering.py (value mapping, flag/zero-chunk no-ops, best-effort failure swallowed). - 🟡 SonarQube hotspots (python:S5332) were 3 http:// URLs in the new test fixtures (mock hosts, never contacted) blocking the quality gate (new_security_hotspots_reviewed). Switched them to https:// so no hotspot is raised. - 🟢 Zero-chunk guard: record_indexing_usage() no-ops when chunk_count == 0, so an empty document no longer writes zero-value billing rows. Deferred (stated on the PR): Mistral x.index-or-0 sort key (pre-existing, equivalent), CHANGELOG note for the Ollama /api/embed switch (CHANGELOG is commitizen-generated from commit bodies, which document it), class-var query_token_count (safe under the per-request instance pattern). Deck #67. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
70 lines
2.3 KiB
Python
70 lines
2.3 KiB
Python
"""Unit tests for Ollama provider token-usage surfacing.
|
|
|
|
The provider has no other unit coverage; these focus on the ``*_with_usage``
|
|
methods added for usage metering (Deck #67) — provider-reported
|
|
``prompt_eval_count`` and the char-based estimate fallback when it's absent.
|
|
"""
|
|
|
|
from unittest.mock import AsyncMock, MagicMock
|
|
|
|
import pytest
|
|
|
|
from nextcloud_mcp_server.providers.ollama import OllamaProvider
|
|
|
|
|
|
@pytest.fixture
|
|
def ollama_provider():
|
|
# Construct with no models so __init__ skips _check_model_is_loaded (no
|
|
# network call), then enable embeddings post-construction. https mock host
|
|
# (never contacted — client.post is patched in each test).
|
|
provider = OllamaProvider(base_url="https://ollama:11434")
|
|
provider.embedding_model = "nomic-embed-text"
|
|
return provider
|
|
|
|
|
|
def _embed_response(embeddings, prompt_eval_count=None):
|
|
payload = {"embeddings": embeddings}
|
|
if prompt_eval_count is not None:
|
|
payload["prompt_eval_count"] = prompt_eval_count
|
|
resp = MagicMock()
|
|
resp.json = MagicMock(return_value=payload)
|
|
resp.raise_for_status = MagicMock()
|
|
return resp
|
|
|
|
|
|
@pytest.mark.unit
|
|
async def test_ollama_embed_batch_with_usage_reports_prompt_eval_count(ollama_provider):
|
|
"""prompt_eval_count from /api/embed is surfaced as the token count."""
|
|
ollama_provider.client.post = AsyncMock(
|
|
return_value=_embed_response([[0.1, 0.2], [0.3, 0.4]], prompt_eval_count=7)
|
|
)
|
|
|
|
embeddings, tokens = await ollama_provider.embed_batch_with_usage(["a", "b"])
|
|
|
|
assert embeddings == [[0.1, 0.2], [0.3, 0.4]]
|
|
assert tokens == 7
|
|
|
|
|
|
@pytest.mark.unit
|
|
async def test_ollama_with_usage_estimates_when_count_absent(ollama_provider):
|
|
"""Older Ollama omits prompt_eval_count → char-based estimate."""
|
|
ollama_provider.client.post = AsyncMock(
|
|
return_value=_embed_response([[0.1]], prompt_eval_count=None)
|
|
)
|
|
|
|
_, tokens = await ollama_provider.embed_with_usage("abcdefgh") # 8 chars → 2
|
|
|
|
assert tokens == 2
|
|
|
|
|
|
@pytest.mark.unit
|
|
async def test_ollama_empty_batch_with_usage(ollama_provider):
|
|
"""Empty batch returns no embeddings, zero tokens, and makes no request."""
|
|
ollama_provider.client.post = AsyncMock()
|
|
|
|
embeddings, tokens = await ollama_provider.embed_batch_with_usage([])
|
|
|
|
assert embeddings == []
|
|
assert tokens == 0
|
|
ollama_provider.client.post.assert_not_called()
|