refactor(mail): address PR #935 round-1 review

- models/mail.py: lowercase `list` generics per CLAUDE.md convention.
- client/mail.py: _ocs_get now inspects ocs.meta.statuscode (re-raises >=400
  as HTTPStatusError carrying the OCS code so callers' 404/403 handling
  applies) and guards response.json() against non-JSON bodies (RequestError).
- Extract the duplicated _format_addresses + content reconstruction into
  vector/mail_content.py, used by both processor.py and context.py (fixes the
  SonarCloud new_duplicated_lines_density gate).
- processor.py: add the missing mail_message Qdrant payload block so the
  computed mail metadata (subject/from/to/cc/date_int/has_attachments/
  account_id/mailbox_id) is actually stored, not dropped.
- Rename the list_messages `filter` param to `search_filter` (avoid shadowing
  builtins.filter); still maps to the OCS `filter` query param.
- Docstring notes: has_more heuristic, attachment content size.
- Tests: OCS meta-failure + non-JSON client paths; initial-sync scanner tests
  (tests/unit/vector/test_scanner_mail.py).

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
Chris Coutinho
2026-06-20 12:31:32 +02:00
co-authored by Claude Opus 4.8
parent 3074622455
commit 62ee3e9f32
8 changed files with 303 additions and 94 deletions
+4 -33
View File
@@ -17,6 +17,7 @@ from nextcloud_mcp_server.models.deck import DeckCard
from nextcloud_mcp_server.search.access_filter import build_ownership_filter
from nextcloud_mcp_server.utils.validation import is_valid_nextcloud_doc_id
from nextcloud_mcp_server.vector.html_processor import html_to_markdown
from nextcloud_mcp_server.vector.mail_content import build_mail_content
from nextcloud_mcp_server.vector.placeholder import get_placeholder_filter
from nextcloud_mcp_server.vector.qdrant_client import get_qdrant_client
@@ -830,40 +831,10 @@ async def _fetch_document_text(
doc_id,
)
return None
# Reconstruct full content as indexed by the processor (subject +
# From + To + blank line + body) so chunk offsets align. Keep this in
# sync with the mail_message branch in vector/processor.py.
# Reconstruct full content via the shared helper so chunk offsets
# match what the processor indexed (single source of truth).
message = await nc_client.mail.get_message(int(doc_id))
def _format_addresses(addrs: list[dict] | None) -> str:
parts = []
for addr in addrs or []:
label = addr.get("label")
email = addr.get("email")
if label and email and label != email:
parts.append(f"{label} <{email}>")
elif email:
parts.append(email)
elif label:
parts.append(label)
return ", ".join(parts)
subject = message.get("subject") or ""
from_str = _format_addresses(message.get("from"))
to_str = _format_addresses(message.get("to"))
raw_body = message.get("body") or ""
body_text = (
html_to_markdown(raw_body) if message.get("hasHtmlBody") else raw_body
)
content_parts = [subject]
if from_str:
content_parts.append(f"From: {from_str}")
if to_str:
content_parts.append(f"To: {to_str}")
content_parts.append("") # Blank line
content_parts.append(body_text)
return "\n".join(content_parts)
return build_mail_content(message)
else:
logger.warning("Unsupported doc_type for context expansion: %s", doc_type)
return None