Sweep all 1676 G004 violations across 112 files, converting
`logger.<level>(f"…{x}…")` to `logger.<level>("…%s…", x)`.
Why: ruff rule G004 was added to pyproject.toml to enforce lazy
%-style logging — defers formatting until the log level is enabled
and lets structured log tooling match the unformatted template.
Conversion preserves rendered output byte-for-byte:
- `{x}` → `%s` + `x`
- `{x!r}` / `{x!s}` / `{x!a}` → `%r` / `%s` / `%a`
- Format specs (`{x:.2f}`, `{x:>10}`) → `%s` + `format(x, 'spec')`
(printf-style specs aren't 1:1 with Python format specs, so we
delegate to `format()` to keep identical output)
- Literal `%` → `%%`
- Concatenated f-strings (`f"a {x} " "b"`) flattened
- Trailing kwargs (`exc_info=True`) preserved
Verified:
- `uv run ruff check --select G004` → 0 violations
- `uv run ty check -- nextcloud_mcp_server` → passes
- `uv run pytest tests/unit/` → 1010 passed
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
50 lines
1.6 KiB
Python
50 lines
1.6 KiB
Python
"""HTML to Markdown conversion utilities for vector sync."""
|
|
|
|
import logging
|
|
import re
|
|
|
|
from markdownify import markdownify as md
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
def html_to_markdown(html_content: str | None) -> str:
|
|
"""Convert HTML content to Markdown, preserving semantic structure.
|
|
|
|
This function converts HTML (typically from RSS/Atom feed items) to Markdown
|
|
for better text embedding. Markdown preserves:
|
|
- Heading hierarchy (important for document structure)
|
|
- Lists (bullet and numbered)
|
|
- Links (as [text](url))
|
|
- Bold/italic emphasis
|
|
- Paragraphs and line breaks
|
|
|
|
Args:
|
|
html_content: HTML string to convert (may be None or empty)
|
|
|
|
Returns:
|
|
Markdown string, or empty string if input is None/empty
|
|
|
|
Example:
|
|
>>> html_to_markdown("<h1>Title</h1><p>Content with <b>bold</b>.</p>")
|
|
'# Title\\n\\nContent with **bold**.\\n\\n'
|
|
"""
|
|
if not html_content:
|
|
return ""
|
|
|
|
try:
|
|
markdown = md(
|
|
html_content,
|
|
heading_style="ATX", # Use # style headings
|
|
strip=["script", "style", "iframe", "noscript"], # Remove unsafe elements
|
|
bullets="-", # Use - for unordered lists
|
|
code_language="", # Don't add language hints to code blocks
|
|
)
|
|
return markdown.strip()
|
|
except Exception as e:
|
|
logger.warning("Failed to convert HTML to Markdown: %s", e)
|
|
# Fallback: strip all HTML tags as a last resort
|
|
|
|
text = re.sub(r"<[^>]+>", " ", html_content)
|
|
return " ".join(text.split()) # Normalize whitespace
|