fix(documents): guard Unix-only resource import for Windows (#877)

`document_processors/_isolation.py` did an unconditional module-level
`import resource`, a POSIX-only stdlib module absent on Windows. It was
pulled into the API startup path via
`server/webdav.py -> utils/document_parser -> document_processors`, so
the MCP server failed to start on Windows since 0.101.2 with
`ModuleNotFoundError: No module named 'resource'`.

- Guard the import behind `sys.platform`; bind `resource = None` on
  win32. `_apply_mem_limit()` degrades to a logged no-op when the module
  is unavailable (the RLIMIT_AS cap is a Linux-pod safety measure, not a
  correctness requirement).
- Make the document-parser import in `server/webdav.py` lazy so server
  startup never loads the ingest document stack
  (document_processors -> pymupdf -> _isolation) at all -- it is only
  needed when a file is actually read and parsed. This both fixes #877
  and decouples the API layer from ingest-only deps.
- Add unit regressions for the no-op path and the win32 import guard.
- Add a cross-platform `package-smoke` CI job (ubuntu + windows) that
  installs the package isolated and runs the CLI, exercising the
  cli -> server -> webdav import chain that crashed in #877.

Fixes #877

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
Chris Coutinho
2026-06-08 14:59:03 +02:00
co-authored by Claude Opus 4.8
parent 96c0491f14
commit fc8a4e4dfa
4 changed files with 85 additions and 5 deletions
@@ -13,7 +13,7 @@ in its process pool.
"""
import logging
import resource
import sys
from pathlib import Path
from typing import Any
@@ -21,6 +21,15 @@ import anyio
import anyio.to_process
from anyio import BrokenWorkerProcess
# ``resource`` is a Unix-only stdlib module -- it does not exist on Windows, and
# importing it unconditionally crashed Windows startup (#877). The RLIMIT_AS cap
# it provides is a Linux-pod safety measure, not a correctness requirement, so on
# platforms without it we fall back to a no-op (``resource is None``).
if sys.platform == "win32": # pragma: no cover - exercised only on Windows
resource = None
else:
import resource
logger = logging.getLogger(__name__)
# Guard so the address-space limit is applied once per (reused) worker process.
@@ -44,10 +53,18 @@ def _apply_mem_limit(mem_limit_mb: int) -> None:
Applied once per worker process. The hard limit is left untouched (we only
lower the soft limit), and we never set a soft limit above the hard limit.
On a platform without the Unix-only ``resource`` module (e.g. Windows, see
#877) the cap is skipped -- the worker still runs, just without the
address-space limit.
"""
global _MEM_LIMIT_APPLIED
if _MEM_LIMIT_APPLIED or mem_limit_mb <= 0:
return
if resource is None:
logger.debug("resource module unavailable; skipping RLIMIT_AS cap")
_MEM_LIMIT_APPLIED = True
return
target = mem_limit_mb * 1024 * 1024
soft, hard = resource.getrlimit(resource.RLIMIT_AS)
soft_target = target if hard == resource.RLIM_INFINITY else min(target, hard)
+10 -4
View File
@@ -13,10 +13,6 @@ from nextcloud_mcp_server.server.tag_exclusion import (
get_excluded_file_paths,
is_path_excluded,
)
from nextcloud_mcp_server.utils.document_parser import (
is_parseable_document,
parse_document,
)
logger = logging.getLogger(__name__)
@@ -115,6 +111,16 @@ def configure_webdav_tools(mcp: FastMCP):
content, content_type = await client.webdav.read_file(path)
# Imported lazily so server startup never loads the document-parsing
# stack (document_processors -> pymupdf -> _isolation). That stack is an
# ingest-layer concern and, before this, broke Windows startup via a
# Unix-only ``import resource`` (#877). It is only needed when a file is
# actually read and parsed.
from nextcloud_mcp_server.utils.document_parser import ( # noqa: PLC0415
is_parseable_document,
parse_document,
)
# Check if this is a parseable document (PDF, DOCX, etc.)
# is_parseable_document() checks if document processing is enabled
if is_parseable_document(content_type):