* test: reset model-manager shutdown state between backend tests Two leaks, one of them mine. 1. `model_manager._shutting_down` is a module-global Event and the GPU pool is a module-global executor. Any test that runs the app lifespan flips both on the way out (begin_shutdown + _reset_gpu_pool) and nothing puts them back — correct in production, where the process is ending; wrong across a combined session. A test arriving with the flag set finds a shut-down executor, so its first run_in_executor raises "cannot schedule new futures after shutdown", which the preload path classifies as benign and swallows. The symptom is a load that silently never starts. Reset before AND after: before so an inherited flag cannot decide the test, after so a test that legitimately shuts down does not hand it on. 2. tests/test_torch_compile_path_gate.py assigned services.settings_store into sys.modules directly instead of via monkeypatch.setitem. That leaks process-wide out of collection and breaks every later import of the real module. backend/tests/test_no_module_stubs.py exists to catch exactly that, and caught it — I introduced it two commits ago while isolating the Settings gate for a review finding. #1269 stays open for its last failure, which is a different root cause: test_lifespan_shutdown_mid_load fails because a reload fixture in tests/ replaces services.model_manager, so the test patches one module object while main's lifespan uses another (verified: `same=False`). That is the duplicate- module class, not a state leak, and needs its own fix. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> * test: let the shutdown-state reset fail loudly CodeRabbit Major: the broad try/except meant a reset that raised left the next test with stale shutdown or executor state — precisely the order-dependent failure the fixture exists to remove, while looking like it had worked. That is the same silent-fail-open shape as the watermark and ffmpeg bugs fixed earlier in this cycle. If reset_shutdown_flag() or _reset_gpu_pool() can raise, that is a real problem in model_manager and it should be loud. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> * test: assert the shutdown-state reset fixture actually resets (CodeRabbit) Ordered pair: one test leaves the module globals exactly as the lifespan leaves them, the next asserts it arrived clean — delete the fixture and the second fails. Plus a mechanical guard that the reset stays un-swallowed, so a future try/except cannot make the fixture look like it worked. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> --------- Co-authored-by: Claude Opus 5 <noreply@anthropic.com>
122 lines
5.0 KiB
Python
122 lines
5.0 KiB
Python
"""#1266: a torch lib path with whitespace can never survive inductor's linker.
|
|
|
|
Inductor passes the torch library directory to clang++/g++ as an unquoted
|
|
``-L`` flag, so a path containing a space splits into two arguments and the
|
|
compile dies with ``no such file or directory: 'Support/...'``. The bug is
|
|
inside PyTorch; what we control is not paying for a compile attempt that cannot
|
|
succeed and not flooding the log tail with its failure — which is how it turned
|
|
up in #1259, consuming a chunk of the captured crash output while not being the
|
|
actual fault.
|
|
|
|
Not hypothetical anywhere: macOS keeps app data under
|
|
``~/Library/Application Support/`` and a Windows profile is routinely
|
|
``C:/Users/First Last``.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import types
|
|
|
|
import pytest
|
|
|
|
|
|
def _env(monkeypatch, *, torch_file, triton=True):
|
|
"""Bind should_torch_compile's dependencies to a controlled fake."""
|
|
from services import engine_env
|
|
|
|
monkeypatch.setattr(
|
|
engine_env.importlib.util,
|
|
"find_spec",
|
|
lambda name: object() if (triton and name == "triton") else None,
|
|
)
|
|
monkeypatch.setattr(engine_env, "_compile_runtime_failure", None, raising=False)
|
|
# Isolate the Settings gate: should_torch_compile() otherwise reads the real
|
|
# settings_store, so a persisted perf.torch_compile_disabled=1 (or a missing
|
|
# settings table) would decide these tests instead of the path logic.
|
|
monkeypatch.setattr(
|
|
engine_env, "settings_store", None, raising=False
|
|
)
|
|
# monkeypatch.setitem, never a raw assignment: a bare object dropped into
|
|
# sys.modules leaks process-wide out of collection and breaks every later
|
|
# import of the real module in a mixed run. backend/tests/
|
|
# test_no_module_stubs.py exists to catch exactly that, and caught this.
|
|
import sys as _sys
|
|
import types as _types
|
|
|
|
monkeypatch.setitem(
|
|
_sys.modules,
|
|
"services.settings_store",
|
|
_types.SimpleNamespace(get_text=lambda *a, **k: "0"),
|
|
)
|
|
monkeypatch.setattr(
|
|
engine_env, "_cuda_arch_supported_for_compile", lambda: (True, "")
|
|
)
|
|
monkeypatch.setitem(
|
|
__import__("sys").modules, "torch", types.SimpleNamespace(__file__=torch_file)
|
|
)
|
|
return engine_env
|
|
|
|
|
|
CLEAN = "/opt/venv/lib/python3.11/site-packages/torch/__init__.py"
|
|
SPACED = "/Users/x/Library/Application Support/omnivoice/.venv/lib/torch/__init__.py"
|
|
|
|
|
|
def test_compile_allowed_on_a_clean_path(monkeypatch):
|
|
ee = _env(monkeypatch, torch_file=CLEAN)
|
|
assert ee.should_torch_compile("cuda") is True
|
|
|
|
|
|
def test_compile_skipped_when_the_torch_path_has_a_space(monkeypatch, caplog):
|
|
"""The regression: this used to attempt, fail in clang++, and log the wreck."""
|
|
ee = _env(monkeypatch, torch_file=SPACED)
|
|
with caplog.at_level("INFO", logger="omnivoice.engine_env"):
|
|
assert ee.should_torch_compile("cuda") is False
|
|
assert any("whitespace" in r.getMessage() for r in caplog.records), (
|
|
"the skip must say WHY, or it is indistinguishable from the other skips"
|
|
)
|
|
|
|
|
|
def test_force_env_still_overrides(monkeypatch, caplog):
|
|
"""Consistent with the arch gate: the user can insist — and is told."""
|
|
ee = _env(monkeypatch, torch_file=SPACED)
|
|
monkeypatch.setenv(ee._FORCE_COMPILE_ENV, "1")
|
|
with caplog.at_level("WARNING", logger="omnivoice.engine_env"):
|
|
assert ee.should_torch_compile("cuda") is True
|
|
assert any(
|
|
"forced" in r.getMessage().lower() for r in caplog.records
|
|
), "forcing past a known-broken path must be recorded, not silent"
|
|
|
|
|
|
def test_skip_message_names_the_escape_hatch(monkeypatch, caplog):
|
|
"""A skip the user cannot override is a dead end; the hint must be there."""
|
|
ee = _env(monkeypatch, torch_file=SPACED)
|
|
with caplog.at_level("INFO", logger="omnivoice.engine_env"):
|
|
assert ee.should_torch_compile("cuda") is False
|
|
assert any(ee._FORCE_COMPILE_ENV in r.getMessage() for r in caplog.records)
|
|
|
|
|
|
def test_home_directory_is_not_logged_verbatim(monkeypatch, caplog):
|
|
"""The reason string is logged and lands in pasted bug reports, so the path
|
|
goes through the same redaction as every other user-facing failure text."""
|
|
import pathlib as _pl
|
|
|
|
home = str(_pl.Path.home())
|
|
ee = _env(monkeypatch, torch_file=f"{home}/Library/Application Support/x/torch/__init__.py")
|
|
with caplog.at_level("INFO", logger="omnivoice.engine_env"):
|
|
ee.should_torch_compile("cuda")
|
|
joined = " ".join(r.getMessage() for r in caplog.records)
|
|
assert home not in joined, f"home directory leaked into the log: {joined}"
|
|
|
|
|
|
def test_unreadable_torch_path_is_not_a_reason_to_skip(monkeypatch):
|
|
"""No torch metadata means no evidence of a problem — fail open, matching
|
|
the module's other probes."""
|
|
ee = _env(monkeypatch, torch_file=None)
|
|
assert ee.should_torch_compile("cuda") is True
|
|
|
|
|
|
@pytest.mark.parametrize("device", ["mps", "cpu", "xpu"])
|
|
def test_non_cuda_devices_are_unaffected(monkeypatch, device):
|
|
"""The device gate runs first and must keep short-circuiting."""
|
|
ee = _env(monkeypatch, torch_file=SPACED)
|
|
assert ee.should_torch_compile(device) is False
|