Files
VoiceStudio/tests/test_gpu_routing_verdict.py
T
Palash DebnathandClaude Opus 4.8 c6a55794da feat(routing): active-engine GPU verdict in preflight + diagnose (#21 PR 4/5) (#433)
Surfaces a routing verdict for the CURRENTLY-SELECTED TTS engine in the two
system-health surfaces, so a CPU fallback / unavailable-GPU is heard about
before a slow or failed synth — the no-silent-fallback contract, read-only.

- `tts_backend.active_routing()` + `gpu_routing_verdict()`: the active engine's
  routing derived from list_backends() (byte-identical to the matrix) plus the
  host compute summary (family + VRAM from the canonical probe). Never raise.
- `/system/diagnose` gains a `gpu_routing` check: accelerated→ok,
  accelerated-with-caveat / cpu_fallback→warn (+ actionable hint), cpu_only→ok
  (no-GPU host is the expected normal state — never noise-warns), unavailable→
  fail, no-engine→warn. ASCII-safe detail strings (the text dump enforces ASCII).
- `/setup/preflight` gains an "Active engine routing" check + an explicit
  `gpu_routing` object on PreflightResponse (a real field — the response has no
  extra="allow", so it would otherwise be dropped). `device` gains `gpu_family`
  (ROCm-vs-CUDA aware) + `vram_gb`. New `GpuRouting` schema.

Tests: gpu_routing_verdict (host + active-engine + degraded), diagnose status
mapping across all 6 states + never-raises, preflight gpu_routing object +
check + device.gpu_family. Existing diagnose/preflight tests stay green (checks
are additive; the report's top-level key set is unchanged).

Deferred (documented): synth-time routing headers/WS-frames at the 3 synth
entry points. Selection is already hard-gated (PR 3 select_engine), and the
matrix (PR 5) + this preflight/diagnose verdict surface the situation — the
synth-time signal is incremental belt-and-suspenders for the env-var-pinned
edge and is best validated interactively. Tracked as a #21 follow-up.

Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-14 02:07:54 +05:30

87 lines
3.8 KiB
Python

"""Active-engine GPU routing verdict (#21 PR 4) — the no-silent-fallback
surface for preflight + diagnose.
Covers `tts_backend.gpu_routing_verdict()` (built from active_routing + the host
probe) and `diagnose._check_gpu_routing()`'s status mapping. Both are exercised
with mocks so no torch / full app import is needed.
"""
from __future__ import annotations
import pytest
from core.device_caps import HostCaps
from core.diagnose import OK, WARN, FAIL
# ── gpu_routing_verdict ─────────────────────────────────────────────────────
def test_verdict_uses_active_routing_and_host(monkeypatch):
from services import tts_backend as tb
monkeypatch.setattr(tb, "active_routing", lambda: {
"engine": "omnivoice", "available": True,
"effective_device": "cpu", "routing_status": "cpu_fallback",
"routing_reason": "engine has no CUDA path; running on CPU",
})
monkeypatch.setattr(
"core.device_caps.detect_host_caps",
lambda: HostCaps(family="cuda", available_families=("cuda", "cpu"), vram_gb=24.0),
)
v = tb.gpu_routing_verdict()
assert v["engine"] == "omnivoice"
assert v["routing_status"] == "cpu_fallback"
assert v["host_family"] == "cuda"
assert v["vram_gb"] == 24.0
def test_verdict_degrades_when_no_active_engine(monkeypatch):
from services import tts_backend as tb
monkeypatch.setattr(tb, "active_routing", lambda: None)
monkeypatch.setattr(
"core.device_caps.detect_host_caps",
lambda: HostCaps(family="cpu", available_families=("cpu",)),
)
v = tb.gpu_routing_verdict()
assert v["engine"] is None
assert v["routing_status"] == "none"
assert v["host_family"] == "cpu"
# ── diagnose._check_gpu_routing status mapping ──────────────────────────────
@pytest.mark.parametrize("verdict,expected_status", [
({"engine": "e", "effective_device": "cuda", "routing_status": "accelerated",
"routing_reason": None, "host_family": "cuda", "vram_gb": 24.0}, OK),
({"engine": "e", "effective_device": "cuda", "routing_status": "accelerated",
"routing_reason": "CUDA selected, but: may fail at kernel launch",
"host_family": "cuda", "vram_gb": 24.0}, WARN),
({"engine": "e", "effective_device": "cpu", "routing_status": "cpu_fallback",
"routing_reason": "engine has no CUDA path; running on CPU",
"host_family": "cuda", "vram_gb": 24.0}, WARN),
({"engine": "e", "effective_device": "cpu", "routing_status": "cpu_only",
"routing_reason": None, "host_family": "cpu", "vram_gb": 0.0}, OK),
({"engine": "e", "effective_device": "cuda", "routing_status": "unavailable",
"routing_reason": "requires cuda; this host has cpu",
"host_family": "cpu", "vram_gb": 0.0}, FAIL),
({"engine": None, "effective_device": None, "routing_status": "none",
"routing_reason": None, "host_family": "cpu", "vram_gb": 0.0}, WARN),
])
def test_diagnose_routing_status_mapping(monkeypatch, verdict, expected_status):
import core.diagnose as diag
monkeypatch.setattr("services.tts_backend.gpu_routing_verdict", lambda: verdict)
check = diag._check_gpu_routing()
assert check["id"] == "gpu_routing"
assert check["status"] == expected_status
if expected_status in (WARN, FAIL):
assert check["hint"], "actionable hint required on warn/fail"
def test_diagnose_routing_never_raises(monkeypatch):
import core.diagnose as diag
def _boom():
raise RuntimeError("probe exploded")
monkeypatch.setattr("services.tts_backend.gpu_routing_verdict", _boom)
check = diag._check_gpu_routing()
assert check["status"] == WARN # degrades, doesn't propagate