From cd58bbded030f84c60c3dd48fc8f603f672a52b9 Mon Sep 17 00:00:00 2001 From: Palash Debnath <4178343+debpalash@users.noreply.github.com> Date: Mon, 7 Sep 2026 11:13:33 +0530 Subject: [PATCH] fix(moss): align accelerator detection and routing status --- CHANGELOG.md | 2 + backend/api/routers/settings.py | 2 +- backend/core/device_caps.py | 26 +++++++++-- backend/engines/moss_tts_v15/__init__.py | 16 +++---- backend/services/tts_backend.py | 2 +- docs/engines/moss-tts-v15.md | 9 ++-- frontend/src/i18n/locales/ar.json | 1 + frontend/src/i18n/locales/de.json | 1 + frontend/src/i18n/locales/en.json | 1 + frontend/src/i18n/locales/es.json | 1 + frontend/src/i18n/locales/fr.json | 1 + frontend/src/i18n/locales/hi.json | 1 + frontend/src/i18n/locales/id.json | 1 + frontend/src/i18n/locales/it.json | 1 + frontend/src/i18n/locales/ja.json | 1 + frontend/src/i18n/locales/ko.json | 1 + frontend/src/i18n/locales/nl.json | 1 + frontend/src/i18n/locales/pl.json | 1 + frontend/src/i18n/locales/pt.json | 1 + frontend/src/i18n/locales/ru.json | 1 + frontend/src/i18n/locales/sv.json | 1 + frontend/src/i18n/locales/th.json | 1 + frontend/src/i18n/locales/tr.json | 1 + frontend/src/i18n/locales/uk.json | 1 + frontend/src/i18n/locales/vi.json | 1 + frontend/src/i18n/locales/zh-CN.json | 1 + frontend/src/i18n/locales/zh-TW.json | 1 + tests/backend/api/test_engines_route_shape.py | 2 +- tests/backend/test_compute_device_settings.py | 11 +++++ tests/test_asr_gpu_compat.py | 2 +- tests/test_device_caps.py | 21 +++++++++ tests/test_engine_routing.py | 2 +- tests/test_moss_tts_v15.py | 44 +++++++++++++++++-- tests/test_setup_preflight.py | 2 +- 34 files changed, 135 insertions(+), 27 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 6e096417..31ac46e7 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -20,6 +20,8 @@ the frozen-backend fallback mirror it for their toolchains. ### Fixed +- MOSS accelerator routing now matches runtime device selection, including native XPU and registered NPU detection (#1830) — thanks @li-lizhe! + ## [0.5.2] — 2026-09-02 diff --git a/backend/api/routers/settings.py b/backend/api/routers/settings.py index e3e7152f..b952e6cb 100644 --- a/backend/api/routers/settings.py +++ b/backend/api/routers/settings.py @@ -150,7 +150,7 @@ def _compute_device_state() -> dict: caps = device_caps.detect_host_caps() env_pin = (os.environ.get("OMNIVOICE_DEVICE") or "").strip().lower() auto_family = next( - (f for f in ("cuda", "rocm", "xpu", "mps") if f in caps.available_families), + (f for f in device_caps.ACCELERATOR_PRIORITY if f in caps.available_families), "cpu", ) value = device_caps.requested_device_override() diff --git a/backend/core/device_caps.py b/backend/core/device_caps.py index 2c535474..6749d4ed 100644 --- a/backend/core/device_caps.py +++ b/backend/core/device_caps.py @@ -35,7 +35,8 @@ import sys from dataclasses import dataclass from typing import Literal -DeviceFamily = Literal["cuda", "rocm", "mps", "xpu", "cpu"] +DeviceFamily = Literal["cuda", "rocm", "mps", "xpu", "npu", "cpu"] +ACCELERATOR_PRIORITY = ("cuda", "rocm", "xpu", "npu", "mps") # Stable substring stamped onto notes that represent a real kernel-launch risk # (arch/driver mismatch) — as opposed to advisory notes (multi-GPU, VRAM query @@ -533,9 +534,12 @@ def _probe() -> HostCaps: # is the whole truth in that case (CodeRabbit, #1425). notes.extend(why_no_gpu(torch)) - # ── Intel XPU via IPEX ─────────────────────────────────────────────── + # Older builds register XPU through IPEX; modern torch exposes it directly. try: import intel_extension_for_pytorch # noqa: F401 + except Exception: + pass + try: if hasattr(torch, "xpu") and torch.xpu.is_available(): detected.append("xpu") if not device_name: @@ -546,7 +550,21 @@ def _probe() -> HostCaps: pass notes.append("XPU VRAM not queried (unreliable across IPEX versions)") except Exception: - # IPEX absent or XPU probe failed — no XPU on this host. + # XPU probe failed — no usable XPU on this host. + pass + + # Vendor extensions may register an NPU with torch. Probe only an already + # registered backend; never install or import an optional vendor package. + try: + if hasattr(torch, "npu") and torch.npu.is_available(): + detected.append("npu") + if not device_name: + try: + device_name = torch.npu.get_device_name(0) + except Exception: + pass + notes.append("NPU VRAM not queried") + except Exception: pass # ── Apple Silicon MPS ──────────────────────────────────────────────── @@ -579,7 +597,7 @@ def _probe() -> HostCaps: # Preferred family by priority; cpu when nothing accelerated was detected. family: DeviceFamily = "cpu" - for pref in ("cuda", "rocm", "xpu", "mps"): + for pref in ACCELERATOR_PRIORITY: if pref in detected: family = pref # type: ignore[assignment] break diff --git a/backend/engines/moss_tts_v15/__init__.py b/backend/engines/moss_tts_v15/__init__.py index f586d699..0577d315 100644 --- a/backend/engines/moss_tts_v15/__init__.py +++ b/backend/engines/moss_tts_v15/__init__.py @@ -29,15 +29,11 @@ Do NOT import ``main.py`` from the parent process — it runs under a different venv (``transformers==5.0.0``) and importing it in-process would re-introduce the exact conflict this isolation exists to avoid. -Hardware honesty (cross-platform rule): MOSS-TTS-v1.5's upstream documents -only CUDA and CPU. There is **no documented or tested MPS path** — the -custom ``trust_remote_code`` modelling code and the separate audio -tokenizer are unverified on Apple Silicon. We therefore advertise -``gpu_compat = ("cuda", "cpu")`` and the sidecar selects ``cuda`` when -present else ``cpu`` — it never silently routes to MPS where it might -crash. On Apple Silicon the engine honestly resolves to CPU (slow but -correct), and the engine is opt-in regardless, so it never becomes a -broken default on any platform. +Hardware routing follows the sidecar's runtime-available PyTorch accelerator: +CUDA/ROCm, XPU, or a registered NPU. MPS remains excluded; CPU is the fallback. +XPU/NPU routing is covered with mocked device contracts, not physical-hardware +synthesis certification; users need a compatible torch/vendor runtime in the +isolated engine venv. """ from __future__ import annotations @@ -91,7 +87,7 @@ class MossTTSV15Backend(SubprocessBackend): _DEFAULT_SAMPLE_RATE = 24000 # Honest hardware surface: upstream documents CUDA + CPU only. MPS is # undocumented / untested, so we do NOT claim it (cross-platform rule). - gpu_compat = ("cuda", "cpu") + gpu_compat = ("cuda", "rocm", "xpu", "npu", "cpu") # ── availability ─────────────────────────────────────────────────────── diff --git a/backend/services/tts_backend.py b/backend/services/tts_backend.py index f84bb406..6a5b712f 100644 --- a/backend/services/tts_backend.py +++ b/backend/services/tts_backend.py @@ -2371,7 +2371,7 @@ def list_backends() -> list[dict]: "one_click_install": bool, # services.sidecar_install can provision it in-app "last_error": Optional[str], # cached most-recent failure "isolation_mode": "in-process" | "subprocess", - "gpu_compat": list[str], # subset of {cuda, rocm, mps, xpu, cpu} + "gpu_compat": list[str], # subset of {cuda, rocm, mps, xpu, npu, cpu} "supports_cloning": Optional[bool], # True/False from the class attr; None when # model-dependent (property, e.g. mlx-audio) "effective_device": str, # device this engine uses on THIS host diff --git a/docs/engines/moss-tts-v15.md b/docs/engines/moss-tts-v15.md index 95f2cc76..c0cf5cc0 100644 --- a/docs/engines/moss-tts-v15.md +++ b/docs/engines/moss-tts-v15.md @@ -23,10 +23,11 @@ interpreter, so MOSS runs behind 8 GB GPUs when quantized; the bf16 Transformers path used here is ~16 GB of weights, so a 16 GB+ GPU is the realistic CUDA target. It also runs on **CPU** (fp32) — correct but slow. -- **Device:** CUDA when present, else CPU. **There is no MPS path** — - upstream documents only CUDA/CPU and the custom modelling code is - untested on Apple Silicon, so VoiceStudio never routes MOSS to MPS. On a - Mac it runs on CPU. +- **Device:** the sidecar uses a runtime-available PyTorch CUDA/ROCm, XPU, + or registered NPU backend, otherwise CPU. The isolated engine venv needs the + matching torch/vendor integration. MPS still uses CPU. XPU/NPU routing is + covered by mocked loader tests; physical-device synthesis has not been + validated by this change. ## Install diff --git a/frontend/src/i18n/locales/ar.json b/frontend/src/i18n/locales/ar.json index dfdd3bb7..43df5833 100644 --- a/frontend/src/i18n/locales/ar.json +++ b/frontend/src/i18n/locales/ar.json @@ -334,6 +334,7 @@ "device_family_cuda": "NVIDIA GPU (CUDA)", "device_family_rocm": "AMD GPU (ROCm)", "device_family_xpu": "Intel GPU (XPU)", + "device_family_npu": "NPU", "device_family_mps": "Apple GPU (MPS)", "device_family_cpu": "CPU", "device_load_failed": "تعذّر تحميل إعداد الجهاز", diff --git a/frontend/src/i18n/locales/de.json b/frontend/src/i18n/locales/de.json index a05aa549..5ab70363 100644 --- a/frontend/src/i18n/locales/de.json +++ b/frontend/src/i18n/locales/de.json @@ -334,6 +334,7 @@ "device_family_cuda": "NVIDIA GPU (CUDA)", "device_family_rocm": "AMD GPU (ROCm)", "device_family_xpu": "Intel GPU (XPU)", + "device_family_npu": "NPU", "device_family_mps": "Apple GPU (MPS)", "device_family_cpu": "CPU", "device_load_failed": "Geräteeinstellung konnte nicht geladen werden", diff --git a/frontend/src/i18n/locales/en.json b/frontend/src/i18n/locales/en.json index cba85f5b..8e1d908a 100644 --- a/frontend/src/i18n/locales/en.json +++ b/frontend/src/i18n/locales/en.json @@ -934,6 +934,7 @@ "device_family_cuda": "NVIDIA GPU (CUDA)", "device_family_rocm": "AMD GPU (ROCm)", "device_family_xpu": "Intel GPU (XPU)", + "device_family_npu": "NPU", "device_family_mps": "Apple GPU (MPS)", "device_family_cpu": "CPU", "device_load_failed": "Failed to load device setting", diff --git a/frontend/src/i18n/locales/es.json b/frontend/src/i18n/locales/es.json index 39286cbd..765a2432 100644 --- a/frontend/src/i18n/locales/es.json +++ b/frontend/src/i18n/locales/es.json @@ -334,6 +334,7 @@ "device_family_cuda": "NVIDIA GPU (CUDA)", "device_family_rocm": "AMD GPU (ROCm)", "device_family_xpu": "Intel GPU (XPU)", + "device_family_npu": "NPU", "device_family_mps": "Apple GPU (MPS)", "device_family_cpu": "CPU", "device_load_failed": "No se pudo cargar el ajuste del dispositivo", diff --git a/frontend/src/i18n/locales/fr.json b/frontend/src/i18n/locales/fr.json index 1e309a34..874f3631 100644 --- a/frontend/src/i18n/locales/fr.json +++ b/frontend/src/i18n/locales/fr.json @@ -334,6 +334,7 @@ "device_family_cuda": "NVIDIA GPU (CUDA)", "device_family_rocm": "AMD GPU (ROCm)", "device_family_xpu": "Intel GPU (XPU)", + "device_family_npu": "NPU", "device_family_mps": "Apple GPU (MPS)", "device_family_cpu": "CPU", "device_load_failed": "Impossible de charger le réglage du périphérique", diff --git a/frontend/src/i18n/locales/hi.json b/frontend/src/i18n/locales/hi.json index 9c72fe06..5857dbdb 100644 --- a/frontend/src/i18n/locales/hi.json +++ b/frontend/src/i18n/locales/hi.json @@ -334,6 +334,7 @@ "device_family_cuda": "NVIDIA GPU (CUDA)", "device_family_rocm": "AMD GPU (ROCm)", "device_family_xpu": "Intel GPU (XPU)", + "device_family_npu": "NPU", "device_family_mps": "Apple GPU (MPS)", "device_family_cpu": "CPU", "device_load_failed": "डिवाइस सेटिंग लोड नहीं हो सकी", diff --git a/frontend/src/i18n/locales/id.json b/frontend/src/i18n/locales/id.json index 6f7014d0..5c6c3f63 100644 --- a/frontend/src/i18n/locales/id.json +++ b/frontend/src/i18n/locales/id.json @@ -334,6 +334,7 @@ "device_family_cuda": "NVIDIA GPU (CUDA)", "device_family_rocm": "AMD GPU (ROCm)", "device_family_xpu": "Intel GPU (XPU)", + "device_family_npu": "NPU", "device_family_mps": "Apple GPU (MPS)", "device_family_cpu": "CPU", "device_load_failed": "Gagal memuat pengaturan perangkat", diff --git a/frontend/src/i18n/locales/it.json b/frontend/src/i18n/locales/it.json index 9a29dda4..0eb4c8df 100644 --- a/frontend/src/i18n/locales/it.json +++ b/frontend/src/i18n/locales/it.json @@ -334,6 +334,7 @@ "device_family_cuda": "NVIDIA GPU (CUDA)", "device_family_rocm": "AMD GPU (ROCm)", "device_family_xpu": "Intel GPU (XPU)", + "device_family_npu": "NPU", "device_family_mps": "Apple GPU (MPS)", "device_family_cpu": "CPU", "device_load_failed": "Impossibile caricare l'impostazione del dispositivo", diff --git a/frontend/src/i18n/locales/ja.json b/frontend/src/i18n/locales/ja.json index d0c7bda2..d32798f4 100644 --- a/frontend/src/i18n/locales/ja.json +++ b/frontend/src/i18n/locales/ja.json @@ -334,6 +334,7 @@ "device_family_cuda": "NVIDIA GPU (CUDA)", "device_family_rocm": "AMD GPU (ROCm)", "device_family_xpu": "Intel GPU (XPU)", + "device_family_npu": "NPU", "device_family_mps": "Apple GPU (MPS)", "device_family_cpu": "CPU", "device_load_failed": "デバイス設定を読み込めませんでした", diff --git a/frontend/src/i18n/locales/ko.json b/frontend/src/i18n/locales/ko.json index d8918462..0a34362f 100644 --- a/frontend/src/i18n/locales/ko.json +++ b/frontend/src/i18n/locales/ko.json @@ -334,6 +334,7 @@ "device_family_cuda": "NVIDIA GPU (CUDA)", "device_family_rocm": "AMD GPU (ROCm)", "device_family_xpu": "Intel GPU (XPU)", + "device_family_npu": "NPU", "device_family_mps": "Apple GPU (MPS)", "device_family_cpu": "CPU", "device_load_failed": "장치 설정을 불러오지 못했습니다", diff --git a/frontend/src/i18n/locales/nl.json b/frontend/src/i18n/locales/nl.json index d076f1ab..31edd767 100644 --- a/frontend/src/i18n/locales/nl.json +++ b/frontend/src/i18n/locales/nl.json @@ -334,6 +334,7 @@ "device_family_cuda": "NVIDIA GPU (CUDA)", "device_family_rocm": "AMD GPU (ROCm)", "device_family_xpu": "Intel GPU (XPU)", + "device_family_npu": "NPU", "device_family_mps": "Apple GPU (MPS)", "device_family_cpu": "CPU", "device_load_failed": "Apparaatinstelling kon niet worden geladen", diff --git a/frontend/src/i18n/locales/pl.json b/frontend/src/i18n/locales/pl.json index c0055e44..72158d36 100644 --- a/frontend/src/i18n/locales/pl.json +++ b/frontend/src/i18n/locales/pl.json @@ -334,6 +334,7 @@ "device_family_cuda": "NVIDIA GPU (CUDA)", "device_family_rocm": "AMD GPU (ROCm)", "device_family_xpu": "Intel GPU (XPU)", + "device_family_npu": "NPU", "device_family_mps": "Apple GPU (MPS)", "device_family_cpu": "CPU", "device_load_failed": "Nie udało się wczytać ustawienia urządzenia", diff --git a/frontend/src/i18n/locales/pt.json b/frontend/src/i18n/locales/pt.json index 6b3f8996..e6547902 100644 --- a/frontend/src/i18n/locales/pt.json +++ b/frontend/src/i18n/locales/pt.json @@ -334,6 +334,7 @@ "device_family_cuda": "NVIDIA GPU (CUDA)", "device_family_rocm": "AMD GPU (ROCm)", "device_family_xpu": "Intel GPU (XPU)", + "device_family_npu": "NPU", "device_family_mps": "Apple GPU (MPS)", "device_family_cpu": "CPU", "device_load_failed": "Falha ao carregar a configuração do dispositivo", diff --git a/frontend/src/i18n/locales/ru.json b/frontend/src/i18n/locales/ru.json index a677e4e9..81628c23 100644 --- a/frontend/src/i18n/locales/ru.json +++ b/frontend/src/i18n/locales/ru.json @@ -334,6 +334,7 @@ "device_family_cuda": "NVIDIA GPU (CUDA)", "device_family_rocm": "AMD GPU (ROCm)", "device_family_xpu": "Intel GPU (XPU)", + "device_family_npu": "NPU", "device_family_mps": "Apple GPU (MPS)", "device_family_cpu": "CPU", "device_load_failed": "Не удалось загрузить настройку устройства", diff --git a/frontend/src/i18n/locales/sv.json b/frontend/src/i18n/locales/sv.json index 66efb85c..ea41596d 100644 --- a/frontend/src/i18n/locales/sv.json +++ b/frontend/src/i18n/locales/sv.json @@ -334,6 +334,7 @@ "device_family_cuda": "NVIDIA GPU (CUDA)", "device_family_rocm": "AMD GPU (ROCm)", "device_family_xpu": "Intel GPU (XPU)", + "device_family_npu": "NPU", "device_family_mps": "Apple GPU (MPS)", "device_family_cpu": "CPU", "device_load_failed": "Kunde inte läsa in enhetsinställningen", diff --git a/frontend/src/i18n/locales/th.json b/frontend/src/i18n/locales/th.json index 6fdad619..d30d77bb 100644 --- a/frontend/src/i18n/locales/th.json +++ b/frontend/src/i18n/locales/th.json @@ -334,6 +334,7 @@ "device_family_cuda": "NVIDIA GPU (CUDA)", "device_family_rocm": "AMD GPU (ROCm)", "device_family_xpu": "Intel GPU (XPU)", + "device_family_npu": "NPU", "device_family_mps": "Apple GPU (MPS)", "device_family_cpu": "CPU", "device_load_failed": "โหลดการตั้งค่าอุปกรณ์ไม่สำเร็จ", diff --git a/frontend/src/i18n/locales/tr.json b/frontend/src/i18n/locales/tr.json index d4769d4d..16f1bcac 100644 --- a/frontend/src/i18n/locales/tr.json +++ b/frontend/src/i18n/locales/tr.json @@ -334,6 +334,7 @@ "device_family_cuda": "NVIDIA GPU (CUDA)", "device_family_rocm": "AMD GPU (ROCm)", "device_family_xpu": "Intel GPU (XPU)", + "device_family_npu": "NPU", "device_family_mps": "Apple GPU (MPS)", "device_family_cpu": "CPU", "device_load_failed": "Aygıt ayarı yüklenemedi", diff --git a/frontend/src/i18n/locales/uk.json b/frontend/src/i18n/locales/uk.json index 0df2456e..b0a05e25 100644 --- a/frontend/src/i18n/locales/uk.json +++ b/frontend/src/i18n/locales/uk.json @@ -334,6 +334,7 @@ "device_family_cuda": "NVIDIA GPU (CUDA)", "device_family_rocm": "AMD GPU (ROCm)", "device_family_xpu": "Intel GPU (XPU)", + "device_family_npu": "NPU", "device_family_mps": "Apple GPU (MPS)", "device_family_cpu": "CPU", "device_load_failed": "Не вдалося завантажити налаштування пристрою", diff --git a/frontend/src/i18n/locales/vi.json b/frontend/src/i18n/locales/vi.json index 9fb54a41..10926cbc 100644 --- a/frontend/src/i18n/locales/vi.json +++ b/frontend/src/i18n/locales/vi.json @@ -334,6 +334,7 @@ "device_family_cuda": "NVIDIA GPU (CUDA)", "device_family_rocm": "AMD GPU (ROCm)", "device_family_xpu": "Intel GPU (XPU)", + "device_family_npu": "NPU", "device_family_mps": "Apple GPU (MPS)", "device_family_cpu": "CPU", "device_load_failed": "Không tải được cài đặt thiết bị", diff --git a/frontend/src/i18n/locales/zh-CN.json b/frontend/src/i18n/locales/zh-CN.json index 934ef94f..792257b0 100644 --- a/frontend/src/i18n/locales/zh-CN.json +++ b/frontend/src/i18n/locales/zh-CN.json @@ -586,6 +586,7 @@ "device_family_cuda": "NVIDIA GPU (CUDA)", "device_family_rocm": "AMD GPU (ROCm)", "device_family_xpu": "Intel GPU (XPU)", + "device_family_npu": "NPU", "device_family_mps": "Apple GPU (MPS)", "device_family_cpu": "CPU", "device_load_failed": "无法加载设备设置", diff --git a/frontend/src/i18n/locales/zh-TW.json b/frontend/src/i18n/locales/zh-TW.json index 2fe6db21..4d9b15d3 100644 --- a/frontend/src/i18n/locales/zh-TW.json +++ b/frontend/src/i18n/locales/zh-TW.json @@ -334,6 +334,7 @@ "device_family_cuda": "NVIDIA GPU (CUDA)", "device_family_rocm": "AMD GPU (ROCm)", "device_family_xpu": "Intel GPU (XPU)", + "device_family_npu": "NPU", "device_family_mps": "Apple GPU (MPS)", "device_family_cpu": "CPU", "device_load_failed": "無法載入裝置設定", diff --git a/tests/backend/api/test_engines_route_shape.py b/tests/backend/api/test_engines_route_shape.py index 36bf1462..2bd74fb9 100644 --- a/tests/backend/api/test_engines_route_shape.py +++ b/tests/backend/api/test_engines_route_shape.py @@ -82,7 +82,7 @@ _REQUIRED_KEYS = { "effective_device", "routing_status", "routing_reason", } _TTS_ASR_STATUSES = {"accelerated", "cpu_fallback", "cpu_only", "unavailable"} -_VALID_FAMILIES = {"cuda", "rocm", "mps", "xpu", "cpu"} +_VALID_FAMILIES = {"cuda", "rocm", "mps", "xpu", "npu", "cpu"} def test_engines_response_includes_new_fields(fresh_app): diff --git a/tests/backend/test_compute_device_settings.py b/tests/backend/test_compute_device_settings.py index 8ece0a57..9e13631f 100644 --- a/tests/backend/test_compute_device_settings.py +++ b/tests/backend/test_compute_device_settings.py @@ -109,3 +109,14 @@ def test_env_pin_is_reported_and_wins(fresh_app, monkeypatch): r = c.put("/api/settings/compute-device", json={"value": "auto"}) assert r.status_code == 200 assert r.json()["value"] == "cpu" + + +def test_auto_summary_reports_registered_npu(fresh_app, monkeypatch): + from core import device_caps + monkeypatch.setattr(device_caps, 'detect_host_caps', lambda: device_caps.HostCaps( + family='npu', available_families=('npu', 'cpu'), + )) + body = _client(fresh_app).get('/api/settings/compute-device').json() + assert body['auto_family'] == body['effective_family'] == 'npu' + # Detection does not globally opt unrelated engines into NPU execution. + assert 'npu' not in body['choices'] diff --git a/tests/test_asr_gpu_compat.py b/tests/test_asr_gpu_compat.py index 186feb99..bc4fab22 100644 --- a/tests/test_asr_gpu_compat.py +++ b/tests/test_asr_gpu_compat.py @@ -34,7 +34,7 @@ _EXPECTED = { # is_available — e.g. parakeet-mlx gates on Apple Silicon via mlx_supported()). _GPU_ONLY: set[str] = {"parakeet-mlx"} -_VALID = {"cuda", "rocm", "mps", "xpu", "cpu"} +_VALID = {"cuda", "rocm", "mps", "xpu", "npu", "cpu"} def _cls(engine_id): diff --git a/tests/test_device_caps.py b/tests/test_device_caps.py index 32fa642d..f4fc7387 100644 --- a/tests/test_device_caps.py +++ b/tests/test_device_caps.py @@ -235,3 +235,24 @@ def test_result_is_cached_until_refresh(): def teardown_module(_module): # Drop any cached mock-derived result so other test modules re-probe clean. device_caps.detect_host_caps.cache_clear() + + +def test_builtin_xpu_does_not_require_ipex(): + caps = _probe_with({'torch': _torch_mock(xpu_available=True), 'intel_extension_for_pytorch': None}) + assert caps.family == 'xpu' + + +def test_registered_npu_is_reported_without_importing_vendor_packages(): + torch = _torch_mock() + torch.npu = types.SimpleNamespace(is_available=lambda: True, get_device_name=lambda i: 'Ascend') + caps = _probe_with({'torch': torch, 'intel_extension_for_pytorch': None}) + assert caps.family == 'npu' + assert caps.available_families == ('npu', 'cpu') + assert caps.device_name == 'Ascend' + + +def test_unavailable_npu_does_not_claim_acceleration(): + torch = _torch_mock() + torch.npu = types.SimpleNamespace(is_available=lambda: False) + caps = _probe_with({'torch': torch, 'intel_extension_for_pytorch': None}) + assert caps.family == 'cpu' diff --git a/tests/test_engine_routing.py b/tests/test_engine_routing.py index 902758a2..467c3444 100644 --- a/tests/test_engine_routing.py +++ b/tests/test_engine_routing.py @@ -132,7 +132,7 @@ def test_empty_compat_is_defensive_cpu_only(): # ── Contract guarantees ─────────────────────────────────────────────────── def test_never_emits_n_a(): - for fam in ("cuda", "rocm", "mps", "xpu", "cpu"): + for fam in ("cuda", "rocm", "mps", "xpu", "npu", "cpu"): for compat in ((), ("cpu",), ("cuda",), ("cuda", "cpu"), ("mps", "cpu")): assert resolve_routing(compat, _caps(fam))["routing_status"] != "n/a" diff --git a/tests/test_moss_tts_v15.py b/tests/test_moss_tts_v15.py index 7592c4b9..3b25357f 100644 --- a/tests/test_moss_tts_v15.py +++ b/tests/test_moss_tts_v15.py @@ -106,12 +106,12 @@ def test_audited_custom_remote_code_requires_both_opt_ins(monkeypatch): # ── hardware honesty (cross-platform rule) ───────────────────────────────── -def test_gpu_compat_cuda_cpu_no_mps(): +def test_gpu_compat_matches_accelerator_paths_without_mps(): """MPS is undocumented/untested upstream — we must not claim it.""" from engines.moss_tts_v15 import MossTTSV15Backend - assert MossTTSV15Backend.gpu_compat == ("cuda", "cpu"), ( - f"expected ('cuda', 'cpu'), got {MossTTSV15Backend.gpu_compat!r}" + assert MossTTSV15Backend.gpu_compat == ("cuda", "rocm", "xpu", "npu", "cpu"), ( + f"unexpected device targets: {MossTTSV15Backend.gpu_compat!r}" ) @@ -199,3 +199,41 @@ def test_generate_without_ref_audio_omits_reference(monkeypatch): MossTTSV15Backend().generate("just text") assert "ref_audio" not in captured assert "tokens" not in captured + + +@pytest.mark.parametrize('family', [None, 'cuda', 'xpu', 'npu', 'mps']) +def test_loader_device_matches_routing(monkeypatch, family): + """Exercise model + tokenizer placement without importing optional weights.""" + import io + from types import SimpleNamespace + from unittest.mock import Mock + from engines.moss_tts_v15 import main, MossTTSV15Backend + from core.device_caps import HostCaps + from services.engine_routing import resolve_routing + + expected = family if family not in (None, 'mps') else 'cpu' + accelerator = Mock(return_value=SimpleNamespace(type=family) if family else None) + monkeypatch.setitem(sys.modules, 'torch', SimpleNamespace( + accelerator=SimpleNamespace(current_accelerator=accelerator), + bfloat16='bf16', float32='fp32', + )) + processor = Mock() + processor.model_config.sampling_rate = 24000 + tokenizer = processor.audio_tokenizer + model = Mock() + model.to.return_value = model + factory = Mock() + factory.from_pretrained.return_value = model + monkeypatch.setitem(sys.modules, 'transformers', SimpleNamespace( + AutoModel=factory, + AutoProcessor=SimpleNamespace(from_pretrained=lambda *a, **kw: processor), + )) + monkeypatch.setattr(main, '_state', None) + monkeypatch.setattr(main, '_model_source', lambda: ('local-fixture', 'a' * 40)) + state = main._load_model(io.BytesIO()) + accelerator.assert_called_once_with(check_available=True) + model.to.assert_called_once_with(expected) + tokenizer.to.assert_called_once_with(expected) + assert factory.from_pretrained.call_args.kwargs['torch_dtype'] == ('fp32' if expected == 'cpu' else 'bf16') + caps = HostCaps(family=family or 'cpu', available_families=(family, 'cpu') if family else ('cpu',)) + assert resolve_routing(MossTTSV15Backend.gpu_compat, caps)['effective_device'] == state[2] diff --git a/tests/test_setup_preflight.py b/tests/test_setup_preflight.py index 707de334..4ec54e41 100644 --- a/tests/test_setup_preflight.py +++ b/tests/test_setup_preflight.py @@ -145,7 +145,7 @@ def test_preflight_device_summary(client): assert d["gpu_backend"] in {"cuda", "rocm", "mps", "cpu"} assert d["gpu_vendor"] in {"nvidia", "amd", "apple", "intel", "unknown", "none"} # #21: canonical-probe family + VRAM joined the device summary. - assert d["gpu_family"] in {"cuda", "rocm", "mps", "xpu", "cpu"} + assert d["gpu_family"] in {"cuda", "rocm", "mps", "xpu", "npu", "cpu"} assert isinstance(d["vram_gb"], (int, float))