diff --git a/CHANGELOG.md b/CHANGELOG.md index ba2c34da..773b7aba 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -73,6 +73,7 @@ the frozen-backend fallback mirror it for their toolchains. - Onboarding recognizes locally saved Hugging Face tokens without contacting Hugging Face, and replacing one requires an explicit click (#1852) — thanks @psiberfunk! - The logs panel no longer reports “All clear” before log retrieval succeeds or while logs contain warnings or errors (#1870) — thanks @motodriver! - MOSS accelerator routing now matches runtime device selection, including native XPU and registered NPU detection (#1830) — thanks @li-lizhe! +- Confucius accelerator routing matches device selection, and dots.tts keeps safe CPU precision on non-CUDA accelerators (#1831) — thanks @li-lizhe! ## [0.5.2] — 2026-09-02 diff --git a/backend/engines/confucius4/__init__.py b/backend/engines/confucius4/__init__.py index 3e2a1815..81c7b3e3 100644 --- a/backend/engines/confucius4/__init__.py +++ b/backend/engines/confucius4/__init__.py @@ -62,9 +62,9 @@ class Confucius4Backend(SubprocessBackend): # Upstream vocoder rate (config target_sample_rate) — confirmed 22 050 Hz by # a live run (2026-07-02); still re-read from the sidecar's ready/audio frames. _DEFAULT_SAMPLE_RATE = 22050 - # CUDA fast path + CPU fallback, both exercised (CPU end-to-end validated). - # No MPS claim — upstream has no Metal path. - gpu_compat = ("cuda", "cpu") + # Match device propagation into upstream .to(device). XPU/NPU routing is + # contract-tested, not a claim of physical-hardware synthesis validation. + gpu_compat = ("cuda", "rocm", "xpu", "npu", "cpu") @classmethod def is_available(cls) -> tuple[bool, str]: diff --git a/backend/engines/confucius4/main.py b/backend/engines/confucius4/main.py index 2b188f9b..d98d34db 100644 --- a/backend/engines/confucius4/main.py +++ b/backend/engines/confucius4/main.py @@ -104,7 +104,7 @@ def _ensure_clone_on_sys_path() -> None: def _load_model(stdout): - """Cold-construct the Confucius4 model (CUDA, else CPU — both validated).""" + """Cold-construct using an available torch accelerator, with CPU fallback.""" global _model if _model is not None: return _model @@ -115,7 +115,10 @@ def _load_model(stdout): import torch from confuciustts.cli.inference import ConfuciusTTS # type: ignore[import-not-found] - device = "cuda" if torch.cuda.is_available() else "cpu" + device = torch.accelerator.current_accelerator(check_available=True) + device = device.type if device is not None else "cpu" # 'cuda', 'npu', 'mps', 'xpu', 'cpu' + if device == "mps": + device = "cpu" # MPS was slower than CPU in the existing validation run _send(stdout, {"op": "progress", "stage": "loading_model", "percent": 50}) _model = ConfuciusTTS(config_path=_config_path(), device=device) diff --git a/backend/engines/dots_tts/main.py b/backend/engines/dots_tts/main.py index 36a490b1..597923de 100644 --- a/backend/engines/dots_tts/main.py +++ b/backend/engines/dots_tts/main.py @@ -115,6 +115,8 @@ def _load_runtime(stdout): from dots_tts.runtime import DotsTtsRuntime # type: ignore[import-not-found] repo = os.environ.get("OMNIVOICE_DOTS_TTS_MODEL", _DEFAULT_REPO) + # Match DotsTtsRuntime's own CUDA/CPU selection. Its _check_torch_env + # rejects half precision without CUDA, even when an XPU/NPU is available. default_precision = "bfloat16" if torch.cuda.is_available() else "float32" precision = os.environ.get("OMNIVOICE_DOTS_TTS_PRECISION", default_precision) optimize = os.environ.get("OMNIVOICE_DOTS_TTS_OPTIMIZE", "0") == "1" diff --git a/docs/engines/confucius4-tts.md b/docs/engines/confucius4-tts.md index 7a32f58c..94dcb238 100644 --- a/docs/engines/confucius4-tts.md +++ b/docs/engines/confucius4-tts.md @@ -88,3 +88,11 @@ sr = model.sample_rate # 22050 - ✅ **Sidecar logic unit-tested** (`tests/test_confucius4_sidecar.py`): language normalization, tensor→PCM (mono/stereo/clip), config-path resolution, clone sys.path injection, wire framing, synthesize dispatch. + +## Accelerator routing + +The sidecar passes a runtime-available CUDA/ROCm, XPU, or registered NPU +through upstream's device-aware model loading. The engine venv needs a matching +PyTorch/vendor runtime. XPU/NPU selection is covered by mocked loader and routing +tests; this change does not certify synthesis on physical XPU/NPU hardware. +MPS keeps the existing CPU fallback described in the validation record above. diff --git a/docs/engines/dots-tts.md b/docs/engines/dots-tts.md index c21f5f03..a24d80f1 100644 --- a/docs/engines/dots-tts.md +++ b/docs/engines/dots-tts.md @@ -115,3 +115,8 @@ dots.tts runs in a dedicated sidecar venv (it pins `transformers==4.57`, which conflicts with the parent's `transformers>=5.3`). For why that adds disk and how uv keeps the cost down, see [Engine venvs & disk usage](disk-usage.md). + +The upstream runtime selects CUDA or CPU internally. Automatic precision follows +that selection: bfloat16 on CUDA, float32 otherwise, including XPU/NPU/MPS hosts +where this runtime executes on CPU. `OMNIVOICE_DOTS_TTS_PRECISION` remains an +explicit override. diff --git a/tests/test_confucius4_scaffold.py b/tests/test_confucius4_scaffold.py index 38784822..08144707 100644 --- a/tests/test_confucius4_scaffold.py +++ b/tests/test_confucius4_scaffold.py @@ -19,7 +19,7 @@ def test_registered_in_lazy_registry(): def test_backend_class_metadata(): from engines.confucius4 import Confucius4Backend assert Confucius4Backend.id == "confucius4-tts" - assert Confucius4Backend.gpu_compat == ("cuda", "cpu") # CPU validated E2E; no MPS claim + assert Confucius4Backend.gpu_compat == ("cuda", "rocm", "xpu", "npu", "cpu") assert Confucius4Backend.supports_voice_design is False diff --git a/tests/test_confucius4_sidecar.py b/tests/test_confucius4_sidecar.py index a7385837..9328a963 100644 --- a/tests/test_confucius4_sidecar.py +++ b/tests/test_confucius4_sidecar.py @@ -181,3 +181,29 @@ def test_synthesize_calls_generate_and_emits_audio(sc, monkeypatch): def test_synthesize_rejects_empty_text(sc): with pytest.raises(ValueError, match="text"): sc._handle_synthesize({"text": ""}, io.BytesIO()) + + +@pytest.mark.parametrize('family', [None, 'cuda', 'xpu', 'npu', 'mps']) +def test_load_model_matches_routing(sc, monkeypatch, family): + import sys + from types import SimpleNamespace + from unittest.mock import Mock + from core.device_caps import HostCaps + from services.engine_routing import resolve_routing + from engines.confucius4 import Confucius4Backend + + accelerator = Mock(return_value=SimpleNamespace(type=family) if family else None) + monkeypatch.setitem(sys.modules, 'torch', SimpleNamespace( + accelerator=SimpleNamespace(current_accelerator=accelerator), + )) + constructor = Mock() + monkeypatch.setitem(sys.modules, 'confuciustts.cli.inference', SimpleNamespace(ConfuciusTTS=constructor)) + monkeypatch.setattr(sc, '_model', None) + monkeypatch.setattr(sc, '_config_path', lambda: 'fixture.yaml') + monkeypatch.setattr(sc, '_ensure_clone_on_sys_path', lambda: None) + sc._load_model(io.BytesIO()) + expected = family if family not in (None, 'mps') else 'cpu' + constructor.assert_called_once_with(config_path='fixture.yaml', device=expected) + accelerator.assert_called_once_with(check_available=True) + caps = HostCaps(family=family or 'cpu', available_families=(family, 'cpu') if family else ('cpu',)) + assert resolve_routing(Confucius4Backend.gpu_compat, caps)['effective_device'] == expected diff --git a/tests/test_dots_tts_accelerator_precision.py b/tests/test_dots_tts_accelerator_precision.py new file mode 100644 index 00000000..6cc0452c --- /dev/null +++ b/tests/test_dots_tts_accelerator_precision.py @@ -0,0 +1,24 @@ +"""DOTS picks CUDA/CPU internally; another accelerator must not imply bf16.""" +import io +import sys +from types import SimpleNamespace +from unittest.mock import Mock + +import pytest + + +@pytest.mark.parametrize('family', [None, 'cuda', 'xpu', 'npu', 'mps']) +def test_precision_matches_runtime_device(monkeypatch, family): + from engines.dots_tts import main + + accelerator = Mock(return_value=SimpleNamespace(type=family) if family else None) + monkeypatch.setitem(sys.modules, 'torch', SimpleNamespace( + accelerator=SimpleNamespace(current_accelerator=accelerator), + cuda=SimpleNamespace(is_available=lambda: family == 'cuda'), + )) + loader = Mock() + monkeypatch.setitem(sys.modules, 'dots_tts.runtime', SimpleNamespace(DotsTtsRuntime=loader)) + monkeypatch.setattr(main, '_runtime', None) + monkeypatch.delenv('OMNIVOICE_DOTS_TTS_PRECISION', raising=False) + main._load_runtime(io.BytesIO()) + assert loader.from_pretrained.call_args.kwargs['precision'] == ('bfloat16' if family == 'cuda' else 'float32')