From ed6fca46a97d2dd90351fe7bc9e0be3c6e2b0896 Mon Sep 17 00:00:00 2001 From: li-lizhe <147392333@qq.com> Date: Sun, 6 Sep 2026 09:24:39 +0800 Subject: [PATCH 1/3] fix(tts-engines): select device via torch.accelerator in confucius4 and dots_tts MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Both Confucius4 and DOTS-TTS engine sidecars hardcode device selection to `torch.cuda.is_available()`, which returns False on Ascend NPU, Intel XPU, and other non-CUDA accelerators — causing the models to silently run on CPU (in fp32) instead of the available accelerator. Replace with `torch.accelerator.current_accelerator().type`, the device-agnostic API that auto-detects CUDA, NPU, XPU, MPS, and CPU. MPS is excluded for Confucius4 (upstream untested on Apple Silicon). dtype stays bf16 for any GPU-class accelerator and fp32 on CPU. Also verified on Ascend 910B (torch 2.14, torch_npu, 4 NPU): before: cuda_available=False → device "cpu", precision "float32" after: accelerator → confucius4 device="npu", dots_tts precision="bfloat16" --- backend/engines/confucius4/main.py | 4 +++- backend/engines/dots_tts/main.py | 2 +- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/backend/engines/confucius4/main.py b/backend/engines/confucius4/main.py index 2b188f9b..3983bf6a 100644 --- a/backend/engines/confucius4/main.py +++ b/backend/engines/confucius4/main.py @@ -115,7 +115,9 @@ def _load_model(stdout): import torch from confuciustts.cli.inference import ConfuciusTTS # type: ignore[import-not-found] - device = "cuda" if torch.cuda.is_available() else "cpu" + device = torch.accelerator.current_accelerator().type # 'cuda', 'npu', 'mps', 'xpu', 'cpu' + if device == "mps": + device = "cpu" # ConfuciusTTS is untested on MPS; fall back to CPU for safety _send(stdout, {"op": "progress", "stage": "loading_model", "percent": 50}) _model = ConfuciusTTS(config_path=_config_path(), device=device) diff --git a/backend/engines/dots_tts/main.py b/backend/engines/dots_tts/main.py index 36a490b1..19d055d1 100644 --- a/backend/engines/dots_tts/main.py +++ b/backend/engines/dots_tts/main.py @@ -115,7 +115,7 @@ def _load_runtime(stdout): from dots_tts.runtime import DotsTtsRuntime # type: ignore[import-not-found] repo = os.environ.get("OMNIVOICE_DOTS_TTS_MODEL", _DEFAULT_REPO) - default_precision = "bfloat16" if torch.cuda.is_available() else "float32" + default_precision = "bfloat16" if torch.accelerator.current_accelerator().type != "cpu" else "float32" precision = os.environ.get("OMNIVOICE_DOTS_TTS_PRECISION", default_precision) optimize = os.environ.get("OMNIVOICE_DOTS_TTS_OPTIMIZE", "0") == "1" From ae2fa83b027e91ce7cba02126180191e4eeacea9 Mon Sep 17 00:00:00 2001 From: li-lizhe <147392333@qq.com> Date: Sun, 6 Sep 2026 21:06:33 +0800 Subject: [PATCH 2/3] fix: handle None from current_accelerator(); exclude MPS in dots_tts current_accelerator() returns None on CPU-only builds (no accelerator compiled in), so .type would crash. Use check_available=True and fall back to 'cpu' when None. For dots_tts, also select fp32 on MPS since DotsTtsRuntime is untested on MPS. Addresses greptile P1 + coderabbit Functional Correctness review comments. --- backend/engines/confucius4/main.py | 3 ++- backend/engines/dots_tts/main.py | 6 +++++- 2 files changed, 7 insertions(+), 2 deletions(-) diff --git a/backend/engines/confucius4/main.py b/backend/engines/confucius4/main.py index 3983bf6a..bc0b5fe9 100644 --- a/backend/engines/confucius4/main.py +++ b/backend/engines/confucius4/main.py @@ -115,7 +115,8 @@ def _load_model(stdout): import torch from confuciustts.cli.inference import ConfuciusTTS # type: ignore[import-not-found] - device = torch.accelerator.current_accelerator().type # 'cuda', 'npu', 'mps', 'xpu', 'cpu' + device = torch.accelerator.current_accelerator(check_available=True) + device = device.type if device is not None else "cpu" # 'cuda', 'npu', 'mps', 'xpu', 'cpu' if device == "mps": device = "cpu" # ConfuciusTTS is untested on MPS; fall back to CPU for safety _send(stdout, {"op": "progress", "stage": "loading_model", "percent": 50}) diff --git a/backend/engines/dots_tts/main.py b/backend/engines/dots_tts/main.py index 19d055d1..35df28c1 100644 --- a/backend/engines/dots_tts/main.py +++ b/backend/engines/dots_tts/main.py @@ -115,7 +115,11 @@ def _load_runtime(stdout): from dots_tts.runtime import DotsTtsRuntime # type: ignore[import-not-found] repo = os.environ.get("OMNIVOICE_DOTS_TTS_MODEL", _DEFAULT_REPO) - default_precision = "bfloat16" if torch.accelerator.current_accelerator().type != "cpu" else "float32" + # current_accelerator() returns None on CPU-only builds; MPS is also + # excluded because DotsTtsRuntime is untested on MPS (use fp32). + accel = torch.accelerator.current_accelerator(check_available=True) + accel_type = accel.type if accel is not None else "cpu" + default_precision = "bfloat16" if accel_type not in ("cpu", "mps") else "float32" precision = os.environ.get("OMNIVOICE_DOTS_TTS_PRECISION", default_precision) optimize = os.environ.get("OMNIVOICE_DOTS_TTS_OPTIMIZE", "0") == "1" From b9183bb2a693ae61adb2c8425ef40906a730a586 Mon Sep 17 00:00:00 2001 From: Palash Debnath <4178343+debpalash@users.noreply.github.com> Date: Mon, 7 Sep 2026 11:12:38 +0530 Subject: [PATCH 3/3] fix(engines): align accelerator metadata and DOTS runtime precision --- CHANGELOG.md | 2 ++ backend/engines/confucius4/__init__.py | 6 ++--- backend/engines/confucius4/main.py | 4 +-- backend/engines/dots_tts/main.py | 8 +++--- docs/engines/confucius4-tts.md | 8 ++++++ docs/engines/dots-tts.md | 5 ++++ tests/test_confucius4_scaffold.py | 2 +- tests/test_confucius4_sidecar.py | 26 ++++++++++++++++++++ tests/test_dots_tts_accelerator_precision.py | 24 ++++++++++++++++++ 9 files changed, 74 insertions(+), 11 deletions(-) create mode 100644 tests/test_dots_tts_accelerator_precision.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 6e096417..c2aff17c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -20,6 +20,8 @@ the frozen-backend fallback mirror it for their toolchains. ### Fixed +- Confucius accelerator routing matches device selection, and dots.tts keeps safe CPU precision on non-CUDA accelerators (#1831) — thanks @li-lizhe! + ## [0.5.2] — 2026-09-02 diff --git a/backend/engines/confucius4/__init__.py b/backend/engines/confucius4/__init__.py index 3e2a1815..81c7b3e3 100644 --- a/backend/engines/confucius4/__init__.py +++ b/backend/engines/confucius4/__init__.py @@ -62,9 +62,9 @@ class Confucius4Backend(SubprocessBackend): # Upstream vocoder rate (config target_sample_rate) — confirmed 22 050 Hz by # a live run (2026-07-02); still re-read from the sidecar's ready/audio frames. _DEFAULT_SAMPLE_RATE = 22050 - # CUDA fast path + CPU fallback, both exercised (CPU end-to-end validated). - # No MPS claim — upstream has no Metal path. - gpu_compat = ("cuda", "cpu") + # Match device propagation into upstream .to(device). XPU/NPU routing is + # contract-tested, not a claim of physical-hardware synthesis validation. + gpu_compat = ("cuda", "rocm", "xpu", "npu", "cpu") @classmethod def is_available(cls) -> tuple[bool, str]: diff --git a/backend/engines/confucius4/main.py b/backend/engines/confucius4/main.py index bc0b5fe9..d98d34db 100644 --- a/backend/engines/confucius4/main.py +++ b/backend/engines/confucius4/main.py @@ -104,7 +104,7 @@ def _ensure_clone_on_sys_path() -> None: def _load_model(stdout): - """Cold-construct the Confucius4 model (CUDA, else CPU — both validated).""" + """Cold-construct using an available torch accelerator, with CPU fallback.""" global _model if _model is not None: return _model @@ -118,7 +118,7 @@ def _load_model(stdout): device = torch.accelerator.current_accelerator(check_available=True) device = device.type if device is not None else "cpu" # 'cuda', 'npu', 'mps', 'xpu', 'cpu' if device == "mps": - device = "cpu" # ConfuciusTTS is untested on MPS; fall back to CPU for safety + device = "cpu" # MPS was slower than CPU in the existing validation run _send(stdout, {"op": "progress", "stage": "loading_model", "percent": 50}) _model = ConfuciusTTS(config_path=_config_path(), device=device) diff --git a/backend/engines/dots_tts/main.py b/backend/engines/dots_tts/main.py index 35df28c1..597923de 100644 --- a/backend/engines/dots_tts/main.py +++ b/backend/engines/dots_tts/main.py @@ -115,11 +115,9 @@ def _load_runtime(stdout): from dots_tts.runtime import DotsTtsRuntime # type: ignore[import-not-found] repo = os.environ.get("OMNIVOICE_DOTS_TTS_MODEL", _DEFAULT_REPO) - # current_accelerator() returns None on CPU-only builds; MPS is also - # excluded because DotsTtsRuntime is untested on MPS (use fp32). - accel = torch.accelerator.current_accelerator(check_available=True) - accel_type = accel.type if accel is not None else "cpu" - default_precision = "bfloat16" if accel_type not in ("cpu", "mps") else "float32" + # Match DotsTtsRuntime's own CUDA/CPU selection. Its _check_torch_env + # rejects half precision without CUDA, even when an XPU/NPU is available. + default_precision = "bfloat16" if torch.cuda.is_available() else "float32" precision = os.environ.get("OMNIVOICE_DOTS_TTS_PRECISION", default_precision) optimize = os.environ.get("OMNIVOICE_DOTS_TTS_OPTIMIZE", "0") == "1" diff --git a/docs/engines/confucius4-tts.md b/docs/engines/confucius4-tts.md index 7a32f58c..94dcb238 100644 --- a/docs/engines/confucius4-tts.md +++ b/docs/engines/confucius4-tts.md @@ -88,3 +88,11 @@ sr = model.sample_rate # 22050 - ✅ **Sidecar logic unit-tested** (`tests/test_confucius4_sidecar.py`): language normalization, tensor→PCM (mono/stereo/clip), config-path resolution, clone sys.path injection, wire framing, synthesize dispatch. + +## Accelerator routing + +The sidecar passes a runtime-available CUDA/ROCm, XPU, or registered NPU +through upstream's device-aware model loading. The engine venv needs a matching +PyTorch/vendor runtime. XPU/NPU selection is covered by mocked loader and routing +tests; this change does not certify synthesis on physical XPU/NPU hardware. +MPS keeps the existing CPU fallback described in the validation record above. diff --git a/docs/engines/dots-tts.md b/docs/engines/dots-tts.md index c21f5f03..a24d80f1 100644 --- a/docs/engines/dots-tts.md +++ b/docs/engines/dots-tts.md @@ -115,3 +115,8 @@ dots.tts runs in a dedicated sidecar venv (it pins `transformers==4.57`, which conflicts with the parent's `transformers>=5.3`). For why that adds disk and how uv keeps the cost down, see [Engine venvs & disk usage](disk-usage.md). + +The upstream runtime selects CUDA or CPU internally. Automatic precision follows +that selection: bfloat16 on CUDA, float32 otherwise, including XPU/NPU/MPS hosts +where this runtime executes on CPU. `OMNIVOICE_DOTS_TTS_PRECISION` remains an +explicit override. diff --git a/tests/test_confucius4_scaffold.py b/tests/test_confucius4_scaffold.py index 38784822..08144707 100644 --- a/tests/test_confucius4_scaffold.py +++ b/tests/test_confucius4_scaffold.py @@ -19,7 +19,7 @@ def test_registered_in_lazy_registry(): def test_backend_class_metadata(): from engines.confucius4 import Confucius4Backend assert Confucius4Backend.id == "confucius4-tts" - assert Confucius4Backend.gpu_compat == ("cuda", "cpu") # CPU validated E2E; no MPS claim + assert Confucius4Backend.gpu_compat == ("cuda", "rocm", "xpu", "npu", "cpu") assert Confucius4Backend.supports_voice_design is False diff --git a/tests/test_confucius4_sidecar.py b/tests/test_confucius4_sidecar.py index a7385837..9328a963 100644 --- a/tests/test_confucius4_sidecar.py +++ b/tests/test_confucius4_sidecar.py @@ -181,3 +181,29 @@ def test_synthesize_calls_generate_and_emits_audio(sc, monkeypatch): def test_synthesize_rejects_empty_text(sc): with pytest.raises(ValueError, match="text"): sc._handle_synthesize({"text": ""}, io.BytesIO()) + + +@pytest.mark.parametrize('family', [None, 'cuda', 'xpu', 'npu', 'mps']) +def test_load_model_matches_routing(sc, monkeypatch, family): + import sys + from types import SimpleNamespace + from unittest.mock import Mock + from core.device_caps import HostCaps + from services.engine_routing import resolve_routing + from engines.confucius4 import Confucius4Backend + + accelerator = Mock(return_value=SimpleNamespace(type=family) if family else None) + monkeypatch.setitem(sys.modules, 'torch', SimpleNamespace( + accelerator=SimpleNamespace(current_accelerator=accelerator), + )) + constructor = Mock() + monkeypatch.setitem(sys.modules, 'confuciustts.cli.inference', SimpleNamespace(ConfuciusTTS=constructor)) + monkeypatch.setattr(sc, '_model', None) + monkeypatch.setattr(sc, '_config_path', lambda: 'fixture.yaml') + monkeypatch.setattr(sc, '_ensure_clone_on_sys_path', lambda: None) + sc._load_model(io.BytesIO()) + expected = family if family not in (None, 'mps') else 'cpu' + constructor.assert_called_once_with(config_path='fixture.yaml', device=expected) + accelerator.assert_called_once_with(check_available=True) + caps = HostCaps(family=family or 'cpu', available_families=(family, 'cpu') if family else ('cpu',)) + assert resolve_routing(Confucius4Backend.gpu_compat, caps)['effective_device'] == expected diff --git a/tests/test_dots_tts_accelerator_precision.py b/tests/test_dots_tts_accelerator_precision.py new file mode 100644 index 00000000..6cc0452c --- /dev/null +++ b/tests/test_dots_tts_accelerator_precision.py @@ -0,0 +1,24 @@ +"""DOTS picks CUDA/CPU internally; another accelerator must not imply bf16.""" +import io +import sys +from types import SimpleNamespace +from unittest.mock import Mock + +import pytest + + +@pytest.mark.parametrize('family', [None, 'cuda', 'xpu', 'npu', 'mps']) +def test_precision_matches_runtime_device(monkeypatch, family): + from engines.dots_tts import main + + accelerator = Mock(return_value=SimpleNamespace(type=family) if family else None) + monkeypatch.setitem(sys.modules, 'torch', SimpleNamespace( + accelerator=SimpleNamespace(current_accelerator=accelerator), + cuda=SimpleNamespace(is_available=lambda: family == 'cuda'), + )) + loader = Mock() + monkeypatch.setitem(sys.modules, 'dots_tts.runtime', SimpleNamespace(DotsTtsRuntime=loader)) + monkeypatch.setattr(main, '_runtime', None) + monkeypatch.delenv('OMNIVOICE_DOTS_TTS_PRECISION', raising=False) + main._load_runtime(io.BytesIO()) + assert loader.from_pretrained.call_args.kwargs['precision'] == ('bfloat16' if family == 'cuda' else 'float32')