Compare commits

...
225 Commits
Author SHA1 Message Date
Palash Debnath 8c6cb9a9ac Merge pull request #2186 from debpalash/chore/electron-0.5.4
chore(release): prepare Electron 0.5.4 reliability update
2026-09-17 23:57:12 +05:30
Palash Debnath da05973f78 test(release): enforce concise Highlights and repair guide anchor 2026-09-17 23:33:00 +05:30
Palash Debnath 860f62a9a0 docs(release): reconcile introduction and change-entry ordering 2026-09-17 23:21:30 +05:30
Palash Debnath e63e0a6634 docs(release): sync owner presentation and contributor-credit requirements 2026-09-17 23:18:41 +05:30
Palash Debnath c38b7d0bbb docs(release): use the contributor current GitHub handle 2026-09-17 23:13:09 +05:30
Palash Debnath 1746868fc7 Merge remote-tracking branch 'origin/main' into chore/electron-0.5.4 2026-09-17 23:11:41 +05:30
Palash Debnath 3b9ca6401d Merge pull request #2175 from shivsin25/fix/2156-auto-language-profile-override
fix(generate): name the voice profile behind a refused language
2026-09-17 10:41:36 -07:00
Palash Debnath 07e9b8dc84 docs(release): explain the runtime refresh after upgrading 2026-09-17 22:55:02 +05:30
Palash Debnath 478590b689 docs(release): describe 0.5.4 fixes and verified community credits 2026-09-17 22:47:44 +05:30
Palash Debnath da3667a369 Merge branch 'fix/review-2175' into chore/electron-0.5.4 2026-09-17 22:47:26 +05:30
Palash Debnath 77802c5f87 Merge remote-tracking branch 'origin/main' into chore/electron-0.5.4 2026-09-17 22:46:34 +05:30
Palash Debnath 6ae92dbc47 Merge remote-tracking branch 'origin/main' into fix/review-2175
# Conflicts:
#	docs/engines/mlx-audio.md
2026-09-17 22:46:11 +05:30
Palash Debnath 9dc41252c1 Merge pull request #2174 from paranoyouz-collab/fix/kokoro-supported-language-list
fix(tts): name every Kokoro language from the installed table
2026-09-17 10:15:40 -07:00
Palash Debnath 2aa5f38012 Merge remote-tracking branch 'origin/main' into fix/review-2175
# Conflicts:
#	CHANGELOG.md
2026-09-17 22:25:24 +05:30
Palash Debnath 082d7260d5 docs: credit Kokoro fix author alongside the bug reporter 2026-09-17 22:23:44 +05:30
Palash Debnath 86c7f6bb35 chore(release): prepare Electron 0.5.4 metadata and authored notes 2026-09-17 22:19:11 +05:30
Palash Debnath eed4632933 Merge remote-tracking branch 'origin/main' into fix/review-2174 2026-09-17 21:57:28 +05:30
Palash Debnath 1fd321d145 Merge pull request #2173 from shivsin25/fix/2163-gated-install-token
fix(setup): send the HF token when installing a gated model
2026-09-17 09:27:20 -07:00
Palash Debnath 2b6fd847ce fix(kokoro): accept every advertised installed language label 2026-09-17 21:52:05 +05:30
Palash Debnath 50a71fd176 Merge remote-tracking branch 'origin/main' into fix/review-2173 2026-09-17 21:32:02 +05:30
Palash Debnath 750e81bcf7 Merge pull request #2179 from debpalash/fix/runtime-validation-2176
fix(electron): validate reused runtimes before startup
2026-09-17 09:01:54 -07:00
Palash Debnath 8a7ab51784 fix(electron): isolate runtime probes and reuse healthy fallback 2026-09-17 21:06:12 +05:30
Palash Debnath 7ed58cb5b7 Merge remote-tracking branch 'origin/main' into fix/runtime-validation-2176
# Conflicts:
#	CHANGELOG.md
2026-09-17 20:42:36 +05:30
Palash Debnath 20f7329f59 Merge pull request #2172 from debpalash/fix/voice-conversion-capability-2147
fix(electron): check cloning capability before voice conversion
2026-09-17 08:12:07 -07:00
Palash Debnath c52441188e fix(electron): retain fresh custom runtime destinations 2026-09-17 20:36:10 +05:30
Palash Debnath 49f9fe3b06 docs: explain recovery from gated installer token failures 2026-09-17 20:31:30 +05:30
Palash Debnath f6d1e0c2df fix(kokoro): resolve the advertised British English label 2026-09-17 20:30:40 +05:30
Palash Debnath d217553e2e fix(electron): preserve unowned roots without project folders 2026-09-17 20:30:38 +05:30
Palash Debnath d6efb14849 fix(electron): preserve selected runtime ownership during repair 2026-09-17 20:23:05 +05:30
Palash Debnath a222974568 fix(generate): localize terminal language refusals across streaming paths 2026-09-17 20:21:30 +05:30
Palash Debnath ed9e40253c Merge remote-tracking branch 'origin/main' into fix/review-2175
# Conflicts:
#	CHANGELOG.md
2026-09-17 20:15:20 +05:30
Palash Debnath 86c37c8d04 style: format engine retry guidance 2026-09-17 20:14:09 +05:30
Palash Debnath a4ffd5229b Merge remote-tracking branch 'origin/main' into fix/review-2174 2026-09-17 20:13:54 +05:30
Palash Debnath 4b3e998de3 test(setup): validate file lists and isolate token fixtures 2026-09-17 20:13:38 +05:30
Palash Debnath 8b2ec7ba51 Merge remote-tracking branch 'origin/main' into fix/voice-conversion-capability-2147
# Conflicts:
#	CHANGELOG.md
2026-09-17 20:13:37 +05:30
Palash Debnath af8810be53 fix(electron): offer retry when engine capability lookup fails 2026-09-17 20:13:18 +05:30
Palash Debnath 1ef42a6b21 Merge remote-tracking branch 'origin/main' into fix/review-2173 2026-09-17 20:12:15 +05:30
Palash Debnath 31baa43650 Merge remote-tracking branch 'origin/main' into fix/runtime-validation-2176
# Conflicts:
#	CHANGELOG.md
2026-09-17 20:12:01 +05:30
Palash Debnath 834b305c05 fix(electron): validate reused Python runtimes before startup 2026-09-17 20:11:38 +05:30
Palash Debnath 418b1b271d Merge pull request #2171 from debpalash/fix/signal-diagnostic-ci
test(lifecycle): handle both valid SIGKILL diagnostics
2026-09-17 07:41:16 -07:00
Shivendra-CoherentandClaude Opus 5 a6113762a4 fix(generate): answer a refused language on the remote path too
Review findings from greptile on this PR.

P1 — a profile-derived language refused by a *remote* worker never
reached the new branch. The worker's ValueError travels home as
RemoteJobFailed, a GatewayError caught ahead of the ValueError handler,
so the user got a retryable 503 offering "run it on this machine
instead" — which cannot help, because the same engine refuses the same
language wherever it runs. Both handlers now build the response through
one helper, so the local and remote paths cannot drift again. Only
language refusals change class there; a genuine worker failure keeps its
retryable 503 and its "run it here" offer.

P2 — a *generic* rejection was wrapped twice. `_language_rejection_or`
adds the engine remedy before the handler sees it, and the profile
message then quoted that wrapper, repeating both the engine-switch
advice and "Engine's own message:". The wrapper now keeps the engine's
own error reachable and the profile message quotes that, once. The
Kokoro path never showed this because its wording is self-describing and
skips the first wrapper, which is why the first cut's assertion passed.

Three regression tests, each failing before this commit.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-09-17 19:50:07 +05:30
Shivendra-CoherentandClaude Opus 5 ad6dae49cd docs(changelog): correct this PR's number to #2175
Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-09-17 19:34:14 +05:30
Shivendra-CoherentandClaude Opus 5 fbede643c1 fix(generate): name the voice profile behind a refused language
A user on mlx-audio with the language picker on "Auto" got:

    400: mlx-audio's Kokoro model doesn't support language='Persian'.
    … Pick one of those, leave language as 'Auto', or switch to a
    multilingual engine …

They had left it on Auto. The UI omits `language` entirely while its
picker reads "Auto" (useProfiles.js only appends a non-Auto value), and
#533 fills that gap from the selected voice profile. So Auto is exactly
how 'Persian' got there: the remedy the message leads with is the state
the user was already in, and nothing points at the profile that actually
supplied the language. Three generate attempts in their action log, a
detour through Settings, then the report.

Provenance was the missing fact, and only the request scope has it — the
engine adapters are handed a language with no idea who chose it. So
_resolve_profile_conditioning now reports whether it filled the language,
and /generate uses that to answer a refused profile language by naming
the profile and the remedies that exist: change the profile's language,
pick a supported one explicitly, or switch engine.

An explicitly requested language is untouched — the user really did pick
it, so blaming the profile would be a lie — and #533 still drives
generation whenever the engine can speak the profile's language.

Recognising the refusal needed one more thing: Kokoro's wording
("doesn't support language=…") matched none of #1257's signatures, so
the engine that issue was written for was the one engine its rewrite
never fired for. That wording is now recognised, but kept out of #1257's
rewrite path — it already names its engine and its languages, and
re-wrapping it only nests "Engine's own message:" twice.

No per-model language map: #1257 weighed that and chose engine-naming
over "a brittle map that goes stale on each engine update". This follows
the same principle — say where the language came from, don't enumerate.

docs/engines/mlx-audio.md repeated the same "leave language on Auto"
advice and is corrected here.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-09-17 18:04:51 +05:30
Claude 743f4918cb test(tts): document the Kokoro label tests
Raises docstring coverage on the changed files past the threshold the PR
check enforces; the reasoning each test carried in a leading comment now
lives where the checker and a reader both find it.
2026-09-17 11:43:08 +00:00
shivsin25 543f0f6f18 Merge branch 'main' into fix/2163-gated-install-token 2026-09-17 17:08:59 +05:30
paranoyouz-collab ec3e12cd6e fix(tts): name every Kokoro language from the installed table
resolve_kokoro_lang_code validates against mlx-audio's own LANG_CODES —
its docstring is explicit that the vendored table is the only source of
truth — but built the "Kokoro supports: …" list in the rejection message
from the hardcoded _KOKORO_ISO_BY_FULL_NAME instead.

That map exists to translate full names into Kokoro's codes, and it has
no entry that produces "b". British English is reachable as en-gb, and
the existing test asserts resolve("en-gb") == "b", yet the message never
mentioned it: the installed table lists nine languages and the message
named eight. Any language a later mlx-audio adds would be dropped the
same way, telling a user to switch engines when they need not.

Derives the labels from ALIASES and LANG_CODES instead, preferring the
full name a caller can actually pass and falling back to the table's own
description for codes no full name reaches.

The derivation takes the tables as arguments, so its tests run on every
platform. The resolution tests around it need the Apple-Silicon-only
package and skip on every CI runner, which is why a message drifting
from the table it documents went unnoticed.

Fixes #2156
2026-09-17 11:22:09 +00:00
Shivendra-CoherentandClaude Opus 5 37b498c891 fix(setup): send the HF token when installing a gated model
Installing a gated model failed the fast download path with "401
Unauthorized" even with a valid token and the licence accepted, then fell
back to snapshot_download and logged a 401 that reads like the token or
the licence grant is at fault when it is neither (#2163).

`token_resolver.resolve()` returns a ResolvedToken record, not the bearer
string. Two call sites handed that record straight to consumers typed
`token: str | None`, and both fail silently rather than loudly:

- `_segmented_snapshot` passes it to HfApi, get_hf_file_metadata and our
  own segmented_download. huggingface_hub's build_hf_headers ignores a
  non-str token and falls back to its own ambient discovery, so a token
  held only in VoiceStudio's Settings produces NO Authorization header
  and every gated file 401s. segmented_download instead interpolates it
  into `f"Bearer {token}"`, sending a malformed header that also inlines
  the raw secret into the request.
- `_step_fetch_weights` passes it to snapshot_download, so gated engine
  weights 401 the same way.

Every other resolve() caller already unwraps `.token`; these two were the
outliers. Both now unwrap once, at the seam.

The existing weights tests all stubbed resolve() to return None, so no
test ever exercised a resolved token — which is why this went unnoticed.
The new tests drive a real ResolvedToken through both seams and assert a
`str` reaches every consumer, plus an integration test that installs the
pyannote diarisation pipeline end to end: a weightless config_only repo
validates, both dependency repos are fetched, and every call carries the
bearer string.

Two catalogue invariants keep the rest of #2163 from returning by edit:
dependency repos must be revision-pinned (revision_for raises otherwise,
so an unpinned one ships an always-failing install), and a config_only
entry must declare config_required_files (without them the completeness
check can never pass and the error lists no files at all — the shape the
report hit on 0.5.2).

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-09-17 16:50:16 +05:30
Palash Debnath e6e8f16f55 fix: guard voice conversion with active model cloning capability 2026-09-17 16:29:51 +05:30
Palash Debnath b1f900c996 test: accept startup and supervisor SIGKILL diagnostics 2026-09-17 16:28:42 +05:30
Palash Debnath a8581f899e Merge pull request #2170 from debpalash/fix/community-integration
fix(runtime): integrate reviewed engine, transcription, and import reliability fixes
2026-09-17 03:40:31 -07:00
Palash Debnath eda99113ec docs(dubbing): deduplicate merged stream guidance 2026-09-17 15:41:15 +05:30
Palash Debnath d2e2d6073a Merge branch 'fix/review-2138' into fix/community-integration
# Conflicts:
#	docs/dubbing/translation-engines.md
2026-09-17 15:40:36 +05:30
Palash Debnath 80a1f22175 Merge branch 'fix/review-2165' into fix/community-integration 2026-09-17 15:40:36 +05:30
Palash Debnath 0f13d49363 fix(dubbing): guard diarization lifetime and task stream delivery 2026-09-17 15:40:23 +05:30
Palash Debnath a5ebdebf5d test(worker): observe artifact cleanup completion directly 2026-09-17 15:39:56 +05:30
Palash Debnath fa745366ab Merge branch 'fix/review-2165' into fix/community-integration 2026-09-17 15:18:05 +05:30
Palash Debnath c1bc91fe5b test(worker): await acknowledged artifact cleanup 2026-09-17 15:17:57 +05:30
Palash Debnath 5b01b65d1e Merge branch 'fix/review-2107' into fix/community-integration 2026-09-17 15:14:01 +05:30
Palash Debnath fecde7fca5 fix(media): require successful tool version probes 2026-09-17 15:13:49 +05:30
Palash Debnath c7883c107c Merge branch 'fix/media-error-diagnostics' into fix/community-integration 2026-09-17 15:03:56 +05:30
Palash Debnath 17c0894a85 Merge branch 'fix/review-2073' into fix/community-integration 2026-09-17 15:03:56 +05:30
Palash Debnath 39cdd4b4db Merge branch 'fix/review-2106' into fix/community-integration 2026-09-17 15:03:56 +05:30
Palash Debnath 8d903eed8e Merge branch 'fix/review-2107' into fix/community-integration 2026-09-17 15:03:56 +05:30
Palash Debnath e477251b35 Merge branch 'fix/review-2138' into fix/community-integration 2026-09-17 15:03:56 +05:30
Palash Debnath 57501ef8ef Merge branch 'fix/review-2109' into fix/community-integration 2026-09-17 15:03:56 +05:30
Palash Debnath 8b1b1e9b51 Merge branch 'fix/review-2165' into fix/community-integration 2026-09-17 15:03:56 +05:30
Palash Debnath 863d304cd5 Merge branch 'fix/review-2083' into fix/community-integration 2026-09-17 15:03:56 +05:30
Palash Debnath 11b3b87b02 test(media): resolve diagnostics module at runtime 2026-09-17 15:03:30 +05:30
Palash Debnath 80e7de7e96 test(import): restore application state after text imports 2026-09-17 15:03:26 +05:30
Palash Debnath 7a7a1d379d test(sidecar): isolate imports and await process termination 2026-09-17 15:03:22 +05:30
Palash Debnath 65044b6cfd test(asr): resolve repair module at runtime 2026-09-17 15:03:18 +05:30
Palash Debnath 06c675fbc8 fix(mlx): retain resampling context between chunks 2026-09-17 15:03:15 +05:30
Palash Debnath 737f04f42d fix(media): reject unusable sibling ffprobe 2026-09-17 15:03:11 +05:30
Palash Debnath 7182c8d312 fix(audio): share bit-depth-aware segment conversion 2026-09-17 15:03:07 +05:30
Palash Debnath 9ee1f9c1ed fix(dubbing): make disconnect cleanup single-shot 2026-09-17 15:03:03 +05:30
Palash Debnath fa16980938 Merge branch 'fix/review-2109' into fix/community-integration 2026-09-17 14:40:22 +05:30
Palash Debnath 6450fd931f Merge branch 'fix/review-2165' into fix/community-integration 2026-09-17 14:40:22 +05:30
Palash Debnath a0154778a4 test: keep long-text case IDs within Windows environment limits 2026-09-17 14:40:20 +05:30
Palash Debnath 1c5c16e21e fix: select native file locking by host capability 2026-09-17 14:36:57 +05:30
Palash Debnath 49892a6066 docs: remove duplicate release entry and enforce new CI references 2026-09-17 14:08:03 +05:30
Palash Debnath 05a8b58b53 Merge branch 'fix/review-2077' into fix/community-integration 2026-09-17 14:01:28 +05:30
Palash Debnath 1f17560cde fix: normalize blank separators before WebVTT metadata classification 2026-09-17 14:01:25 +05:30
Palash Debnath 7a7cccbea2 docs: reconcile reviewed subtitle guidance and release references 2026-09-17 13:57:28 +05:30
Palash Debnath e3d3913f52 Merge branch 'fix/review-2077' into fix/community-integration
# Conflicts:
#	CHANGELOG.md
#	docs/dubbing/translation-engines.md
2026-09-17 13:57:07 +05:30
Palash Debnath ad52d32760 fix: distinguish WebVTT identifiers and empty cue boundaries 2026-09-17 13:57:05 +05:30
Palash Debnath 1ddf79a6de Merge branch 'fix/review-2077' into fix/community-integration
# Conflicts:
#	docs/dubbing/translation-engines.md
2026-09-17 13:48:47 +05:30
Palash Debnath 308dda0cac fix: exclude WebVTT metadata before scanning cue timestamps 2026-09-17 13:48:45 +05:30
Palash Debnath 3775a2acd9 Merge branch 'fix/review-2077' into fix/community-integration 2026-09-17 13:29:25 +05:30
Palash Debnath fb80d88f34 fix: require complete subtitle timing pairs for paste detection 2026-09-17 13:29:23 +05:30
Palash Debnath d874239c73 Merge branch 'fix/review-2165' into fix/community-integration 2026-09-17 13:28:29 +05:30
Palash Debnath 85f8328b7c test: always restore translation spies after client tests 2026-09-17 13:28:27 +05:30
Palash Debnath 6a15030e31 Merge branch 'fix/review-2077' into fix/community-integration 2026-09-17 13:15:43 +05:30
Palash Debnath cef375e700 Merge branch 'fix/review-2165' into fix/community-integration
# Conflicts:
#	docs/dubbing/translation-engines.md
#	electron/src/renderer/src/i18n/locales/ar.json
#	electron/src/renderer/src/i18n/locales/de.json
#	electron/src/renderer/src/i18n/locales/en.json
#	electron/src/renderer/src/i18n/locales/es.json
#	electron/src/renderer/src/i18n/locales/fr.json
#	electron/src/renderer/src/i18n/locales/hi.json
#	electron/src/renderer/src/i18n/locales/id.json
#	electron/src/renderer/src/i18n/locales/it.json
#	electron/src/renderer/src/i18n/locales/ja.json
#	electron/src/renderer/src/i18n/locales/ko.json
#	electron/src/renderer/src/i18n/locales/nl.json
#	electron/src/renderer/src/i18n/locales/pl.json
#	electron/src/renderer/src/i18n/locales/pt.json
#	electron/src/renderer/src/i18n/locales/ru.json
#	electron/src/renderer/src/i18n/locales/sv.json
#	electron/src/renderer/src/i18n/locales/th.json
#	electron/src/renderer/src/i18n/locales/tr.json
#	electron/src/renderer/src/i18n/locales/uk.json
#	electron/src/renderer/src/i18n/locales/vi.json
#	electron/src/renderer/src/i18n/locales/zh-CN.json
#	electron/src/renderer/src/i18n/locales/zh-TW.json
#	frontend/src/i18n/locales/ar.json
#	frontend/src/i18n/locales/de.json
#	frontend/src/i18n/locales/en.json
#	frontend/src/i18n/locales/es.json
#	frontend/src/i18n/locales/fr.json
#	frontend/src/i18n/locales/hi.json
#	frontend/src/i18n/locales/id.json
#	frontend/src/i18n/locales/it.json
#	frontend/src/i18n/locales/ja.json
#	frontend/src/i18n/locales/ko.json
#	frontend/src/i18n/locales/nl.json
#	frontend/src/i18n/locales/pl.json
#	frontend/src/i18n/locales/pt.json
#	frontend/src/i18n/locales/ru.json
#	frontend/src/i18n/locales/sv.json
#	frontend/src/i18n/locales/th.json
#	frontend/src/i18n/locales/tr.json
#	frontend/src/i18n/locales/uk.json
#	frontend/src/i18n/locales/vi.json
#	frontend/src/i18n/locales/zh-CN.json
#	frontend/src/i18n/locales/zh-TW.json
#	tests/test_dub_translate.py
2026-09-17 13:15:43 +05:30
Palash Debnath 84290db784 fix: localize Argos runtime recovery across desktop and browser 2026-09-17 13:15:15 +05:30
Palash Debnath 810e5e350b fix: preserve numeric-only subtitle dialogue after skipped cues 2026-09-17 13:14:05 +05:30
Palash Debnath c9fefe39f3 Merge branch 'fix/review-2077' into fix/community-integration 2026-09-17 12:56:48 +05:30
Palash Debnath c3137da680 fix(subtitles): advance index state across invalid cues 2026-09-17 12:56:29 +05:30
Palash Debnath 281287d3e7 Merge branch 'fix/review-2165' into fix/community-integration 2026-09-17 12:52:33 +05:30
Palash Debnath 5c6d0f7f8f test(asr): isolate OOM recovery from host free VRAM 2026-09-17 12:52:31 +05:30
Palash Debnath 819fd6e4a3 Merge branch 'fix/review-2109' into fix/community-integration 2026-09-17 12:51:55 +05:30
Palash Debnath ebc03bb117 chore: clarify timeout metadata fallback and remove unused import 2026-09-17 12:51:20 +05:30
Palash Debnath e040d19513 docs: account for shared import and cancellation services 2026-09-17 12:48:46 +05:30
Palash Debnath c3b4cc8de6 Merge branch 'fix/media-error-diagnostics' into fix/community-integration 2026-09-17 12:48:22 +05:30
Palash Debnath 68c48e384b Merge branch 'fix/catalogue-install-capability' into fix/community-integration 2026-09-17 12:48:22 +05:30
Palash Debnath 9110b01c75 Merge branch 'fix/review-2077' into fix/community-integration
# Conflicts:
#	docs/dubbing/translation-engines.md
2026-09-17 12:48:22 +05:30
Palash Debnath 9579d1c82e Merge branch 'fix/review-2109' into fix/community-integration
# Conflicts:
#	docs/install/troubleshooting.md
2026-09-17 12:48:22 +05:30
Palash Debnath de1d9d66eb Merge branch 'fix/review-2083' into fix/community-integration 2026-09-17 12:48:22 +05:30
Palash Debnath 38b6d11815 fix(subtitles): preserve ambiguous numbers in mixed cue files 2026-09-17 12:48:21 +05:30
Palash Debnath a17b6a8172 fix: close lifecycle and diagnostic review regressions 2026-09-17 12:47:12 +05:30
Palash Debnath a9195b8e40 fix: close lifecycle and diagnostic review regressions 2026-09-17 12:47:08 +05:30
Palash Debnath 0be2859f75 fix: close lifecycle and diagnostic review regressions 2026-09-17 12:47:04 +05:30
Palash Debnath 40e52a1fa7 fix: close lifecycle and diagnostic review regressions 2026-09-17 12:46:56 +05:30
Palash Debnath 4c4c23d69b Merge branch 'fix/review-2165' into fix/community-integration
# Conflicts:
#	CHANGELOG.md
2026-09-17 12:40:40 +05:30
Palash Debnath 6bf4316554 docs: consolidate community fix highlights 2026-09-17 12:40:10 +05:30
Palash Debnath a8c90343b7 docs: consolidate credited unreleased repair note 2026-09-17 12:39:59 +05:30
Palash Debnath 06389e1ae2 Merge branch 'fix/media-error-diagnostics' into fix/community-integration
# Conflicts:
#	CHANGELOG.md
#	docs/install/troubleshooting.md
2026-09-17 12:38:57 +05:30
Palash Debnath 8f06b4968f Merge branch 'fix/catalogue-install-capability' into fix/community-integration
# Conflicts:
#	CHANGELOG.md
#	docs/install/troubleshooting.md
2026-09-17 12:38:56 +05:30
Palash Debnath 020febe327 Merge branch 'fix/review-2066' into fix/community-integration
# Conflicts:
#	CHANGELOG.md
2026-09-17 12:38:56 +05:30
Palash Debnath 6a7bb1a508 Merge branch 'fix/review-2115' into fix/community-integration
# Conflicts:
#	CHANGELOG.md
2026-09-17 12:38:56 +05:30
Palash Debnath 5f519cbfd0 Merge branch 'fix/review-2085' into fix/community-integration
# Conflicts:
#	CHANGELOG.md
#	docs/install/troubleshooting.md
2026-09-17 12:38:56 +05:30
Palash Debnath e272217561 Merge branch 'fix/review-2084' into fix/community-integration
# Conflicts:
#	CHANGELOG.md
2026-09-17 12:38:56 +05:30
Palash Debnath 4002801c00 Merge branch 'fix/review-2143' into fix/community-integration
# Conflicts:
#	CHANGELOG.md
#	docs/dubbing/translation-engines.md
#	tests/test_dub_translate.py
2026-09-17 12:38:56 +05:30
Palash Debnath 645aa3d451 Merge branch 'fix/review-2077' into fix/community-integration
# Conflicts:
#	CHANGELOG.md
#	docs/dubbing/translation-engines.md
2026-09-17 12:38:37 +05:30
Palash Debnath e1e329e7f0 Merge branch 'fix/review-2073' into fix/community-integration
# Conflicts:
#	CHANGELOG.md
#	docs/install/troubleshooting.md
2026-09-17 12:38:36 +05:30
Palash Debnath 8ba4407d8a Merge branch 'fix/review-2074' into fix/community-integration
# Conflicts:
#	CHANGELOG.md
#	docs/dubbing/translation-engines.md
2026-09-17 12:38:36 +05:30
Palash Debnath 47d441311e Merge branch 'fix/review-2075' into fix/community-integration
# Conflicts:
#	CHANGELOG.md
2026-09-17 12:38:36 +05:30
Palash Debnath 9ed477c0ca Merge branch 'fix/review-2106' into fix/community-integration
# Conflicts:
#	CHANGELOG.md
2026-09-17 12:38:36 +05:30
Palash Debnath dadb642e19 Merge branch 'fix/review-2107' into fix/community-integration
# Conflicts:
#	CHANGELOG.md
#	docs/install/troubleshooting.md
2026-09-17 12:38:36 +05:30
Palash Debnath b12fa3a13c Merge branch 'fix/review-2138' into fix/community-integration
# Conflicts:
#	CHANGELOG.md
#	docs/dubbing/translation-engines.md
2026-09-17 12:38:36 +05:30
Palash Debnath 119816d2b1 Merge branch 'fix/review-2109' into fix/community-integration
# Conflicts:
#	CHANGELOG.md
#	docs/install/troubleshooting.md
2026-09-17 12:38:36 +05:30
Palash Debnath ba77bd9443 Merge branch 'fix/review-2165' into fix/community-integration 2026-09-17 12:38:36 +05:30
Palash Debnath 8e97b2bd01 Merge branch 'fix/review-2083' into fix/community-integration
# Conflicts:
#	CHANGELOG.md
#	docs/install/troubleshooting.md
2026-09-17 12:38:36 +05:30
Palash Debnath 13d3f3fb13 Merge branch 'fix/review-2072' into fix/community-integration
# Conflicts:
#	CHANGELOG.md
2026-09-17 12:38:08 +05:30
Palash Debnath 9482788898 fix(dubbing): retain actionable native failure diagnostics 2026-09-17 12:37:43 +05:30
Palash Debnath a390da1590 fix(catalogue): explain local-only native installation 2026-09-17 12:37:34 +05:30
Palash Debnath f53be49e86 style: apply frontend formatter to import regression tests 2026-09-17 12:35:55 +05:30
Palash Debnath 724e573746 style: apply frontend formatter to import regression tests 2026-09-17 12:35:55 +05:30
Palash Debnath 6c5a88260b fix: address updated review findings and regressions 2026-09-17 12:32:14 +05:30
Palash Debnath 75a7b21193 fix: address updated review findings and regressions 2026-09-17 12:32:11 +05:30
Palash Debnath df3b6f402a fix: address updated review findings and regressions 2026-09-17 12:32:05 +05:30
Palash Debnath f26d5cc041 fix: address updated review findings and regressions 2026-09-17 12:31:57 +05:30
Palash Debnath b900130f29 fix: address updated review findings and regressions 2026-09-17 12:31:49 +05:30
Palash Debnath 75efda84b1 fix(sidecars): reconcile receive deadlines with outer job budgets 2026-09-17 12:27:56 +05:30
Palash Debnath 8538702ecb fix(imports): preserve legacy text encodings in Electron and web 2026-09-17 12:27:56 +05:30
Palash Debnath 11584c6f4c fix(dubbing): retain ASR model until native work finishes on disconnect 2026-09-17 12:27:30 +05:30
Palash Debnath a9d1464acb docs: preserve release notes and qualify packaged Metal availability 2026-09-17 12:18:02 +05:30
Palash Debnath 3c4c8ed679 fix: harden native repair and preserve offline Whisper fallback 2026-09-17 12:18:02 +05:30
Palash Debnath 33a6fc1d03 fix: validate WebVTT and mixed-index numeric dialogue together 2026-09-17 12:18:02 +05:30
Palash Debnath a33db40613 fix: align WebVTT detection with cue boundaries 2026-09-17 12:16:41 +05:30
Palash Debnath 8c47e8b92f fix: consolidate Argos coverage and reject silent script changes 2026-09-17 12:16:09 +05:30
Palash Debnath d6f034c754 fix: preserve numeric dialogue across mixed subtitle cue boundaries 2026-09-17 12:15:36 +05:30
Palash Debnath f840303549 Merge branch 'fix/review-2152' into fix/review-2143 2026-09-17 12:14:19 +05:30
Palash Debnath fc98d07b39 fix: integrate current main and finish review requirements for #2066 2026-09-17 12:10:32 +05:30
Palash Debnath 637e24d8a1 fix: integrate current main and finish review requirements for #2085 2026-09-17 12:10:27 +05:30
Palash Debnath 08e5d0d5cd fix: integrate current main and finish review requirements for #2084 2026-09-17 12:10:23 +05:30
Palash Debnath c52fdd263f fix: integrate current main and finish review requirements for #2074 2026-09-17 12:10:19 +05:30
Palash Debnath 2a3257be78 fix: integrate current main and finish review requirements for #2075 2026-09-17 12:10:15 +05:30
Palash Debnath 7d77f1532b fix: integrate current main and finish review requirements for #2106 2026-09-17 12:10:10 +05:30
Palash Debnath 519cc9c1c9 fix: integrate current main and finish review requirements for #2083 2026-09-17 12:10:07 +05:30
Palash Debnath 4273f8149b fix: integrate current main and finish review requirements for #2072 2026-09-17 12:10:02 +05:30
Palash Debnath 65a535a638 fix: integrate current main and finish review requirements for #2080 2026-09-17 12:09:57 +05:30
Palash Debnath b5e9e1bf58 Merge remote-tracking branch 'origin/main' into fix/review-2066 2026-09-17 12:08:09 +05:30
Palash Debnath dad6bb1cef Merge remote-tracking branch 'origin/main' into fix/review-2085 2026-09-17 12:08:01 +05:30
Palash Debnath 904d90d6a7 Merge remote-tracking branch 'origin/main' into fix/review-2084 2026-09-17 12:07:56 +05:30
Palash Debnath 583189a478 Merge remote-tracking branch 'origin/main' into fix/review-2152 2026-09-17 12:07:52 +05:30
Palash Debnath 58eba1e0cd Merge remote-tracking branch 'origin/main' into fix/review-2143 2026-09-17 12:07:48 +05:30
Palash Debnath 52962e47c6 Merge remote-tracking branch 'origin/main' into fix/review-2077 2026-09-17 12:07:40 +05:30
Palash Debnath 6772ad363f Merge remote-tracking branch 'origin/main' into fix/review-2074 2026-09-17 12:07:31 +05:30
Palash Debnath 63a8bf4165 Merge remote-tracking branch 'origin/main' into fix/review-2075 2026-09-17 12:07:27 +05:30
Palash Debnath f3da7b0ef3 Merge remote-tracking branch 'origin/main' into fix/review-2106 2026-09-17 12:07:22 +05:30
Palash Debnath 552711b1d8 Merge remote-tracking branch 'origin/main' into fix/review-2111 2026-09-17 12:07:11 +05:30
Palash Debnath eca0972511 Merge remote-tracking branch 'origin/main' into fix/review-2109 2026-09-17 12:07:07 +05:30
Palash Debnath 1685f6aba9 Merge remote-tracking branch 'origin/main' into fix/review-2072 2026-09-17 12:06:58 +05:30
Palash Debnath c780744b11 docs: document sibling media tool resolution and credit contributor 2026-09-17 12:06:32 +05:30
Palash Debnath aeb0c4ea6a Merge remote-tracking branch 'origin/main' into fix/review-2107 2026-09-17 12:06:01 +05:30
sean park e35f066a9b docs(changelog): note the CTranslate2 exec-stack repair and the PyTorch Whisper OOM ladder (#2165) 2026-09-17 13:37:00 +09:00
sean park 67c61fe357 fix(asr): repair CTranslate2's exec-stack request instead of only routing around it (#692)
ctranslate2 <=4.4.0 marks its native library's stack RWE; kernels that
refuse the request fail the dlopen, taking whisperx, faster-whisper and
Argos translation down. main learned to detect and fall back; this also
fixes the library: core.execstack clears the one ELF bit in place on
first probe (Linux-only, memoized, never raises), so the engines load
normally on hardened kernels. Argos's probe moves to
argostranslate.translate so it cannot advertise an engine whose every
request 500s, and the dub route returns one actionable 400.

Also: pytorch-whisper survives a CUDA OOM at transcribe time by
stepping the batch down (16->4->1, 8->2->1 with word timestamps) and
finishing on CPU rather than dropping the chunk.
2026-09-17 13:19:30 +09:00
rollroyces 50c6bfbd01 fix(dub): drop redundant blank-input branch; expand docstring
Addresses the CodeRabbit 🟡 Minor nit on the empty-input branch added in
#2143: the explicit `if not raw: raise ValueError(...)` is fully shadowed
by the `re.fullmatch(r"[a-z]{2,3}", code)` check two lines below, which
raises the same message for any input that does not look like a language
code (including empty and whitespace-only after strip()). Removing the
shortcut is the smallest correct change — same caller-observable
behavior, one less branch.

The expanded docstring enumerates the alias shapes the function now
recognises, so future readers do not have to reverse-engineer the
behaviour from the alias dicts.

The test `test_argos_lang_code_rejects_invalid_inputs` is widened to
cover a non-empty but invalid shape (`"Chinese ("` — looks like a label,
fails the regex) alongside the original empty/whitespace cases, so it
asserts behavior unique to the regex branch rather than passing for the
wrong reason.
2026-09-17 09:55:38 +08:00
Palash Debnath 18c543484b Merge pull request #2161 from debpalash/fix/electron-release-empty-signing
Fix Electron release packaging with missing signing credentials
2026-09-16 16:08:26 -07:00
Palash Debnath 54c89453a4 Handle empty signing secrets and retry immutable Electron release tags 2026-09-17 04:16:51 +05:30
SurefireStudios 8c69dff00f Trim the deadline comments to what the code needs 2026-09-16 14:36:19 -07:00
SurefireStudios 968d05faea fix(sidecar): derive the generate deadline from the request's own budget
Greptile on #2109: a flat 600s default still undercuts the budget the job
was granted. generate_timeout_s adds 1s per 40 characters past a 1200-char
allowance and can be raised by env, so a long passage is granted more than
600s and a constant deadline only moves the cliff to longer inputs rather
than removing it. The comment on GENERATE_RECV_TIMEOUT_S claimed the inner
deadline "can never fire before the outer budget", which was not true for
those inputs.

_effective_recv_timeout_s(text) now takes the larger of the engine's own
value and the budget this specific request was granted, and generate()
uses it for both the first frame and the progress loop.

Only for engines that expressed no opinion. An override is a deliberate
statement about that model, and #2103 asks for "fast engines opt down" to
keep working, so a subclass that sets recv_timeout_s keeps exactly that
value — including a smaller one. Detected by walking __mro__ for a
recv_timeout_s in a subclass __dict__ rather than comparing values, so an
engine that happens to pick the same number as the floor is still honoured.

Budget probing is advisory: any failure falls back to the class floor
rather than turning a working generate into an error.

Tests: a long passage raises the deadline above the flat floor and tracks
generate_timeout_s; an opt-down engine keeps its value at any text length;
a failing probe falls back. Verified load-bearing by making the derivation
return the engine value unconditionally (the long-passage test fails on
600.0 > 600.0).
2026-09-16 14:36:19 -07:00
SurefireStudios 07f2976756 docs(changelog): reference the PR alongside the issue 2026-09-16 14:36:19 -07:00
SurefireStudios a65a0a95af fix(sidecar): stop killing generations at the health-check ping budget
SubprocessBackend.recv_timeout_s defaulted to RECV_TIMEOUT_S (60s), the
budget for a health_check ping. Four engines never overrode it and so used
a ping's deadline as their generation deadline: confucius4-tts, dots-tts,
moss-tts-v15 and supertonic3. All four fall back to CPU off CUDA, where a
normal sentence does not finish in 60s, and the watchdog killed them
mid-synthesis.

That deadline sat 5-10x below the wall-clock budget the same job was
granted by model_manager.generate_timeout_s (300s accelerated, 600s CPU),
so the sidecar was reclaimed while its caller still considered it well
inside budget. Every engine that did override the hook chose 300s..900s,
at or above the accelerated budget; omnivoice_subprocess documents the
intent as "aligns the kill deadline with the generate budget". The 60s
default contradicted that intent for anyone who did not opt out of it.

which left the class default and the four silent engines untouched.

Default to GENERATE_RECV_TIMEOUT_S (600s, the CPU budget floor) so an
engine that expresses no opinion can no longer be cut off before its own
job budget. RECV_TIMEOUT_S stays 60s for health_check, where a ping must
stay fast. Overriding is still how an engine asks for more.

Second half: the watchdog logged "exceeded recv timeout; killing" but
generate() raised "sidecar closed pipe mid-generate", so the one fact that
explained the failure never left the backend log and reporters read it as
a crash. _last_recv_timed_out already distinguished the two for the spawn
handshake (#2026); use it here too and name the deadline, the elapsed
time and the sidecar's last stderr. Crash wording is unchanged.

Tests: a registry invariant that no SubprocessBackend can be dispatched
with a deadline under the accelerated budget, so a new engine cannot
inherit this by omission the way these four did; lockstep with
model_manager's budgets; and a wedging sidecar asserting the timeout
message. Verified fail-before/pass-after by reverting the default to 60s
(all 7 fail, naming exactly the four engines) and by forcing the crash
wording (the message test fails).

Fixes #2103
2026-09-16 14:36:09 -07:00
LMG XENON be1be89a69 Merge branch 'main' into fix/apple-silicon-metal-docs-ci-2105 2026-09-16 19:36:08 +01:00
LMG XENON 15ef5da87d Merge branch 'debpalash:main' into fix/sidecar-recv-timeouts-2103 2026-09-16 19:34:58 +01:00
rollroyces e59bd03150 test(dub): cover remaining argos_lang_code aliases and empty input
Extends the existing parametrized test on top of #2143 to cover every
new normalization branch the function introduces (Chinese Traditional,
Mandarin, zho, in→id, iw→he, fil→tl), and adds a parametrized
ValueError check for empty and whitespace-only input.

This addresses the CodeRabbit quick-win nit on #2143 so its coverage
matches the production branches. All 47 dub_translate tests pass.
2026-09-16 23:40:12 +08:00
Shivendra-CoherentandClaude Opus 5 34b5eaabf1 fix(subtitles): keep cue lines that are only a number
An imported or pasted subtitle whose line is just a number — a year, a
score, a street number, a "3 / 2 / 1" countdown — was silently deleted,
and when that line was the cue's only text the whole cue disappeared.

Cue bodies are sliced from a timing line up to the next one, which
swallows the next cue's index. The parser clawed that back by popping
every trailing digit-only line, a rule that cannot tell an index from
numeric dialogue. Three shapes fell out of it:

- the last cue of a file has no next index to pop, so a closing "1999"
  was taken for one and the cue was dropped as empty;
- an index-less export has no indices at all, yet still lost its
  closing numeric line — and lost it without counting a skip, so the
  import reported itself lossless while dropping a line;
- the `while` loop popped digit lines until it hit a non-digit, eating
  a multi-line countdown cue whole.

Give back exactly the one line that was swallowed: a single line, only
when a next cue exists to own it, and only in a file that indexes its
cues at all — decided once from the preamble before the first timing
line, since a body of "42" is indistinguishable from an index on its
own. Index detection is ASCII-only, because `str.isdigit()` is also
true for Arabic-Indic and Devanagari numerals, which in a 646-language
dubbing app are dialogue rather than SubRip indices.

Affects both subtitle entry points: POST /dub/import-srt/{job_id} and
POST /dub/parse-subtitle-text.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-09-16 18:10:51 +05:30
Gyanu Mayank b6c4640b7d fix: Argos language names and BCP-47 tags map to ISO 639-1
Argos Translate ships pairs as zh, not Chinese. Passing the UI
label or a tag like cmn-Hans made get_translation_from_codes
return None and every subtitle stayed in the source language.

Fixes #2140
2026-09-16 08:42:35 +05:30
denemon d6b1326aae fix(dub): observe late _ping_while failures and fix locale log labels
- _ping_while retrieves failures that occur after the consumer disconnects,
  preventing "exception was never retrieved" at GC; the early-exit test
  covers this case
- stream_cut_backend_alive_local now builds the Settings → Logs → Backend
  path from each locale's actual UI labels (`settings.title`,
  `settings.logs`, `common.backend`)
- Fix incorrect section names in 15 locales, including ru, de, and zh-TW

Addresses CodeRabbit review on #2138
2026-09-16 02:38:55 +09:00
denemon 052eb1050b fix(dub): cancel wrapped work when the transcribe stream closes early
`_ping_while` runs awaited work as its own task, so a client disconnect
previously cancelled only the ping loop. Reference-text refinement could
continue running while `run_transcribe_guarded` skipped its abandon path
and the stream finalizer unloaded the ASR model underneath it.

Cancel unfinished work on early exit, matching the cancellation behavior
of the bare `await` it replaced. Do not await it in `finally`, which can
also run under `GeneratorExit`.

- **dub_core**: `_ping_while` cancels an unfinished future in `finally`,
  covering all seven call sites
- **test**: verify that `_ping_while` cancels work on `aclose()` while
  leaving completed work untouched (fail-before / pass-after)
- **CHANGELOG**: add a Highlights entry for the user-visible fix; keep
  Fixed entries under `### Fixed` per `CLAUDE.md`

Addresses Greptile P1/P2 on #2138
2026-09-16 02:30:16 +09:00
denemon 22e52c64a4 fix(dub): keep SSE streams alive through byte-silent steps (#2108)
After transcription, several minutes of backend work could produce no
SSE bytes, causing the desktop webview to close the idle connection while
the job continued and eventually completed. This also led to a misleading
reverse-proxy error on local connections.

- dub_core: keep post-transcript awaits alive with `_ping_while` (5s pings)
- dub_export: send SSE comments after 15s of stream silence
- backendCrash.ts: show deployment-aware connection-loss guidance
- Add regression tests for post-transcript pings, task-stream keepalive,
  and local-mode error messaging

Fixes #2108
2026-09-16 02:08:54 +09:00
LMGXENON 0c71d506ec docs: document Apple Silicon Metal build parity and drop continue-on-error (#2105)
- Drops experimental flag (continue-on-error) on macos-14 in build-omnivoice-tts.yml now that omnivoice.cpp builds cleanly with -DGGML_METAL=ON at the pinned SHA.
- Updates bin/README.md, backend/engines/omnivoice_gguf/README.md, and SPIKE-01 ADR to document verified Apple Silicon Metal acceleration.
- Corrects workflow reference in bin/README.md.
- Adds changelog credit for @martinezpl.

Closes #2105
2026-09-15 01:02:01 +01:00
LMGXENON 50cba2881c fix(engine): initialize timed_out before empty-reply check and move credit to Highlights (#2103) 2026-09-15 00:43:25 +01:00
LMGXENON 750aa2eb57 fix(engine): validate ASR timeouts, reserve outer guard grace, and enhance diagnostics (#2103) 2026-09-15 00:34:18 +01:00
LMGXENON ddb8e4dd86 fix(engine): address review findings on sidecar deadlines, ASR timeout, and changelog credit (#2103) 2026-09-15 00:22:55 +01:00
LMGXENON ba041b0723 fix(engine): report sidecar receive timeouts and raise engine deadlines (#2103)
- SubprocessBackend and SubprocessASRBackend now check _last_recv_timed_out
  when a sidecar stream closes unexpectedly, reporting an actionable timeout
  error with the deadline rather than describing a generic pipe-closed crash.
- Confucius4Backend, DotsTTSBackend, MossTTSV15Backend, and Supertonic3Backend
  now define generous recv_timeout_s properties with OMNIVOICE_* env overrides,
  preventing mid-generation termination on slower CPU/MPS devices.
- Adds regression tests in backend/tests/test_omnivoice_subprocess.py.

Closes #2103
2026-09-15 00:09:17 +01:00
kapelame c5f3355951 fix(audio): preserve the parent directory when locating ffprobe 2026-09-14 11:29:42 -04:00
kapelame 68653ca408 fix(audio): resample MLX outputs to the declared sample rate 2026-09-14 11:29:38 -04:00
moep90 df18ac177e docs: correct the Blackwell/Triton attribution behind the #278 fallback
Three comments name Blackwell sm_120 as an architecture Triton/Inductor does
not support. Measured on an sm_120 device with the pinned torch 2.8.0+cu128
and triton 3.4.0: torch.compile default and reduce-overhead (the
cudagraph_trees path #278 names), an attention module over growing sequence
lengths, and a raw Triton kernel all run, with compiled output matching eager
(maxdiff 1.19e-06 and 0.0). The app's own probe agrees — arch_unsupported
returns None, so compile is already attempted there.

The error text #278 quotes is also misattributed. "Detected that you are using
FX to symbolically trace a dynamo-optimized function" reproduces with CUDA
unavailable: Dynamo raises it whenever FX traces a compiled function,
regardless of device. It belongs in the compile-stack classifier, not in the
evidence for a missing-architecture failure.

Comments only. The fallback contract and the arch-list gate are unchanged and
still correct: the gate is generic rather than a Blackwell blocklist, and on a
build whose arch list lacks the device the described mechanism holds. Only the
example and the FX attribution are stale.

Signed-off-by: Moep90 <volleyballlive@googlemail.com>

Signed-off-by: moep90 <volleyballlive@googlemail.com>
2026-09-14 06:54:15 +02:00
moep90 6cc67c31f6 docs: fix the arch-list check and qualify the Pascal drop by torch release
Two review findings on the troubleshooting section.

The get_arch_list() check told users to run `import torch` in a section whose
documented symptom is `import torch` crashing natively, so it was useless for
exactly its intended readers. It now leads with that: if the import crashes,
that is the symptom, skip to the fix. The check keeps its remaining purpose of
ruling the missing-kernel theory out where torch does import.

"The cu128 wheels dropped Maxwell and Pascal" attributed the drop to the CUDA
variant when it tracks the torch release. CU128_ARCHS carries sm_61 from the
#1285 build, which predates 2.8; the pinned 2.8.0+cu128 build reports sm_70
first. Both are right for their own release, so the blanket claim was not.
Stated as the two measurements instead, and deliberately not extended to
2.9.x, which is unmeasured here.

Signed-off-by: Moep90 <volleyballlive@googlemail.com>

Signed-off-by: moep90 <volleyballlive@googlemail.com>
2026-09-14 06:45:27 +02:00
moep90 1bf772466c docs: drop the last "a build that adds sm_120" phrasing
Audit leftover in the same file: the "Keeping the change" note explained that
the default pin cannot move because "a build that adds sm_120 can drop older
architectures". That still implies the pinned build lacks sm_120, sixty lines
below the corrected cause. The reason the default cannot move is simply that a
newer build can drop older GPUs, which cu128 already did for Maxwell and
Pascal.

Deliberately not restating the resulting floor. docs/competitive-analysis.md
says Turing sm_75, but the pinned 2.8.0+cu128 build reports sm_70 first, so the
claim is unverified and left out rather than copied.

Signed-off-by: Moep90 <volleyballlive@googlemail.com>

Signed-off-by: moep90 <volleyballlive@googlemail.com>
2026-09-14 06:41:17 +02:00
moep90 42acd7d12f docs: correct the same sm_120 claim in the troubleshooting guide
Review catch: the code comments were corrected while
docs/install/troubleshooting.md still gave users the false diagnosis in two
places, which is the docs-sync rule exactly. The user-facing wording was the
stronger of the two:

  5b:   "the wheel does not contain code for the GPU" and "no setting works
        around it"
  1052: "This is a property of the pinned build, not of your driver or your
        install"

The second is actively misleading. Since torch 2.8.0+cu128 does list sm_120 in
get_arch_list(), the driver, the platform and the native init path are exactly
where the cause plausibly is, and the doc steered readers away from them.

Both sections now state that the cause is not established, that the pinned
build does contain Blackwell code, and that moving the trio to 2.9.x is the
known workaround for the users who hit it. 5b also gains a get_arch_list()
command so a reader can check their own build instead of trusting a blanket
claim. The fix steps are unchanged.

Also softens the CU128_ARCHS citation in the three comments. That list is
fixture data fed to a mocked torch, captured verbatim from the #1285 report,
so it corroborates the arch list but does not assert what the installed wheel
contains. Saying it "already asserts" overstated it.

Signed-off-by: Moep90 <volleyballlive@googlemail.com>

Signed-off-by: moep90 <volleyballlive@googlemail.com>
2026-09-14 06:37:01 +02:00
moep90 7fe80a5ab8 docs: correct the sm_120 kernel claim behind the #1931 torchaudio guard
Three comments justify the set_audio_backend() hasattr guard by asserting
that the pinned torch 2.8.0 carries no sm_120 kernels, so Blackwell owners
have no choice but to upgrade. The pinned build does carry them.

On a Blackwell GPU (capability (12, 0)) running this repo's own resolved
environment, torch 2.8.0+cu128 reports:

    arch_list   ['sm_70', 'sm_75', 'sm_80', 'sm_86', 'sm_90', 'sm_100', 'sm_120']
    capability  (12, 0)
    cuda_matmul OK

import torch succeeds and CUDA runs. pyproject.toml:282 resolves torch from
the cu128 index, and tests/test_cuda_arch_compat.py:39 already lists sm_120
in CU128_ARCHS, so the repo asserted the opposite of these comments in two
places at once.

#1931 itself reported an `import torch` access violation on Windows with an
sm_120 card, and attributed it to missing kernel support. The symptom was
real and the reporter did resolve it by moving to torch 2.9.1; the stated
mechanism is what got copied into the comments.

The guard's justification survives intact: sm_120 users did land on torch
2.9.x, which brings torchaudio 2.9, which is what removed set_audio_backend().
Only the reason changes. Also drops two related over-attributions in the same
files, where the crash was described as hitting Blackwell or RTX 50-series
machines rather than any machine on torchaudio 2.9.

Comments and one assertion message only. No behaviour change, no test logic
touched. Not re-litigating #1931's fix.

Signed-off-by: Moep90 <volleyballlive@googlemail.com>

Signed-off-by: moep90 <volleyballlive@googlemail.com>
2026-09-14 06:30:19 +02:00
moep90 23df0e80ad docs(audio): correct the torch 2.8.0 sm_120 claim in the TorchCodec comments
Both comments said #1931's cohort has no choice but to leave the torch 2.8.0
pin because it carries no sm_120 kernels. That is wrong for this repo's pin:
pyproject resolves torch==2.8.0 from the cu128 index, whose arch_list
includes sm_120 (confirmed on an sm_120 device, and already asserted by
CU128_ARCHS in tests/test_cuda_arch_compat.py).

State the forcing function that is actually verifiable instead: torch 2.8.0
publishes no aarch64 wheel, so arm64 CUDA hosts land on 2.9 with no 2.8.0
option at all. Comments only, no behaviour change.

Signed-off-by: Moep90 <volleyballlive@googlemail.com>

Signed-off-by: moep90 <volleyballlive@googlemail.com>
2026-09-14 06:22:51 +02:00
moep90 6dafebeebe fix(audio): scale the pydub fallback by the decoded sample width
The fallback divided every decode by 32768.0, which is only correct for
16-bit input. pydub reports 8-bit as sample_width 1 and widens 24-bit to a
full-range int32 (sample_width 4), so a reference clip came back 32768x too
loud for 24- and 32-bit sources and 256x too quiet for 8-bit ones. Nothing
downstream clamps, so the clip silently became noise.

Measured on x86_64 with pydub 0.25.1, peak of a 0.5-amplitude sine:

    PCM_U8   0.00195    PCM_24   32768.0
    PCM_16   0.5        PCM_32   32768.0

Catching ImportError made this reachable for everyone: before, the fallback
ran only for formats torchaudio could not open, but on torchaudio >= 2.9
without TorchCodec it is the only path, turning a loud failure into a silent
wrong result. Fixing it here rather than leaving it for the next reader.

pydub already exposes the right divisor as max_possible_amplitude. No 24-bit
special case is needed: its own comment claims 24-bit values are "not scaled
up to the 32 bit range", but the conversion writes the pad byte first, so it
lands in the LSB and the sample does occupy the full int32 range. Verified
empirically across all five subtypes above.

Also corrects this file's motivation: torch 2.8.0 from the pinned cu128
index does carry sm_120 (arch_list confirmed on an sm_120 device, matching
CU128_ARCHS in tests/test_cuda_arch_compat.py), so the earlier claim that
RTX 50-series owners must leave the pin was wrong.

Raised by CodeRabbit and Greptile on the PR.

Signed-off-by: Moep90 <volleyballlive@googlemail.com>

Signed-off-by: moep90 <volleyballlive@googlemail.com>
2026-09-14 06:22:51 +02:00
Moep90 05ee8c90b2 fix(audio): fall back to soundfile when torchaudio 2.9 routes I/O through TorchCodec
torchaudio >= 2.9 sends save()/load() through TorchCodec, which requires
FFmpeg shared libraries. Without them every generation returns 500 and every
reference-audio read (cloning, watermark, dub) raises ImportError.

Follow-up to #1931: that issue's fix guarded set_audio_backend() but left
load()/save() unprotected, so the RTX 50-series users it identified as having
no choice but to leave the torch 2.8.0 pin still hit a hard failure.

- _safe_torchaudio_save: fall back to the audited _safe_soundfile_write
- _safe_soundfile_write: optional format= passthrough (BytesIO needs it)
- load_audio: catch ImportError so its existing pydub fallback actually runs

backend="soundfile" does not avoid this; 2.9 accepts and ignores it.

Signed-off-by: Moep90 <volleyballlive@googlemail.com>
2026-09-14 06:05:49 +02:00
STINGandCursor 5bad0e765a fix(deps): cap pedalboard below the SIGILL wheels
0.9.21+ Linux wheels are built with -march=native and crash first
synthesis on many CPUs. Keep the last known-good 0.9.20 in lock.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-09-14 05:03:08 +03:00
kevin9327andClaude Opus 5 fc2d811471 docs(changelog): file the entry apart from the other open PRs' lines
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-14 08:51:22 +09:00
kevin9327andClaude Opus 5 d83c4c39b6 docs(changelog): file the entry apart from the other open PRs' lines
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-14 08:51:17 +09:00
kevin9327andClaude Opus 5 dc28c4d52d docs(changelog): file the entry apart from the other open subtitle PRs
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-14 08:51:08 +09:00
kevin9327andClaude Opus 5 b33d9452ee docs(changelog): note the WebVTT paste fix (#2077)
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-14 08:50:25 +09:00
kevin9327andClaude Opus 5 596a0861c2 docs(changelog): note the alembic.ini locale fix (#2075)
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-14 08:50:18 +09:00
kevin9327andClaude Opus 5 651b621e4a test(db): say which installs read alembic.ini
Review follow-up: packaged desktop installs do not ship alembic.ini, so the
locale decode only reaches from-source runs.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-14 08:49:24 +09:00
kevin9327andClaude Opus 5 151d58f024 fix(dub): a pasted WebVTT file maps its cues, not its markup
Dub -> Paste translation -> Load file accepts .vtt and sends timestamped
text to the lenient SRT parser, which is meant to take VTT too. Two
ordinary WebVTT files broke it:

- Cues without an hours field (00:01.000 --> 00:04.500) matched neither
  the frontend's timing detector nor the backend pattern, so the dialog
  mapped WEBVTT, the timing lines and the dialogue as plain translations,
  and the endpoint itself answered "No timed cues found".
- A cue identifier or NOTE block after a cue became part of that cue's
  text, because only digit-only index lines were trimmed.

The hours are now optional in both patterns, as dub_pipeline's yt-dlp
caption parser already allows. For WebVTT input a cue's text ends at its
first blank line, as the format specifies; SRT keeps its lenient
blank-line handling.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-14 08:48:05 +09:00
kevin9327andClaude Opus 5 61f99debb4 fix(db): run migrations on Chinese, Japanese and Korean Windows
alembic reads alembic.ini with encoding="locale". On Python 3.11 that is
the Windows ANSI code page even under PYTHONUTF8=1, which the desktop
shell sets, so the em dashes in the ini's comments raised
UnicodeDecodeError inside Config() on cp932/cp936/cp949/cp950 systems.
_run_alembic_upgrade treats that as "alembic unavailable": every startup
logged "alembic upgrade head skipped: 'cp949' codec can't decode byte
0xe2", never took the pre-migration backup, and never ran a migration --
including the data-healing ones (0006/0007) that the additive column
reconcile cannot replace.

The ini is now ASCII, with a note saying why, and a test parses it in
each of those code pages with the parser alembic builds and pins it
ASCII. Same Python 3.11 locale-decoding class as the .pth fix in #1795.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-14 08:40:02 +09:00
kevin9327andClaude Opus 5 2ce74581c2 docs(changelog): note the subtitle timestamp fix (#2074)
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-14 08:33:10 +09:00
kevin9327andClaude Opus 5 3cd67811e5 fix(subtitles): exported cue times keep their milliseconds
The SRT/VTT formatters in dub_export and openai_compat truncated
(seconds % 1) * 1000. Most decimal times are not exact in binary (2.3 is
2.29999...), so a cue imported as 00:00:02,300 exported as 00:00:02,299:
every such cue moved a millisecond early in the /dub/srt and /dub/vtt
downloads, burned-in subtitles, and /v1/audio/transcriptions srt/vtt.

All four now call srt_parser.format_cue_timestamp, which rounds the whole
value to milliseconds once and splits it, so 59.9996 carries to
00:01:00,000 rather than printing ",1000" -- the same round-then-divmod
shape karaoke_ass._ass_time already uses.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-14 08:31:19 +09:00
kevin9327andClaude Opus 5 d9f51da3c6 docs(structure): count the new text_upload service
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-14 08:30:22 +09:00
kevin9327andClaude Opus 5 ad4b3547d4 fix(import): an undefined Windows-1252 byte no longer spoils the whole file
Review follow-up. A Windows-1252 upload with one of the five bytes cp1252
leaves undefined fell back to decoding the entire file as Latin-1, so its
curly quotes, dashes and euro signs became C1 control characters. Only
the undefined bytes now take their Latin-1 code point, which is what the
browser's windows-1252 decoder does, so readTextFile and
decode_text_upload agree byte for byte.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-14 08:21:52 +09:00
kevin9327andClaude Opus 5 41c50ed9c8 fix(import): the browser-side file pickers read the same encodings
Stories -> Import read the picked file with File.text(), which decodes
UTF-8 only, so a UTF-16 script came back NUL-riddled and a Windows-1252
one as replacement characters. Dub -> Paste translation -> Load file used
FileReader.readAsText(), which handles a UTF-16 BOM but still turned
Windows-1252 accents, dashes and quotes into replacement characters.

Both now go through readTextFile, which applies decode_text_upload's
rule with TextDecoder: a BOM names the encoding, valid UTF-8 stays
UTF-8, anything else is Windows-1252.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-14 08:20:02 +09:00
kevin9327andClaude Opus 5 9d30de27b4 docs(changelog): note the upload encoding fix (#2073)
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-14 08:15:46 +09:00
kevin9327andClaude Opus 5 c8c833062b fix(import): read UTF-16 and Windows-1252 subtitles and manuscripts
/dub/import-srt fell back to Latin-1 when UTF-8 failed, so a UTF-16 .srt
(Notepad's "Unicode", many subtitle editors) decoded with a NUL between
every character and was rejected as having no cues, and a Windows-1252
one turned curly quotes and dashes into C1 control characters.
/audiobook/import decoded .txt/.md with errors="ignore", silently dropping
every accent, dash and curly quote from a Windows-1252 manuscript,
returning NUL-interleaved text for a UTF-16 one, and keeping a UTF-8 BOM
at the start of the editor text.

Both now use decode_text_upload: a BOM names the encoding, valid UTF-8
stays UTF-8, and anything else is read as Windows-1252, with Latin-1 for
the bytes cp1252 leaves undefined so the decode never raises.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-14 08:14:46 +09:00
basil-k-aji-dev a628ba161e fix: install cuDNN 8 compat libraries in the CUDA Docker image
The CUDA base image is pytorch/pytorch:2.8.0-cuda12.8-cudnn9-runtime, so it
ships cuDNN 9. CTranslate2 — WhisperX and faster-whisper — links cuDNN 8, and
its absence aborts the backend process outright rather than raising (#1371).

scripts/setup.py side-loads the cuDNN 8 libraries for source installs, but the
Dockerfile never did, so every CTranslate2 ASR engine was unavailable in Docker
and the demo synthesis timed out with libcudnn_ops_infer.so.8 missing.

Install the same nvidia-cudnn-cu12==8.9.7.29 shim during the image build,
deriving the target from sys.prefix so it matches where backend/core/cudnn8.py
searches rather than hardcoding the conda path — sys.prefix differs between the
conda-based CUDA image and the ROCm venv. Guarded to GPU_FLAVOR=cuda, since
ROCm does not use cuDNN, and --no-deps keeps the base image's torch stack
untouched. A post-install assert fails the build if no .so.8 libraries landed,
rather than letting it resurface as the same runtime warning.

Fixes #2050
2026-09-13 23:51:57 +05:30
rukhaam19 63e647bd91 feat: add desktop toolchain PATH healing script and unit tests 2026-09-13 13:39:16 +05:30
179 changed files with 6133 additions and 443 deletions
+3 -4
View File
@@ -63,10 +63,9 @@ jobs:
experimental: false
- os: macos-14
platform: darwin-arm64
# Metal build path is unpublished upstream (Pitfall 1 in
# 04-RESEARCH.md); experimental so a failed Metal build doesn't
# block — the SPIKE-01 ADR records the in-process fallback.
experimental: true
# Apple Silicon Metal build compiles cleanly with -DGGML_METAL=ON
# at the pinned SHA (#2105); non-experimental to catch regressions.
experimental: false
steps:
- uses: actions/checkout@v4
with:
+28 -4
View File
@@ -6,6 +6,10 @@ on:
tags: ['v*']
workflow_dispatch:
inputs:
release_tag:
description: "Existing version tag to package using this workflow from main (optional)"
type: string
default: ''
publish:
description: "Publish the tagged Electron release after all platforms pass"
type: boolean
@@ -19,9 +23,12 @@ permissions:
contents: read
concurrency:
group: electron-release-${{ github.ref }}
group: electron-release-${{ inputs.release_tag || github.ref_name }}
cancel-in-progress: false
env:
RELEASE_REF: ${{ inputs.release_tag && format('refs/tags/{0}', inputs.release_tag) || github.ref }}
jobs:
validate:
# The transition tag is assembled after the manual Tauri draft succeeds.
@@ -29,9 +36,13 @@ jobs:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
with:
ref: ${{ env.RELEASE_REF }}
- name: Require an exact version tag
env:
REF: ${{ github.ref }}
REF: ${{ env.RELEASE_REF }}
WORKFLOW_REF: ${{ github.ref }}
RELEASE_TAG_OVERRIDE: ${{ inputs.release_tag }}
ALLOW_UNSIGNED: ${{ inputs.allow_unsigned }}
DISPATCH_ACTOR: ${{ github.actor }}
RERUN_ACTOR: ${{ github.triggering_actor }}
@@ -42,8 +53,12 @@ jobs:
echo "Only the repository owner may accept unsigned installers"; exit 1;
}
fi
if [ -n "$RELEASE_TAG_OVERRIDE" ]; then
test "$WORKFLOW_REF" = refs/heads/main || { echo "Tag overrides require the workflow from main"; exit 1; }
fi
VERSION=$(node -p "require('./frontend/package.json').version")
test "$REF" = "refs/tags/v$VERSION" || { echo "Dispatch on the exact version tag"; exit 1; }
test "$REF" = "refs/tags/v$VERSION" || { echo "Select the exact version tag"; exit 1; }
test "$(git rev-parse HEAD)" = "$(git rev-parse "$REF^{commit}")" || { echo "Checkout does not match the release tag"; exit 1; }
package:
needs: validate
runs-on: ${{ matrix.runner }}
@@ -81,6 +96,8 @@ jobs:
CSC_IDENTITY_AUTO_DISCOVERY: 'false'
steps:
- uses: actions/checkout@v4
with:
ref: ${{ env.RELEASE_REF }}
- uses: actions/setup-node@v4
with:
node-version: '22'
@@ -129,6 +146,11 @@ jobs:
CSC_KEY_PASSWORD: ${{ secrets.ELECTRON_CSC_KEY_PASSWORD }}
working-directory: electron
run: |
# An empty CSC_LINK is interpreted as the working directory by the
# signer. Omit absent credentials rather than passing empty strings.
if [ -z "${CSC_LINK:-}" ]; then
unset CSC_LINK CSC_KEY_PASSWORD
fi
bun x electron-builder --config electron-builder.config.mjs ${{ matrix.flags }} --publish never
node tests/packaging-contract.mjs --artifact
node tests/update-package-contract.mjs --platform ${{ matrix.platform }} --arch ${{ matrix.arch }}
@@ -174,11 +196,13 @@ jobs:
contents: write
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
TAG: ${{ github.ref_name }}
TAG: ${{ inputs.release_tag || github.ref_name }}
SUNSET_TAG: ${{ vars.TAURI_SUNSET_TAG }}
PUBLISH: ${{ inputs.publish }}
steps:
- uses: actions/checkout@v4
with:
ref: ${{ env.RELEASE_REF }}
- uses: actions/download-artifact@v4
with:
pattern: electron-release-*
+1
View File
@@ -29,6 +29,7 @@ Binding for every AI agent (Claude, Codex, Cursor, review bots, …). CLAUDE.md
- Local-first: no new required network calls; any HF download gated on installed-ness or explicit user action; all synthetic audio through the `mark_synthetic` chokepoint.
- Every user-facing string via i18n, present in ALL 21 `frontend/src/i18n/locales/*.json` with real translations.
- Docs-sync in the same PR. CHANGELOG Unreleased: quiet one-liners ending `(#N)` + `— thanks @user!` for community work, under a short `**Highlights**` list.
- Tagged release announcements lead with the biggest user-visible change; redesigns need real UI screenshots and migrations need installer links and steps. Verify all contributor credits from the tag comparison and included PRs; list authors and bug reporters separately (see `docs/RELEASING.md`).
- Versioning: `frontend/package.json` is the single source of truth; never bump without the owner asking.
- `frontend/package.json` dep changes require regenerating root `bun.lock` (Docker runs `--frozen-lockfile`).
- Issues: absorb or decline — never defer to a future version. Check the open-PR queue before implementing community-reported fixes.
+79
View File
@@ -8,6 +8,84 @@ the frozen-backend fallback mirror it for their toolchains.
## [Unreleased]
## [0.5.4] — 2026-09-17
**More reliable setup, generation, and dubbing.** This patch detects incomplete desktop runtimes before startup, repairs them without overwriting an existing Tauri installation, and makes engine failures easier to recover from. It also fixes gated model downloads, subtitle imports, and saved-voice language errors.
**Download**
| Platform | Installer |
| --- | --- |
| Windows x64 | [Installer](https://github.com/debpalash/VoiceStudio/releases/download/v0.5.4/VoiceStudio-Electron-0.5.4-win-x64.exe) |
| macOS Apple Silicon | [DMG](https://github.com/debpalash/VoiceStudio/releases/download/v0.5.4/VoiceStudio-Electron-0.5.4-mac-arm64.dmg) |
| macOS Intel | [DMG](https://github.com/debpalash/VoiceStudio/releases/download/v0.5.4/VoiceStudio-Electron-0.5.4-mac-x64.dmg) |
| Linux x64 | [AppImage](https://github.com/debpalash/VoiceStudio/releases/download/v0.5.4/VoiceStudio-Electron-0.5.4-linux-x64.AppImage) · [deb](https://github.com/debpalash/VoiceStudio/releases/download/v0.5.4/VoiceStudio-Electron-0.5.4-linux-x64.deb) |
Already using Electron? Install over your existing app and keep your data. If prompted, choose **Install local runtime** to refresh its dependencies. Moving from Tauri? Back up your data directory with the app closed, install Electron, then verify your voices and projects before removing Tauri. Follow the [migration guide](https://github.com/debpalash/VoiceStudio/blob/v0.5.4/docs/electron-migration.md). Tauri v0.5.3 remains the final Tauri release; its updater cannot install Electron.
**Highlights**
- Repair incomplete Electron runtimes and recover gated model downloads (#2179, #2173)
- More reliable transcription, engine deadlines, and dubbing streams (#2165, #2109, #2111, #2138)
- Preserve subtitle text and legacy manuscript encodings across desktop and web (#2077, #2151, #2073)
- Clear language guidance and capability checks before voice generation (#2172, #2174, #2175)
### Fixed
- Forward saved Hugging Face tokens when downloading gated model weights and dependencies (#2173) — thanks @shivsin25!
- Explain unsupported saved-profile languages consistently in Electron, web, and streaming generation (#2175) — thanks @shivsin25!
- List every installed Kokoro language and accept its displayed name, including British English (#2174) — thanks @drakeo338!
- Validate Python dependencies before reusing a desktop runtime and offer setup for incomplete environments (#2176)
- Check active model cloning support before starting voice conversion (#2147)
- Accept both valid SIGKILL diagnostics in the desktop lifecycle regression check (#2170)
- Repair CTranslate2 loading safely across ASR and translation, and retain the loaded Whisper model during CPU fallback (#2165) — thanks @guruthechosen!
- Avoid pedalboard wheels that crash on unsupported CPU instructions (#2080) — thanks @D3nii!
- Include cuDNN 8 compatibility libraries for CTranslate2 in CUDA containers (#2072) — thanks @basil-k-aji-dev!
- Preserve audio reads, writes, and reference amplitude without TorchCodec (#2083) — thanks @Moep90!
- Give isolated engines request-sized deadlines, validate timeout overrides, and distinguish hangs from crashes (#2109) (#2111) — thanks @SurefireStudios and @LMGXENON!
- Keep dubbing streams alive during quiet steps and delay model cleanup until native refinement ends (#2138) — thanks @denemon!
- Locate ffprobe beside ffmpeg without changing parent directory names (#2107) — thanks @kapelame!
- Resample MLX output chunks to the declared rate before joining them (#2106) — thanks @kapelame!
- Read database migration configuration on Chinese, Japanese, and Korean Windows (#2075) — thanks @kevin9327!
- Preserve milliseconds and carry rounded subtitle timestamps across second boundaries (#2074) — thanks @kevin9327!
- Decode UTF-16 and Windows-1252 subtitle and manuscript imports in Electron, web, and backend routes (#2073) — thanks @kevin9327!
- Preserve numeric subtitle dialogue while recognizing mixed indexed and unindexed cues (#2151) — thanks @shivsin25!
- Parse pasted WebVTT cues while separating metadata, identifiers, empty cues, and complete timing lines (#2077) — thanks @kevin9327!
- Normalize Argos language aliases without silently changing Traditional Chinese to Simplified (#2143, #2152) — thanks @gyanu2507 and @rollroyces!
- Clarify Blackwell import-crash diagnostics without blaming missing kernels (#2084) — thanks @Moep90!
- Distinguish architecture preflight rejection from independent compile-stack failures (#2085) — thanks @Moep90!
- Require the pinned Apple Silicon GGUF build to pass and document runtime preflight conditions (#2115) — thanks @LMGXENON and @martinezpl!
- Correct the Windows Rustup installation command in tooling and documentation (#2066) — thanks @Rukhaam!
- Show local setup guidance when remote native engine installation is unavailable (#2166)
- Show scrubbed native error tails and exit codes for failed dubbing extraction (#2167)
### CI
- Handle missing Electron signing credentials and retry packaging fixes without moving release tags (#2157)
### Contributors
- @D3nii — compatibility with CPUs unsupported by newer pedalboard wheels.
- @LMGXENON — engine deadlines, timeout diagnostics, and Apple Silicon build documentation.
- @Moep90 — audio I/O fallbacks and GPU compatibility guidance.
- @Rukhaam — Windows toolchain setup guidance.
- @Shivendra-Coherent and @shivsin25 — numeric subtitle dialogue, gated downloads, and profile-language errors.
- @SurefireStudios — request-sized sidecar generation deadlines.
- @basil-k-aji-dev — cuDNN compatibility libraries in CUDA containers.
- @denemon — dubbing stream keepalives and cancellation cleanup.
- @guruthechosen — CTranslate2 repair and Whisper CPU recovery.
- @gyanu2507 and @rollroyces — Argos language normalization and regression coverage.
- @kapelame — ffprobe discovery and MLX audio resampling.
- @kevin9327 — subtitle timing, text encodings, WebVTT, and Windows database migrations.
- @drakeo338 — Kokoro supported-language reporting.
- @debpalash — integration, Electron runtime recovery, localization, regression coverage, and release maintenance.
### Bug reports
- Thanks to @YChhunsann, @martinezpl, @denemon, @adeelahmadsiddique, @TehSmoo, @kmsitcomputer, @raya-mansouri, @infinitete, and @kor1998 for the reports behind the fixes above.
## [0.5.3] — 2026-09-17
**Highlights**
@@ -67,6 +145,7 @@ the frozen-backend fallback mirror it for their toolchains.
- Transcribing an M4A file with PyTorch Whisper works, instead of failing with "Format not recognised" (#2042, #2039)
- PyTorch Whisper runs on 6 GB NVIDIA cards instead of falling back to CPU, because its memory check now fits the model it loads (#2044, #2041)
- MCP tools wait as long as the backend does, so a long transcription no longer fails at 120 s with an empty error (#2043, #2040)
- Installing a gated model now sends your Hugging Face token on the fast download path too, so pyannote diarisation and gated engine weights stop failing with "401 Unauthorized" when the token is saved in Settings (#2173, #2163)
- Generating on an older NVIDIA GPU (Tesla T4, and other pre-Ampere cards) no longer kills the backend on the first request — CUDA graphs are not captured below sm_80 (#2135)
- "Disable torch.compile" in Settings → Performance now works on macOS and Linux, not only Windows; it was greyed out on the platforms that needed it (#2135)
- Setting `TORCH_COMPILE_DISABLE=1` in the environment now actually disables torch.compile, for the in-process engine and engine subprocesses alike (#2135)
+3 -1
View File
@@ -44,7 +44,9 @@ For anything new: prefer what's already pinned in `pyproject.toml` / `frontend/p
**Docs-sync (hard rule, owner-set 2026-06-11):** any change that alters something these docs describe — README.md, `.github/CONTRIBUTING.md`, `.github/SECURITY.md`, `.github/SUPPORT.md`, LICENSE, or `docs/**` (install flows, Docker tag semantics, platform support, versioning/release behavior, review process, supported versions) — must update those docs **in the same PR** as the change. If a doc impact is discovered after merge, the docs fix is the immediate next commit, not backlog. Stale docs are treated as bugs.
**Release notes / changelog (hard rule, owner-set 2026-06-16):** every tagged release gets a **high-quality, user-facing `## [X.Y.Z] — DATE` section in `CHANGELOG.md`** before (or in the same hour as) the tag — never the "Auto-generated release for vX.Y.Z…" fallback. `release.yml` extracts that section verbatim as the GitHub Release body (the `Extract CHANGELOG section for tag` step), so a missing/empty section ships a bare release. Quality bar (owner-restyled 2026-07-17, replaces the old bold-lead paragraphs): **quiet and scannable** a short `**Highlights**` bullet list first (plain words, one line each), then `### Changed` / `### Added` / `### Docs` / `### Fixed` / `### License` / `### CI` subsections where each entry is a **single one-liner** with the `(#NNN)` issue/PR ref and contributor credit (`— thanks @user!`) where applicable. Written for users, grouped by theme, no multi-line paragraphs, **not** raw commit dumps. This applies to **preview builds too**: preview release notes summarize what's new on `main` since the last stable, in the same style. Workflow: as features merge, keep `## [Unreleased]` current; at release time rename it to the version + date. If a release was already cut with the fallback body, the next action is to backfill `CHANGELOG.md` **and** `gh release edit <tag>` the live body — not backlog.
**Release notes / changelog (hard rule, owner-set 2026-06-16):** every tagged release gets a **high-quality, user-facing `## [X.Y.Z] — DATE` section in `CHANGELOG.md`** before (or in the same hour as) the tag — never the "Auto-generated release for vX.Y.Z…" fallback. The desktop release workflow extracts that section verbatim as the GitHub Release body, so a missing/empty section ships a bare release. Quality bar (owner-restyled 2026-07-17, replaces the old bold-lead paragraphs): **quiet and scannable change entries** — after an optional tagged-release introduction (see the presentation rule below), a short `**Highlights**` bullet list (plain words, one line each), then `### Changed` / `### Added` / `### Docs` / `### Fixed` / `### License` / `### CI` subsections where each entry is a **single one-liner** with the `(#NNN)` issue/PR ref and contributor credit (`— thanks @user!`) where applicable. Change entries are written for users, grouped by theme, with no multi-line entry paragraphs or raw commit dumps. This applies to **preview builds too**: preview release notes summarize what's new on `main` since the last stable, in the same style. Workflow: as features merge, keep `## [Unreleased]` current; at release time rename it to the version + date. If a release was already cut with the fallback body, the next action is to backfill `CHANGELOG.md` **and** `gh release edit <tag>` the live body — not backlog.
**Release presentation and credits (owner-set 2026-09-17):** Tagged release announcements lead with the biggest user-visible change; redesigns need real UI screenshots and migrations need installer links and steps. Verify all contributor credits from the tag comparison and included PRs; list authors and bug reporters separately (see `docs/RELEASING.md`). Keep Highlights to 35 bullets; the release introduction can include prose, images, and a download table before the concise change entries.
**Localization (hard rule):** No hardcoded non-English (CJK) **user-facing text** anywhere in the codebase except the translation layer (`frontend/src/i18n/`). All UI strings go through i18n (`t('...')` keys in `locales/*.json`); native language names live in `i18n/index.ts` (`LANGUAGES`). Functional CJK is allowed and tracked via the allowlist in `tests/test_no_hardcoded_cjk.py` — text-processing regexes, model/engine vocabulary & identifiers (e.g. CosyVoice speaker IDs), localized error matching, demo/eval data, and test fixtures. CI fails on any hardcoded CJK outside the allowlist; to add legitimate functional CJK, extend `_ALLOWED_FILES` there with a justification.
+10 -6
View File
@@ -1,25 +1,29 @@
# Alembic configuration for VoiceStudio.
# Run from anywhere: alembic -c <repo>/alembic.ini <command>
# Default commands:
# alembic upgrade head apply all pending migrations
# alembic revision -m "" create a new migration
# alembic current show current schema version
# alembic upgrade head - apply all pending migrations
# alembic revision -m "..." - create a new migration
# alembic current - show current schema version
#
# Keep this file ASCII: alembic reads it in the locale code page, which on a
# Chinese, Japanese or Korean Windows cannot decode UTF-8 punctuation
# (tests/test_alembic_ini_locale.py).
#
# DB URL is resolved dynamically from core.config (env aware).
# See backend/migrations/env.py.
[alembic]
# %(here)s = this file's directory. Alembic resolves bare relative paths
# against the process CWD, not the ini and the app doesn't always start
# against the process CWD, not the ini - and the app doesn't always start
# from the repo root (`tauri dev` runs the backend with
# cwd=frontend/src-tauri), which made startup migrations die with
# "Path doesn't exist: backend/migrations" the first time one was pending.
script_location = %(here)s/backend/migrations
prepend_sys_path = %(here)s/backend
# Split multi-path options on os.pathsep, not the legacy space/comma/colon
# set a colon-split would shred "C:\..." absolute paths on Windows.
# set - a colon-split would shred "C:\..." absolute paths on Windows.
path_separator = os
# sqlalchemy.url is set programmatically in env.py do NOT set it here.
# sqlalchemy.url is set programmatically in env.py - do NOT set it here.
sqlalchemy.url =
[loggers]
+2 -1
View File
@@ -186,7 +186,8 @@ async def audiobook_import(file: UploadFile = File(...)) -> dict:
except ValueError as e:
raise HTTPException(status_code=400, detail=f"couldn't parse PDF: {e}")
else:
script = chapterize_plaintext(data.decode("utf-8", "ignore"))
from services.text_upload import decode_text_upload
script = chapterize_plaintext(decode_text_upload(data))
if not script.strip():
raise HTTPException(status_code=400, detail="no text found in the file")
plan = parse_audiobook_script(script)
+8 -5
View File
@@ -1030,11 +1030,14 @@ async def retry_batch_job(job_id: str):
if not translation_engines.is_ready(provider):
raise HTTPException(409, "Configure the selected translation provider before retrying")
if provider == "argos" and job.get("source_lang"):
status = await asyncio.to_thread(
translation_engines.argos_pack_status,
job["source_lang"],
job["langs"],
)
try:
status = await asyncio.to_thread(
translation_engines.argos_pack_status,
job["source_lang"],
job["langs"],
)
except ValueError as exc:
raise HTTPException(status_code=422, detail=str(exc)) from exc
if any(not pair["installed"] for pair in status["pairs"]):
raise HTTPException(409, "Install the required Argos language packs before retrying")
+104 -34
View File
@@ -272,12 +272,10 @@ async def dub_import_srt(job_id: str, file: UploadFile = File(...)):
raise HTTPException(status_code=400, detail=f"Could not read uploaded file: {e}") from e
if not raw_bytes:
raise HTTPException(status_code=400, detail="Uploaded SRT file is empty.")
# Most SRT files are UTF-8 (with or without BOM); fall back to latin-1
# so legacy Windows-encoded subs don't blow up the import.
try:
text = raw_bytes.decode("utf-8-sig")
except UnicodeDecodeError:
text = raw_bytes.decode("latin-1", errors="replace")
# Most SRT files are UTF-8, but Windows subtitle tools also save UTF-16
# (with a BOM) and Windows-1252; decode those instead of finding no cues.
from services.text_upload import decode_text_upload
text = decode_text_upload(raw_bytes)
from services.srt_parser import parse_srt
result = parse_srt(text)
@@ -869,9 +867,73 @@ _CHUNK_TRANSCRIBE_ATTEMPTS = max(1, int(os.environ.get("OMNIVOICE_TRANSCRIBE_CHU
#: by Chrome's ~5 min no-response cap and by reverse-proxy idle timeouts,
#: which the UI can only report as the generic "stream dropped" guess.
ASR_LOAD_KEEPALIVE_S = float(os.environ.get("OMNIVOICE_ASR_LOAD_KEEPALIVE_S", "15.0"))
#: Seconds between `ping` events while a post-transcript step runs on an
#: executor (diarization, clone extraction, reference-text refinement, the
#: ASR unload / TTS restore). #2108: the per-segment refinement re-runs ASR
#: once per segment — 113 passes, ~19 min on an M1 Pro CPU — and was awaited
#: bare, so the stream went byte-silent for the whole stretch, the webview
#: severed it, and the UI could only say "stream ended early".
POST_ASR_PING_S = 5.0
_sse_event = dub_pipeline.sse_event
class _ASRWorkLifetime:
"""Keep model cleanup behind native work even after its waiter is cancelled."""
def __init__(self):
import threading
self._lock = threading.Lock()
self._closed = threading.Event()
self._cleaned = False
def run(self, fn):
with self._lock:
if self._closed.is_set():
raise RuntimeError("Transcription stream has ended")
return fn()
def stop(self):
# Reject queued work before it can touch an unloaded model.
self._closed.set()
def cleanup(self, fn):
self.stop()
with self._lock:
if self._cleaned:
return
# Claim cleanup before invoking even a non-idempotent unload.
self._cleaned = True
return fn()
async def _ping_while(fut):
"""Yield `ping` events every POST_ASR_PING_S until ``fut`` settles.
Every await in the transcribe stream body that can outlast a few seconds
goes through here so the connection never goes byte-silent. The result
(or exception) stays on ``fut`` for the caller to read.
Leaving early — the client disconnected, or the body raised at a `yield` —
cancels ``fut`` exactly as a bare ``await fut`` would have, so a wrapped
run_transcribe_guarded still runs its abandon path instead of refining on
after disconnection. Native threads are not cancelled by Future.cancel();
_ASRWorkLifetime orders model cleanup behind their actual completion. Nothing is awaited in the finally: it also runs under GeneratorExit.
A failure that lands after we left is still marked retrieved, so it cannot
surface later as "exception was never retrieved" (CodeRabbit, #2138) —
the same done-callback the TTS-load keepalive uses.
"""
fut.add_done_callback(lambda f: f.cancelled() or f.exception())
try:
while True:
done, _ = await asyncio.wait({fut}, timeout=POST_ASR_PING_S)
if done:
return
yield _sse_event("ping", {})
finally:
if not fut.done():
fut.cancel()
_prep_event_helper = dub_pipeline.prep_event # alias; we keep the module-local _prep_event below for the inline one-liner shape
#: User-facing warning emitted when auto voice cloning is skipped because the
@@ -1045,6 +1107,7 @@ async def dub_transcribe_stream(
# _gen_body parks the loaded backend here; the normal unload clears it;
# gen()'s `finally` unloads whatever is still parked, on EVERY exit.
_loaded_asr: dict = {"backend": None}
_asr_work = _ASRWorkLifetime()
# Same shape, same reason, for the TTS offload (#1191): offload_tts_for_asr()
# moves the TTS model to CPU, and only _gen_body's success path moved it
# back — so an abort/error/disconnect stranded it there, silently making
@@ -1447,7 +1510,7 @@ async def dub_transcribe_stream(
for _attempt in range(1, _CHUNK_TRANSCRIBE_ATTEMPTS + 1):
# Run as a task and poll so pings keep the EventSource alive.
task = asyncio.ensure_future(run_transcribe_guarded(
_gpu_pool, _transcribe_chunk,
_gpu_pool, lambda: _asr_work.run(_transcribe_chunk),
what=f"Dub chunk {i + 1}/{chunks_n}",
timeout=transcribe_timeout_s,
timeout_env=transcribe_timeout_env,
@@ -1855,16 +1918,10 @@ async def dub_transcribe_stream(
"heuristic",
)
fut_diar = loop.run_in_executor(_gpu_pool, _diarize)
final_segs = None
diar_warning = None
labels_source = "heuristic"
while True:
done, pending = await asyncio.wait([fut_diar], timeout=5.0)
if done:
final_segs, diar_warning, labels_source = done.pop().result()
break
yield _sse_event("ping", {})
fut_diar = loop.run_in_executor(_gpu_pool, lambda: _asr_work.run(_diarize))
async for _ping in _ping_while(fut_diar):
yield _ping
final_segs, diar_warning, labels_source = fut_diar.result()
if job.get("aborted") or task_manager.is_cancelled(job_id):
yield _sse_event("aborted", {})
return
@@ -1931,12 +1988,9 @@ async def dub_transcribe_stream(
labels_source=labels_source,
),
)
while True:
done, pending = await asyncio.wait([fut_clones], timeout=5.0)
if done:
clones = done.pop().result()
break
yield _sse_event("ping", {})
async for _ping in _ping_while(fut_clones):
yield _ping
clones = fut_clones.result()
if clones:
from services.speaker_clone import refine_ref_texts
# Bound the re-transcribe like every other ASR dispatch in
@@ -1946,11 +2000,14 @@ async def dub_transcribe_stream(
# and raises — keep the original (unrefined) clones, matching
# refine_ref_text's own "failure is a strict no-op" fallback.
try:
clones = await run_transcribe_guarded(
fut_refine = asyncio.ensure_future(run_transcribe_guarded(
_gpu_pool,
lambda: refine_ref_texts(clones, _asr_backend),
lambda: _asr_work.run(lambda: refine_ref_texts(clones, _asr_backend)),
what="Dub clone ref-text refine",
)
))
async for _ping in _ping_while(fut_refine):
yield _ping
clones = fut_refine.result()
except ASRTimeoutError as e:
logger.warning(
"clone ref-text refine timed out; keeping original ref_text: %s", e
@@ -1973,23 +2030,29 @@ async def dub_transcribe_stream(
# reviewers, on the first version of this fix).
_seg_clone_dir = _safe_job_dir(job_id) or os.path.dirname(vocals_for_clone)
os.makedirs(_seg_clone_dir, exist_ok=True)
seg_clones = await loop.run_in_executor(
fut_seg_refs = loop.run_in_executor(
_cpu_pool, lambda: extract_segment_refs(
vocals_for_clone, final_segs,
_seg_clone_dir,
seg_ids=seg_ids_for_clone,
),
)
async for _ping in _ping_while(fut_seg_refs):
yield _ping
seg_clones = fut_seg_refs.result()
if seg_clones:
from services.speaker_clone import refine_ref_texts
# Same guard as the per-speaker refine above (#730):
# keep the original seg_clones on a wedge/timeout.
try:
seg_clones = await run_transcribe_guarded(
fut_refine = asyncio.ensure_future(run_transcribe_guarded(
_gpu_pool,
lambda: refine_ref_texts(seg_clones, _asr_backend),
lambda: _asr_work.run(lambda: refine_ref_texts(seg_clones, _asr_backend)),
what="Dub segment ref-text refine",
)
))
async for _ping in _ping_while(fut_refine):
yield _ping
seg_clones = fut_refine.result()
except ASRTimeoutError as e:
logger.warning(
"segment ref-text refine timed out; keeping original ref_text: %s", e
@@ -2044,13 +2107,19 @@ async def dub_transcribe_stream(
# (CodeRabbit review, #1198 — normal-completion half).
if _asr_backend:
try:
await loop.run_in_executor(_gpu_pool, _asr_backend.unload)
fut_unload = loop.run_in_executor(_gpu_pool, lambda: _asr_work.cleanup(_asr_backend.unload))
async for _ping in _ping_while(fut_unload):
yield _ping
fut_unload.result()
except Exception as e:
logger.warning("Failed to unload ASR backend: %s", e)
# Unload attempted once — don't retry from gen()'s finally.
_loaded_asr["backend"] = None
await loop.run_in_executor(_cpu_pool, restore_tts_after_asr)
fut_restore = loop.run_in_executor(_cpu_pool, restore_tts_after_asr)
async for _ping in _ping_while(fut_restore):
yield _ping
fut_restore.result()
# Debt paid — don't make gen()'s finally repeat it.
_tts_offloaded["v"] = False
@@ -2092,6 +2161,7 @@ async def dub_transcribe_stream(
yield _sse_event("error", stream_failure("transcription_failed"))
yield _sse_event("done", {})
finally:
_asr_work.stop()
# Last-resort VRAM release (see _loaded_asr above): covers crashes,
# early terminal-error returns, and client disconnects
# (GeneratorExit bypasses the except, never this finally).
@@ -2117,7 +2187,7 @@ async def dub_transcribe_stream(
# (CodeRabbit review, #1198).
try:
_fut = asyncio.get_running_loop().run_in_executor(
_gpu_pool, _b.unload
_gpu_pool, lambda: _asr_work.cleanup(_b.unload)
)
# Restore the TTS model only AFTER the ASR weights are
# freed — the same ordering the success path enforces, so
@@ -2126,7 +2196,7 @@ async def dub_transcribe_stream(
except RuntimeError:
# No running loop (interpreter teardown) — best effort.
try:
_b.unload()
_asr_work.cleanup(_b.unload)
except Exception as e:
logger.warning("Failed to unload ASR backend: %s", e)
_submit_tts_restore()
+20 -12
View File
@@ -70,6 +70,13 @@ def _unique_stamp() -> str:
_SAFE_LANG = re.compile(r"^[A-Za-z0-9_-]{1,32}$")
#: Seconds of silence on a `/tasks/stream` before a keepalive comment goes out.
#: A task that is busy but quiet — ffmpeg on a long video, a slow TTS segment,
#: a job queued behind another — leaves the stream byte-silent, and byte-silent
#: SSE gets severed by the desktop webview, Chrome's ~5 min cap or a proxy's
#: idle timeout (#1196, #2108). Comments are invisible to every consumer.
TASK_STREAM_KEEPALIVE_S = 15.0
def _job_dir_or_400(job_id: str) -> str:
if not re.fullmatch(r"[A-Za-z0-9_-]{1,64}", job_id or ""):
@@ -273,14 +280,21 @@ async def stream_task(task_id: str, after_seq: int = 0):
await task_manager.add_listener(task_id, q)
try:
while True:
evt = await q.get()
try:
evt = await asyncio.wait_for(q.get(), timeout=TASK_STREAM_KEEPALIVE_S)
except asyncio.TimeoutError:
yield ": keepalive\n\n"
continue
if evt is None:
break
yield evt
finally:
await task_manager.remove_listener(task_id, q)
return StreamingResponse(_reader(), media_type="text/event-stream")
return StreamingResponse(
_reader(), media_type="text/event-stream",
headers={"Cache-Control": "no-cache, no-transform", "X-Accel-Buffering": "no"},
)
@router.get("/jobs")
@@ -1722,11 +1736,8 @@ async def dub_download_audio(
def _format_srt_time(seconds):
h = int(seconds // 3600)
m = int((seconds % 3600) // 60)
s = int(seconds % 60)
ms = int((seconds % 1) * 1000)
return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}"
from services.srt_parser import format_cue_timestamp
return format_cue_timestamp(seconds, ",")
def _pick_subtitle_text(seg: dict, dual: bool) -> str:
"""One line per subtitle cue, unless dual=true and an original exists.
@@ -1810,11 +1821,8 @@ async def dub_export_srt(
)
def _format_vtt_time(seconds):
h = int(seconds // 3600)
m = int((seconds % 3600) // 60)
s = int(seconds % 60)
ms = int((seconds % 1) * 1000)
return f"{h:02d}:{m:02d}:{s:02d}.{ms:03d}"
from services.srt_parser import format_cue_timestamp
return format_cue_timestamp(seconds, ".")
@router.get("/dub/vtt/{job_id}")
@router.get("/dub/vtt/{job_id}/{filename}")
+22
View File
@@ -790,6 +790,28 @@ async def dub_translate(req: TranslateRequest):
f"switch the Engine dropdown to another provider."
)
return JSONResponse(status_code=400, content={"error": friendly})
# The package imports without its native dep; the *translator*
# needs CTranslate2, whose library is rejected outright by kernels
# that refuse an executable stack (#692). Repair it (a one-bit ELF
# patch), and if that is impossible say so in one actionable 400
# instead of the opaque 500 every segment used to produce.
try:
from core.execstack import ensure_ctranslate2_loadable
ensure_ctranslate2_loadable()
except Exception as e: # noqa: BLE001 — repair must not block translation
logger.debug("exec-stack repair unavailable (%s) — continuing", e)
try:
import argostranslate.translate # noqa: F401
except Exception as e: # noqa: BLE001 — OSError here, not ImportError
friendly = (
f"The '{provider}' engine's CTranslate2 runtime could not be "
"loaded in this backend."
+ " Switch the Engine dropdown to NLLB (local) or an online "
"provider, or reinstall the backend, then retry."
)
return JSONResponse(status_code=400, content={"error": friendly, "detail": {"code": "argos_runtime_unavailable", "message": friendly}})
target_codes = list(dict.fromkeys(
seg.target_lang if seg.target_lang else req.target_lang
for seg in req.segments
+24 -10
View File
@@ -21,12 +21,12 @@ import os
import threading
from time import perf_counter
from fastapi import APIRouter, Depends, HTTPException
from fastapi import APIRouter, Depends, HTTPException, Request
from huggingface_hub import utils as hf_utils
from huggingface_hub.errors import HFValidationError
from pydantic import BaseModel, Field
from api.dependencies import require_admin, require_admin_action, require_desktop
from api.dependencies import require_admin, require_admin_action, require_desktop, is_loopback
from core import prefs
from core.engine_licenses import LICENSE_GATED_ENGINES
from services import tts_backend, asr_backend, llm_backend, translation_engines
@@ -85,6 +85,9 @@ def _family_payload(family: str, module):
for backend in backends:
engine_id = backend.get("id")
if engine_id == active == "mlx-audio":
# Constructor resolves model preferences only; never loads weights.
backend["supports_cloning"] = tts_backend.MLXAudioBackend().supports_cloning
if engine_id in LICENSE_GATED_ENGINES:
backend["license_required"] = True
try:
@@ -116,18 +119,29 @@ def _is_hf_repo_id(value: str) -> bool:
return True
def _request_install_capability(payload, request):
allowed = bool(request and request.client and is_loopback(request.client.host))
result = dict(payload)
result["backends"] = [dict(entry) for entry in payload["backends"]]
for entry in result["backends"]:
if entry.get("one_click_install") and not allowed:
entry["one_click_install"] = False
entry["local_install_required"] = True
return result
@router.get("/engines")
def list_all_engines():
def list_all_engines(request: Request):
return {
"tts": _family_payload("tts", tts_backend),
"tts": _request_install_capability(_family_payload("tts", tts_backend), request),
"asr": _family_payload("asr", asr_backend),
"llm": _family_payload("llm", llm_backend),
}
@router.get("/engines/tts")
def list_tts_backends():
return _family_payload("tts", tts_backend)
def list_tts_backends(request: Request):
return _request_install_capability(_family_payload("tts", tts_backend), request)
@router.get(
@@ -409,10 +423,10 @@ async def uninstall_translation_engine(engine_id: str):
"/engines/audiocpp/runtime/install/status",
dependencies=[Depends(require_admin)],
)
def audiocpp_runtime_install_status():
def audiocpp_runtime_install_status(request: Request):
from services import audiocpp_runtime_install
return audiocpp_runtime_install.status()
return {**audiocpp_runtime_install.status(), "install_allowed": bool(request.client and is_loopback(request.client.host))}
@router.post(
@@ -485,7 +499,7 @@ def install_sidecar_engine(engine_id: str):
"/engines/sidecar/{engine_id}/install/status",
dependencies=[Depends(require_admin)],
)
def sidecar_install_status(engine_id: str):
def sidecar_install_status(engine_id: str, request: Request = None):
"""Step-by-step status of the sidecar install job (poll while running).
Shape: ``{engine_id, installed, managed, install_dir, job}`` where job is
@@ -494,7 +508,7 @@ def sidecar_install_status(engine_id: str):
"""
from services import sidecar_install
try:
return sidecar_install.get_status(engine_id)
return {**sidecar_install.get_status(engine_id), "install_allowed": bool(request and request.client and is_loopback(request.client.host))}
except KeyError:
raise HTTPException(
status_code=404,
+155 -15
View File
@@ -143,7 +143,7 @@ def _resolve_profile_conditioning(row, *, ref_text=None, instruct=None,
out = {
"ref_audio_path": None, "ref_text": ref_text, "instruct": instruct,
"seed": seed, "language": language, "kind": None,
"persist_ref_text": False,
"persist_ref_text": False, "language_from_profile": False,
}
# `kind` is authoritative (0005): 'design' profiles condition on their
# deterministic rendered sample + instruct; 'clone' on the user's
@@ -212,6 +212,12 @@ def _resolve_profile_conditioning(row, *, ref_text=None, instruct=None,
prof_lang = None
if prof_lang and prof_lang != "Auto":
out["language"] = prof_lang
# #2156: record that the caller never asked for this language. The
# UI omits `language` entirely while its picker reads "Auto", so a
# profile-filled language must not be reported back as if the user
# had picked it — an engine that can't speak it would otherwise
# tell them to "leave language as Auto", which is what they did.
out["language_from_profile"] = True
return out
@@ -807,6 +813,7 @@ def _oom_friendly_reraise(e):
def _generate_timeout_s(
text: str,
*,
engine: object = None,
execution_device=None,
min_vram_gb=0.0,
hardware_family=None,
@@ -827,6 +834,7 @@ def _generate_timeout_s(
from services.model_manager import generate_timeout_s
return generate_timeout_s(
text,
engine=engine,
execution_device=execution_device,
min_vram_gb=min_vram_gb,
hardware_family=hardware_family,
@@ -1041,6 +1049,18 @@ _LANGUAGE_REJECTION_SIGNATURES = (
"unsupported language code",
)
# Engine-specific rejections that ALREADY name the engine and what it supports,
# so #1257's generic rewrite deliberately leaves them alone — re-wrapping them
# only nests "Engine's own message:" twice. They still have to be recognised as
# language rejections for #2156's provenance check, which cares about the
# *cause* of the language, not the quality of the wording.
_SELF_DESCRIBING_LANGUAGE_REJECTIONS = (
# services/tts_backend.py: "…doesn't support language='Persian'. Kokoro
# supports: …" — mlx-audio's Kokoro, the engine reported in #2156.
"doesn't support language",
"does not support language",
)
#: `unsupported language: xx` / `unsupported language 'xx'` — but not
#: `unsupported language model ...`.
_LANGUAGE_REJECTION_RE = re.compile(
@@ -1050,15 +1070,87 @@ _LANGUAGE_REJECTION_RE = re.compile(
)
def _is_language_rejection(text: str) -> bool:
"""True when an engine failure is about the LANGUAGE it was handed.
Matched on the message, not the type: the engines multiplex third-party
libraries that each raise their own class. Covers the self-describing
wordings too #1257's rewrite skips those, but #2156 still needs to know a
language was refused so it can say where that language came from.
"""
low = text.lower()
return (
any(sig in low for sig in _LANGUAGE_REJECTION_SIGNATURES)
or any(sig in low for sig in _SELF_DESCRIBING_LANGUAGE_REJECTIONS)
or bool(_LANGUAGE_REJECTION_RE.search(text))
)
def _profile_language_rejection_detail(exc: BaseException, language) -> str:
"""The 400 body for a language the *voice profile* supplied, not the user.
#2156: the UI omits `language` while its picker reads "Auto", and #533
fills that gap from the selected profile. When the active engine can't
speak the profile's language the engine's own message tells the user to
"leave language as 'Auto'" which is exactly what they did, so the advice
cannot be acted on. Name the real source and the remedies that exist.
"""
return (
f"This voice profile is saved with the language '{language}', and the "
f"active engine can't speak it. The language picker being on \"Auto\" "
f"does not override that — Auto fills the language in from the "
f"profile. Set this voice profile's language to one the engine "
f"supports, pick a supported language explicitly for this render, or "
f"switch engine in Model Catalogue (the VoiceStudio engine has the "
f"widest coverage). Engine's own message: {exc}"
)
def _language_rejection_payload(exc, language, *, from_profile):
"""Stable, non-retryable error metadata for both response transports."""
from core.public_errors import stream_failure
failure = stream_failure("invalid_request")
failure["terminal"] = True
if from_profile:
failure.update(
code="profile_language_rejected", language=language,
detail=_profile_language_rejection_detail(_root_language_error(exc), language),
)
return failure
def _language_rejection_http_error(exc: BaseException, language, *, from_profile):
"""The 400 a refused language deserves, wherever the refusal was raised.
A language an engine cannot speak is never retryable not by waiting, and
not by re-running the same request on another machine. Built here so the
local (`ValueError`) and remote (`RemoteJobFailed`) handlers cannot drift:
#2156 shipped the profile-aware branch on the local path only, and a remote
render kept answering with a retryable 503 that offered "run it on this
machine instead", which cannot help.
"""
root = _root_language_error(exc)
detail = (
_profile_language_rejection_detail(root, language)
if from_profile else str(exc)
)
if from_profile:
detail = {"code": "profile_language_rejected", "language": language, "message": detail}
return HTTPException(status_code=400, detail=detail)
def _language_rejection_or(e: BaseException, backend, language):
"""``e`` rewritten with engine context when it's a language rejection.
Returns ``e`` unchanged otherwise, so this is safe to wrap any failure in.
Matched on the message, not the type: the engines multiplex third-party
libraries that each raise their own class.
Deliberately narrower than :func:`_is_language_rejection`: a message that
already names its engine and the languages it supports is left alone rather
than nested inside a second "Engine's own message:".
"""
text = str(e)
low = text.lower()
if any(sig in low for sig in _SELF_DESCRIBING_LANGUAGE_REJECTIONS):
return e
if not any(sig in low for sig in _LANGUAGE_REJECTION_SIGNATURES) and not (
_LANGUAGE_REJECTION_RE.search(text)
):
@@ -1067,13 +1159,23 @@ def _language_rejection_or(e: BaseException, backend, language):
type(backend), "id", type(backend).__name__
)
requested = f" '{language}'" if language else ""
return ValueError(
rewritten = ValueError(
f"The {engine} engine can't speak{requested}. VoiceStudio offers every "
f"language its default engine supports, but each engine covers a "
f"different set — pick one this engine supports, or switch engine in "
f"Model Catalogue (the VoiceStudio engine has the widest coverage) "
f"and generate again. Engine's own message: {e}"
)
# Keep the engine's own text reachable. #2156's profile message quotes the
# engine once; without this it would quote THIS wrapper, repeating both the
# engine-switch remedy and "Engine's own message:" twice.
rewritten.engine_language_error = e
return rewritten
def _root_language_error(exc: BaseException) -> BaseException:
"""The engine's own rejection, unwrapping :func:`_language_rejection_or`."""
return getattr(exc, "engine_language_error", exc)
def _persist_profile_ref_text(profile_id: str, ref_text: str) -> None:
@@ -1568,6 +1670,9 @@ async def generate_speech(
ref_lease = None
used_seed = seed
resolved_profile_id = None
# #2156: True once a profile's stored language fills a language the caller
# never sent, so a rejection can name the profile instead of the picker.
language_from_profile = False
history_mode = None # profile.kind when a profile drives; else inferred at insert
# #1032: profile id to persist an auto-transcribed reference transcript to.
# Set only for a plain (unlocked) clone profile whose stored ref_text is
@@ -1596,6 +1701,7 @@ async def generate_speech(
instruct = _cond["instruct"]
used_seed = _cond["seed"]
language = _cond["language"]
language_from_profile = _cond["language_from_profile"]
if _cond["persist_ref_text"]:
persist_ref_text_profile_id = profile_id
elif ref_audio is not None:
@@ -1752,6 +1858,7 @@ async def generate_speech(
what="TTS generate",
timeout=_generate_timeout_s(
text,
engine=_backend,
execution_device=_routing["effective_device"],
min_vram_gb=_engine_min_vram_gb,
hardware_family=_routing_hardware_family,
@@ -1869,10 +1976,14 @@ async def generate_speech(
# holding what is often its only slot until the lease lapses.
render.cancel()
raise
except ValueError:
except ValueError as e:
logger.error("Remote generation request rejected")
from core.public_errors import stream_failure
yield _line({"type": "error", **stream_failure("invalid_request")})
failure = (
_language_rejection_payload(e, language, from_profile=language_from_profile)
if _is_language_rejection(str(e)) else stream_failure("invalid_request")
)
yield _line({"type": "error", **failure})
except gpu_gateway.ModelNotDownloaded as e:
logger.warning("Remote model missing on %s", _target_label)
from core.public_errors import stream_failure
@@ -1888,13 +1999,16 @@ async def generate_speech(
except gpu_gateway.RemoteJobFailed as e:
logger.error("Remote generate failed on %s", _target_label)
from core.public_errors import stream_failure
yield _line({
"type": "error",
**stream_failure("generation_failed"),
"retryable": True,
"target_label": e.worker_label or _target_label,
"hint": e.hint,
})
if _is_language_rejection(str(e)):
yield _line({"type": "error", **_language_rejection_payload(
e, language, from_profile=language_from_profile,
)})
else:
yield _line({
"type": "error", **stream_failure("generation_failed"),
"retryable": True, "target_label": e.worker_label or _target_label,
"hint": e.hint,
})
except Exception as exc:
# Mid-job remote failure is NOT quietly redone here: the client
# treats a retryable error as "surface it", so the user decides
@@ -2052,6 +2166,7 @@ async def generate_speech(
min_vram_gb=_engine_min_vram_gb,
timeout=_generate_timeout_s(
text,
engine=_backend,
execution_device=_routing["effective_device"],
min_vram_gb=_engine_min_vram_gb,
hardware_family=_routing_hardware_family,
@@ -2078,6 +2193,7 @@ async def generate_speech(
min_vram_gb=_engine_min_vram_gb,
timeout=_generate_timeout_s(
text,
engine=_backend,
execution_device=_routing["effective_device"],
min_vram_gb=_engine_min_vram_gb,
hardware_family=_routing_hardware_family,
@@ -2124,6 +2240,7 @@ async def generate_speech(
# even after the v0.3.22 scaled budget shipped.
timeout=_generate_timeout_s(
chunk_text,
engine=_backend,
execution_device=_routing["effective_device"],
min_vram_gb=_engine_min_vram_gb,
hardware_family=_routing_hardware_family,
@@ -2200,10 +2317,14 @@ async def generate_speech(
failure = stream_failure("generation_timeout")
failure["retry_after"] = 30
yield _line({"type": "error", **failure})
except ValueError:
except ValueError as e:
logger.error("Streaming generation request rejected")
from core.public_errors import stream_failure
yield _line({"type": "error", **stream_failure("invalid_request")})
failure = (
_language_rejection_payload(e, language, from_profile=language_from_profile)
if _is_language_rejection(str(e)) else stream_failure("invalid_request")
)
yield _line({"type": "error", **failure})
except Exception as exc:
# A streaming request answers 200 and carries its failure as an
# in-band error frame, so it never reaches the global 500
@@ -2290,6 +2411,7 @@ async def generate_speech(
_local_render, what="TTS generate",
timeout=_generate_timeout_s(
text,
engine=_backend,
execution_device=_routing["effective_device"],
min_vram_gb=_engine_min_vram_gb,
hardware_family=_routing_hardware_family,
@@ -2385,6 +2507,15 @@ async def generate_speech(
# the client can offer "run it on this machine instead" — a resubmit
# the user chose, with a wait they were told about.
logger.error("Remote generate failed on %s: %s", _target_label, e)
# #2156: a language the engine can't speak is a request problem, not a
# worker problem. It travels home as RemoteJobFailed — caught here,
# ahead of the ValueError branch below — so without this the user is
# told to retry on this machine, where the same engine refuses the same
# language. Answer it as the 400 it is, on either path.
if _is_language_rejection(str(e)):
raise _language_rejection_http_error(
e, language, from_profile=language_from_profile
) from e
raise HTTPException(
status_code=503,
detail=f"{e} {e.hint or 'Run it on this machine instead, or pick another GPU.'}",
@@ -2419,6 +2550,15 @@ async def generate_speech(
raise HTTPException(status_code=503, detail=str(e)) from e
except ValueError as e:
logger.error("Validation failed: %s", e)
# #2156: the language the engine refused was never chosen by the user —
# it came from the selected voice profile because the picker was on
# "Auto". The engine's own remedy ("leave language as 'Auto'") is then
# unfollowable, so say where the language actually came from. Only this
# scope knows that; the engine adapters never see the provenance.
if language_from_profile and _is_language_rejection(str(e)):
raise _language_rejection_http_error(
e, language, from_profile=True
) from e
# Most ValueErrors here are VoiceStudio's own validation messages and
# are exactly what the user should read. A few are raw library text
# naming parameters and files the user cannot act on — those get the
+4 -10
View File
@@ -721,17 +721,11 @@ def list_voices():
def _format_ts_srt(seconds: float) -> str:
"""Format seconds as SRT timestamp: HH:MM:SS,mmm"""
h = int(seconds // 3600)
m = int((seconds % 3600) // 60)
s = int(seconds % 60)
ms = int((seconds % 1) * 1000)
return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}"
from services.srt_parser import format_cue_timestamp
return format_cue_timestamp(seconds, ",")
def _format_ts_vtt(seconds: float) -> str:
"""Format seconds as VTT timestamp: HH:MM:SS.mmm"""
h = int(seconds // 3600)
m = int((seconds % 3600) // 60)
s = int(seconds % 60)
ms = int((seconds % 1) * 1000)
return f"{h:02d}:{m:02d}:{s:02d}.{ms:03d}"
from services.srt_parser import format_cue_timestamp
return format_cue_timestamp(seconds, ".")
+10 -1
View File
@@ -230,7 +230,16 @@ def _segmented_snapshot(repo_id: str, *, endpoint: "str | None", revision: str)
from services.segmented_download import segmented_download
from services.token_resolver import resolve as _resolve_token
token = _resolve_token()
# `resolve()` returns a ResolvedToken record, not the bearer string, and
# every consumer below is typed `token: str | None`. Handing over the
# record fails silently rather than loudly (#2163): huggingface_hub's
# build_hf_headers ignores a non-str token and falls back to its own
# ambient discovery, so a token held only in VoiceStudio's settings sends
# NO Authorization header at all and every gated file 401s; our own
# segmented_download interpolates it into `f"Bearer {token}"` and sends a
# malformed header carrying the raw secret. Unwrap once, here.
_resolved = _resolve_token()
token = _resolved.token if _resolved else None
api = HfApi(endpoint=endpoint, token=token)
info = api.repo_info(repo_id, repo_type="model", revision=revision)
commit = info.sha
+1
View File
@@ -337,6 +337,7 @@ async def convert_speech(
what="Voice convert",
timeout=_generate_timeout_s(
text,
engine=backend,
execution_device=compute_profile["effective_device"],
min_vram_gb=compute_profile["min_vram_gb"],
hardware_family=compute_profile.get("runtime_hardware_family"),
+245
View File
@@ -0,0 +1,245 @@
"""Repair wheel-shipped shared libraries that request an executable stack.
Why this exists
---------------
CTranslate2 wheels up to and including 4.4.0 ship
``ctranslate2.libs/libctranslate2-*.so`` with ``PT_GNU_STACK`` marked
``RWE`` a request for an executable stack. Linux kernels that refuse to
grant it (hardened kernels, and mainline from 6.x onwards) fail the
``dlopen`` outright::
ImportError: libctranslate2-d3638643.so.4.4.0: cannot enable executable
stack as shared object requires: Invalid argument
Everything that links CTranslate2 dies with it: the **whisperx** and
**faster-whisper** ASR engines (#692) *and* Argos translation, which is the
default dub translation engine (``argostranslate.translate`` imports
``ctranslate2``). #692 taught the ASR selector to fall through to another
engine; it never fixed the library, so Linux users on Python 3.11 lost both
engines. The pin is upstream and not ours to lift: whisperx 3.4.5 the last
release that supports Python 3.11, which is what ``.python-version``, CI and
the installers use requires ``ctranslate2<4.5.0``, and 4.5.0 is the first
release whose ``.so`` drops the exec-stack request.
The flag is a single bit in the ELF program header, so we clear it in place
rather than shipping a patched wheel or asking users for ``patchelf`` (which
is not installed on a typical desktop). Inspection is a ~100-byte read with
no imports, so :func:`ensure_ctranslate2_loadable` is cheap enough to call
from an availability probe: it only rewrites a file when that file would
otherwise refuse to load.
Everything here is a no-op off Linux (macOS/Windows have no such rejection)
and handles malformed ELF data a repair that cannot happen returns a reason, and the
caller degrades exactly as it did before.
"""
from __future__ import annotations
import glob
import logging
import os
import struct
import sys
import threading
logger = logging.getLogger("omnivoice.execstack")
#: ELF segment type for the stack-permission marker, and its executable bit.
_PT_GNU_STACK = 0x6474E551
_PF_X = 0x1
#: Serialize in-process writes; flock also coordinates sidecar processes.
_REPAIR_LOCK = threading.RLock()
def _elf_header(fh) -> tuple[str, int, int, int, bool] | None:
"""Return ``(endian_prefix, e_phoff, e_phentsize, e_phnum, is_64)`` or None.
None means "not an ELF file we understand" which is a normal answer
(a ``.so`` stub, a text file, a Mach-O), never an error.
"""
fh.seek(0)
ident = fh.read(16)
if len(ident) < 16 or ident[:4] != b"\x7fELF":
return None
if ident[4] not in (1, 2) or ident[5] not in (1, 2):
return None
is_64 = ident[4] == 2
endian = "<" if ident[5] == 1 else ">"
fh.seek(0, os.SEEK_END)
size = fh.tell()
header_size = 64 if is_64 else 52
if size < header_size:
return None
fh.seek(0)
header = fh.read(header_size)
if len(header) != header_size:
return None
e_phoff = struct.unpack_from(endian + ("Q" if is_64 else "I"), header, 0x20 if is_64 else 0x1C)[0]
e_phentsize, e_phnum = struct.unpack_from(endian + "HH", header, 0x36 if is_64 else 0x2A)
if (not e_phnum or e_phoff < header_size or
e_phentsize < (56 if is_64 else 32) or
e_phoff + e_phentsize * e_phnum > size):
return None
# p_flags sits at a different offset per class (ELF64 puts it right after
# p_type; ELF32 puts it last), so the caller needs the class too.
return endian, e_phoff, e_phentsize, e_phnum, is_64
def _gnu_stack_flags_offset(fh) -> tuple[int, int, str] | None:
"""Locate the ``PT_GNU_STACK`` ``p_flags`` field.
Returns ``(file_offset, flags_value, endian_prefix)``, or None when the
file is not an ELF or carries no such segment.
"""
parsed = _elf_header(fh)
if parsed is None:
return None
endian, e_phoff, e_phentsize, e_phnum, is_64 = parsed
flags_rel = 4 if is_64 else 24 # p_flags offset inside the program header
for i in range(e_phnum):
base = e_phoff + i * e_phentsize
fh.seek(base)
raw = fh.read(e_phentsize)
if len(raw) < flags_rel + 4:
continue
(p_type,) = struct.unpack_from(endian + "I", raw, 0)
if p_type != _PT_GNU_STACK:
continue
(p_flags,) = struct.unpack_from(endian + "I", raw, flags_rel)
return base + flags_rel, p_flags, endian
return None
def has_execstack(path: str) -> bool | None:
"""True when ``path`` requests an executable stack.
None when the question does not apply: unreadable, not an ELF, or no
``PT_GNU_STACK`` segment.
"""
try:
with open(path, "rb") as fh:
found = _gnu_stack_flags_offset(fh)
except OSError:
return None
if found is None:
return None
_offset, flags, _endian = found
return bool(flags & _PF_X)
def clear_execstack(path: str) -> tuple[bool, str]:
"""Clear the executable-stack request on ``path``.
Returns ``(changed, detail)``. ``changed`` is False both when there was
nothing to do and when the write was refused (a read-only bundle, for
instance) ``detail`` says which.
"""
try:
# Lock and inspect the same descriptor we write: another process may
# already have repaired it, or the wheel may have been replaced.
with _REPAIR_LOCK, open(path, "r+b") as fh:
# Use host capability, not an emulated target platform.
if os.name == "posix":
import fcntl
fcntl.flock(fh, fcntl.LOCK_EX)
found = _gnu_stack_flags_offset(fh)
if found is None:
return False, "no PT_GNU_STACK segment"
offset, flags, endian = found
if not flags & _PF_X:
return False, "already non-executable"
fh.seek(offset)
fh.write(struct.pack(endian + "I", flags & ~_PF_X))
fh.flush()
os.fsync(fh.fileno())
except OSError as e:
return False, f"unreadable or not writable ({e.__class__.__name__})"
return True, "cleared PT_GNU_STACK executable bit"
def ctranslate2_library_paths() -> list[str]:
"""Native libraries shipped with the installed ``ctranslate2`` wheel.
Found without importing ``ctranslate2`` importing it is the very thing
that fails when the exec-stack bit is set.
"""
import importlib.util
roots: list[str] = []
try:
spec = importlib.util.find_spec("ctranslate2")
except (ImportError, ValueError): # pragma: no cover — defensive
spec = None
locations = list(getattr(spec, "submodule_search_locations", None) or []) if spec else []
for pkg_dir in locations:
roots.append(pkg_dir)
roots.append(os.path.join(os.path.dirname(pkg_dir), "ctranslate2.libs"))
# Frozen builds flatten the wheel into the bundle directory.
meipass = getattr(sys, "_MEIPASS", None)
if meipass:
roots.append(meipass)
roots.append(os.path.join(meipass, "ctranslate2.libs"))
out: list[str] = []
for root in roots:
if not os.path.isdir(root):
continue
for pattern in ("libctranslate2*.so*", "libctranslate2*.dylib"):
out.extend(sorted(glob.glob(os.path.join(root, pattern))))
# Dedupe, preserving order.
return list(dict.fromkeys(out))
def ensure_ctranslate2_loadable() -> tuple[bool, str]:
"""Make ``import ctranslate2`` possible on kernels that refuse exec stacks.
Returns ``(ok, detail)`` where ``ok`` is False only when a library needs
the repair and could not get it the caller should then report its
engine unavailable with ``detail`` as the reason. The repair is idempotent. Re-probe on each call so installation or
external repair takes effect without restarting.
"""
result: tuple[bool, str]
if sys.platform != "linux":
# Only Linux rejects an exec-stack request at dlopen time.
result = (True, "not applicable off Linux")
else:
libs = ctranslate2_library_paths()
if not libs:
result = (True, "no ctranslate2 library found")
else:
repaired: list[str] = []
blocked: list[str] = []
for lib in libs:
if has_execstack(lib) is not True:
continue
changed, detail = clear_execstack(lib)
if changed:
repaired.append(os.path.basename(lib))
logger.warning(
"Repaired %s: %s — its executable-stack request is "
"rejected by this kernel, which broke whisperx, "
"faster-whisper and Argos translation (#692)",
os.path.basename(lib), detail,
)
elif has_execstack(lib) is not False:
blocked.append(f"{os.path.basename(lib)} ({detail})")
if blocked:
result = (
False,
"ctranslate2's native library requests an executable stack, "
"which this kernel refuses, and it could not be patched: "
+ "; ".join(blocked)
+ ". Reinstall the backend on Python 3.12+ (which resolves "
"ctranslate2 4.8+, without the exec-stack request), or run "
"`patchelf --clear-execstack <library>` once.",
)
elif repaired:
result = (True, "repaired " + ", ".join(repaired))
else:
result = (True, "no exec-stack request")
return result
def reset_ctranslate2_cache() -> None:
"""Compatibility hook; recoverable probe results are no longer cached."""
+1 -1
View File
@@ -24,7 +24,7 @@ from pathlib import Path
# tests/test_app_version.py::test_all_version_files_in_lockstep and bumped by
# release.yml's version-bump job, so it stays equal to
# pyproject/tauri.conf/Cargo/package.json.
_FALLBACK_VERSION = "0.5.3"
_FALLBACK_VERSION = "0.5.4"
def _fallback_version() -> str:
+4
View File
@@ -84,6 +84,10 @@ def _is_ct_error(msg):
def _get_model():
global _model
if _model is None:
from core.execstack import ensure_ctranslate2_loadable
ok, detail = ensure_ctranslate2_loadable()
if not ok:
raise ImportError(f"faster-whisper cannot load CTranslate2: {detail}")
from faster_whisper import WhisperModel
# Same weights as in-process faster-whisper: ASR_MODEL_FASTER selects
# for BOTH variants, ASR_MODEL_FW stays as a sidecar-only override.
+15
View File
@@ -25,6 +25,8 @@ runs under the Confucius4 venv — never imported by the parent), and
from __future__ import annotations
import logging
import math
import os
from typing import TYPE_CHECKING
from services.subprocess_backend import SubprocessBackend
@@ -99,6 +101,19 @@ class Confucius4Backend(SubprocessBackend):
from engines.confucius4.bootstrap import CONFUCIUS4_SIDECAR_SCRIPT
return CONFUCIUS4_SIDECAR_SCRIPT
@property
def recv_timeout_s(self) -> float:
"""Receive timeout in seconds for the Confucius4 sidecar process (#2103)."""
# Confucius4 is an LLM-based TTS (~17x realtime on CPU); synthesis legitimately
# outruns the 60s class default. OMNIVOICE_CONFUCIUS4_RECV_TIMEOUT_S tunes it (#2103).
try:
v = float(os.environ.get("OMNIVOICE_CONFUCIUS4_RECV_TIMEOUT_S", "900"))
except (ValueError, TypeError):
return 900.0
if not math.isfinite(v):
return 900.0
return max(30.0, v)
@property
def sample_rate(self) -> int:
return self._DEFAULT_SAMPLE_RATE
+15
View File
@@ -31,6 +31,8 @@ by the parent), and ``bootstrap.py`` (venv probe + lazy bootstrap).
from __future__ import annotations
import logging
import math
import os
import sys
from typing import TYPE_CHECKING
@@ -121,6 +123,19 @@ class DotsTTSBackend(SubprocessBackend):
from engines.dots_tts.bootstrap import DOTS_TTS_SIDECAR_SCRIPT
return DOTS_TTS_SIDECAR_SCRIPT
@property
def recv_timeout_s(self) -> float:
"""Receive timeout in seconds for the dots.tts sidecar process (#2103)."""
# dots.tts is a 2B autoregressive model; synthesis on CPU legitimately
# outruns the 60s class default. OMNIVOICE_DOTS_TTS_RECV_TIMEOUT_S tunes it (#2103).
try:
v = float(os.environ.get("OMNIVOICE_DOTS_TTS_RECV_TIMEOUT_S", "900"))
except (ValueError, TypeError):
return 900.0
if not math.isfinite(v):
return 900.0
return max(30.0, v)
# ── TTSBackend protocol ────────────────────────────────────────────────
@property
+15
View File
@@ -38,6 +38,8 @@ isolated engine venv.
from __future__ import annotations
import logging
import math
import os
from typing import TYPE_CHECKING
from services.subprocess_backend import SubprocessBackend
@@ -127,6 +129,19 @@ class MossTTSV15Backend(SubprocessBackend):
from engines.moss_tts_v15.bootstrap import MOSS_TTS_V15_SIDECAR_SCRIPT
return MOSS_TTS_V15_SIDECAR_SCRIPT
@property
def recv_timeout_s(self) -> float:
"""Receive timeout in seconds for the MOSS-TTS-v1.5 sidecar process (#2103)."""
# MOSS-TTS-v1.5 is an 8B model; synthesis legitimately outruns the
# 60s class default. OMNIVOICE_MOSS_TTS_V15_RECV_TIMEOUT_S tunes it (#2103).
try:
v = float(os.environ.get("OMNIVOICE_MOSS_TTS_V15_RECV_TIMEOUT_S", "900"))
except (ValueError, TypeError):
return 900.0
if not math.isfinite(v):
return 900.0
return max(30.0, v)
# ── TTSBackend protocol ────────────────────────────────────────────────
@property
+5 -5
View File
@@ -113,11 +113,11 @@ This clears the quarantine xattr recursively — including on the bundled
returns a clear error message pointing at this command rather than
silently hanging on a Gatekeeper-killed spawn.
If the macOS Apple Silicon Metal build fails to materialize in Wave 1
(no published `buildmetal.sh` in `omnivoice.cpp` per Pitfall 1), the
GGUF engine is unavailable on Apple Silicon and the existing in-process
`VoiceStudioBackend` remains the cloning default on that platform — no
hard block, no error toast on launch.
The macOS Apple Silicon Metal build compiles cleanly with `-DGGML_METAL=ON`
at the pinned `omnivoice.cpp` SHA (#2105), enabling GPU-accelerated GGUF
voice cloning when the packaged binary passes preflight and is permitted by
macOS. Missing binaries, placeholders, or Gatekeeper rejection leave
`VoiceStudioBackend` available as the in-process fallback.
## Smoke test
+14
View File
@@ -35,6 +35,7 @@ Threat model (per Plan 03-01 frontmatter):
from __future__ import annotations
import logging
import math
import os
import sys
from pathlib import Path
@@ -99,6 +100,19 @@ class Supertonic3Backend(SubprocessBackend):
def sidecar_script(cls) -> Path:
return SUPERTONIC3_SIDECAR_SCRIPT
@property
def recv_timeout_s(self) -> float:
"""Receive timeout in seconds for the Supertonic-3 sidecar process (#2103)."""
# Supertonic-3 runs ONNX on CPU; cold load downloads ~400MB and long
# synthesis benefits from more headroom than 60s. OMNIVOICE_SUPERTONIC3_RECV_TIMEOUT_S (#2103).
try:
v = float(os.environ.get("OMNIVOICE_SUPERTONIC3_RECV_TIMEOUT_S", "300"))
except (ValueError, TypeError):
return 300.0
if not math.isfinite(v):
return 300.0
return max(30.0, v)
# ── availability ───────────────────────────────────────────────────
@classmethod
+10 -5
View File
@@ -636,11 +636,16 @@ def _phase_a_build_inner() -> None:
# failure in that phase takes the whole backend down — the desktop app sits
# on "starting backend" forever and /health stays 503.
#
# That is not a hypothetical version: RTX 50-series (Blackwell, sm_120)
# users have no choice but to move off the pinned torch 2.8.0, which has no
# sm_120 kernels, and the torch 2.9.x they land on brings torchaudio 2.9
# with it. So the one group forced to upgrade hit a hard startup crash for
# a line that does nothing (#1931).
# That is not a hypothetical version: #1931 came from an sm_120 (Blackwell)
# user whose torch import crashed on Windows and who fixed it by moving to
# torch 2.9.1, which brings torchaudio 2.9 with it. Someone already working
# around one problem then hit a hard startup crash on a line that does
# nothing (#1931).
#
# The pin is not missing sm_120 kernels: torch 2.8.0 from the cu128 index
# lists sm_120 in get_arch_list(). CU128_ARCHS in
# tests/test_cuda_arch_compat.py records the same list, captured verbatim
# from a real cu128 build in #1285.
if hasattr(torchaudio, "set_audio_backend"):
torchaudio.set_audio_backend("soundfile")
from utils import hf_progress
+135 -9
View File
@@ -167,6 +167,8 @@ async def run_transcribe_guarded(executor, fn, *, what: str = "ASR",
immediately; running work calls it from the worker finalizer. Normal
completion leaves cleanup with the caller.
"""
from services.inference_cancellation import InferenceCancellation
cancellation = InferenceCancellation()
loop = asyncio.get_running_loop()
# Same SystemExit containment as the TTS pool (#1133 class): an ASR
# dependency written as a CLI must not be able to shut the backend down.
@@ -192,7 +194,8 @@ async def run_transcribe_guarded(executor, fn, *, what: str = "ASR",
def _job():
try:
return inner()
with cancellation.activate():
return inner()
finally:
with abandon_lock:
abandon_state["finished"] = True
@@ -204,6 +207,7 @@ async def run_transcribe_guarded(executor, fn, *, what: str = "ASR",
fut = asyncio.wrap_future(concurrent_fut, loop=loop)
def _abandon() -> None:
cancellation.cancel()
cancelled_before_start = concurrent_fut.cancel()
with abandon_lock:
abandon_state["requested"] = True
@@ -286,6 +290,26 @@ def _ctranslate2_cudnn_ok() -> tuple[bool, str]:
return True, "ready"
def _ctranslate2_execstack_ok() -> tuple[bool, str]:
"""Make CTranslate2 importable on kernels that refuse an executable stack.
ctranslate2 4.4.0 the version whisperx 3.4.5 pins, and 3.4.5 is the
newest release that supports the Python 3.11 we ship marks its native
library's stack ``RWE``. Kernels that refuse the request fail the dlopen
with "cannot enable executable stack", killing whisperx, faster-whisper
*and* Argos translation (#692). :mod:`core.execstack` clears that one bit
in place, so call this BEFORE importing either engine; it cheaply rechecks the library and
only writes when the library would otherwise refuse to load.
"""
try:
from core.execstack import ensure_ctranslate2_loadable
return ensure_ctranslate2_loadable()
except Exception as e: # noqa: BLE001 — a broken repair must not block ASR
logger.debug("exec-stack repair unavailable (%s) — continuing", e)
return True, "repair probe unavailable"
def _decode_audio_16k_mono(audio_path: str):
"""Decode `audio_path` to a 16 kHz mono float32 waveform using VoiceStudio's
*validated* ffmpeg, instead of whisperx.load_audio's bare ``"ffmpeg"`` PATH
@@ -657,6 +681,9 @@ class WhisperXBackend(ASRBackend):
@classmethod
def is_available(cls) -> tuple[bool, str]:
ct2_ok, ct2_detail = _ctranslate2_execstack_ok()
if not ct2_ok:
return False, f"whisperx cannot load CTranslate2: {ct2_detail}"
try:
import whisperx # noqa: F401
except ImportError as e:
@@ -684,6 +711,14 @@ class WhisperXBackend(ASRBackend):
# → speechbrain, or a stray k2_fsa redirect import aborts ASR on Windows
# (#630/#611/#647). No-op on macOS/Linux and when speechbrain is absent.
_harden_speechbrain_lazy_imports()
# #692: repair CTranslate2's exec-stack request before the import that
# would be rejected by it. Memoized, so this is free after the probe.
ct2_ok, ct2_detail = _ctranslate2_execstack_ok()
if not ct2_ok:
# ImportError (not RuntimeError): this IS a native-import failure,
# and the sentinel lets load_active_asr_backend degrade to the next
# engine instead of failing ASR wholesale (#1185).
raise ImportError(f"whisperx cannot load CTranslate2: {ct2_detail}")
import whisperx
# #723: re-check the CUDA pick against *currently free* VRAM — the TTS
# model may have claimed the card since __init__. A too-big load dies
@@ -1020,6 +1055,9 @@ class FasterWhisperBackend(ASRBackend):
@classmethod
def is_available(cls) -> tuple[bool, str]:
ct2_ok, ct2_detail = _ctranslate2_execstack_ok()
if not ct2_ok:
return False, f"faster-whisper cannot load CTranslate2: {ct2_detail}"
try:
import faster_whisper # noqa: F401
except ImportError as e:
@@ -1034,6 +1072,9 @@ class FasterWhisperBackend(ASRBackend):
def _ensure_model(self):
if self._model is not None:
return
ct2_ok, ct2_detail = _ctranslate2_execstack_ok() # #692, see WhisperX
if not ct2_ok:
raise ImportError(f"faster-whisper cannot load CTranslate2: {ct2_detail}")
from faster_whisper import WhisperModel
# Device / compute-type auto-pick:
# - CUDA present → GPU fp16
@@ -1445,9 +1486,42 @@ class PyTorchWhisperBackend(ASRBackend):
f"Underlying: {e}"
) from e
#: Batch sizes to try on CUDA, largest first. The VRAM preflight only sizes
#: the *weights*; generation adds an encoder/decoder workspace that scales
#: with the batch, and `return_timestamps="word"` keeps every layer's
#: cross-attention for the whole batch — gigabytes at batch 16. A card with
#: room for the model can therefore still OOM at the first transcribe, which
#: used to lose that chunk entirely (the dub retried the same batch size and
#: gave up, leaving a hole in the transcript). Step down, then use CPU.
_CUDA_BATCH_LADDER = (16, 4, 1)
_CUDA_BATCH_LADDER_WORD_TS = (8, 2, 1)
@staticmethod
def _is_oom(exc: BaseException) -> bool:
try:
import torch
if isinstance(exc, torch.cuda.OutOfMemoryError):
return True
except Exception: # noqa: BLE001 — classification must not raise
pass
return "out of memory" in str(exc).lower()
def _rebuild_on_cpu(self) -> None:
"""Move the existing pipeline to CPU without resolving any model files."""
import torch
# Keep the loaded checkpoint, tokenizer and feature extractor. Looking
# up the default model here could download a different model offline.
self._pipe.model.to(device="cpu", dtype=torch.float32)
self._pipe.device = torch.device("cpu")
try:
torch.cuda.empty_cache()
except Exception:
pass # Some builds have no CUDA cache to release.
def transcribe(self, audio_path: str, *, word_timestamps: bool = True) -> dict:
import soundfile as sf
import torch
self._ensure_pipe()
# #2039: libsndfile cannot open MP4/M4A (AAC), which /transcribe and
# the MCP tool both accept. Those decode through the validated ffmpeg
@@ -1460,15 +1534,67 @@ class PyTorchWhisperBackend(ASRBackend):
audio_np, sr = _decode_audio_16k_mono(audio_path), 16000
if audio_np.ndim > 1:
audio_np = audio_np.mean(axis=1)
bs = 16 if torch.cuda.is_available() else 2
result = self._pipe(
{"array": audio_np, "sampling_rate": sr},
return_timestamps="word" if word_timestamps else True,
chunk_length_s=15,
batch_size=bs,
)
def _run(batch_size: int):
return self._pipe(
{"array": audio_np, "sampling_rate": sr},
return_timestamps="word" if word_timestamps else True,
chunk_length_s=15,
batch_size=batch_size,
)
if self._on_cuda():
ladder = (
self._CUDA_BATCH_LADDER_WORD_TS if word_timestamps
else self._CUDA_BATCH_LADDER
)
for i, bs in enumerate(ladder):
try:
result = _run(bs)
break
except Exception as e: # noqa: BLE001 — only OOM is retryable
if not self._is_oom(e):
raise
try:
import torch
torch.cuda.empty_cache()
except Exception: # noqa: BLE001
pass
if i + 1 < len(ladder):
logger.warning(
"PyTorch Whisper CUDA OOM at batch_size=%d"
"retrying at %d. Free VRAM (Flush models, close "
"other GPU apps) for full-speed ASR.",
bs, ladder[i + 1],
)
continue
# Smallest batch still OOMs: finish on CPU rather than
# return an empty chunk the caller cannot distinguish
# from silence.
logger.warning(
"PyTorch Whisper CUDA OOM even at batch_size=1 — "
"transcribing on CPU (slower, same model). Detail: %s", e,
)
self._rebuild_on_cpu()
result = _run(2)
else:
result = _run(2)
return result if isinstance(result, dict) else {"chunks": [], "raw": result}
def _on_cuda(self) -> bool:
"""Whether the built pipeline actually sits on a CUDA device.
`torch.cuda.is_available()` is the wrong question: `_pick_device()` may
have chosen CPU on a CUDA host (low free VRAM), and a CPU pipeline must
not be handed a CUDA-sized batch.
"""
try:
device = getattr(self._pipe, "device", None)
return "cuda" in str(device).lower()
except Exception: # noqa: BLE001
return False
# ── NeMo Parakeet TDT (NVIDIA — Open ASR Leaderboard SOTA, 25 langs) ────────
+42 -1
View File
@@ -204,6 +204,42 @@ def _safe_torchaudio_save(
fmt, e,
)
torchaudio.save(path_or_buf, tensor, sample_rate, format=fmt)
except (ImportError, RuntimeError) as e:
if isinstance(e, RuntimeError) and "could not load libtorchcodec" not in str(e).lower():
raise _describe_write_failure(e, path_or_buf) from e
# torchaudio >= 2.9 routes save() through TorchCodec, which needs
# FFmpeg *shared libraries* on the system. Where those are absent the
# write raises ImportError and every generation fails. #1931 guarded
# set_audio_backend() against that torchaudio but left save() itself
# unprotected; arm64 CUDA hosts reach it unavoidably, since torch
# 2.8.0 publishes no aarch64 wheel. soundfile is already a locked
# dependency and the tensor is normalized by this point, so hand it to
# the audited sibling helper rather than failing the request.
logger.warning(
"torchaudio.save needs TorchCodec (%s); writing via soundfile", e
)
if hasattr(path_or_buf, "seek") and hasattr(path_or_buf, "truncate"):
try:
path_or_buf.seek(0)
path_or_buf.truncate(0)
except (OSError, io.UnsupportedOperation):
pass # Non-seekable streams cannot be rewound; preserve fallback behavior.
_subtype = {
"wav": "FLOAT" if bits_per_sample == 32 else "PCM_16",
"flac": "PCM_16",
"ogg": "VORBIS",
"mp3": "MPEG_LAYER_III",
}.get(fmt, "PCM_16")
try:
_safe_soundfile_write(
path_or_buf,
tensor.transpose(0, 1).contiguous().numpy(),
sample_rate,
subtype=_subtype,
format=fmt.upper(),
)
except Exception as e2:
raise _describe_write_failure(e2, path_or_buf) from e2
except Exception as e:
# #1221: libsndfile reports OS-level write failures as a bare
# "LibsndfileError: System error." — no path, no errno, nothing the
@@ -267,6 +303,7 @@ def _safe_soundfile_write(
sample_rate: int,
*,
subtype: str = "PCM_16",
format: str | None = None,
) -> None:
"""Sibling helper for the one in-tree ``sf.write`` site.
@@ -284,6 +321,10 @@ def _safe_soundfile_write(
subtype: Soundfile subtype string. ``"PCM_16"`` (default) for
standard 16-bit PCM WAV; ``"PCM_24"``, ``"FLOAT"`` etc.
also work.
format: Container format (``"WAV"``, ``"FLAC"``, ``"OGG"``,
``"MP3"``). ``None`` lets soundfile infer it from the path's
extension which it cannot do for a file-like object, so
callers passing a buffer must name it.
Raises:
ValueError: if the array is empty.
@@ -321,7 +362,7 @@ def _safe_soundfile_write(
samples = np.ascontiguousarray(samples)
_ensure_audio_parent(path)
sf.write(path, samples, sample_rate, subtype=subtype)
sf.write(path, samples, sample_rate, subtype=subtype, format=format)
def atomic_save_wav(
+19 -2
View File
@@ -64,6 +64,23 @@ from core.logging_utils import log_safe
logger = logging.getLogger("omnivoice.dub_pipeline")
def _media_process_error(tool: str, returncode: int, stderr: bytes, *, paths=()) -> str:
"""Keep the actionable end of native diagnostics without leaking paths."""
from core.scrub import scrub_text
detail = stderr.decode(errors="replace")
# Input/output may live outside home directories (mounted media, Windows
# drive roots). Remove the exact command paths before generic scrubbing.
for path in sorted((str(p) for p in paths if p), key=len, reverse=True):
for variant in {path, path.replace("\\", "/"), path.replace("/", "\\")}:
detail = detail.replace(variant, "[redacted path]")
detail = scrub_text(detail).strip()
tail = detail[-2000:]
if len(detail) > 2000:
tail = "" + tail
return f"{tool} exited with code {returncode}" + (f": {tail}" if tail else ". No diagnostic output.")
# ── Module-level state ──────────────────────────────────────────────────────
# These used to live in dub_core.py. The router now re-exports them for
# backward compat during the transition.
@@ -1326,7 +1343,7 @@ async def ingest_pipeline(
"-ar", "16000", "-ac", "1", audio_path, "-y",
])
if p.returncode != 0:
msg = (stderr.decode(errors="replace") or f"ffmpeg returned exit code {p.returncode}").strip()[:500]
msg = _media_process_error("FFmpeg", p.returncode, stderr, paths=(video_path, audio_path, job_dir))
raise Exception(msg)
# Second, FULL-QUALITY extraction for source separation. audio.wav
# is deliberately 16 kHz mono — that's what ASR wants — but Demucs
@@ -1476,7 +1493,7 @@ async def ingest_pipeline(
elif evt[0] == "done":
rc, stderr_full = evt[1], evt[2]
if rc != 0:
raise Exception(stderr_full.decode(errors="replace")[:500])
raise Exception(_media_process_error("Demucs", rc, stderr_full, paths=(audio_hq_path, audio_path, job_dir)))
# Stems land under the INPUT's basename ("audio_hq" when the
# full-quality extraction succeeded, "audio" on its fallback).
demucs_out = os.path.join(
+6 -4
View File
@@ -199,10 +199,12 @@ def mark_flashinfer_runtime_failure(reason: str) -> None:
def _cuda_arch_supported_for_compile() -> "tuple[bool, str]":
"""Check the GPU's architecture against this torch build's arch list.
New GPU architectures (e.g. Blackwell sm_120, issue #278) routinely break
torch.compile/Triton before upstream support lands: the eager model runs
via PTX forward-compat, but Inductor/Triton kernel compilation targets the
new arch directly and fails mid-generation. If the device's arch tag is
A new GPU architecture routinely breaks torch.compile/Triton before
upstream support lands (issue #278): the eager model runs via PTX
forward-compat, but Inductor/Triton kernel compilation targets the new arch
directly and fails mid-generation. Blackwell sm_120 was that case; it no
longer is on the pinned torch 2.8.0+cu128, where this probe can return
supported; independent compiler/runtime failures still need eager fallback. If the device's arch tag is
absent from this build's arch list we treat compile as unsupported and use
eager. The comparison is delegated to ``core.device_caps.arch_unsupported``
so it stays CUDA/ROCm-aware a ROCm build lists ``gfx`` names, and the
+7 -4
View File
@@ -174,12 +174,12 @@ def _binary_runs(path: str) -> bool:
if cached is not None:
return cached
try:
subprocess.run(
result = subprocess.run(
[path, "-version"],
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
timeout=10, check=False,
)
ok = True
ok = result.returncode == 0
except (OSError, subprocess.TimeoutExpired, subprocess.SubprocessError) as e:
logger.warning(
"Rejecting non-runnable ffmpeg/ffprobe candidate %s: %s",
@@ -309,8 +309,11 @@ def find_ffprobe():
try:
ffmpeg_path = find_ffmpeg()
if ffmpeg_path:
candidate = ffmpeg_path.replace("ffmpeg", "ffprobe")
if os.path.isfile(candidate):
candidate = os.path.join(
os.path.dirname(ffmpeg_path),
os.path.basename(ffmpeg_path).replace("ffmpeg", "ffprobe"),
)
if os.path.isfile(candidate) and _binary_runs(candidate):
return candidate
except Exception:
pass
@@ -0,0 +1,32 @@
"""Request cancellation visible to killable sidecars on executor threads.
In-process native calls remain accounted for until they return; only sidecars
can safely terminate early. A scope belongs to one job, never to a pool thread.
"""
from contextlib import contextmanager
import threading
_local = threading.local()
class InferenceCancellation:
def __init__(self):
self.cancelled = threading.Event()
def cancel(self):
self.cancelled.set()
@contextmanager
def activate(self):
previous = getattr(_local, "scope", None)
_local.scope = self
try:
if self.cancelled.is_set():
raise RuntimeError("Inference request was cancelled before execution")
yield
finally:
_local.scope = previous
def current_cancellation():
return getattr(_local, "scope", None)
+37 -10
View File
@@ -1,11 +1,12 @@
import os
import re
import sys
import time
import asyncio
import logging
import math
import os
import queue
import re
import sys
import threading
import time
from concurrent.futures import Executor, Future, ThreadPoolExecutor
from utils.containment import contain_system_exit
@@ -527,6 +528,7 @@ def generate_timeout_s(
text: "str | None", *, engine: object = None, execution_device: "str | None" = None,
min_vram_gb: float = 0.0, hardware_family: "str | None" = None,
vram_gb: "float | None" = None,
_include_sidecar_grace: bool = True,
) -> float:
"""THE wall-clock execution budget for one synthesis job, scaled to input.
@@ -562,6 +564,7 @@ def generate_timeout_s(
claiming the card is under-provisioned in user-facing diagnostics.
"""
base = GPU_JOB_TIMEOUT_S
explicit_budget = _GENERATE_TIMEOUT_EXPLICIT or GPU_JOB_TIMEOUT_S != _CONFIGURED_GPU_JOB_TIMEOUT_S
try:
from core.device_caps import detect_host_caps
caps = detect_host_caps()
@@ -588,6 +591,7 @@ def generate_timeout_s(
)
if family == "cpu" and (cpu_explicit or not universal_override):
base = CPU_JOB_TIMEOUT_S
explicit_budget = cpu_explicit
elif not universal_override and family in (
"cuda", "rocm", "vulkan", "xpu",
):
@@ -610,7 +614,23 @@ def generate_timeout_s(
# Device probing is advisory here; the configured universal bound is
# still safe when a platform probe is unavailable during startup.
pass
return base + (max(0, len(text or "") - 1200) / 40.0)
# If the engine specifies its own sidecar receive timeout (e.g. SubprocessBackend
# engines like Confucius, Dots, Moss, Supertonic), the outer execution budget
# must not cut the sidecar off early (#2103). A bounded 5s grace period ensures
# the sidecar's watchdog timer fires and surfaces its actionable timeout error
# before the outer pool cancellation cuts it off.
sidecar_grace = 0.0
if not explicit_budget and engine is not None and hasattr(engine, "recv_timeout_s"):
try:
sidecar_timeout = float(engine.recv_timeout_s)
if math.isfinite(sidecar_timeout) and sidecar_timeout > 0:
base = max(base, sidecar_timeout)
sidecar_grace = 5.0 if _include_sidecar_grace else 0.0
except (TypeError, ValueError):
pass # Invalid optional engine metadata cannot disable the outer guard.
return base + (max(0, len(text or "") - 1200) / 40.0) + sidecar_grace
def _retry_after_estimate(stats: dict) -> float:
@@ -726,6 +746,8 @@ async def run_on_gpu_pool_guarded(fn, *, what: str = "GPU job",
finalizer. Normal completion never calls it. This lets request-owned temp
files outlive abandoned workers without delaying ordinary requests (#1668).
"""
from services.inference_cancellation import InferenceCancellation
cancellation = InferenceCancellation()
loop = asyncio.get_running_loop()
ex = executor if executor is not None else _get_gpu_pool()
# Resolved at CALL time, not def time, so monkeypatching/reloading the
@@ -768,7 +790,8 @@ async def run_on_gpu_pool_guarded(fn, *, what: str = "GPU job",
except RuntimeError:
pass # loop already closed (caller vanished) — still run the job
try:
return _inner()
with cancellation.activate():
return _inner()
finally:
# Idents are reused by the OS; a stale heartbeat under this ident
# must not vouch for some future job on the same thread.
@@ -783,6 +806,7 @@ async def run_on_gpu_pool_guarded(fn, *, what: str = "GPU job",
fut = asyncio.wrap_future(concurrent_fut, loop=loop)
def _abandon() -> None:
cancellation.cancel()
# Keep the concurrent future so we can distinguish a job cancelled out
# of the queue from a thread that Python cannot stop once it has begun.
cancelled_before_start = concurrent_fut.cancel()
@@ -1565,10 +1589,13 @@ def _is_compile_runtime_failure(exc: BaseException) -> bool:
"""True when an exception originates in the torch.compile stack (Dynamo /
Inductor / Triton / FX / CUDA-graph trees) rather than in the model itself.
#278: on GPU architectures Triton doesn't support yet (e.g. Blackwell
sm_120), the compiled model dies mid-generation with errors like
"Detected that you are using FX to symbolically trace a dynamo-optimized
function" or an AssertionError out of torch/_inductor/cudagraph_trees.py.
#278: an independent compile-stack failure can surface during generation
as an AssertionError out of torch/_inductor/cudagraph_trees.py. An
architecture missing from the running torch build's arch list is rejected
earlier by should_torch_compile(), before this runtime fallback applies.
#278 also quotes "Detected that you are using FX to symbolically trace a
dynamo-optimized function"; Dynamo raises that on any device, CPU included,
so it is a compile-stack error to catch here but never an arch signal.
Walks the exception chain and checks (a) the exception type's module,
(b) the message, (c) the traceback file paths the cudagraph case is a
bare AssertionError, so the traceback check is load-bearing.
+6 -1
View File
@@ -1543,10 +1543,15 @@ def _step_fetch_weights(spec: SidecarSpec, job: dict) -> None:
# other model download in the app — see setup/download.py): the
# source checkout is unpinned upstream `main` anyway, and hf_hub
# checksum-verifies each artifact. Hence the B615 waiver below.
# Unwrap to the bearer string: snapshot_download takes `token: str |
# None`, and a ResolvedToken record is ignored in favour of ambient
# discovery, so gated engine weights 401 for a user whose token lives
# in VoiceStudio's settings rather than HF's own cache (#2163).
_resolved = resolve_token()
kwargs: dict = {
"repo_id": spec.weights_repo_id,
"local_dir": str(wdir),
"token": resolve_token(),
"token": _resolved.token if _resolved else None,
}
if spec.weights_revision:
kwargs["revision"] = spec.weights_revision
+80 -10
View File
@@ -24,7 +24,7 @@ from dataclasses import dataclass
# Captures: HH MM SS sep(`,` or `.`) ms (1-3 digits)
_TS = r"(\d{1,2}):([0-5]?\d):([0-5]?\d)[,.](\d{1,3})"
_TS = r"(?:(\d{1,2}):)?([0-5]?\d):([0-5]?\d)[,.](\d{1,3})"
# Horizontal whitespace only — NEVER plain `\s`, which matches newlines.
# A timing line lives on ONE line, so `\s*` bought nothing but catastrophic
# backtracking: under re.MULTILINE the engine restarts at every line start,
@@ -41,7 +41,19 @@ _TIMING_RE = re.compile(rf"^{_H}{_TS}{_H}-->{_H}{_TS}.*$", re.MULTILINE)
def _ts_to_seconds(h: str, m: str, s: str, ms: str) -> float:
# Pad ms to 3 digits so "5" -> 0.005, "50" -> 0.050.
ms_padded = (ms + "000")[:3]
return int(h) * 3600 + int(m) * 60 + int(s) + int(ms_padded) / 1000.0
return int(h or 0) * 3600 + int(m) * 60 + int(s) + int(ms_padded) / 1000.0
def _is_index_line(line: str) -> bool:
"""True when `line` is a bare SubRip cue number.
Stricter than `str.isdigit()` on purpose: that also accepts non-ASCII
numerals (Arabic-Indic "١٩٩٩", Devanagari "२०२६", and the full-width
forms), which in a 646-language dubbing app are dialogue, never the
ASCII cue indices SubRip actually writes.
"""
stripped = line.strip()
return stripped.isascii() and stripped.isdigit()
@dataclass
@@ -68,13 +80,62 @@ def parse_srt(content: str) -> SrtParseResult:
# Strip BOM and normalise line endings; many editors save SRTs as CRLF.
text = content.lstrip("").replace("\r\n", "\n").replace("\r", "\n")
is_webvtt = bool(re.match(r"WEBVTT(?:[ \t]|\n|$)", text.lstrip()))
if is_webvtt:
# Metadata is block-scoped. Filter it BEFORE scanning timings so an
# example timestamp inside a NOTE/STYLE/REGION cannot become speech.
blocks = []
for block in re.split(r"\n[^\S\n]*\n", text):
lines = block.strip().split("\n")
first = lines[0].strip()
# WebVTT's block parser gives a timing line in position two
# precedence over the identifier (including STYLE/REGION/NOTE).
# https://www.w3.org/TR/webvtt1/#file-parsing
identifies_cue = len(lines) > 1 and _TIMING_RE.match(lines[1])
metadata = first in {"STYLE", "REGION"} or re.match(r"NOTE(?:[ \t]|$)", first)
if metadata and not identifies_cue:
continue
blocks.append(block)
text = "\n\n".join(blocks)
raw: list[dict] = []
skipped = 0
# Find every timing line, slice the cue text from there to the next
# timing line (or end of file). This is robust to missing index
# numbers and to spec deviations in the blank-line separator.
matches = list(_TIMING_RE.finditer(text))
# A mixed file can stop numbering at any cue. Track each boundary;
# never treat an initial index as permission to discard later numbers.
head = text[:matches[0].start()].strip() if matches else ""
first_marker = head.split("\n")[-1].strip() if head else ""
cue_index = int(first_marker) if _is_index_line(first_marker) and len(first_marker) <= 12 else None
for i, m in enumerate(matches):
body_start = m.end()
has_next = i + 1 < len(matches)
body_end = matches[i + 1].start() if has_next else len(text)
body = text[body_start:body_end]
if is_webvtt:
# The blank separator ends WebVTT dialogue; following identifiers,
# NOTE/STYLE blocks belong outside the cue, even when numeric.
body = re.split(r"\n[^\S\n]*\n", body, maxsplit=1)[0]
# An index must directly precede the next timing line. A blank line
# AFTER a number instead marks that number as preceding dialogue.
next_index = None
if has_next and not is_webvtt:
# Inspect lines rather than a backtracking regex on uploaded text.
# One newline terminates the marker; a second means it is dialogue.
marker_lines = body.split("\n")
if marker_lines and not marker_lines[-1].strip(" \t"):
marker_lines.pop()
marker = marker_lines[-1].strip(" \t") if marker_lines else ""
numeric = bool(marker) and marker.isascii() and marker.isdecimal()
separated = len(marker_lines) > 2 and not marker_lines[-2].strip()
expected_index = cue_index + 1 if cue_index is not None else i + 2
expected = marker.lstrip("0") == str(expected_index)
has_dialogue = any(line.strip() for line in marker_lines[:-1])
if numeric and expected and (has_dialogue or separated) and (cue_index is not None or separated):
body = "\n".join(marker_lines[:-1])
next_index = expected_index
cue_index = next_index
try:
start = _ts_to_seconds(m.group(1), m.group(2), m.group(3), m.group(4))
end = _ts_to_seconds(m.group(5), m.group(6), m.group(7), m.group(8))
@@ -84,14 +145,7 @@ def parse_srt(content: str) -> SrtParseResult:
if end <= start:
skipped += 1
continue
body_start = m.end()
body_end = matches[i + 1].start() if i + 1 < len(matches) else len(text)
body = text[body_start:body_end].strip("\n")
# Drop the trailing index number of the NEXT cue (which got eaten
# into our body) by trimming trailing digit-only lines.
lines = body.split("\n")
while lines and lines[-1].strip().isdigit():
lines.pop()
lines = body.strip("\n").split("\n")
cue_text = "\n".join(line.strip() for line in lines if line.strip())
if not cue_text:
skipped += 1
@@ -126,3 +180,19 @@ def parse_srt(content: str) -> SrtParseResult:
for i, seg in enumerate(out)
]
return SrtParseResult(segments=segments, skipped_cues=skipped, dropped_overlaps=dropped)
def format_cue_timestamp(seconds: float, ms_separator: str) -> str:
"""`HH:MM:SS<sep>mmm` for `seconds`, rounded to the millisecond.
Rounds the whole value once, then splits it, so a time that is not exact
in binary (2.3 is 2.29999...) stays 2.300 instead of truncating to 2.299,
which moved every such cue a millisecond early on export, and 59.9996
carries to the next second instead of printing `,1000`. SRT separates
the milliseconds with `,`; WebVTT with `.`.
"""
total_ms = int(round(seconds * 1000))
h, rem = divmod(total_ms, 3_600_000)
m, rem = divmod(rem, 60_000)
s, ms = divmod(rem, 1000)
return f"{h:02d}:{m:02d}:{s:02d}{ms_separator}{ms:03d}"
+40 -12
View File
@@ -22,6 +22,8 @@ only process isolation.
from __future__ import annotations
import logging
import math
import os
import sys
import threading
from pathlib import Path
@@ -34,7 +36,8 @@ from services.subprocess_backend import (
logger = logging.getLogger("omnivoice.asr.subprocess")
# A model load + transcription can take a while on CPU for a long clip; give
# the transcribe round-trip more headroom than the TTS default.
# the transcribe round-trip more headroom than the TTS default. Configurable via
# OMNIVOICE_ASR_RECV_TIMEOUT_S (#2103).
ASR_RECV_TIMEOUT_S = 600.0
@@ -78,6 +81,17 @@ class SubprocessASRBackend(SubprocessBackend):
pass
return "cpu"
@property
def recv_timeout_s(self) -> float:
"""Wall-clock timeout in seconds waiting for an ASR sidecar response (#2103)."""
try:
v = float(os.environ.get("OMNIVOICE_ASR_RECV_TIMEOUT_S", str(ASR_RECV_TIMEOUT_S)))
except (ValueError, TypeError):
return ASR_RECV_TIMEOUT_S
if not math.isfinite(v):
return ASR_RECV_TIMEOUT_S
return max(30.0, v)
def transcribe(self, audio_path: str, *, word_timestamps: bool = True) -> dict:
"""Transcribe ``audio_path`` in the sidecar. Returns the engine's
result dict ({"segments": [...], "language": ...}).
@@ -112,6 +126,7 @@ class SubprocessASRBackend(SubprocessBackend):
if slot_future is not None:
slot_future.cancel()
raise TimeoutError("timed out waiting for a free GPU worker")
timeout_s = self.recv_timeout_s
with self._lock:
self._spawn()
from services.performance_profiles import asr_decode_defaults
@@ -121,17 +136,25 @@ class SubprocessASRBackend(SubprocessBackend):
"word_timestamps": bool(word_timestamps),
"decode_options": asr_decode_defaults(),
})
reply = self._recv_with_timeout(ASR_RECV_TIMEOUT_S)
if not reply:
# EOF can arrive before Windows updates poll(); retire the
# stale handle so an immediate retry respawns the sidecar.
self.shutdown()
# Pipe closed mid-transcription → the child crashed.
raise RuntimeError(
f"{self.id} ASR sidecar crashed mid-transcription "
f"(device={self._device()}); the job failed but the backend "
f"stayed up — retry to respawn a fresh sidecar."
)
reply = self._recv_with_timeout(timeout_s)
timed_out = self._last_recv_timed_out
if not reply:
# EOF can arrive before Windows updates poll(); retire the
# stale handle so an immediate retry respawns the sidecar.
self.shutdown()
if timed_out:
raise RuntimeError(
f"{self.id} ASR sidecar exceeded receive timeout "
f"({timeout_s:g}s); killed mid-transcription "
f"(device={self._device()}) — retry or raise "
f"OMNIVOICE_ASR_RECV_TIMEOUT_S."
)
# Pipe closed mid-transcription → the child crashed.
raise RuntimeError(
f"{self.id} ASR sidecar crashed mid-transcription "
f"(device={self._device()}); the job failed but the backend "
f"stayed up — retry to respawn a fresh sidecar."
)
if reply.get("op") == "error":
raise RuntimeError(
f"{self.id} ASR sidecar error (device={self._device()}): "
@@ -165,6 +188,11 @@ class IsolatedFasterWhisperBackend(SubprocessASRBackend):
@classmethod
def is_available(cls) -> tuple[bool, str]:
from core.execstack import ensure_ctranslate2_loadable
ok, detail = ensure_ctranslate2_loadable()
if not ok:
return False, f"faster-whisper cannot load CTranslate2: {detail}"
try:
import faster_whisper # noqa: F401
except Exception as e:
+119 -14
View File
@@ -44,6 +44,7 @@ import contextlib
import collections
import json
import logging
import math
import os
import struct
import subprocess
@@ -103,10 +104,32 @@ _STDERR_TAIL_LINES = 12
_STDERR_TAIL_CHARS = 800
#: Per-frame _recv read timeout (best-effort — applies to header read; body
#: read is uninterruptible on a stdlib BufferedReader). Used in health_check
#: and generate to bound a hung sidecar.
#: read is uninterruptible on a stdlib BufferedReader). Used by health_check
#: to bound a hung sidecar: a ping must stay fast, so this stays short.
RECV_TIMEOUT_S = 60.0
#: Floor for the generate() deadline of a sidecar that does not choose its own.
#:
#: It must not undercut the budget the job was already granted by
#: services.model_manager.generate_timeout_s — 300s on an accelerated host,
#: 600s on a CPU one — or the watchdog kills a synthesis the caller still
#: considers valid, which is #2103. 600s is that CPU floor.
#:
#: A floor, not the whole answer: that budget also scales with text length, so
#: _effective_recv_timeout_s derives the real per-request deadline and falls
#: back to this value when the budget cannot be computed.
#:
#: Every engine that overrode the hook picked somewhere in 300s..900s, i.e. at
#: or above the accelerated budget; only the ones that stayed silent got 60s.
#:
#: A literal rather than an import of model_manager.CPU_JOB_TIMEOUT_S: that
#: value is read from the environment at import time and monkeypatched by
#: tests, and a class-attribute default that moves with the environment is
#: harder to reason about than one that does not. This module also reaches
#: model_manager only lazily, from inside functions. The two are held in
#: lockstep by backend/tests/test_subprocess_recv_timeout.py instead.
GENERATE_RECV_TIMEOUT_S = 600.0
# ── Idle sidecar reaping (parity Action 13) ─────────────────────────────────
#
@@ -378,12 +401,17 @@ class SubprocessBackend(TTSBackend):
# Per-engine recv timeout for generate(): how long the parent waits for the
# sidecar's audio frame before the watchdog hard-kills the child and reclaims
# its VRAM/device. Default is the conservative RECV_TIMEOUT_S (60s). A
# subclass whose legitimate generates run longer overrides it (or exposes it
# as a property) so a slow-but-valid synth is not falsely killed, while a
# genuinely wedged one is still reclaimed. health_check() keeps using
# RECV_TIMEOUT_S directly, since a ping must stay fast.
recv_timeout_s: float = RECV_TIMEOUT_S
# its VRAM/device. A subclass whose legitimate generates run longer overrides
# it (or exposes it as a property) so a slow-but-valid synth is not falsely
# killed, while a genuinely wedged one is still reclaimed. health_check()
# keeps using RECV_TIMEOUT_S directly, since a ping must stay fast.
#
# The default is GENERATE_RECV_TIMEOUT_S, not RECV_TIMEOUT_S: 60s is a
# health-check ping budget, and inheriting it as a *generation* deadline
# killed four engines mid-sentence (#2103). Overriding remains the way to
# ask for more; inheriting no longer means asking for less than the job's
# own budget.
recv_timeout_s: float = GENERATE_RECV_TIMEOUT_S
# ── instance state (initialised in __init__) ───────────────────────────
@@ -598,6 +626,58 @@ class SubprocessBackend(TTSBackend):
tail = self._stderr_tail_text()
return f"{reason}. Last stderr: {tail}" if tail else f"{reason} (no stderr output)"
def _effective_recv_timeout_s(self, text: str) -> float:
"""This request's silence deadline: never under the budget it was granted.
``recv_timeout_s`` is a per-engine constant, but ``generate_timeout_s``
scales the wall-clock budget with text length, so only a per-request
deadline can satisfy "the watchdog must not fire before the caller's
own budget expires".
An engine that overrode the hook keeps exactly its own value, including
a smaller one: #1611 asks for more and #2103 asks that opting down stay
possible.
"""
own = self.recv_timeout_s
for klass in type(self).__mro__:
if klass is SubprocessBackend:
break # reached the base without finding an override
if "recv_timeout_s" in klass.__dict__:
return own # the engine chose; that choice is the answer
try:
from services.model_manager import generate_timeout_s
budget = float(generate_timeout_s(text, engine=self, _include_sidecar_grace=False))
except Exception:
# Budget probing is advisory: a failure here must not turn a
# working generate into an error. Fall back to the class floor.
return own
if not math.isfinite(budget):
return own
return max(own, budget)
def _generate_failure_reason(self, elapsed_s: float, deadline_s: float) -> str:
"""Say whether generate() lost the sidecar to the deadline or a crash.
The spawn handshake already distinguishes these (#2026); generate() did
not, so a watchdog kill surfaced as "sidecar closed pipe mid-generate"
and every reporter reasonably read it as a crash (#2103). The deadline
is the one fact that explains the failure, so it belongs in the message
the caller sees, not only in the backend log.
"""
if not self._last_recv_timed_out:
# Unchanged wording for a genuine crash: only the deadline case was
# misreported, and #2026 already gave the spawn path its own tail.
return f"{self.id} sidecar closed pipe mid-generate"
reason = (
f"{self.id} sidecar sent nothing for {deadline_s:g}s "
f"(elapsed {elapsed_s:.0f}s), so VoiceStudio stopped it. It may "
f"simply be slower than that deadline on this host; retry or increase "
f"this engine's receive timeout (see Troubleshooting)."
)
tail = self._stderr_tail_text()
return f"{reason}. Last stderr: {tail}" if tail else reason
def _stderr_tail_text(self) -> str:
"""The sidecar's last stderr lines, scrubbed for a user-visible error."""
from core.scrub import scrub_text
@@ -757,9 +837,11 @@ class SubprocessBackend(TTSBackend):
for k, v in kw.items():
if _is_jsonable(v):
msg[k] = v
deadline_s = self._effective_recv_timeout_s(text)
started_at = time.monotonic()
try:
self._send(msg)
reply = self._recv_with_timeout(self.recv_timeout_s)
reply = self._recv_with_timeout(deadline_s)
except (RuntimeError, OSError):
# A broken or malformed protocol stream cannot be reused.
# Reap it before releasing the request lock so an immediate
@@ -790,13 +872,18 @@ class SubprocessBackend(TTSBackend):
except Exception:
pass # the heartbeat is best-effort; never fail a synth over it
try:
reply = self._recv_with_timeout(self.recv_timeout_s)
reply = self._recv_with_timeout(deadline_s)
except (RuntimeError, OSError):
self._reap_unusable_process(proc)
raise
if not reply:
# Same EOF for a deadline kill and a crash; _last_recv_timed_out
# is what tells them apart (#2026's spawn path does the same).
reason = self._generate_failure_reason(
time.monotonic() - started_at, deadline_s
)
self._reap_unusable_process(proc)
raise RuntimeError(f"{self.id} sidecar closed pipe mid-generate")
raise RuntimeError(reason)
if reply.get("op") == "error":
stage = str(reply.get("stage") or "unknown")
message = str(reply.get("message") or "unknown sidecar error")
@@ -896,13 +983,31 @@ class SubprocessBackend(TTSBackend):
fired.set()
self._timeout_kill(proc)
watchdog = threading.Timer(timeout_s, _on_timeout)
watchdog.daemon = True
from services.inference_cancellation import current_cancellation
cancellation = current_cancellation()
stop_watchdog = threading.Event()
def _watch() -> None:
deadline = time.monotonic() + timeout_s
while not stop_watchdog.is_set():
remaining = deadline - time.monotonic()
if remaining <= 0 or (cancellation and cancellation.cancelled.is_set()):
_on_timeout() # captured process only; never a later retry
return
stop_watchdog.wait(min(remaining, 0.05))
if cancellation is None:
watchdog = threading.Timer(timeout_s, _on_timeout)
watchdog.daemon = True
else:
watchdog = threading.Thread(target=_watch, daemon=True)
watchdog.start()
try:
return self._recv()
finally:
watchdog.cancel()
stop_watchdog.set()
if cancellation is None:
watchdog.cancel()
# cancel() cannot stop an already-running callback. Finish its
# bounded reap before another receive or generation starts.
watchdog.join()
+46
View File
@@ -0,0 +1,46 @@
"""Decode an uploaded text file whose encoding nobody declared.
Subtitle and manuscript uploads arrive as raw bytes. Windows tools commonly
save them as UTF-16 with a byte-order mark (Notepad's "Unicode", many subtitle
editors) or in the legacy Windows-1252 code page. A UTF-8 decode turns the
first into NUL-interleaved text and, lossily, drops every accent, dash and
curly quote from the second.
"""
from __future__ import annotations
import codecs
_BOMS = (
(codecs.BOM_UTF8, "utf-8"),
(codecs.BOM_UTF16_LE, "utf-16-le"),
(codecs.BOM_UTF16_BE, "utf-16-be"),
)
_CP1252_UNDEFINED = "voicestudio-cp1252-undefined"
def _undefined_as_latin1(exc: UnicodeDecodeError) -> tuple[str, int]:
# Only the offending bytes take their Latin-1 code point — what the
# browser's windows-1252 decoder does — so the rest of the file keeps its
# curly quotes and dashes.
return exc.object[exc.start:exc.end].decode("latin-1"), exc.end
codecs.register_error(_CP1252_UNDEFINED, _undefined_as_latin1)
def decode_text_upload(data: bytes) -> str:
"""Return the text of ``data``, without its byte-order mark.
A BOM names the encoding. Without one, valid UTF-8 is UTF-8; anything else
is read as Windows-1252, the legacy code page such files come from. Each of
the five bytes Windows-1252 leaves undefined takes its Latin-1 code point,
so the decode never raises.
"""
for bom, encoding in _BOMS:
if data.startswith(bom):
return data[len(bom):].decode(encoding, errors="replace")
try:
return data.decode("utf-8")
except UnicodeDecodeError:
return data.decode("cp1252", errors=_CP1252_UNDEFINED)
+62 -4
View File
@@ -29,7 +29,19 @@ logger = logging.getLogger("omnivoice.translation_engines")
_NLLB_REPO_ID = "facebook/nllb-200-distilled-600M"
_ARGOS_INSTALL_LOCK = threading.Lock()
_ARGOS_LANG_ALIASES = {"cmn": "zh"}
_ARGOS_LANG_ALIASES = {
"cmn": "zh",
"zho": "zh",
"in": "id",
"iw": "he",
"fil": "tl",
}
# Human names (and the UI's own labels) that are not ISO 639-1 tokens.
_ARGOS_NAME_ALIASES = {
"chinese": "zh",
"chinese (simplified)": "zh",
"mandarin": "zh",
}
# Engine ID → registry entry. Keyed by the `provider` string sent from the
@@ -39,7 +51,12 @@ REGISTRY: dict[str, dict] = {
"id": "argos",
"display_name": "Argos (Local, Fast)",
"pip_package": "argostranslate",
"probe_module": "argostranslate",
# `argostranslate.translate`, not the bare package: the translator runs
# on CTranslate2, and the bare package imports fine on a host whose
# kernel rejects CTranslate2's native library (#692) — so a shallow
# probe advertised Argos as ready and every translate 500'd. Probe the
# module that actually pulls the native dep (same lesson as #1185).
"probe_module": "argostranslate.translate",
"category": "offline",
"needs_key": False,
"builtin": True,
@@ -124,6 +141,18 @@ def _probe(entry: dict) -> tuple[bool, str | None]:
mod = entry.get("probe_module")
if not mod:
return True, None
if mod.startswith("argostranslate"):
# Repair CTranslate2's exec-stack request before the import that would
# be rejected by it (#692) — otherwise Argos, the default offline
# engine, is unusable on kernels that refuse an executable stack.
try:
from core.execstack import ensure_ctranslate2_loadable
ok, detail = ensure_ctranslate2_loadable()
if not ok:
return False, detail
except Exception as e: # noqa: BLE001 — a broken repair must not hide the engine
logger.debug("exec-stack repair unavailable (%s) — probing anyway", e)
try:
importlib.import_module(mod)
if entry.get("id") == "nllb":
@@ -141,6 +170,11 @@ def _probe(entry: dict) -> tuple[bool, str | None]:
return True, None
except ImportError as e:
return False, f"import {mod!r} failed: {e}"
except Exception as e: # noqa: BLE001
# A native library that refuses to load raises OSError, not ImportError
# (#692). An availability probe must report "unusable here", never take
# the engine list down with it.
return False, f"import {mod!r} failed ({type(e).__name__}): {e}"
def install_command(engine: "str | dict | None") -> str | None:
@@ -299,9 +333,33 @@ def is_ready(engine_id: str) -> bool:
def argos_lang_code(value: str) -> str:
"""Return the base language token used by Argos package metadata."""
code = str(value or "").strip().lower().split("-", 1)[0]
"""Return the base language token used by Argos package metadata.
Accepts ISO 639-1 codes (e.g. ``"zh"``), BCP-47 tags with a region or script
suffix (e.g. ``"zh-CN"``, ``"cmn-Hans"``), human names from the dub UI's own
label list (e.g. ``"Chinese"``, ``"Mandarin"``), legacy / deprecated ISO
639-1 codes still seen in older corpora (``"in"````"id"`` for Indonesian,
``"iw"````"he"`` for Hebrew), and ISO 639-2/T (e.g. ``"zho"````"zh"``,
``"fil"````"tl"`` for Tagalog). Empty or whitespace-only input raises
``ValueError`` so the caller sees an actionable error instead of a
silently-empty language token.
"""
raw = str(value or "").strip()
key = raw.lower()
parts = key.replace("_", "-").split("-")
if key == "chinese (traditional)" or (
parts[0] in {"zh", "zho", "cmn"} and
any(part in {"hant", "tw", "hk", "mo"} for part in parts[1:])
):
raise ValueError("Argos does not provide Traditional Chinese; choose NLLB for this script")
named = _ARGOS_NAME_ALIASES.get(key)
if named:
return named
code = parts[0]
code = _ARGOS_LANG_ALIASES.get(code, code)
named = _ARGOS_NAME_ALIASES.get(code)
if named:
return named
if not re.fullmatch(r"[a-z]{2,3}", code):
raise ValueError("Choose a valid source and target language")
return code
+64 -15
View File
@@ -1624,6 +1624,7 @@ class KittenTTSBackend(TTSBackend):
# resolved unchanged by `resolve_kokoro_lang_code()` below.
_KOKORO_ISO_BY_FULL_NAME = {
"english": "en",
"british english": "en-gb",
"spanish": "es",
"french": "fr",
"hindi": "hi",
@@ -1634,6 +1635,31 @@ _KOKORO_ISO_BY_FULL_NAME = {
}
def _kokoro_supported_labels(aliases: dict, lang_codes: dict) -> list[str]:
"""Human labels for every language Kokoro accepts, read from its own tables.
Derived from the installed package, never from
``_KOKORO_ISO_BY_FULL_NAME``: that map exists to translate full names
*into* Kokoro's codes, and reusing it to describe what Kokoro supports
understates the model when a newly supported code has no full-name alias.
Read every installed code so new languages remain visible without changing
the input-name map.
"""
labels: dict[str, str] = {}
# Prefer the full name a caller can actually pass.
for name, iso in _KOKORO_ISO_BY_FULL_NAME.items():
code = aliases.get(iso, iso)
if code in lang_codes:
labels.setdefault(code, name.title())
# Then whatever the installed table supports that no full name reaches. Its
# own description is a display name for some codes ("British English") and
# an ISO tag for others ("pt-br"); either names the language better than
# dropping it.
for code, described in lang_codes.items():
labels.setdefault(code, str(described))
return sorted(labels.values())
def resolve_kokoro_lang_code(language: str) -> str:
"""Map a full language name / ISO code to Kokoro's single-letter
`lang_code`, against the AUTHORITATIVE table read from the installed
@@ -1651,7 +1677,12 @@ def resolve_kokoro_lang_code(language: str) -> str:
iso = _KOKORO_ISO_BY_FULL_NAME.get(key, key)
code = ALIASES.get(iso, iso)
if code not in LANG_CODES:
supported = ", ".join(sorted(name.title() for name in _KOKORO_ISO_BY_FULL_NAME))
# Labels from newer installed tables must remain selectable even when
# they have no entry in our compatibility map of full names.
code = next((candidate for candidate, label in LANG_CODES.items()
if str(label).strip().lower() == key), code)
if code not in LANG_CODES:
supported = ", ".join(_kokoro_supported_labels(ALIASES, LANG_CODES))
raise ValueError(
f"mlx-audio's Kokoro model (mlx-community/Kokoro-82M-bf16) doesn't "
f"support language={language!r}. Kokoro supports: {supported}. "
@@ -1840,22 +1871,40 @@ class MLXAudioBackend(TTSBackend):
# model is actually active.
kwargs["lang_code"] = language[:2].lower()
pieces = []
def collect(results):
groups = []
rate = None
pending = []
def flush():
if not pending:
return
audio = np.concatenate(pending, axis=-1)
if rate != self.sample_rate:
import torchaudio
audio = torchaudio.functional.resample(
torch.from_numpy(audio), rate, self.sample_rate,
).numpy()
groups.append(audio)
pending.clear()
for result in results:
audio = getattr(result, "audio", result)
if hasattr(audio, "numpy"):
audio = audio.numpy()
sr = getattr(result, "sample_rate", self.sample_rate)
if sr != rate:
flush()
rate = sr
pending.append(np.asarray(audio, dtype=np.float32))
flush()
return groups
try:
for result in self._model.generate(**kwargs):
audio = getattr(result, "audio", result)
if hasattr(audio, "numpy"):
audio = audio.numpy()
pieces.append(np.asarray(audio, dtype=np.float32))
pieces = collect(self._model.generate(**kwargs))
except TypeError:
# Some engines don't accept lang_code / ref_audio. Retry with
# only the universal kwargs.
pieces = []
for result in self._model.generate(text=text, speed=speed):
audio = getattr(result, "audio", result)
if hasattr(audio, "numpy"):
audio = audio.numpy()
pieces.append(np.asarray(audio, dtype=np.float32))
# Retry engines that accept only the universal arguments.
pieces = collect(self._model.generate(text=text, speed=speed))
if not pieces:
raise RuntimeError(f"mlx-audio ({self._model_id}) produced no audio")
+2
View File
@@ -41,6 +41,8 @@ def test_cuda_oom_falls_back_to_cpu(monkeypatch):
return object() # CPU load succeeds
monkeypatch.setattr(whisperx, "load_model", fake_load_model)
# Exercise load-time OOM, independent of actual free VRAM on this host.
monkeypatch.setattr(WhisperXBackend, "_free_vram_gb", staticmethod(lambda: 10.0))
be = WhisperXBackend()
# Force the CUDA starting point regardless of the CI host's hardware.
+247
View File
@@ -0,0 +1,247 @@
"""Regression tests for #692 — the CTranslate2 exec-stack rejection is now
*repaired*, not merely routed around.
ctranslate2 4.4.0 (what whisperx 3.4.5 pins, and 3.4.5 is the last release
supporting the Python 3.11 we ship) marks its native library's stack as
executable. Kernels that refuse the request fail the dlopen outright, which
took out both CTranslate2 ASR engines *and* Argos the default offline dub
translation engine, whose bare `argostranslate` import succeeds without the
native dep, so the engine advertised itself as ready and every translate
request 500'd with an opaque ImportError.
`core.execstack` clears the one ELF bit that causes it. These tests pin the
patcher on synthetic ELFs (no ctranslate2 needed), the memoization, and the
wiring into every consumer probe.
"""
import struct
import pytest
_PT_GNU_STACK = 0x6474E551
_PT_LOAD = 1
def _elf(path, *, bits=64, endian="<", flags=0x7, phdr_type=_PT_GNU_STACK):
"""Write a minimal ELF whose second program header is `phdr_type`.
Only the fields the patcher reads are meaningful a real linker would emit
far more, but the point is to prove the offsets are computed correctly for
both ELF classes, and to fail loudly if they ever drift.
"""
is_64 = bits == 64
phentsize = 56 if is_64 else 32
phoff = 64 if is_64 else 52
ident = b"\x7fELF" + bytes([2 if is_64 else 1, 1 if endian == "<" else 2, 1]) + b"\0" * 9
header = bytearray(phoff)
header[: len(ident)] = ident
if is_64:
struct.pack_into(endian + "Q", header, 0x20, phoff)
struct.pack_into(endian + "HH", header, 0x36, phentsize, 2)
else:
struct.pack_into(endian + "I", header, 0x1C, phoff)
struct.pack_into(endian + "HH", header, 0x2A, phentsize, 2)
def _phdr(p_type, p_flags):
raw = bytearray(phentsize)
struct.pack_into(endian + "I", raw, 0, p_type)
struct.pack_into(endian + "I", raw, 4 if is_64 else 24, p_flags)
return bytes(raw)
path.write_bytes(bytes(header) + _phdr(_PT_LOAD, 0x5) + _phdr(phdr_type, flags))
return str(path)
@pytest.mark.parametrize("bits", [64, 32])
@pytest.mark.parametrize("endian", ["<", ">"])
def test_clear_execstack_flips_only_the_x_bit(tmp_path, bits, endian):
from core import execstack
lib = _elf(tmp_path / "libfake.so", bits=bits, endian=endian, flags=0x7)
assert execstack.has_execstack(lib) is True
changed, detail = execstack.clear_execstack(lib)
assert changed is True and "cleared" in detail
assert execstack.has_execstack(lib) is False
# Read + write permissions survive; only PF_X is gone.
with open(lib, "rb") as fh:
offset, flags, _ = execstack._gnu_stack_flags_offset(fh)
assert flags == 0x6
# Idempotent: a second pass is a no-op, so a restart never rewrites.
assert execstack.clear_execstack(lib) == (False, "already non-executable")
def test_non_executable_stack_is_left_alone(tmp_path):
from core import execstack
lib = _elf(tmp_path / "libok.so", flags=0x6)
assert execstack.has_execstack(lib) is False
before = (tmp_path / "libok.so").read_bytes()
assert execstack.clear_execstack(lib) == (False, "already non-executable")
assert (tmp_path / "libok.so").read_bytes() == before
def test_elf_without_gnu_stack_segment(tmp_path):
from core import execstack
lib = _elf(tmp_path / "libnostack.so", phdr_type=_PT_LOAD, flags=0x7)
assert execstack.has_execstack(lib) is None
assert execstack.clear_execstack(lib) == (False, "no PT_GNU_STACK segment")
def test_non_elf_and_missing_files_are_not_errors(tmp_path):
from core import execstack
text = tmp_path / "notelf.so"
text.write_bytes(b"#!/bin/sh\necho hi\n")
assert execstack.has_execstack(str(text)) is None
assert execstack.clear_execstack(str(text))[0] is False
missing = str(tmp_path / "nope" / "libghost.so")
assert execstack.has_execstack(missing) is None
changed, detail = execstack.clear_execstack(missing)
assert changed is False and "unreadable" in detail
def test_ensure_rechecks_unrepairable_libraries(tmp_path, monkeypatch):
from core import execstack
lib = _elf(tmp_path / "libctranslate2-test.so.4.4.0", flags=0x7)
monkeypatch.setattr(execstack.sys, "platform", "linux")
monkeypatch.setattr(execstack, "ctranslate2_library_paths", lambda: [lib])
monkeypatch.setattr(
execstack, "clear_execstack", lambda p: (False, "not writable (PermissionError)")
)
execstack.reset_ctranslate2_cache()
ok, detail = execstack.ensure_ctranslate2_loadable()
assert ok is False
# Actionable: names the library, why, and both ways out.
assert "executable stack" in detail and "patchelf" in detail and "3.12" in detail
# An external repair must become visible without a backend restart.
_elf(tmp_path / "libctranslate2-test.so.4.4.0", flags=0x6)
assert execstack.ensure_ctranslate2_loadable()[0] is True
def test_ensure_repairs_then_reports_ok(tmp_path, monkeypatch):
from core import execstack
lib = _elf(tmp_path / "libctranslate2-test.so.4.4.0", flags=0x7)
monkeypatch.setattr(execstack.sys, "platform", "linux")
monkeypatch.setattr(execstack, "ctranslate2_library_paths", lambda: [lib])
execstack.reset_ctranslate2_cache()
ok, detail = execstack.ensure_ctranslate2_loadable()
assert ok is True and "repaired" in detail
assert execstack.has_execstack(lib) is False
execstack.reset_ctranslate2_cache()
def test_ensure_is_a_noop_off_linux(tmp_path, monkeypatch):
"""macOS/Windows never reject an exec-stack request — don't touch signed
bundles looking for a problem that cannot exist there."""
from core import execstack
monkeypatch.setattr(execstack.sys, "platform", "darwin")
monkeypatch.setattr(
execstack, "ctranslate2_library_paths", lambda: pytest.fail("probed off Linux")
)
execstack.reset_ctranslate2_cache()
ok, detail = execstack.ensure_ctranslate2_loadable()
assert ok is True and "off Linux" in detail
execstack.reset_ctranslate2_cache()
# ── Wiring: every consumer of the native lib must consult the repair ─────────
def test_asr_probes_report_unavailable_when_repair_impossible(monkeypatch):
from services import asr_backend as ab
monkeypatch.setattr(
ab, "_ctranslate2_execstack_ok", lambda: (False, "kernel refuses it")
)
okx, msgx = ab.WhisperXBackend.is_available()
okf, msgf = ab.FasterWhisperBackend.is_available()
assert okx is False and "CTranslate2" in msgx and "kernel refuses it" in msgx
assert okf is False and "CTranslate2" in msgf and "kernel refuses it" in msgf
def test_argos_probe_module_pulls_the_native_dep():
"""`argostranslate` alone imports fine with a broken CTranslate2 — the
registry must probe the module that actually loads it, or the Engine
selector advertises an engine whose every request fails."""
from services.translation_engines import REGISTRY
assert REGISTRY["argos"]["probe_module"] == "argostranslate.translate"
def test_engine_probe_survives_a_native_load_failure(monkeypatch):
from services import translation_engines as te
def boom(name):
raise OSError("libctranslate2-x.so: cannot enable executable stack")
monkeypatch.setattr(te.importlib, "import_module", boom)
ok, detail = te._probe({"probe_module": "deep_translator"})
assert ok is False and "OSError" in detail
@pytest.mark.parametrize("bits", [32, 64])
@pytest.mark.parametrize("length", [16, 31, 45, 63])
def test_truncated_elf_is_not_an_error(tmp_path, bits, length):
from core import execstack
path = tmp_path / "short.so"
_elf(path, bits=bits)
path.write_bytes(path.read_bytes()[:length])
before = path.read_bytes()
assert execstack.has_execstack(str(path)) is None
assert execstack.clear_execstack(str(path))[0] is False
assert path.read_bytes() == before
def test_install_after_absent_probe_is_detected(tmp_path, monkeypatch):
from core import execstack
monkeypatch.setattr(execstack.sys, "platform", "linux")
libs = []
monkeypatch.setattr(execstack, "ctranslate2_library_paths", lambda: libs)
assert execstack.ensure_ctranslate2_loadable()[0]
libs.append(_elf(tmp_path / "new.so"))
assert execstack.ensure_ctranslate2_loadable()[0]
assert execstack.has_execstack(libs[0]) is False
def test_concurrent_repairs_remain_available(tmp_path, monkeypatch):
from core import execstack
from concurrent.futures import ThreadPoolExecutor
lib = _elf(tmp_path / "parallel.so")
monkeypatch.setattr(execstack.sys, "platform", "linux")
monkeypatch.setattr(execstack, "ctranslate2_library_paths", lambda: [lib])
with ThreadPoolExecutor(max_workers=8) as pool:
results = list(pool.map(lambda _: execstack.ensure_ctranslate2_loadable(), range(24)))
assert all(ok for ok, _ in results)
assert execstack.has_execstack(lib) is False
def test_isolated_probe_checks_repair_before_import(monkeypatch):
from core import execstack
from services.subprocess_asr import IsolatedFasterWhisperBackend
monkeypatch.setattr(execstack, "ensure_ctranslate2_loadable", lambda: (False, "repair blocked"))
ok, detail = IsolatedFasterWhisperBackend.is_available()
assert not ok and "repair blocked" in detail
def test_repair_uses_host_locking_capability_when_target_platform_is_emulated(tmp_path, monkeypatch):
from core import execstack
import builtins
import os
from types import SimpleNamespace
lib = _elf(tmp_path / 'libctranslate2-test.so', flags=0x7)
monkeypatch.setattr(execstack.sys, 'platform', 'linux')
monkeypatch.setattr(execstack, 'os', SimpleNamespace(**{**vars(os), 'name': 'nt'}))
original_import = builtins.__import__
def windows_import(name, *args, **kwargs):
if name == 'fcntl':
raise ModuleNotFoundError('fcntl is unavailable on Windows')
return original_import(name, *args, **kwargs)
monkeypatch.setattr(builtins, '__import__', windows_import)
changed, _ = execstack.clear_execstack(lib)
assert changed
assert execstack.has_execstack(lib) is False
+192 -9
View File
@@ -27,10 +27,7 @@ from pathlib import Path
import pytest
from services.subprocess_backend import (
RECV_TIMEOUT_S,
SubprocessBackend,
)
from services.subprocess_backend import RECV_TIMEOUT_S, SubprocessBackend
from services.tts_backend import OmniVoiceBackend, get_backend_class, list_backends
from engines.omnivoice_subprocess import (
OmniVoiceMPSSubprocessBackend,
@@ -321,11 +318,14 @@ class _PlainBackend(SubprocessBackend):
return ["multi"]
def test_base_default_recv_timeout_is_60s():
# A subclass that does NOT override keeps the conservative default, so the
# existing subprocess engines (IndexTTS, dots.tts, ...) are byte-identical.
assert SubprocessBackend.recv_timeout_s == RECV_TIMEOUT_S == 60.0
assert _PlainBackend().recv_timeout_s == 60.0
def test_base_default_recv_timeout_covers_a_generation():
# A sidecar that does not choose gets a deadline that outlasts the
# wall-clock budget its own job was granted (#2103).
from services.subprocess_backend import GENERATE_RECV_TIMEOUT_S
assert SubprocessBackend.recv_timeout_s == GENERATE_RECV_TIMEOUT_S == 600.0
assert _PlainBackend().recv_timeout_s == 600.0
# The ping budget itself is unchanged: health_check() still wants 60s.
assert RECV_TIMEOUT_S == 60.0
def test_sidecar_spawn_delegates_all_containment_to_nested_owner(monkeypatch, tmp_path):
@@ -414,6 +414,189 @@ def test_omnivoice_subprocess_recv_timeout_floors_at_30s(monkeypatch):
assert OmniVoiceSubprocessBackend().recv_timeout_s == 30.0
def test_subprocess_sidecar_timeout_error_message(monkeypatch):
"""When a sidecar hits its receive timeout, generate() must report the
timeout and deadline rather than describing a generic pipe-closed crash (#2103)."""
b = _PlainBackend()
class _FakeProc:
def poll(self):
return None
b._proc = _FakeProc()
monkeypatch.setattr(b, "_send", lambda msg: None)
def fake_recv_timeout(timeout_s):
b._last_recv_timed_out = True
return None
monkeypatch.setattr(b, "_recv_with_timeout", fake_recv_timeout)
monkeypatch.setattr(b, "_reap_unusable_process", lambda proc: None)
with pytest.raises(RuntimeError) as exc:
b.generate("hello")
assert "600s" in str(exc.value)
assert "stopped it" in str(exc.value)
b._proc = None
def test_subprocess_sidecar_initial_empty_crash_reports_pipe_closed(monkeypatch):
"""When a sidecar process terminates without timing out, generate() reports pipe closure (#2103)."""
b = _PlainBackend()
class _FakeProc:
def poll(self):
return None
b._proc = _FakeProc()
monkeypatch.setattr(b, "_send", lambda msg: None)
def fake_recv_crash(timeout_s):
b._last_recv_timed_out = False
return None
monkeypatch.setattr(b, "_recv_with_timeout", fake_recv_crash)
monkeypatch.setattr(b, "_reap_unusable_process", lambda proc: None)
with pytest.raises(RuntimeError) as exc:
b.generate("hello")
assert "sidecar closed pipe mid-generate" in str(exc.value)
b._proc = None
def test_subprocess_engine_timeouts_raised():
"""All large subprocess TTS engines must declare generous timeouts rather
than inheriting the 60s class default (#2103)."""
from engines.confucius4 import Confucius4Backend
from engines.dots_tts import DotsTTSBackend
from engines.moss_tts_v15 import MossTTSV15Backend
from engines.supertonic3.backend import Supertonic3Backend
assert Confucius4Backend().recv_timeout_s >= 300.0
assert DotsTTSBackend().recv_timeout_s >= 300.0
assert MossTTSV15Backend().recv_timeout_s >= 300.0
assert Supertonic3Backend().recv_timeout_s >= 300.0
def test_subprocess_engine_timeout_env_overrides(monkeypatch):
"""Subprocess TTS engines honor their engine-specific receive timeout env overrides (#2103)."""
from engines.confucius4 import Confucius4Backend
from engines.dots_tts import DotsTTSBackend
from engines.moss_tts_v15 import MossTTSV15Backend
from engines.supertonic3.backend import Supertonic3Backend
monkeypatch.setenv("OMNIVOICE_CONFUCIUS4_RECV_TIMEOUT_S", "1200")
monkeypatch.setenv("OMNIVOICE_DOTS_TTS_RECV_TIMEOUT_S", "1000")
monkeypatch.setenv("OMNIVOICE_MOSS_TTS_V15_RECV_TIMEOUT_S", "1100")
monkeypatch.setenv("OMNIVOICE_SUPERTONIC3_RECV_TIMEOUT_S", "500")
assert Confucius4Backend().recv_timeout_s == 1200.0
assert DotsTTSBackend().recv_timeout_s == 1000.0
assert MossTTSV15Backend().recv_timeout_s == 1100.0
assert Supertonic3Backend().recv_timeout_s == 500.0
def test_subprocess_asr_recv_timeout_env_override(monkeypatch):
"""SubprocessASRBackend.recv_timeout_s honors OMNIVOICE_ASR_RECV_TIMEOUT_S
and safely rejects non-finite, malformed, zero, or negative inputs (#2103)."""
from pathlib import Path
from services.subprocess_asr import SubprocessASRBackend
class _FakeASR(SubprocessASRBackend):
id = "fake-asr"
display_name = "fake-asr"
gpu_compat = ("cuda", "mps", "cpu")
@classmethod
def is_available(cls): return True, "ok"
@classmethod
def venv_python(cls): return Path(sys.executable)
@classmethod
def sidecar_script(cls): return Path("fake")
b = _FakeASR()
assert b.recv_timeout_s == 600.0
# Valid override
monkeypatch.setenv("OMNIVOICE_ASR_RECV_TIMEOUT_S", "750")
assert b.recv_timeout_s == 750.0
# Malformed value falls back to default
monkeypatch.setenv("OMNIVOICE_ASR_RECV_TIMEOUT_S", "not-a-number")
assert b.recv_timeout_s == 600.0
# Non-finite values fall back to default
for invalid in ("nan", "inf", "-inf"):
monkeypatch.setenv("OMNIVOICE_ASR_RECV_TIMEOUT_S", invalid)
assert b.recv_timeout_s == 600.0
# Zero or negative values clamped to 30.0 minimum
for low in ("0", "-10", "15"):
monkeypatch.setenv("OMNIVOICE_ASR_RECV_TIMEOUT_S", low)
assert b.recv_timeout_s == 30.0
def test_subprocess_asr_timeout_error_message(monkeypatch):
"""SubprocessASRBackend.transcribe() raises actionable timeout guidance when watchdog fires (#2103)."""
from pathlib import Path
from services.subprocess_asr import SubprocessASRBackend
class _FakeASR(SubprocessASRBackend):
id = "fake-asr"
display_name = "fake-asr"
gpu_compat = ("cuda", "mps", "cpu")
@classmethod
def is_available(cls): return True, "ok"
@classmethod
def venv_python(cls): return Path(sys.executable)
@classmethod
def sidecar_script(cls): return Path("fake")
class _FakeProc:
def poll(self): return None
def wait(self, timeout=None): return 0
def kill(self): pass
def terminate(self): pass
b = _FakeASR()
b._proc = _FakeProc()
monkeypatch.setattr(b, "_spawn", lambda: None)
monkeypatch.setattr(b, "_send", lambda msg: None)
def fake_recv_timeout(timeout_s):
b._last_recv_timed_out = True
return None
monkeypatch.setattr(b, "_recv_with_timeout", fake_recv_timeout)
monkeypatch.setattr(b, "shutdown", lambda: None)
monkeypatch.setattr("services.model_manager.running_on_gpu_pool", lambda: True)
with pytest.raises(RuntimeError) as exc:
b.transcribe("test.wav")
assert "fake-asr ASR sidecar exceeded receive timeout" in str(exc.value)
assert "OMNIVOICE_ASR_RECV_TIMEOUT_S" in str(exc.value)
def test_generate_timeout_s_coordinates_with_engine_recv_timeout():
"""Outer generation timeout must coordinate with engine sidecar timeout with bounded grace (#2103)."""
from services.model_manager import generate_timeout_s
class _SlowEngine:
recv_timeout_s = 900.0
# With 900s sidecar timeout, outer budget must include at least 5s grace (>= 905s).
budget = generate_timeout_s("short text", engine=_SlowEngine())
assert budget >= 905.0
class _FastEngine:
recv_timeout_s = 60.0
# For fast engines, the default GPU/CPU budget still applies as the floor.
budget_fast = generate_timeout_s("short text", engine=_FastEngine(), execution_device="cuda")
assert budget_fast >= 300.0
# ── roundtrip via the stub sidecar ─────────────────────────────────────────
@@ -0,0 +1,118 @@
"""The pytorch-whisper fallback must survive a CUDA OOM instead of losing a chunk.
The VRAM preflight sizes the *weights*; generation adds a workspace that scales
with the batch, and word timestamps keep every layer's cross-attention for the
whole batch. So a card with room for the model still OOMs at the first
transcribe and the dub path merely retried the identical call, gave up, and
emitted nothing for that chunk: a silent hole in the transcript, indistinguishable
from silence. Step the batch down, then finish on CPU.
"""
import pytest
from services.asr_backend import PyTorchWhisperBackend
_OOM = RuntimeError("CUDA out of memory. Tried to allocate 2.00 GiB")
class _FakePipe:
"""Stands in for a transformers ASR pipeline on CUDA."""
def __init__(self, oom_below_batch):
self.device = "cuda:0"
self.oom_below_batch = oom_below_batch
self.calls = []
def __call__(self, _audio, *, return_timestamps, chunk_length_s, batch_size):
self.calls.append(batch_size)
if batch_size > self.oom_below_batch:
raise _OOM
return {"text": "ok", "chunks": [{"text": "ok", "timestamp": (0.0, 1.0)}]}
@pytest.fixture()
def audio(tmp_path):
import numpy as np
import soundfile as sf
path = tmp_path / "a.wav"
sf.write(path, np.zeros(16000, dtype="float32"), 16000)
return str(path)
def test_oom_steps_down_the_batch_and_still_returns_text(audio):
pipe = _FakePipe(oom_below_batch=2)
backend = PyTorchWhisperBackend(asr_pipe=pipe)
out = backend.transcribe(audio, word_timestamps=True)
assert out["text"] == "ok"
# Tried the word-timestamp ladder in order, stopping at the first that fits.
assert pipe.calls == [8, 2]
@pytest.mark.parametrize(
"word_timestamps,expected", [(True, [8, 2, 1]), (False, [16, 4, 1])]
)
def test_batch_ladder_is_smaller_when_word_timestamps_are_requested(
audio, monkeypatch, word_timestamps, expected
):
"""Word timestamps retain per-layer cross-attention for the whole batch, so
the ladder must start lower than for plain transcription."""
pipe = _FakePipe(oom_below_batch=0)
backend = PyTorchWhisperBackend(asr_pipe=pipe)
sentinel = RuntimeError("cpu rebuild reached")
monkeypatch.setattr(
backend, "_rebuild_on_cpu", lambda: (_ for _ in ()).throw(sentinel)
)
with pytest.raises(RuntimeError, match="cpu rebuild reached"):
backend.transcribe(audio, word_timestamps=word_timestamps)
assert pipe.calls == expected
def test_exhausted_ladder_falls_back_to_cpu(audio, monkeypatch):
pipe = _FakePipe(oom_below_batch=0)
backend = PyTorchWhisperBackend(asr_pipe=pipe)
cpu = _FakePipe(oom_below_batch=99)
cpu.device = "cpu"
def _rebuild():
backend._pipe = cpu
monkeypatch.setattr(backend, "_rebuild_on_cpu", _rebuild)
out = backend.transcribe(audio, word_timestamps=True)
assert out["text"] == "ok"
assert pipe.calls == [8, 2, 1] and cpu.calls == [2]
def test_non_oom_errors_are_not_retried(audio):
class _Boom(_FakePipe):
def __call__(self, *a, **kw):
self.calls.append(kw["batch_size"])
raise ValueError("bad audio")
pipe = _Boom(oom_below_batch=99)
with pytest.raises(ValueError):
PyTorchWhisperBackend(asr_pipe=pipe).transcribe(audio)
assert pipe.calls == [8] # one attempt, no ladder
def test_cpu_pipeline_keeps_the_small_batch(audio):
pipe = _FakePipe(oom_below_batch=99)
pipe.device = "cpu"
PyTorchWhisperBackend(asr_pipe=pipe).transcribe(audio)
assert pipe.calls == [2]
def test_cpu_fallback_preserves_injected_model_and_processor(monkeypatch):
import torch
from types import SimpleNamespace
calls = []
model = SimpleNamespace(to=lambda **kw: calls.append(kw))
processor = object()
pipe = SimpleNamespace(model=model, tokenizer=processor, feature_extractor=processor, device="cuda:0")
backend = PyTorchWhisperBackend(asr_pipe=pipe)
monkeypatch.setattr(backend, "_model_name", lambda: pytest.fail("must not resolve another checkpoint"))
backend._rebuild_on_cpu()
assert backend._pipe is pipe and pipe.model is model
assert pipe.tokenizer is processor and pipe.feature_extractor is processor
assert str(pipe.device) == "cpu"
assert calls == [{"device": "cpu", "dtype": torch.float32}]
@@ -0,0 +1,249 @@
"""Every sidecar's generate deadline outlasts the job budget it was granted (#2103).
#1611 raised IndexTTS's deadline because a healthy synthesis was being killed at
60s. That fixed the reported engine and left the class default alone, so four
more engines confucius4, dots_tts, moss_tts_v15, supertonic3 inherited the
same 60s and were killed the same way.
60s is the ``health_check`` ping budget. Inheriting it as a *generation*
deadline puts the sidecar watchdog five to ten times below
``model_manager.generate_timeout_s`` (300s accelerated, 600s CPU), so the
watchdog reclaims a sidecar the caller still considers well inside its budget.
Every engine that overrode the hook picked 300s..900s, i.e. at or above the
accelerated budget; the four that stayed silent are the whole bug.
The invariant below is what keeps a new engine from re-entering that state by
omission, which is the part #1611 could not do by fixing one engine.
"""
import pytest
# The engines named in #2103 that inherited the ping budget. Listed explicitly
# so the regression is legible even if the registry is reorganised later.
REGRESSED_ENGINE_IDS = ("confucius4-tts", "dots-tts", "moss-tts-v15", "supertonic3")
def _subprocess_backend_classes():
"""Every SubprocessBackend the registry can hand a user, by id."""
from services.tts_backend import get_backend_class
from services.tts_backend import list_backends
found = {}
for row in list_backends(include_hidden=True):
try:
cls = get_backend_class(row["id"])
except Exception:
continue # an engine whose optional import is absent cannot be dispatched
if isinstance(cls, type) and getattr(cls, "_is_subprocess_isolated", False) and hasattr(cls, "recv_timeout_s"):
found[row["id"]] = cls
return found
def test_ping_budget_and_generate_budget_are_separate_constants():
# A ping must stay fast; a generation must not be cut off at a ping's deadline.
from services.subprocess_backend import GENERATE_RECV_TIMEOUT_S
from services.subprocess_backend import RECV_TIMEOUT_S
assert RECV_TIMEOUT_S == 60.0
assert GENERATE_RECV_TIMEOUT_S > RECV_TIMEOUT_S
def test_default_generate_deadline_covers_the_cpu_job_budget():
# Lockstep with model_manager: raising either budget there without raising
# this one re-opens #2103 for every engine that does not override.
# Imported rather than duplicated so the two cannot drift silently.
from services.subprocess_backend import GENERATE_RECV_TIMEOUT_S
from services.subprocess_backend import SubprocessBackend
assert GENERATE_RECV_TIMEOUT_S >= 600.0
assert SubprocessBackend.recv_timeout_s == GENERATE_RECV_TIMEOUT_S
@pytest.mark.parametrize("engine_id", REGRESSED_ENGINE_IDS)
def test_regressed_engines_no_longer_inherit_the_ping_budget(engine_id):
from services.subprocess_backend import RECV_TIMEOUT_S
cls = _subprocess_backend_classes().get(engine_id)
if cls is None:
pytest.fail(f"{engine_id} is not registered in this build")
# Read through an instance: several engines expose the hook as a property.
assert cls.__new__(cls).recv_timeout_s > RECV_TIMEOUT_S
def test_no_registered_sidecar_undercuts_the_accelerated_job_budget():
"""The class-level guard #1611 was missing.
A new SubprocessBackend that simply does not think about ``recv_timeout_s``
now inherits a deadline that already satisfies this; one that overrides it
with something too small fails here rather than in a user's generation.
"""
from services.model_manager import GPU_JOB_TIMEOUT_S
too_short = {}
for engine_id, cls in _subprocess_backend_classes().items():
deadline = cls.__new__(cls).recv_timeout_s
if deadline < GPU_JOB_TIMEOUT_S:
too_short[engine_id] = deadline
assert not too_short, (
"these sidecars would be killed before their own job budget expires: "
f"{too_short} (accelerated budget is {GPU_JOB_TIMEOUT_S:g}s)"
)
# ── a constant is not enough: the budget scales with the text (#2109 review) ─
def _SilentBackend():
from services.subprocess_backend import SubprocessBackend
from services.subprocess_backend import SubprocessBackend
class SilentBackend(SubprocessBackend):
"""A sidecar with no custom deadline."""
id = "silent"
@classmethod
def is_available(cls):
return True, "ok"
@property
def sample_rate(self):
return 24000
@property
def supported_languages(self):
return ["multi"]
return SilentBackend()
def _OpinionatedBackend():
backend = _SilentBackend()
type(backend).id = "opinionated"
type(backend).recv_timeout_s = 45.0
return backend
def test_a_long_passage_raises_the_deadline_past_the_flat_default():
# generate_timeout_s adds 1s per 40 characters past a 1200-char allowance,
# so a long passage is granted more than the flat floor.
from services.subprocess_backend import GENERATE_RECV_TIMEOUT_S
backend = _SilentBackend()
short = backend._effective_recv_timeout_s("hello")
long_text = "x" * 200_000
long_deadline = backend._effective_recv_timeout_s(long_text)
assert short == GENERATE_RECV_TIMEOUT_S
assert long_deadline > short
# And it tracks the budget itself, not some second guess at it.
from services.model_manager import generate_timeout_s
assert generate_timeout_s(long_text, engine=backend) >= long_deadline + 5.0
def test_an_engine_that_opts_down_keeps_its_own_deadline():
# #2103 asks that fast engines stay able to opt down, so deriving from the
# budget must not overrule an override in either direction.
backend = _OpinionatedBackend()
assert backend._effective_recv_timeout_s("hello") == 45.0
assert backend._effective_recv_timeout_s("x" * 200_000) == 45.0
def test_budget_probe_failure_falls_back_instead_of_failing_the_generate(monkeypatch):
from services.subprocess_backend import GENERATE_RECV_TIMEOUT_S
import services.model_manager as mm
def _boom(*a, **kw):
raise RuntimeError("device probe unavailable")
monkeypatch.setattr(mm, "generate_timeout_s", _boom)
assert _SilentBackend()._effective_recv_timeout_s("hello") == GENERATE_RECV_TIMEOUT_S
# ── the deadline has to appear in the error the caller sees (#2103) ─────────
# Wedges on the first synthesize, so the parent's watchdog is the only thing
# that can end the request — the exact shape the #1611 and #2103 reporters hit.
WEDGING_SIDECAR = r'''
import sys, json, struct, time
def _send(o):
b = json.dumps(o, separators=(",", ":")).encode()
sys.stdout.buffer.write(struct.pack("!I", len(b)) + b)
sys.stdout.buffer.flush()
_send({"op": "ready", "engine": "omnivoice-subprocess", "sample_rate": 24000})
print("sidecar still alive, just slow", file=sys.stderr, flush=True)
while True:
time.sleep(1)
'''
def test_timeout_error_names_the_deadline_instead_of_blaming_the_pipe(
tmp_path, monkeypatch,
):
"""#2103's second half: the watchdog's own deadline reached the user.
Before this, a kill and a crash both raised "sidecar closed pipe
mid-generate", so the one fact that explains the failure — that
VoiceStudio stopped the sidecar on its own deadline appeared only in the
backend log, and reporters reasonably concluded the engine had crashed.
"""
from engines.omnivoice_subprocess import OmniVoiceSubprocessBackend
script = tmp_path / "wedging_sidecar.py"
script.write_text(WEDGING_SIDECAR)
monkeypatch.setattr(
OmniVoiceSubprocessBackend, "sidecar_script", classmethod(lambda cls: script),
)
# 2s so the test is fast; the property floors env overrides at 30s, so set
# the attribute the base actually reads (as the existing wedge test does).
monkeypatch.setattr(
OmniVoiceSubprocessBackend, "recv_timeout_s", property(lambda self: 2.0),
)
backend = OmniVoiceSubprocessBackend()
try:
with pytest.raises(RuntimeError) as excinfo:
backend.generate("anything")
finally:
backend.shutdown()
message = str(excinfo.value)
assert "2s" in message, message # the deadline that ended it
assert "stopped it" in message, message # who ended it, not "it closed"
assert "closed pipe" not in message, message
# #2026's stderr tail is carried on this path too, so a sidecar that did
# say something before the kill is not silenced by the timeout.
assert "still alive" in message, message
@pytest.mark.parametrize("text", ["short", "x" * 200000], ids=["short", "long"])
@pytest.mark.parametrize("engine_type", [_SilentBackend, _OpinionatedBackend])
def test_outer_guard_outlasts_sidecar_watchdog(text, engine_type):
from services.model_manager import generate_timeout_s
backend = engine_type()
assert generate_timeout_s(text, engine=backend) >= backend._effective_recv_timeout_s(text) + 5.0
def test_explicit_generation_budget_is_authoritative(monkeypatch):
import services.model_manager as mm
monkeypatch.setattr(mm, 'GPU_JOB_TIMEOUT_S', 12.0)
monkeypatch.setattr(mm, '_GENERATE_TIMEOUT_EXPLICIT', True)
assert mm.generate_timeout_s('short', engine=_SilentBackend(), execution_device='cuda') == 12.0
@pytest.mark.asyncio
@pytest.mark.parametrize('guard_kind', ['asr', 'tts'])
async def test_outer_abandonment_terminates_owned_sidecar(tmp_path, monkeypatch, guard_kind):
from engines.omnivoice_subprocess import OmniVoiceSubprocessBackend
import asyncio
from concurrent.futures import ThreadPoolExecutor
from services.model_manager import run_on_gpu_pool_guarded
from services.asr_backend import run_transcribe_guarded
script = tmp_path / 'wedging_sidecar.py'
script.write_text(WEDGING_SIDECAR)
monkeypatch.setattr(OmniVoiceSubprocessBackend, 'sidecar_script', classmethod(lambda cls: script))
monkeypatch.setattr(OmniVoiceSubprocessBackend, 'recv_timeout_s', property(lambda self: 600.0))
backend = OmniVoiceSubprocessBackend()
# Guard lifetime is independent of which protocol operation is waiting.
with ThreadPoolExecutor(max_workers=1) as executor:
try:
if guard_kind == 'asr':
work = run_transcribe_guarded(executor, lambda: backend.generate('hang'), timeout=0.5)
else:
work = run_on_gpu_pool_guarded(lambda: backend.generate('hang'), executor=executor, timeout=0.5)
with pytest.raises(TimeoutError):
await work
assert backend._proc is not None
await asyncio.wait_for(asyncio.to_thread(backend._proc.wait, timeout=5), timeout=6)
assert backend._proc.poll() is not None
finally:
backend.shutdown()
+6 -7
View File
@@ -21,13 +21,12 @@ The pinned commit SHA for `omnivoice.cpp` lives in
scripts/build-omnivoice-tts.sh --platform <slug> --commit-sha <40hex>
```
See `.github/workflows/ci.yml` `build-omnivoice-tts` job for the CI
matrix that produces these artifacts on every PR. The Apple Silicon
slot (`macos-14`) is marked `continue-on-error: true` because the
upstream `omnivoice.cpp` README does not publish a `buildmetal.sh`
(Pitfall 1 in `04-RESEARCH.md`); a failed Metal build is documented
and the macOS Apple Silicon cloning default falls back to the
in-process `VoiceStudioBackend`.
See `.github/workflows/build-omnivoice-tts.yml` `build-omnivoice-tts` job
for the CI matrix that produces these artifacts. Apple Silicon (`macos-14`)
builds cleanly with `-DGGML_METAL=ON` at the pinned SHA (#2105), enabling
hardware-accelerated Metal inference when the packaged artifact passes binary
preflight and macOS permits execution. Missing or blocked binaries retain the
in-process `VoiceStudioBackend` fallback.
## Placeholder note
+1 -1
View File
@@ -75,7 +75,7 @@
},
"frontend": {
"name": "omnivoice-studio",
"version": "0.5.3",
"version": "0.5.4",
"dependencies": {
"@fontsource-variable/inter": "^5.3.0",
"@fontsource-variable/source-serif-4": "^5.3.0",
+21
View File
@@ -103,6 +103,27 @@ RUN python3 -c "import os, torch, torchaudio, torchvision; \
import torchvision.ops; torchvision.ops.nms; \
print('torchvision C++ ops resolve against this torch')"
# CTranslate2 (WhisperX, faster-whisper) links cuDNN 8, but the CUDA base image
# ships cuDNN 9, so libcudnn_ops_infer.so.8 is absent and loading it aborts the
# backend process outright rather than raising (#1371). scripts/setup.py
# side-loads the cuDNN 8 libraries for source installs; the image needs the same
# shim or Docker users lose every CTranslate2 ASR engine (#2050).
#
# The target is derived from sys.prefix rather than hardcoded: backend/core/
# cudnn8.py looks for <sys.prefix>/lib/pythonX.Y/site-packages/cudnn8_compat,
# and sys.prefix differs between the conda-based CUDA image and the ROCm venv.
# --no-deps keeps this to the cuDNN wheels alone, leaving the base image's torch
# stack untouched. Skipped for ROCm, which does not use cuDNN.
RUN if [ "$GPU_FLAVOR" = "cuda" ]; then \
target="$(python3 -c "import os, sys; print(os.path.join(sys.prefix, 'lib', 'python%d.%d' % sys.version_info[:2], 'site-packages', 'cudnn8_compat'))")" && \
uv pip install --python "$(command -v python3)" --no-cache --no-deps \
--target "$target" nvidia-cudnn-cu12==8.9.7.29 && \
python3 -c "import os, sys; d = os.path.join(sys.prefix, 'lib', 'python%d.%d' % sys.version_info[:2], 'site-packages', 'cudnn8_compat', 'nvidia', 'cudnn', 'lib'); \
libs = [f for f in os.listdir(d) if '.so.8' in f]; \
assert libs, 'cudnn8_compat installed but no .so.8 libraries in ' + d; \
print('cuDNN 8 compat libraries: %d' % len(libs))"; \
fi
# Copy application source
COPY backend/ ./backend/
COPY omnivoice/ ./omnivoice/
+29 -3
View File
@@ -58,6 +58,20 @@ does not update clients already on `v0.2.1`.
## 5. Cutting a release
Release notes lead with the biggest user-visible change, not README edits or
packaging internals. For a desktop redesign or migration, include a real UI
screenshot pinned to the release tag, explain what changed, and give direct
installer links and migration steps. Keep Highlights to 35 bullets, followed
by concise themed entries. Preserve any installer-trust disclosures.
Verify credits against the previous-tag-to-new-tag commit comparison and the
included PRs, including contributor branches merged through maintainer branches.
Add a Contributors section naming every human author and their contribution;
thank verified bug reporters separately and identify dependency bots separately.
Do not infer contributors from the existing changelog's `thanks` entries alone.
Electron publishes the authored version section verbatim, with any required
installer-trust disclosure appended by the workflow.
1. **CHANGELOG first (hard rule):** make sure `CHANGELOG.md` has a complete,
user-facing `## [X.Y.Z] — DATE` section (rename `## [Unreleased]`).
`release.yml` extracts that section verbatim as the GitHub Release body —
@@ -81,7 +95,7 @@ on that tag with `draft=true`, and wait for its final Tauri installers and signe
updater feeds. Then dispatch `electron-release.yml` on the same tag. Automatic
Electron builds are skipped for this tag to avoid racing the Tauri draft.
Keep the release draft until both builds and their checks have passed.
See [Electron transition](#electron-transition-next-desktop-release) below for
See [Electron transition](#electron-desktop-releases) below for
signing requirements and the explicit owner-only unsigned exception.
## 5b. Deployment channels — all must ship (hard rule, owner-set 2026-07-16)
@@ -152,7 +166,7 @@ versionless updater archive. Other versions, sibling platforms, and updater
manifests remain intact. Inventory or deletion permission/network failures stop
the job instead of hiding an upload collision.
## Electron transition (next desktop release)
## Electron desktop releases
Electron is the primary desktop distribution. electron-release.yml builds Linux
x64, Windows x64, macOS arm64 and macOS x64, checks packaged startup and updater
@@ -167,7 +181,10 @@ same tag completes. Electron requires the final signed latest.json and
latest-user.json assets; subsequent releases copy those feeds without changing
their immutable sunset payload URLs. Retain the sunset release and its assets.
Write versioned CHANGELOG notes before release. Review all four platform builds,
Write versioned CHANGELOG notes before release. The Electron release body uses
that authored section verbatim, including its introduction and contributor
credits; describe Tauri migration only when relevant, without announcing another
Tauri release. Review all four platform builds,
checksums, signing requirements and docs/electron-migration.md. Existing Electron
artifact names and app IDs remain stable for updater compatibility. This pipeline
ships stable releases; rolling preview publication is paused during transition.
@@ -188,3 +205,12 @@ choice. Tauri's signing keys do not sign Electron packages.
For the transition tag, automatic Electron release jobs are skipped. Build the
manual Tauri sunset draft first, then dispatch Electron on the same tag after
its signed updater feeds exist. Later tags build Electron automatically.
If a packaging-workflow fix is needed after tagging, keep the release tag
immutable. Merge and validate the workflow fix on main, then dispatch
`electron-release.yml` from main with `release_tag=vX.Y.Z`. Validation and every
packaging/release job check out that exact tag; only the workflow comes from
main. Empty signing secrets are omitted from the builder environment so drafts
and explicitly accepted unsigned builds do not interpret the working directory
as a certificate. Publication still requires `publish=true` and the same guards.
+1 -1
View File
@@ -39,7 +39,7 @@ VoiceStudio/
│ │ └── setup/ first-run wizard, model download
│ ├── core/ config, db, job queue, event bus, auth/CSRF, path security,
│ │ opt-in analytics, version, diagnostics
│ ├── services/ 86 modules of business logic — TTS, dubbing pipeline,
│ ├── services/ 88 modules of business logic — TTS, dubbing pipeline,
│ │ audio DSP, GPU gateway, engine routing, model lifecycle
│ ├── engines/ per-engine adapters: indextts, supertonic3, confucius4,
│ │ dots_tts, moss_tts_v15, pockettts, audiocpp,
+3 -3
View File
@@ -30,7 +30,7 @@ The integration shape is `VoiceStudioGGUFBackend(TTSBackend)` wrapping Phase 2's
| License compatible with v0.3.x ship? | YES — Apache-2.0 (model) + MIT (runtime) | Both verified via HF model card + GitHub README. Same Apache-2.0 chain as the upstream model already shipping in v0.2.7. |
| Runtime: llama.cpp / candle / custom? | CUSTOM (`omnivoice.cpp`, MIT) — does NOT load in vanilla llama.cpp | `gguf.architecture = "omnivoice-lm"` from HF API; README states "GGUF weights for omnivoice.cpp, a C++17/GGML port of VoiceStudio". |
| Quant variants and footprints? | 4 quants × 2 files each (base + tokenizer): Q4_K_M (659 MB), Q8_0 (945 MB), BF16 (1.60 GB), F32 (3.19 GB) | HF `siblings` list confirms all 8 files; sizes from model card table. |
| Cross-platform runtime fit? | Linux + Windows + macOS Intel YES via documented build scripts; macOS Apple Silicon Metal CONDITIONAL (no `buildmetal.sh` published, only feature mention) | `buildcpu.sh`, `buildcuda.sh`, `buildvulkan.sh`, `buildall.sh` listed; Metal in description only — Wave 1 Task 3 builds and verifies via `cmake -DGGML_METAL=ON` per A1. |
| Cross-platform runtime fit? | Linux + Windows + macOS Intel YES via documented build scripts; macOS Apple Silicon Metal YES (builds clean via `cmake -DGGML_METAL=ON` at pinned SHA, #2105) | `buildcpu.sh`, `buildcuda.sh`, `buildvulkan.sh`, `buildall.sh` listed; Metal builds cleanly via `cmake -DGGML_METAL=ON` in CI and locally (#2105). |
| Subprocess CLI fits Phase 2 `SubprocessBackend`? | YES | README shows `echo "Hello world." | ./build/omnivoice-tts --model … --codec … --lang … -o …` — line-oriented stdin + argv + output-file pattern is exactly what `SubprocessBackend` is designed for. |
### Pinned SHAs (filled in Wave 1 by Task 1)
@@ -52,13 +52,13 @@ Both SHAs are mirrored in `backend/engines/omnivoice_gguf/quant_map.json` `_meta
- Adds a maintained-by-others C++ runtime to the dependency graph (`omnivoice.cpp`, 42 stars at decision time).
- Adds ~12-16 MB of platform binaries to the installer (must verify against Phase 3 mirror-timing baseline per Pitfall 6).
- macOS code signing scope expands by 4 binaries (track via REL-05; same `xattr -cr` workaround as #54 applies in v0.3.x).
- `omnivoice.cpp` README does not publish a macOS Metal build script — only `buildcpu.sh`, `buildcuda.sh`, `buildvulkan.sh`, `buildall.sh`. Apple Silicon Metal must be verified in Wave 1.
- `omnivoice.cpp` README does not publish a standalone `buildmetal.sh` script; Apple Silicon Metal is built directly via `cmake -DGGML_METAL=ON` (#2105).
**Mitigations:**
- Pin `omnivoice.cpp` by commit SHA (`886fc079838ca7400cb2b42b36e2a65aa1daabe8`); rebuild from pinned SHA in CI for all 4 target platforms.
- Pin every quant file by commit SHA in `quant_map.json` (`361609388ae572a820d085185bbbe2a2aac4b30e`); shippable JSON so the table can update without an app release.
- In-process `VoiceStudioBackend` remains as fallback if any GGUF step fails (probe, download, load, generate).
- macOS Apple Silicon Metal build is verified in Wave 1 with explicit acceptance criteria; if blocked, downgrade SPIKE-01 default on macOS to in-process path and document in this ADR's "Status" line.
- macOS Apple Silicon Metal build is verified via `cmake -DGGML_METAL=ON` (#2105); in-process `VoiceStudioBackend` remains available as general fallback.
- SHA-256 checksums on bundled binaries (per GATE-05); verify at first launch and on every quant load.
- Subprocess arg composition uses typed `Path` objects rooted in app directories; quant override UI is a dropdown over `quant_map.json` entries only (no freeform path input — supply-chain control analogous to INST-09).
+30
View File
@@ -234,3 +234,33 @@ panels instead so neither editor becomes unusably small.
from-source checkout.
- **Installed it but still "needs install"** — restart the backend so Python
picks up the newly-installed module.
- **"The 'argos' engine's CTranslate2 runtime could not be loaded…"** — Argos
translates on CTranslate2, and on Linux kernels that refuse an executable
stack the CTranslate2 library shipped with Python 3.11 installs (4.4.0) is
rejected outright. VoiceStudio repairs that library in place on first use; if
it cannot (read-only install), the message names the fix — reinstall the
backend on Python 3.12+, or run `patchelf --clear-execstack` on the library
once — and NLLB stays available in the meantime
([#692](https://github.com/debpalash/VoiceStudio/issues/692)).
SRT and WebVTT exports round each cue timestamp once to the nearest millisecond, including carry into the next second or minute. The OpenAI-compatible transcription exports use the same formatter.
Paste translation accepts WebVTT files with hourless timestamps. Only timing records at line starts activate timestamp matching; timestamp-like text inside a sentence remains dialogue.
Subtitle import preserves numeric dialogue such as years and countdowns, including files mixing numbered and unnumbered cues. Cue numbers are removed only at identified cue boundaries.
Argos accepts Chinese/Simplified Chinese names, Mandarin aliases, and language tags such as `zh-CN`. Traditional Chinese requests (`zh-TW`, `zh-Hant`, and the display name) are rejected explicitly; select NLLB for Traditional Chinese rather than silently receiving a different script.
Mixed or malformed SRT files can make a bare number indistinguishable from spoken dialogue. The importer removes numbering only when cue boundaries and sequential numbering support it; ambiguous nonsequential numbers are retained as text to avoid silent data loss. Standard indexed SRT and WebVTT exports avoid this ambiguity.
If the Argos native runtime cannot load, both desktop and browser clients show localized recovery guidance: reinstall the backend or select NLLB. The API returns the stable `argos_runtime_unavailable` error code without exposing native library paths.
WebVTT import separates metadata blocks from cue identifiers using the [WebVTT block-parsing rules](https://www.w3.org/TR/webvtt1/#file-parsing): a timing line immediately after an identifier makes a cue, even when that identifier is NOTE, STYLE, or REGION. Later timing examples inside metadata are ignored, and empty cues never borrow the next cues identifier as dialogue.
Dubbing transcription emits keepalives during quiet diarization, reference-refinement, and cleanup steps. Disconnecting stops queued model work; native calls already running retain their model until they finish, then cleanup restores TTS. Task streams also request that proxies disable buffering so keepalives reach the client promptly.
+18
View File
@@ -18,3 +18,21 @@ permissions. No automatic installer-to-installer migration is provided.
The final Tauri updater feeds retain signed Tauri payloads at immutable URLs.
A Tauri updater must never receive an Electron installer.
Electron checks required Python imports before reusing an existing runtime. An
incomplete environment opens setup instead of repeatedly crashing; installation
still requires your explicit action. Automatic selection skips broken legacy
runtimes and uses the Electron runtime location, leaving Tauri data intact.
An explicitly selected runtime is not silently replaced during startup. If that
location contains an incomplete environment Electron does not own, choosing
setup creates a separate runtime in Electrons default location instead of
modifying or taking ownership of the existing environment.
A new custom runtime destination remains selectable. Setup creates and owns it
only when that directory does not already exist; existing unowned roots stay
untouched even if their `project` subdirectory is missing.
Dependency checks ignore inherited `PYTHONPATH` and `PYTHONHOME`, matching backend
startup. If repair switches away from an unowned environment and a healthy
Electron runtime already exists, it is reused without reinstalling dependencies.
+6 -4
View File
@@ -55,10 +55,12 @@ transcription.
process, so the engine checks up front and reports itself unavailable
instead ([#1371](https://github.com/debpalash/VoiceStudio/issues/1371)).
pytorch-whisper covers that case on torch's bundled cuDNN 9.
- On some hardened Linux kernels the CTranslate2 native library is rejected
with "cannot enable executable stack" (an OSError, not an ImportError) —
reported as unavailable rather than crashing engine selection
([#692](https://github.com/debpalash/VoiceStudio/issues/692)).
- On Linux kernels that refuse an executable stack, the CTranslate2 native
library (4.4.0 and older) is rejected with "cannot enable executable stack"
(an OSError, not an ImportError). VoiceStudio clears that ELF flag in place
on first probe so the engine loads; if the file cannot be written it reports
itself unavailable with the repair command rather than crashing engine
selection ([#692](https://github.com/debpalash/VoiceStudio/issues/692)).
- CTranslate2's GPU teardown can rarely segfault the process at unload. If
you hit that, switch to the crash-isolated variant —
[faster-whisper-isolated](faster-whisper-isolated.md)
+22 -2
View File
@@ -43,7 +43,8 @@ HF repo id. The env var overrides the persisted UI choice.
## Behaviour notes
- Output is 24 kHz mono for most hosted models.
- Output is 24 kHz mono. Results from models with a different native rate
(such as Dia at 44.1 kHz) are resampled before stitching and export.
- **Cloning works only with the `csm` model** — it is the only curated model
confirmed to accept a reference clip. Other models silently ignore
reference audio, so the engine reports cloning support only when CSM is
@@ -53,7 +54,14 @@ HF repo id. The env var overrides the persisted UI choice.
- Language support is per-model (Kokoro ~8 languages, others vary). An
unsupported language for Kokoro produces a clear error naming what it
does support ([#977](https://github.com/debpalash/VoiceStudio/issues/977))
leave language on Auto or switch to a multilingual engine.
pick a language it supports or switch to a multilingual engine.
- Auto is not an escape hatch from that. With the picker on Auto the request
carries no language, and a selected voice profile's saved language fills the
gap ([#533](https://github.com/debpalash/VoiceStudio/issues/533)) — so a
profile saved as, say, Persian still reaches Kokoro and is still refused. The
error names the profile as the source in that case
([#2156](https://github.com/debpalash/VoiceStudio/issues/2156)); change the
profile's language, or pick a supported one explicitly for the render.
## Platform notes
@@ -74,3 +82,15 @@ See also: [benchmarks.md](../benchmarks.md),
[languages.md](../languages.md),
[downloading-models.md](../downloading-models.md),
[disk usage](disk-usage.md).
Consecutive chunks with the same native sample rate are resampled together to
preserve filter context at chunk boundaries; rate changes start a new group.
Profile language refusals are terminal request errors on both local and remote
rendering, including streaming. Electron and web show localized guidance that
Auto inherits the profile language; neither silently retries the same refusal.
Kokoro language errors list every language in the installed models table.
“British English” and `en-gb` both select its British English voice pipeline.
Display names from newer installed Kokoro tables are accepted too, so a language
advertised by the error message can be selected without updating a hardcoded map.
+6
View File
@@ -61,6 +61,12 @@ A 6 GB card with nothing else loaded runs the default model on the GPU
([#2041](https://github.com/debpalash/VoiceStudio/issues/2041)). Disable
the check with `OMNIVOICE_ASR_VRAM_PREFLIGHT=0`.
The preflight sizes the weights; the generation workspace grows with the
batch on top. If a transcribe still hits a CUDA out-of-memory, the engine
steps the batch down (16 → 4 → 1, or 8 → 2 → 1 with word timestamps) and
finishes on the CPU rather than dropping the chunk — no silent holes in a
dub transcript.
## Quirks
- If the pipeline fails to import (`AutoFeatureExtractor` errors), the cause
+6 -2
View File
@@ -62,8 +62,12 @@ Two more fallback chains run at load time:
process fast-fails with no traceback, so the engine is reported unavailable
up front and selection falls through to pytorch-whisper, which uses torch's
own cuDNN 9 ([#1371](https://github.com/debpalash/VoiceStudio/issues/1371)).
- On some hardened Linux kernels CTranslate2's native library is rejected with
"cannot enable executable stack" — reported as unavailable, not a crash
- On Linux kernels that refuse an executable stack, CTranslate2's native
library (4.4.0 and older — what whisperx 3.4.5 pins on Python 3.11) is
rejected with "cannot enable executable stack". VoiceStudio now clears that
one ELF flag in place on first probe and the engine loads normally; if the
library cannot be written (a read-only bundle), the engine reports itself
unavailable with the repair command instead of crashing
([#692](https://github.com/debpalash/VoiceStudio/issues/692)).
- A partially-installed environment (interrupted sync, antivirus quarantine)
can break WhisperX's deep import chain (whisperx → pyannote →
+4
View File
@@ -419,3 +419,7 @@ Two paths are worth persisting across container restarts:
[Pull and run (AMD GPU / ROCm)](#pull-and-run-amd-gpu--rocm) above for when
to set one by hand.
- More entries: [docs/install/troubleshooting.md](troubleshooting.md).
### CTranslate2 compatibility
CUDA images include an isolated cuDNN 8 compatibility library directory for WhisperX and faster-whisper alongside the base PyTorch cuDNN 9 runtime. The image build verifies the compatibility libraries exist. ROCm images skip this NVIDIA-only dependency.
+87 -18
View File
@@ -188,7 +188,13 @@ before the token works for downloads.
3. Retry the job. The token state in **Settings → API Keys** should now show
the "App" row with a green check next to your username.
**Linked issue:** [#35](https://github.com/debpalash/VoiceStudio/issues/35)
If the token and license are already valid but an older build still reports
401 during installation, update VoiceStudio and retry. Settings tokens now
reach both the fast download path and engine-weight installers; no token
rotation is needed for that fixed client bug.
**Linked issues:** [#35](https://github.com/debpalash/VoiceStudio/issues/35),
[#2163](https://github.com/debpalash/VoiceStudio/issues/2163)
### PocketTTS gated weights
@@ -291,15 +297,29 @@ ready. The desktop app sits on "starting backend", `/health` returns 503, and
`/startup/progress` shows `ml_imports` active. From source you see `import
torch` die with a native access violation rather than a Python traceback.
**Cause:** VoiceStudio pins `torch 2.8.0`. That build carries no `sm_120`
kernels, so on a Blackwell card the CUDA initializer faults inside the native
library. This is not a VoiceStudio bug and no setting works around it — the
wheel does not contain code for the GPU.
**Cause:** not established. The pinned build is not missing Blackwell code:
`torch 2.8.0+cu128` lists `sm_120` in `torch.cuda.get_arch_list()`, and that
build imports and runs CUDA normally on some Blackwell cards. What is confirmed
is that on the Windows setups in
[#1931](https://github.com/debpalash/VoiceStudio/issues/1931) `import torch`
faults inside the native library before Python can raise an error, and moving
the torch trio to 2.9.x clears it.
**Fix:** move the whole torch trio to a build with `sm_120` kernels. They must
move together — upgrading one past the ABI the others were built against gives
you `RuntimeError: operator torchvision::nms does not exist`, which is the
next section's problem instead.
If `import torch` crashes for you, that crash *is* the symptom — skip straight
to the fix below. Where torch does import, this shows what the build actually
contains:
```bash
uv run python -c "import torch; print(torch.__version__, torch.cuda.get_arch_list())"
```
`sm_120` in that list means the kernels are present and the crash is elsewhere
in the native init path. Either way the upgrade below is the known workaround.
**Fix:** move the whole torch trio to 2.9.x. They must move together —
upgrading one past the ABI the others were built against gives you
`RuntimeError: operator torchvision::nms does not exist`, which is the next
section's problem instead.
Edit **both** pin lists, keeping them identical:
@@ -341,8 +361,10 @@ guarded now, so the upgrade path above is clean on a current checkout.
**Keeping the change:** these are the repo's own pins, so a `git pull` that
touches them will conflict or overwrite. Re-apply after updating until the
default pin moves — the default cannot move for everyone until the newer torch
is verified across the older GPUs VoiceStudio supports, since a build that adds
`sm_120` can drop older architectures.
is verified across the older GPUs VoiceStudio supports, since a newer build can
drop older architectures. Which ones varies by torch release, not by the CUDA
variant alone: the pinned 2.8.0+cu128 build reports `sm_70` first, while the
cu128 arch list captured in #1285 still carried `sm_61`.
**Linked issue:** [#1931](https://github.com/debpalash/VoiceStudio/issues/1931)
— thanks to the reporter for the full diagnosis, including the verification
@@ -1091,14 +1113,16 @@ retrying.
and the backend log ends inside the `ml_imports` phase — often with a native
crash (exit code `0xffffffff` / `-1073741819`) rather than a Python traceback.
**Cause.** VoiceStudio pins `torch 2.8.0+cu128`, which ships no `sm_120`
kernels. On an RTX 50-series card `import torch` dies natively, before any
VoiceStudio code can classify it — which is why the app can only say the
backend did not start. This is a property of the pinned build, not of your
driver or your install.
**Cause.** Not established. `torch 2.8.0+cu128` does contain Blackwell code:
`sm_120` is in `torch.cuda.get_arch_list()`, and that build imports and runs
CUDA normally on some Blackwell cards, so this is not simply a wheel without
kernels for your GPU. What is confirmed is that on the Windows setups in
[#1931](https://github.com/debpalash/VoiceStudio/issues/1931) `import torch`
dies natively before any VoiceStudio code can classify it, which is why the app
can only say the backend did not start. Moving to torch 2.9.x clears it for the
users who hit it.
**Fix.** Move to a torch build that has Blackwell kernels. From a source
checkout, in the project folder:
**Fix.** Move to torch 2.9.x. From a source checkout, in the project folder:
1. Edit `pyproject.toml``[tool.uv] constraint-dependencies` and raise the
torch constraint to `torch==2.9.1+cu128` (matching `torchaudio` /
@@ -1148,3 +1172,48 @@ remove the app binary itself are in
[docs/install/uninstall.md](uninstall.md).
**Linked issue:** [#1089](https://github.com/debpalash/VoiceStudio/issues/1089)
### Pedalboard illegal-instruction crashes
VoiceStudio pins pedalboard to `>=0.9.14,<0.9.21` while [upstream portable-wheel repair #466](https://github.com/spotify/pedalboard/pull/466) remains open. Re-sync the locked environment after updating from a build with newer affected wheels. This keeps the existing effects API floor while avoiding the reported Linux CPU import crash.
### TorchCodec unavailable
When torchaudio requires an unavailable TorchCodec installation, VoiceStudio writes through soundfile and reads reference audio through its FFmpeg fallback. Reference amplitude is normalized using the decoded sample representation, including 8-, 24-, and 32-bit PCM.
### Isolated engine timeouts
Generation has a separate deadline from health checks. Default sidecar deadlines scale with text length and the host execution budget; per-engine timeout overrides remain supported. The outer job guard includes time for sidecar termination and error reporting. A timeout identifies the deadline, while a closed pipe without a timeout indicates a crash.
Per-engine receive overrides include `OMNIVOICE_CONFUCIUS4_RECV_TIMEOUT_S`, `OMNIVOICE_DOTS_TTS_RECV_TIMEOUT_S`, `OMNIVOICE_MOSS_TTS_V15_RECV_TIMEOUT_S`, and `OMNIVOICE_SUPERTONIC3_RECV_TIMEOUT_S` (seconds). Invalid or non-finite values use the default; values below 30 seconds are raised to 30.
### FFprobe alongside FFmpeg
When FFprobe is not on PATH, VoiceStudio also checks beside the selected FFmpeg binary. Parent folders named `ffmpeg` remain unchanged; only the executable name becomes `ffprobe` (or `ffprobe.exe` on Windows).
### Subtitle and manuscript encodings
Electron and web imports accept UTF-8, UTF-16 with a byte-order mark, and Windows-1252 text. The same decoder rules apply to uploaded subtitles and audiobook manuscripts; legacy punctuation is preserved.
### Compile fallback after startup
An architecture accepted by the torch.compile preflight may still encounter independent Dynamo, Inductor, Triton, or CUDA-graph runtime errors. VoiceStudio distinguishes those from GPU memory exhaustion and retries with eager execution; architecture support alone does not guarantee compilation succeeds.
### Native engine installation from a remote client
Sidecar and audio.cpp runtime installation is restricted to requests from the backend computer's loopback interface. An API key does not bypass this restriction. The catalogue now shows local setup guidance instead of offering a remote install that will be rejected. Open the backend through `localhost` on that computer, or follow the engine's setup guide there. In Docker, bridge-network requests may not be loopback even when the browser runs on the host; use the documented container setup rather than weakening the native-install gate.
### Dubbing extraction fails
Extraction errors show the FFmpeg exit code and the end of its diagnostics, with private paths scrubbed. Use the final error line to distinguish missing audio streams, unsupported inputs, permissions, or disk errors. A version banner alone does not identify the cause; include the final diagnostic and source format when reporting a failure.
Explicit generation budgets remain authoritative. If an outer TTS/ASR guard times out or its caller disconnects, the active sidecar receive kills and reaps its captured child; it cannot terminate a later retry. In-process inference keeps its existing lifetime accounting until the native call returns.
### Voice conversion requires a cloning model
In Electron, voice conversion stays disabled until the active text-to-speech
model is ready and supports voice cloning. Use the Models link to choose one;
the source recording and target voice are preserved when returning. Preset-only
models such as MLX Kokoro cannot clone a target voice. This capability check does
not download or load model weights.
+4 -1
View File
@@ -48,7 +48,7 @@ Everything above, plus the toolchain:
with the **"Desktop development with C++"** workload checked.
- **Bun**`powershell -c "irm bun.sh/install.ps1 | iex"`.
- **FFmpeg**`winget install Gyan.FFmpeg`.
- **Rust / Cargo**`winget install Rust.Rustup` or download `rustup-init.exe` from [rustup.rs](https://rustup.rs/).
- **Rust / Cargo**`winget install Rustlang.Rustup` or download `rustup-init.exe` from [rustup.rs](https://rustup.rs/).
After installing Rustup, close and reopen PowerShell before running `bun run tauri:desktop-prod`.
## GPU support on Windows
@@ -350,3 +350,6 @@ This test-host preparation does not change installer privileges or user machines
Verbose MSI logs are printed if installation or removal fails. The Windows CI
job also rejects an invalid MSI and verifies policy absence, value types, account
cleanup, and verbose failure logs using Windows PowerShell 5.1.
Migration configuration is kept ASCII so Alembic can read it under Windows locale code pages as well as UTF-8. This applies to source installs and direct Alembic commands.
+143
View File
@@ -2,6 +2,10 @@
import { afterEach, expect, it, vi } from 'vitest';
import { EventEmitter } from 'node:events';
const mocks = vi.hoisted(() => ({
runtimeConfig: null as { root: string; owned: boolean } | null,
existingProject: false,
existingRoot: false,
dependencies: vi.fn(async (_project?: string) => true),
ready: vi.fn(async () => false),
compatible: vi.fn(async () => false),
interrupted: vi.fn(async () => false),
@@ -13,8 +17,34 @@ const mocks = vi.hoisted(() => ({
}));
vi.mock('electron', () => ({ app: { isPackaged: true, getPath: () => '/private/voicestudio' } }));
vi.mock('node:child_process', () => ({ spawn: mocks.spawn, spawnSync: vi.fn() }));
vi.mock('node:fs', async (importOriginal) => {
const original = await importOriginal<typeof import('node:fs')>();
return {
...original,
readFileSync: (...args: Parameters<typeof original.readFileSync>) =>
String(args[0]).endsWith('runtime-location.json') && mocks.runtimeConfig
? JSON.stringify(mocks.runtimeConfig)
: original.readFileSync(...args),
writeFileSync: (...args: Parameters<typeof original.writeFileSync>) => {
if (String(args[0]).endsWith('runtime-location.json')) {
mocks.runtimeConfig = JSON.parse(String(args[1]));
return;
}
return original.writeFileSync(...args);
},
mkdirSync: (...args: Parameters<typeof original.mkdirSync>) =>
String(args[0]).includes('private') ? undefined : original.mkdirSync(...args),
existsSync: (path: Parameters<typeof original.existsSync>[0]) =>
String(path).includes('selected')
? String(path).endsWith('project')
? mocks.existingProject
: mocks.existingRoot
: original.existsSync(path),
};
});
vi.mock('node:fs/promises', () => ({ rm: mocks.rm }));
vi.mock('./runtime-project', () => ({
runtimeDependenciesReady: mocks.dependencies,
runtimeReady: mocks.ready,
runtimeCompatible: mocks.compatible,
runtimeInstallInterrupted: mocks.interrupted,
@@ -34,6 +64,12 @@ afterEach(() => {
vi.unstubAllGlobals();
vi.unstubAllEnvs();
vi.clearAllMocks();
mocks.runtimeConfig = null;
mocks.existingProject = false;
mocks.existingRoot = false;
mocks.dependencies.mockResolvedValue(true);
mocks.ready.mockResolvedValue(false);
mocks.compatible.mockResolvedValue(false);
mocks.interrupted.mockResolvedValue(false);
mocks.promoteCaches.mockResolvedValue(undefined);
});
@@ -364,3 +400,110 @@ it('reserves setup before asynchronous checks and cancels stale preflight', asyn
expect(mocks.install).not.toHaveBeenCalled();
expect(supervisor.status.stage).toBe('idle');
});
it.each(['ready', 'compatible'] as const)(
'offers setup instead of spawning a %s runtime with missing dependencies',
async (kind) => {
vi.stubGlobal(
'fetch',
vi.fn(async () => {
throw new Error('no backend');
}),
);
vi.stubEnv('OMNIVOICE_BACKEND_CMD', '');
vi.stubEnv('VOICESTUDIO_SKIP_BACKEND', '');
mocks[kind].mockResolvedValue(true);
mocks.dependencies.mockResolvedValue(false);
const supervisor = new BackendSupervisor();
await supervisor.start();
expect(supervisor.status.stage).toBe('setup_required');
expect(mocks.spawn).not.toHaveBeenCalled();
expect(mocks.install).not.toHaveBeenCalled();
expect(mocks.stage).not.toHaveBeenCalled();
await supervisor.shutdown();
},
);
it.each([true, false])(
'preserves a selected unowned runtime (project exists: %s)',
async (projectExists) => {
const { resolve, join } = await import('node:path');
const selected = resolve('/selected/VoiceStudio');
mocks.runtimeConfig = { root: selected, owned: false };
mocks.existingProject = projectExists;
mocks.existingRoot = true;
mocks.ready.mockResolvedValue(true);
mocks.dependencies.mockResolvedValue(false);
mocks.install.mockRejectedValue(new Error('offline'));
vi.stubGlobal(
'fetch',
vi.fn(async () => {
throw new Error('no backend');
}),
);
vi.stubEnv('OMNIVOICE_BACKEND_CMD', '');
vi.stubEnv('VOICESTUDIO_SKIP_BACKEND', '');
const supervisor = new BackendSupervisor();
await supervisor.start();
expect(supervisor.status.stage).toBe('setup_required');
expect(
mocks.dependencies.mock.calls.every(([project]) => String(project).includes('selected')),
).toBe(true);
expect(mocks.install).not.toHaveBeenCalled();
await supervisor.setupRuntime();
expect(mocks.install.mock.calls[0][1]).toBe(join('/private/voicestudio', 'runtime', 'project'));
expect(mocks.runtimeConfig?.root).not.toBe(selected);
expect(mocks.rm).not.toHaveBeenCalled();
expect(mocks.stage).not.toHaveBeenCalled();
await supervisor.shutdown();
},
);
it('installs into a newly selected custom destination that does not yet exist', async () => {
const { resolve, join } = await import('node:path');
const selected = resolve('/selected/VoiceStudio');
mocks.runtimeConfig = { root: selected, owned: false };
mocks.install.mockRejectedValue(new Error('offline'));
vi.stubGlobal(
'fetch',
vi.fn(async () => {
throw new Error('no backend');
}),
);
vi.stubEnv('OMNIVOICE_BACKEND_CMD', '');
vi.stubEnv('VOICESTUDIO_SKIP_BACKEND', '');
const supervisor = new BackendSupervisor();
await supervisor.start();
await supervisor.setupRuntime();
expect(mocks.install.mock.calls[0][1]).toBe(join(selected, 'project'));
expect(mocks.runtimeConfig).toEqual({ root: selected, owned: true });
await supervisor.shutdown();
});
it('reuses a healthy default runtime after explicit setup leaves an unowned environment', async () => {
const { resolve, join } = await import('node:path');
mocks.runtimeConfig = { root: resolve('/selected/VoiceStudio'), owned: false };
mocks.existingRoot = true;
mocks.ready.mockResolvedValue(true);
mocks.dependencies.mockImplementation(async (project?: string) => !project?.includes('selected'));
vi.stubGlobal(
'fetch',
vi.fn(async () => {
throw new Error('no backend');
}),
);
vi.stubEnv('OMNIVOICE_BACKEND_CMD', '');
vi.stubEnv('VOICESTUDIO_SKIP_BACKEND', '');
const supervisor = new BackendSupervisor();
await supervisor.start();
expect(supervisor.status.stage).toBe('setup_required');
const restart = vi.spyOn(supervisor, 'start').mockResolvedValue();
await supervisor.setupRuntime();
expect(mocks.install).not.toHaveBeenCalled();
expect(mocks.dependencies).toHaveBeenCalledWith(
join('/private/voicestudio', 'runtime', 'project'),
);
expect(restart).toHaveBeenCalledOnce();
restart.mockRestore();
await supervisor.shutdown();
});
+51 -20
View File
@@ -2,6 +2,7 @@ import {
installRuntime,
promoteLegacyRuntimeCaches,
runtimeCompatible,
runtimeDependenciesReady,
runtimeInstallInterrupted,
runtimeReady,
runtimePython,
@@ -482,11 +483,8 @@ export class BackendSupervisor extends EventEmitter<{
}
if (app.isPackaged && !parseBackendCmdOverride(process.env.OMNIVOICE_BACKEND_CMD)) {
const project = await this.resolveRuntimeProject();
if (
!(await runtimeReady(backendRoot(), project)) &&
!(await runtimeCompatible(backendRoot(), project))
) {
const { project, ready } = await this.resolveRuntimeProject();
if (!ready) {
if (gen === this.generation) {
this.runtimeInterrupted = await runtimeInstallInterrupted(project);
this.setStage('setup_required');
@@ -522,7 +520,7 @@ export class BackendSupervisor extends EventEmitter<{
this.stage !== 'setup_required'
)
return;
const project =
let project =
this.runtimeProject ?? join(storedRuntimeRoot() ?? defaultRuntimeRoot(), 'project');
this.runtimeProject = project;
const controller = new AbortController();
@@ -537,15 +535,44 @@ export class BackendSupervisor extends EventEmitter<{
this.setStage('installing', { message: undefined });
try {
const reusable =
(await runtimeReady(backendRoot(), project)) ||
(await runtimeCompatible(backendRoot(), project));
((await runtimeReady(backendRoot(), project)) ||
(await runtimeCompatible(backendRoot(), project))) &&
(await runtimeDependenciesReady(project));
if (gen !== this.generation || controller.signal.aborted) return;
if (reusable) {
await this.start();
return;
}
const runtimeRoot = dirname(project);
let runtimeRoot = dirname(project);
const configured = storedRuntimeLocation();
if (
configured &&
!configured.owned &&
samePath(configured.root, runtimeRoot) &&
!samePath(runtimeRoot, defaultRuntimeRoot()) &&
existsSync(runtimeRoot)
) {
// An explicit setup action may create a new runtime, but must never
// take ownership of (or repair in place) another installation's files.
runtimeRoot = defaultRuntimeRoot();
project = join(runtimeRoot, 'project');
this.runtimeProject = project;
writeRuntimeLocation(runtimeRoot, true);
this.pushLog(
'out',
'Creating a separate Electron runtime; existing environment preserved.',
);
this.emitStatus();
const fallbackReusable =
((await runtimeReady(backendRoot(), project)) ||
(await runtimeCompatible(backendRoot(), project))) &&
(await runtimeDependenciesReady(project));
if (gen !== this.generation || controller.signal.aborted) return;
if (fallbackReusable) {
await this.start();
return;
}
}
if (
configured &&
samePath(configured.root, runtimeRoot) &&
@@ -797,27 +824,28 @@ export class BackendSupervisor extends EventEmitter<{
this.emitStatus();
}
private async resolveRuntimeProject(): Promise<string> {
private async resolveRuntimeProject(): Promise<{ project: string; ready: boolean }> {
const bundle = backendRoot();
const own = join(defaultRuntimeRoot(), 'project');
const configuredRoot = storedRuntimeRoot();
const configured = configuredRoot ? join(configuredRoot, 'project') : null;
const candidates = [
this.runtimeProject,
configured,
own,
...legacyTauriRuntimeProjects(),
].filter((candidate): candidate is string => Boolean(candidate));
// Explicit selection is authoritative, including when it needs setup.
const candidates = (
configured ? [configured] : [this.runtimeProject, own, ...legacyTauriRuntimeProjects()]
).filter((candidate): candidate is string => Boolean(candidate));
for (const project of new Set(candidates.map((candidate) => resolve(candidate)))) {
if ((await runtimeReady(bundle, project)) || (await runtimeCompatible(bundle, project))) {
if (
((await runtimeReady(bundle, project)) || (await runtimeCompatible(bundle, project))) &&
(await runtimeDependenciesReady(project))
) {
this.runtimeProject = project;
if (project !== own && project !== configured)
this.pushLog('out', `Reusing compatible Tauri runtime: ${project}`);
return project;
return { project, ready: true };
}
}
this.runtimeProject = configured ?? own;
return this.runtimeProject;
return { project: this.runtimeProject, ready: false };
}
private emitStatus(): void {
@@ -889,7 +917,10 @@ export class BackendSupervisor extends EventEmitter<{
// Python passes this child-side descriptor to every nested operation.
// Reading the parent side keeps the ownership channel live and lets
// Node observe EOF only after the complete backend subtree releases it.
const drain = child.stdio?.[processOptions.drainFd] as NodeJS.ReadableStream | null | undefined;
const drain = child.stdio?.[processOptions.drainFd] as
| NodeJS.ReadableStream
| null
| undefined;
drain?.on('error', (error: unknown) => {
if (!isExpectedPipeClose(error)) {
this.pushLog('err', `Backend drain stream failed: ${errorMessage(error)}`);
@@ -0,0 +1,63 @@
// @vitest-environment node
import { afterEach, expect, it, vi } from 'vitest';
import { execFile } from 'node:child_process';
import { runtimeDependenciesReady, runtimePython } from './runtime-project';
vi.mock('node:child_process', () => ({ execFile: vi.fn() }));
afterEach(() => {
vi.clearAllMocks();
vi.unstubAllEnvs();
});
it.each([null, new Error('No module named uvicorn'), new Error('ETIMEDOUT'), new Error('ENOENT')])(
'validates imports using the selected interpreter and fails closed (%s)',
async (error) => {
vi.mocked(execFile).mockImplementation(((
_command: unknown,
_args: unknown,
_options: unknown,
callback: (error: Error | null) => void,
) => callback(error)) as never);
const project = '/runtime with spaces';
expect(await runtimeDependenciesReady(project)).toBe(error === null);
expect(execFile).toHaveBeenCalledWith(
runtimePython(project),
['-c', 'import fastapi, uvicorn, omnivoice, faster_whisper'],
expect.objectContaining({
cwd: project,
timeout: 30_000,
windowsHide: true,
env: expect.objectContaining({
HF_HUB_OFFLINE: '1',
TRANSFORMERS_OFFLINE: '1',
PYTHONNOUSERSITE: '1',
}),
}),
expect.any(Function),
);
},
);
it.each(['PYTHONPATH', 'PYTHONHOME'] as const)(
'isolates imports from inherited %s',
async (variable) => {
vi.stubEnv(variable, '/unrelated-python');
vi.mocked(execFile).mockImplementation(((
_command: unknown,
_args: unknown,
options: { env: NodeJS.ProcessEnv },
callback: (error: Error | null) => void,
) => {
const contaminated = Boolean(options.env[variable]);
callback(
variable === 'PYTHONPATH'
? contaminated
? null
: new Error('No module named uvicorn')
: contaminated
? new Error('invalid Python home')
: null,
);
}) as never);
expect(await runtimeDependenciesReady('/selected-runtime')).toBe(variable === 'PYTHONHOME');
expect(process.env[variable]).toBe('/unrelated-python');
},
);
+25
View File
@@ -1,3 +1,4 @@
import { execFile } from 'node:child_process';
import { createHash, randomUUID } from 'node:crypto';
import {
cp,
@@ -247,6 +248,30 @@ export async function runtimeInstallInterrupted(project: string): Promise<boolea
}
}
/** Check the selected interpreter locally before trusting runtime metadata. */
export async function runtimeDependenciesReady(project: string): Promise<boolean> {
const env: NodeJS.ProcessEnv = { ...process.env };
delete env.PYTHONHOME;
delete env.PYTHONPATH;
env.HF_HUB_OFFLINE = '1';
env.TRANSFORMERS_OFFLINE = '1';
env.PYTHONNOUSERSITE = '1';
return new Promise((resolve) => {
execFile(
runtimePython(project),
['-c', 'import fastapi, uvicorn, omnivoice, faster_whisper'],
{
cwd: project,
windowsHide: true,
timeout: 30_000,
maxBuffer: 256 * 1024,
env,
},
(error) => resolve(!error),
);
});
}
export async function runtimeReady(bundle: string, project: string): Promise<boolean> {
if (await runtimeIncomplete(project)) return false;
try {
@@ -1,3 +1,4 @@
import { readTextFile } from '../../../../../../frontend/src/utils/readTextFile';
import { useEffect, useMemo, useRef, useState } from 'react';
import { AlertTriangleIcon, ClipboardPasteIcon, FileTextIcon, XIcon } from 'lucide-react';
import { useTranslation } from 'react-i18next';
@@ -73,8 +74,7 @@ export function PasteTranslation({ segments, disabled, onApply, onClose }: Paste
const readFile = (file?: File) => {
if (!file) return;
setReadFailed(false);
void file
.text()
void readTextFile(file)
.then(setText)
.catch(() => setReadFailed(true));
};
@@ -1,3 +1,4 @@
import { readTextFile } from '../../../../../../frontend/src/utils/readTextFile';
import { ProfileAvatar } from '@/components/profile-avatar';
import {
BookOpenTextIcon,
@@ -119,7 +120,7 @@ export function LongformPage({ mode }: { mode: Mode }) {
try {
let text: string;
if (mode === 'stories' && /\.(txt|md|srt)$/i.test(file.name))
text = importToText(file.name, await file.text());
text = importToText(file.name, await readTextFile(file));
else {
const body = new FormData();
body.set('file', file);
@@ -16,3 +16,14 @@ it('splits locally with bounded chunks and stable text order', () => {
expect(chunks.join(' ')).toBe(text);
expect(splitIntoChunks('', 500)).toEqual([]);
});
it('decodes a UTF-16 manuscript before extracting subtitle speech', async () => {
const { readTextFile } = await import('../../../../../../frontend/src/utils/readTextFile');
const text = '1\n00:00:01,000 --> 00:00:02,000\nCafé — hello';
const bytes = new Uint8Array(2 + text.length * 2);
bytes.set([0xff, 0xfe]);
const view = new DataView(bytes.buffer);
for (let i = 0; i < text.length; i++) view.setUint16(2 + i * 2, text.charCodeAt(i), true);
const file = { arrayBuffer: async () => bytes.buffer };
expect(importToText('captions.srt', await readTextFile(file))).toBe('Café — hello');
});
@@ -18,6 +18,7 @@ export function EngineInstall({ id }: { id: string }) {
queryFn: () =>
apiJson<{
installed: boolean;
install_allowed?: boolean;
job: null | {
state: string;
steps: { name?: string; state: string }[];
@@ -36,8 +37,15 @@ export function EngineInstall({ id }: { id: string }) {
<Button
size="sm"
variant="outline"
disabled={running || status.data?.installed}
disabled={
running ||
status.isPending ||
status.isError ||
status.data?.installed ||
status.data?.install_allowed === false
}
onClick={async () => {
if (status.data?.install_allowed === false) return;
setStarting(true);
setFailed(null);
setDismissedFailure(false);
@@ -54,6 +62,11 @@ export function EngineInstall({ id }: { id: string }) {
<DownloadIcon />
{t(running ? 'modelMaintenance.installing' : 'modelMaintenance.install')}
</Button>
{status.data?.install_allowed === false && (
<p className="max-w-sm text-xs text-muted-foreground">
{t('engines.localInstallRequired')}
</p>
)}
{!dismissedFailure && (failed || status.isError || status.data?.job?.state === 'failed') && (
<SettingsActionError
className="max-w-sm text-left"
@@ -107,7 +107,41 @@ it('confirms a diarisation engine selection after the backend accepts it', async
);
fireEvent.click(await screen.findByRole('button', { name: 'modelSettings.select' }));
await waitFor(() =>
expect(mock.toast.success).toHaveBeenCalledWith('settings.engine_switched'),
await waitFor(() => expect(mock.toast.success).toHaveBeenCalledWith('settings.engine_switched'));
});
it('explains remote native installation restrictions before a POST', async () => {
mock.api.mockImplementation((path: string) =>
Promise.resolve(
path === '/engines/diarisation'
? {
active: 'pyannote',
options: [
{
id: 'audiocpp-sortformer',
label: 'Sortformer',
model_installed: true,
runtime_installed: false,
installed: false,
},
],
}
: {
installed: false,
supported: true,
install_allowed: false,
job: { state: 'idle', progress: 0 },
},
),
);
render(
<QueryClientProvider client={new QueryClient()}>
<DiarisationSettings />
</QueryClientProvider>,
);
expect(await screen.findByText('engines.localInstallRequired')).toBeInTheDocument();
const button = screen.getByRole('button', { name: 'modelMaintenance.install' });
expect(button).toBeDisabled();
fireEvent.click(button);
expect(mock.api.mock.calls.some(([, init]) => init?.method === 'POST')).toBe(false);
});
@@ -61,6 +61,7 @@ export function DiarisationSettings() {
apiJson<{
installed: boolean;
supported: boolean;
install_allowed?: boolean;
job: { state: string; progress: number; error?: string | null };
}>('/engines/audiocpp/runtime/install/status'),
refetchInterval: (state) => (state.state.data?.job.state === 'running' ? 1_500 : 10_000),
@@ -75,6 +76,7 @@ export function DiarisationSettings() {
}, [client, runtime.data?.installed]);
const runtimeRunning = busy || runtime.data?.job.state === 'running';
const installRuntime = async () => {
if (runtime.data?.install_allowed === false) return;
setBusy(true);
setFailed(null);
try {
@@ -152,7 +154,12 @@ export function DiarisationSettings() {
<Button
size="sm"
variant="outline"
disabled={runtimeRunning}
disabled={runtimeRunning || runtime.data?.install_allowed === false}
title={
runtime.data?.install_allowed === false
? t('engines.localInstallRequired')
: undefined
}
onClick={() => void installRuntime()}
>
<DownloadIcon />
@@ -163,6 +170,13 @@ export function DiarisationSettings() {
: t('modelMaintenance.install')}
</Button>
)}
{option.id === 'audiocpp-sortformer' &&
!option.runtime_installed &&
runtime.data?.install_allowed === false && (
<p className="max-w-sm text-xs text-muted-foreground">
{t('engines.localInstallRequired')}
</p>
)}
{option.installed && (
<Button
size="sm"
@@ -380,11 +394,13 @@ export function ModelSettings({
name: engine.display_name,
available: engine.available,
detail:
engine.id === selected
? engineState.active_model
: !engine.available
? engine.install_hint || engine.reason || engine.hint || undefined
: undefined,
engine.local_install_required && !engine.available
? t('engines.localInstallRequired')
: engine.id === selected
? engineState.active_model
: !engine.available
? engine.install_hint || engine.reason || engine.hint || undefined
: undefined,
models: engine.curated_models,
installable: engine.one_click_install,
setupSnippet: !engine.available ? engine.setup_snippet || undefined : undefined,
@@ -2,7 +2,26 @@ import { clearConversion } from './conversion-state';
import { cleanup, fireEvent, render, screen, act } from '@testing-library/react';
import { QueryClient, QueryClientProvider } from '@tanstack/react-query';
import { afterEach, expect, it, vi } from 'vitest';
const mock = vi.hoisted(() => ({ convert: vi.fn() }));
const mock = vi.hoisted(() => ({
queryError: false,
retry: vi.fn(),
convert: vi.fn(),
ready: true,
cloning: true as boolean | null,
}));
vi.mock('@/hooks/use-engines', () => ({
useEngines: () => ({
isError: mock.queryError,
error: new Error('Engine query failed'),
retry: mock.retry,
data: mock.queryError ? undefined : {},
activeTtsReady: mock.ready,
activeTts: { supports_cloning: mock.cloning },
}),
}));
vi.mock('@tanstack/react-router', () => ({
Link: ({ children }: { children: React.ReactNode }) => <a>{children}</a>,
}));
vi.mock('@/lib/api/convert', () => ({ convertSpeech: mock.convert }));
vi.mock('@/hooks/use-recording', () => ({
useRecording: () => ({ isRecording: false, isStarting: false, isCleaning: false }),
@@ -22,6 +41,9 @@ import { ConvertVoice } from './convert-voice';
afterEach(() => {
cleanup();
clearConversion();
mock.queryError = false;
mock.ready = true;
mock.cloning = true;
vi.clearAllMocks();
});
function mount() {
@@ -79,3 +101,50 @@ it('retains source and target when returning from model settings', () => {
expect(screen.getByRole('button', { name: 'source.wav' })).toBeInTheDocument();
expect(screen.getByRole('button', { name: 'convert.convert' })).toBeEnabled();
});
it.each([false, null])(
'blocks conversion for unsupported or unknown cloning capability (%s)',
(capability) => {
mock.cloning = capability;
const { upload } = mount();
upload();
fireEvent.click(screen.getByRole('button', { name: 'Alpha' }));
const button = screen.getByRole('button', { name: 'convert.convert' });
expect(button).toBeDisabled();
expect(screen.getByText('convert.cloning_required')).toBeInTheDocument();
fireEvent.click(button);
expect(mock.convert).not.toHaveBeenCalled();
},
);
it('rechecks engine capability when returning from settings without losing inputs', () => {
const first = mount();
first.upload();
fireEvent.click(screen.getByRole('button', { name: 'Alpha' }));
first.unmount();
mock.cloning = false;
const second = mount();
expect(screen.getByRole('button', { name: 'convert.convert' })).toBeDisabled();
second.unmount();
mock.cloning = true;
mock.ready = false;
const third = mount();
expect(screen.getByRole('button', { name: 'convert.convert' })).toBeDisabled();
third.unmount();
mock.ready = true;
mount();
expect(screen.getByRole('button', { name: 'convert.convert' })).toBeEnabled();
expect(screen.getByRole('button', { name: 'source.wav' })).toBeInTheDocument();
});
it('offers retry instead of model guidance when engine lookup fails', () => {
mock.queryError = true;
mock.ready = false;
const { upload } = mount();
upload();
fireEvent.click(screen.getByRole('button', { name: 'Alpha' }));
expect(screen.getByRole('button', { name: 'convert.convert' })).toBeDisabled();
expect(screen.queryByText('convert.cloning_required')).not.toBeInTheDocument();
expect(screen.getByText('Engine query failed')).toBeInTheDocument();
fireEvent.click(screen.getByRole('button', { name: 'common.retry' }));
expect(mock.retry).toHaveBeenCalledOnce();
});
@@ -11,6 +11,7 @@ import { PipelineFailure } from '@/components/pipeline-failure';
import { AgentFixButton } from '@/components/agent-fix-button';
import { ProfileAvatar } from '@/components/profile-avatar';
import { WaveformPlayer } from '@/components/waveform-player';
import { useEngines } from '@/hooks/use-engines';
import { useProfiles } from '@/hooks/use-profiles';
import { useRecording } from '@/hooks/use-recording';
import { ApiError, apiPath, describeError } from '@/lib/api/client';
@@ -21,6 +22,8 @@ import { beginAppActivity } from '@/lib/app-activity';
export function ConvertVoice() {
const { t } = useTranslation();
const profiles = useProfiles();
const engines = useEngines();
const canClone = engines.activeTtsReady && engines.activeTts?.supports_cloning === true;
const client = useQueryClient();
const { file, voice, search, match, result } = useConversion();
const setFile = (value: File | null) => setConversion('file', value);
@@ -64,7 +67,7 @@ export function ConvertVoice() {
}, [file]);
useEffect(() => () => request.current?.abort(), []);
const voices = (profiles.data ?? []).filter((p) => p.kind === 'clone' && p.ref_audio_path);
const valid = file && voices.some((p) => p.id === voice) && !recordingBusy;
const valid = canClone && file && voices.some((p) => p.id === voice) && !recordingBusy;
const run = async () => {
if (!valid || request.current) return;
const controller = new AbortController();
@@ -210,6 +213,32 @@ export function ConvertVoice() {
</label>
<p className="text-xs text-muted-foreground">{t('convert.match_duration_hint')}</p>
</div>
{engines.isError && !engines.data && !busy && (
<div
role="alert"
className="flex flex-wrap items-center gap-2 text-sm text-muted-foreground"
>
<span>{describeError(engines.error)}</span>
<Button variant="outline" size="sm" onClick={engines.retry}>
{t('common.retry')}
</Button>
</div>
)}
{!canClone && !(engines.isError && !engines.data) && !busy && (
<div
role="status"
className="flex flex-wrap items-center gap-2 text-sm text-muted-foreground"
>
<span>{t('convert.cloning_required')}</span>
<Link
to="/settings/models/$family"
params={{ family: 'tts' }}
className="font-medium text-primary hover:underline"
>
{t('modelSettings.models')}
</Link>
</div>
)}
{error && (
<PipelineFailure
fallback={error}
@@ -2245,6 +2245,7 @@
"tailscale_qr_alt": "رمز الاستجابة السريعة لعنوان URL لـ Tailscale"
},
"convert": {
"cloning_required": "اختر نموذجًا جاهزًا لتحويل النص إلى كلام يدعم استنساخ الصوت.",
"source_kicker": "المقطع المصدر",
"drop_audio": "أسقط المقطع المراد تحويله — أو انقر. WAV، MP3، M4A…",
"target_voice": "الصوت الهدف",
@@ -2492,7 +2493,9 @@
"none_ready_title": "تحويل النص إلى كلام غير جاهز",
"none_ready_body": "يمكن لـ VoiceStudio إعداد أفضل خيار متوافق مع هذا الجهاز تلقائيًا.",
"active": "المحرك: {{name}}",
"none": "لا يوجد محرك نشط"
"none": "لا يوجد محرك نشط",
"localInstallRequired": "ثبّت هذا المحرك على جهاز الخادم باستخدام localhost أو دليل الإعداد. التثبيت عن بُعد معطّل.",
"argosRuntimeUnavailable": "بيئة تشغيل Argos غير متاحة. أعد تثبيت الخادم الخلفي أو اختر NLLB."
},
"performanceProfile": {
"requiresEngine": "ثبّت {{engine}} وحدده لضبط السرعة والجودة.",
@@ -2527,6 +2530,7 @@
"recorder_unsupported": "لا يستطيع هذا النظام ترميز صوت الميكروفون."
},
"tts_errors": {
"profile_language_rejected": "يستخدم الوضع التلقائي لغة ملف الصوت ({{language}})، التي لا يدعمها هذا المحرك. غيّر لغة الملف أو اختر لغة مدعومة أو بدّل المحرك.",
"generation_in_progress": "هناك عملية توليد قيد التشغيل بالفعل — انتظر حتى تنتهي قبل بدء عملية أخرى.",
"error_prefix": "خطأ: {{message}}",
"ignored_duplicate": "تم التجاهل (تم تعيين الفئة بالفعل): {{items}}",
@@ -2237,6 +2237,7 @@
"tailscale_qr_alt": "QR-Code für die Tailscale-URL"
},
"convert": {
"cloning_required": "Wähle ein einsatzbereites Sprachsynthesemodell, das Stimmenklonen unterstützt.",
"source_kicker": "Quellclip",
"drop_audio": "Clip zum Konvertieren hier ablegen — oder klicken. WAV, MP3, M4A…",
"target_voice": "Zielstimme",
@@ -2484,7 +2485,9 @@
"none_ready_title": "Sprachausgabe ist nicht bereit",
"none_ready_body": "VoiceStudio kann automatisch die beste kompatible Option für dieses Gerät einrichten.",
"active": "Engine: {{name}}",
"none": "Keine Engine aktiv"
"none": "Keine Engine aktiv",
"localInstallRequired": "Installiere diese Engine auf dem Backend-Rechner über localhost oder die Einrichtungsanleitung. Die Ferninstallation ist deaktiviert.",
"argosRuntimeUnavailable": "Die Argos-Laufzeit ist nicht verfügbar. Installiere das Backend neu oder wähle NLLB."
},
"performanceProfile": {
"requiresEngine": "Installiere und wähle {{engine}}, um Geschwindigkeit und Qualität anzupassen.",
@@ -2519,6 +2522,7 @@
"channels_stereo": "Stereo"
},
"tts_errors": {
"profile_language_rejected": "Automatisch verwendet die Sprache des Stimmprofils ({{language}}), die diese Engine nicht unterstützt. Ändere die Profilsprache, wähle eine unterstützte Sprache oder wechsle die Engine.",
"generation_in_progress": "Eine Generierung läuft bereits warte, bis sie abgeschlossen ist, bevor du eine weitere startest.",
"error_prefix": "Fehler: {{message}}",
"ignored_duplicate": "Ignoriert (Kategorie bereits festgelegt): {{items}}",
@@ -114,7 +114,9 @@
"recheck": "Re-check",
"rechecking": "Re-checking…",
"failed": "failed",
"latencyMs": "{{ms}} ms"
"latencyMs": "{{ms}} ms",
"localInstallRequired": "Install this engine on the backend computer using localhost or its setup guide. Remote installation is disabled.",
"argosRuntimeUnavailable": "Argos runtime unavailable. Reinstall the backend or select NLLB."
},
"clone": {
"title": "Voice cloning",
@@ -290,6 +292,7 @@
"recorder_unsupported": "This system cannot encode microphone audio."
},
"tts_errors": {
"profile_language_rejected": "Auto uses the voice profiles language ({{language}}), which this engine does not support. Change the profile language, choose a supported language, or switch engines.",
"enter_text": "Enter some text first.",
"upload_or_select": "Upload a reference clip or pick a saved voice.",
"trim_hint": "The clip is {{duration}}s — trim to ≤{{max}}s for best cloning.",
@@ -2286,6 +2289,7 @@
"tailscale_qr_alt": "QR code for the Tailscale URL"
},
"convert": {
"cloning_required": "Choose a ready text-to-speech model that supports voice cloning.",
"source_kicker": "Source clip",
"drop_audio": "Drop the clip to convert — or click. WAV, MP3, M4A…",
"target_voice": "Target voice",
@@ -2239,6 +2239,7 @@
"tailscale_qr_alt": "Código QR para la URL de Tailscale"
},
"convert": {
"cloning_required": "Elige un modelo de texto a voz listo para usar que permita clonar voces.",
"source_kicker": "Clip de origen",
"drop_audio": "Suelta el clip a convertir — o haz clic. WAV, MP3, M4A…",
"target_voice": "Voz de destino",
@@ -2486,7 +2487,9 @@
"none_ready_title": "La síntesis de voz no está lista",
"none_ready_body": "VoiceStudio puede configurar automáticamente la mejor opción compatible con este dispositivo.",
"active": "Motor: {{name}}",
"none": "Ningún motor activo"
"none": "Ningún motor activo",
"localInstallRequired": "Instala este motor en el equipo del servidor mediante localhost o su guía de configuración. La instalación remota está desactivada.",
"argosRuntimeUnavailable": "El entorno de ejecución de Argos no está disponible. Reinstala el backend o selecciona NLLB."
},
"performanceProfile": {
"requiresEngine": "Instala y selecciona {{engine}} para ajustar la velocidad y la calidad.",
@@ -2521,6 +2524,7 @@
"channels_mono": "Mono"
},
"tts_errors": {
"profile_language_rejected": "Automático usa el idioma del perfil de voz ({{language}}), que este motor no admite. Cambia el idioma del perfil, elige uno compatible o cambia de motor.",
"generation_in_progress": "Ya hay una generación en curso; espera a que termine antes de iniciar otra.",
"ignored_duplicate": "Ignorado (categoría ya establecida): {{items}}",
"ignored_conflict": "Se ignoró {{items}} — un dialecto chino y un acento inglés no se pueden combinar.",
@@ -2239,6 +2239,7 @@
"tailscale_qr_alt": "Code QR pour l'URL Tailscale"
},
"convert": {
"cloning_required": "Choisissez un modèle de synthèse vocale prêt à utiliser qui prend en charge le clonage vocal.",
"source_kicker": "Clip source",
"drop_audio": "Déposez le clip à convertir — ou cliquez. WAV, MP3, M4A…",
"target_voice": "Voix cible",
@@ -2486,7 +2487,9 @@
"none_ready_title": "La synthèse vocale nest pas prête",
"none_ready_body": "VoiceStudio peut configurer automatiquement la meilleure option compatible avec cet appareil.",
"active": "Moteur : {{name}}",
"none": "Aucun moteur actif"
"none": "Aucun moteur actif",
"localInstallRequired": "Installez ce moteur sur lordinateur du serveur via localhost ou son guide de configuration. Linstallation à distance est désactivée.",
"argosRuntimeUnavailable": "Le moteur Argos est indisponible. Réinstallez le backend ou sélectionnez NLLB."
},
"performanceProfile": {
"requiresEngine": "Installez et sélectionnez {{engine}} pour régler la vitesse et la qualité.",
@@ -2521,6 +2524,7 @@
"microphone_number": "Microphone {{number}}"
},
"tts_errors": {
"profile_language_rejected": "Auto utilise la langue du profil vocal ({{language}}), non prise en charge par ce moteur. Modifiez la langue du profil, choisissez une langue compatible ou changez de moteur.",
"generation_in_progress": "Une génération est déjà en cours — attendez sa fin avant den lancer une autre.",
"error_prefix": "Erreur : {{message}}",
"ignored_duplicate": "Ignoré (catégorie déjà définie) : {{items}}",
@@ -2237,6 +2237,7 @@
"tailscale_qr_alt": "टेलस्केल यूआरएल के लिए क्यूआर कोड"
},
"convert": {
"cloning_required": "वॉइस क्लोनिंग का समर्थन करने वाला तैयार टेक्स्ट-टू-स्पीच मॉडल चुनें।",
"source_kicker": "स्रोत क्लिप",
"drop_audio": "कन्वर्ट करने के लिए क्लिप यहाँ छोड़ें — या क्लिक करें। WAV, MP3, M4A…",
"target_voice": "लक्ष्य आवाज़",
@@ -2484,7 +2485,9 @@
"none_ready_title": "वाक् संश्लेषण तैयार नहीं है",
"none_ready_body": "VoiceStudio इस डिवाइस के लिए सबसे अच्छा संगत विकल्प अपने आप सेट कर सकता है।",
"active": "इंजन: {{name}}",
"none": "कोई इंजन सक्रिय नहीं है"
"none": "कोई इंजन सक्रिय नहीं है",
"localInstallRequired": "इस इंजन को बैकएंड कंप्यूटर पर localhost या सेटअप गाइड से इंस्टॉल करें। रिमोट इंस्टॉलेशन बंद है।",
"argosRuntimeUnavailable": "Argos रनटाइम उपलब्ध नहीं है। बैकएंड फिर से इंस्टॉल करें या NLLB चुनें।"
},
"performanceProfile": {
"requiresEngine": "गति और गुणवत्ता समायोजित करने के लिए {{engine}} इंस्टॉल करके चुनें।",
@@ -2519,6 +2522,7 @@
"recorder_unsupported": "यह सिस्टम माइक्रोफ़ोन ऑडियो एन्कोड नहीं कर सकता।"
},
"tts_errors": {
"profile_language_rejected": "ऑटो वॉइस प्रोफ़ाइल की भाषा ({{language}}) का उपयोग करता है, जिसे यह इंजन समर्थन नहीं करता। प्रोफ़ाइल की भाषा बदलें, समर्थित भाषा चुनें या इंजन बदलें।",
"generation_in_progress": "एक जनरेशन पहले से चल रहा है — दूसरा शुरू करने से पहले उसके पूरा होने की प्रतीक्षा करें।",
"error_prefix": "त्रुटि: {{message}}",
"ignored_duplicate": "अनदेखा (श्रेणी पहले से ही सेट): {{items}}",
@@ -2237,6 +2237,7 @@
"tailscale_qr_alt": "Kode QR untuk URL Tailscale"
},
"convert": {
"cloning_required": "Pilih model teks ke suara yang siap digunakan dan mendukung kloning suara.",
"source_kicker": "Klip sumber",
"drop_audio": "Letakkan klip yang akan dikonversi — atau klik. WAV, MP3, M4A…",
"target_voice": "Suara target",
@@ -2484,7 +2485,9 @@
"none_ready_title": "Sintesis ucapan belum siap",
"none_ready_body": "VoiceStudio dapat menyiapkan opsi kompatibel terbaik untuk perangkat ini secara otomatis.",
"active": "Mesin: {{name}}",
"none": "Tidak ada mesin aktif"
"none": "Tidak ada mesin aktif",
"localInstallRequired": "Instal mesin ini di komputer backend melalui localhost atau panduan penyiapannya. Instalasi jarak jauh dinonaktifkan.",
"argosRuntimeUnavailable": "Runtime Argos tidak tersedia. Instal ulang backend atau pilih NLLB."
},
"performanceProfile": {
"requiresEngine": "Instal dan pilih {{engine}} untuk menyesuaikan kecepatan dan kualitas.",
@@ -2519,6 +2522,7 @@
"channels_stereo": "Stereo"
},
"tts_errors": {
"profile_language_rejected": "Otomatis menggunakan bahasa profil suara ({{language}}), yang tidak didukung mesin ini. Ubah bahasa profil, pilih bahasa yang didukung, atau ganti mesin.",
"generation_in_progress": "Pembuatan sedang berjalan — tunggu hingga selesai sebelum memulai yang lain.",
"error_prefix": "Kesalahan: {{message}}",
"ignored_duplicate": "Diabaikan (kategori sudah ditetapkan): {{items}}",
@@ -2239,6 +2239,7 @@
"tailscale_qr_alt": "Codice QR per l'URL Tailscale"
},
"convert": {
"cloning_required": "Scegli un modello di sintesi vocale pronto che supporti la clonazione della voce.",
"source_kicker": "Clip sorgente",
"drop_audio": "Trascina qui il clip da convertire — o fai clic. WAV, MP3, M4A…",
"target_voice": "Voce di destinazione",
@@ -2486,7 +2487,9 @@
"none_ready_title": "La sintesi vocale non è pronta",
"none_ready_body": "VoiceStudio può configurare automaticamente lopzione compatibile migliore per questo dispositivo.",
"active": "Motore: {{name}}",
"none": "Nessun motore attivo"
"none": "Nessun motore attivo",
"localInstallRequired": "Installa questo motore sul computer del backend tramite localhost o la guida alla configurazione. Linstallazione remota è disabilitata.",
"argosRuntimeUnavailable": "Il runtime di Argos non è disponibile. Reinstalla il backend o seleziona NLLB."
},
"performanceProfile": {
"requiresEngine": "Installa e seleziona {{engine}} per regolare velocità e qualità.",
@@ -2521,6 +2524,7 @@
"channels_stereo": "Stereo"
},
"tts_errors": {
"profile_language_rejected": "Auto usa la lingua del profilo vocale ({{language}}), non supportata da questo motore. Modifica la lingua del profilo, scegli una lingua supportata o cambia motore.",
"generation_in_progress": "È già in corso una generazione: attendi che termini prima di avviarne unaltra.",
"error_prefix": "Errore: {{message}}",
"ignored_duplicate": "Ignorato (categoria già impostata): {{items}}",
@@ -2237,6 +2237,7 @@
"tailscale_qr_alt": "Tailscale URL の QR コード"
},
"convert": {
"cloning_required": "音声クローンに対応した、使用可能な音声合成モデルを選択してください。",
"source_kicker": "ソースクリップ",
"drop_audio": "変換するクリップをドロップ — またはクリック。WAV、MP3、M4A…",
"target_voice": "変換先の声",
@@ -2484,7 +2485,9 @@
"none_ready_title": "音声合成の準備ができていません",
"none_ready_body": "VoiceStudio がこのデバイスに最適な互換オプションを自動で設定できます。",
"active": "エンジン:{{name}}",
"none": "有効なエンジンがありません"
"none": "有効なエンジンがありません",
"localInstallRequired": "バックエンドのコンピューターで localhost またはセットアップガイドを使ってこのエンジンをインストールしてください。リモートインストールは無効です。",
"argosRuntimeUnavailable": "Argos ランタイムを利用できません。バックエンドを再インストールするか、NLLB を選択してください。"
},
"performanceProfile": {
"requiresEngine": "速度と品質を調整するには、{{engine}} をインストールして選択してください。",
@@ -2519,6 +2522,7 @@
"recorder_unsupported": "このシステムではマイク音声をエンコードできません。"
},
"tts_errors": {
"profile_language_rejected": "自動では音声プロファイルの言語({{language}})が使われますが、このエンジンは対応していません。プロファイルの言語を変更するか、対応言語または別のエンジンを選択してください。",
"generation_in_progress": "生成がすでに実行中です。完了してから次の生成を開始してください。",
"error_prefix": "エラー: {{message}}",
"ignored_duplicate": "無視 (カテゴリはすでに設定されています): {{items}}",
@@ -2237,6 +2237,7 @@
"tailscale_qr_alt": "Tailscale URL의 QR 코드"
},
"convert": {
"cloning_required": "음성 복제를 지원하는 사용 가능한 음성 합성 모델을 선택하세요.",
"source_kicker": "소스 클립",
"drop_audio": "변환할 클립을 끌어다 놓거나 클릭하세요. WAV, MP3, M4A…",
"target_voice": "대상 목소리",
@@ -2484,7 +2485,9 @@
"none_ready_title": "음성 합성이 준비되지 않았습니다",
"none_ready_body": "VoiceStudio가 이 기기에 가장 적합한 호환 옵션을 자동으로 설정할 수 있습니다.",
"active": "엔진: {{name}}",
"none": "활성 엔진 없음"
"none": "활성 엔진 없음",
"localInstallRequired": "백엔드 컴퓨터에서 localhost 또는 설정 가이드를 사용하여 이 엔진을 설치하세요. 원격 설치는 비활성화되어 있습니다.",
"argosRuntimeUnavailable": "Argos 런타임을 사용할 수 없습니다. 백엔드를 다시 설치하거나 NLLB를 선택하세요."
},
"performanceProfile": {
"requiresEngine": "속도와 품질을 조절하려면 {{engine}}을 설치하고 선택하세요.",
@@ -2519,6 +2522,7 @@
"recorder_unsupported": "이 시스템에서는 마이크 오디오를 인코딩할 수 없습니다."
},
"tts_errors": {
"profile_language_rejected": "자동은 이 엔진이 지원하지 않는 음성 프로필의 언어({{language}})를 사용합니다. 프로필 언어를 변경하거나 지원되는 언어 또는 다른 엔진을 선택하세요.",
"generation_in_progress": "이미 생성 작업이 실행 중입니다. 완료된 후 새 작업을 시작하세요.",
"error_prefix": "오류: {{message}}",
"ignored_duplicate": "무시됨(카테고리가 이미 설정됨): {{items}}",
@@ -2237,6 +2237,7 @@
"tailscale_qr_alt": "QR-code voor de Tailscale-URL"
},
"convert": {
"cloning_required": "Kies een gebruiksklaar tekst-naar-spraakmodel dat stemmen klonen ondersteunt.",
"source_kicker": "Bronclip",
"drop_audio": "Sleep de te converteren clip hierheen — of klik. WAV, MP3, M4A…",
"target_voice": "Doelstem",
@@ -2484,7 +2485,9 @@
"none_ready_title": "Spraaksynthese is niet gereed",
"none_ready_body": "VoiceStudio kan automatisch de beste compatibele optie voor dit apparaat instellen.",
"active": "Engine: {{name}}",
"none": "Geen engine actief"
"none": "Geen engine actief",
"localInstallRequired": "Installeer deze engine op de backendcomputer via localhost of de installatiehandleiding. Installatie op afstand is uitgeschakeld.",
"argosRuntimeUnavailable": "De Argos-runtime is niet beschikbaar. Installeer de backend opnieuw of kies NLLB."
},
"performanceProfile": {
"requiresEngine": "Installeer en selecteer {{engine}} om snelheid en kwaliteit af te stellen.",
@@ -2519,6 +2522,7 @@
"channels_stereo": "Stereo"
},
"tts_errors": {
"profile_language_rejected": "Automatisch gebruikt de taal van het stemprofiel ({{language}}), die deze engine niet ondersteunt. Wijzig de profieltaal, kies een ondersteunde taal of wissel van engine.",
"generation_in_progress": "Er wordt al audio gegenereerd — wacht tot dit klaar is voordat je opnieuw begint.",
"error_prefix": "Fout: {{message}}",
"ignored_duplicate": "Genegeerd (categorie al ingesteld): {{items}}",
@@ -2241,6 +2241,7 @@
"tailscale_qr_alt": "Kod QR dla adresu URL Tailscale"
},
"convert": {
"cloning_required": "Wybierz gotowy do użycia model syntezy mowy obsługujący klonowanie głosu.",
"source_kicker": "Klip źródłowy",
"drop_audio": "Upuść klip do konwersji — lub kliknij. WAV, MP3, M4A…",
"target_voice": "Głos docelowy",
@@ -2488,7 +2489,9 @@
"none_ready_title": "Synteza mowy nie jest gotowa",
"none_ready_body": "VoiceStudio może automatycznie skonfigurować najlepszą zgodną opcję dla tego urządzenia.",
"active": "Silnik: {{name}}",
"none": "Brak aktywnego silnika"
"none": "Brak aktywnego silnika",
"localInstallRequired": "Zainstaluj ten silnik na komputerze zaplecza przez localhost lub zgodnie z instrukcją konfiguracji. Instalacja zdalna jest wyłączona.",
"argosRuntimeUnavailable": "Środowisko Argos jest niedostępne. Zainstaluj ponownie backend lub wybierz NLLB."
},
"performanceProfile": {
"requiresEngine": "Zainstaluj i wybierz {{engine}}, aby dostosować szybkość i jakość.",
@@ -2523,6 +2526,7 @@
"channels_stereo": "Stereo"
},
"tts_errors": {
"profile_language_rejected": "Tryb automatyczny używa języka profilu głosu ({{language}}), którego ten silnik nie obsługuje. Zmień język profilu, wybierz obsługiwany język lub zmień silnik.",
"generation_in_progress": "Generowanie już trwa — poczekaj na jego zakończenie przed rozpoczęciem kolejnego.",
"error_prefix": "Błąd: {{message}}",
"ignored_duplicate": "Ignorowane (kategoria już ustawiona): {{items}}",
@@ -2239,6 +2239,7 @@
"tailscale_qr_alt": "Código QR para o URL Tailscale"
},
"convert": {
"cloning_required": "Escolha um modelo de síntese de voz pronto para uso que permita clonar vozes.",
"source_kicker": "Clipe de origem",
"drop_audio": "Solte o clipe a converter — ou clique. WAV, MP3, M4A…",
"target_voice": "Voz de destino",
@@ -2486,7 +2487,9 @@
"none_ready_title": "A síntese de voz não está pronta",
"none_ready_body": "O VoiceStudio pode configurar automaticamente a melhor opção compatível com este dispositivo.",
"active": "Motor: {{name}}",
"none": "Nenhum motor ativo"
"none": "Nenhum motor ativo",
"localInstallRequired": "Instale este motor no computador do backend pelo localhost ou pelo guia de configuração. A instalação remota está desativada.",
"argosRuntimeUnavailable": "O ambiente de execução do Argos está indisponível. Reinstale o backend ou selecione NLLB."
},
"performanceProfile": {
"requiresEngine": "Instale e selecione {{engine}} para ajustar a velocidade e a qualidade.",
@@ -2521,6 +2524,7 @@
"channels_mono": "Mono"
},
"tts_errors": {
"profile_language_rejected": "Automático usa o idioma do perfil de voz ({{language}}), não suportado por este motor. Altere o idioma do perfil, escolha um idioma compatível ou mude de motor.",
"generation_in_progress": "Já existe uma geração em andamento — aguarde a conclusão antes de iniciar outra.",
"error_prefix": "Erro: {{message}}",
"ignored_duplicate": "Ignorado (categoria já definida): {{items}}",
@@ -2241,6 +2241,7 @@
"tailscale_qr_alt": "QR-код для URL-адреса Tailscale"
},
"convert": {
"cloning_required": "Выберите готовую к работе модель синтеза речи с поддержкой клонирования голоса.",
"source_kicker": "Исходный клип",
"drop_audio": "Перетащите клип для преобразования — или нажмите. WAV, MP3, M4A…",
"target_voice": "Целевой голос",
@@ -2488,7 +2489,9 @@
"none_ready_title": "Синтез речи не готов",
"none_ready_body": "VoiceStudio может автоматически настроить лучший совместимый вариант для этого устройства.",
"active": "Движок: {{name}}",
"none": "Нет активного движка"
"none": "Нет активного движка",
"localInstallRequired": "Установите этот движок на компьютере сервера через localhost или по инструкции. Удалённая установка отключена.",
"argosRuntimeUnavailable": "Среда выполнения Argos недоступна. Переустановите серверную часть или выберите NLLB."
},
"performanceProfile": {
"requiresEngine": "Установите и выберите {{engine}}, чтобы настроить скорость и качество.",
@@ -2523,6 +2526,7 @@
"recorder_unsupported": "Эта система не может кодировать звук с микрофона."
},
"tts_errors": {
"profile_language_rejected": "Авто использует язык голосового профиля ({{language}}), который этот движок не поддерживает. Измените язык профиля, выберите поддерживаемый язык или другой движок.",
"generation_in_progress": "Генерация уже выполняется — дождитесь её завершения, прежде чем запускать следующую.",
"error_prefix": "Ошибка: {{message}}",
"ignored_duplicate": "Игнорируется (категория уже установлена): {{items}}",
@@ -2237,6 +2237,7 @@
"tailscale_qr_alt": "QR-kod för Tailscale URL"
},
"convert": {
"cloning_required": "Välj en användningsklar talsyntesmodell som stöder röstkloning.",
"source_kicker": "Källklipp",
"drop_audio": "Släpp klippet som ska konverteras — eller klicka. WAV, MP3, M4A…",
"target_voice": "Målröst",
@@ -2484,7 +2485,9 @@
"none_ready_title": "Talsyntesen är inte redo",
"none_ready_body": "VoiceStudio kan automatiskt konfigurera det bästa kompatibla alternativet för den här enheten.",
"active": "Motor: {{name}}",
"none": "Ingen motor aktiv"
"none": "Ingen motor aktiv",
"localInstallRequired": "Installera motorn på backenddatorn via localhost eller installationsguiden. Fjärrinstallation är avstängd.",
"argosRuntimeUnavailable": "Argos körmiljö är inte tillgänglig. Installera om backend eller välj NLLB."
},
"performanceProfile": {
"requiresEngine": "Installera och välj {{engine}} för att justera hastighet och kvalitet.",
@@ -2519,6 +2522,7 @@
"channels_stereo": "Stereo"
},
"tts_errors": {
"profile_language_rejected": "Auto använder röstprofilens språk ({{language}}), som denna motor inte stöder. Ändra profilens språk, välj ett språk som stöds eller byt motor.",
"generation_in_progress": "En generering pågår redan vänta tills den är klar innan du startar en ny.",
"error_prefix": "Fel: {{message}}",
"ignored_duplicate": "Ignorerad (kategori redan inställd): {{items}}",
@@ -2237,6 +2237,7 @@
"tailscale_qr_alt": "รหัส QR สำหรับ Tailscale URL"
},
"convert": {
"cloning_required": "เลือกโมเดลแปลงข้อความเป็นเสียงที่พร้อมใช้งานและรองรับการโคลนเสียง",
"source_kicker": "คลิปต้นฉบับ",
"drop_audio": "วางคลิปที่จะแปลงที่นี่ — หรือคลิก WAV, MP3, M4A…",
"target_voice": "เสียงปลายทาง",
@@ -2484,7 +2485,9 @@
"none_ready_title": "การสังเคราะห์เสียงยังไม่พร้อม",
"none_ready_body": "VoiceStudio สามารถตั้งค่าตัวเลือกที่เข้ากันได้ดีที่สุดสำหรับอุปกรณ์นี้โดยอัตโนมัติ",
"active": "เอนจิน: {{name}}",
"none": "ไม่มีเอนจินที่ใช้งานอยู่"
"none": "ไม่มีเอนจินที่ใช้งานอยู่",
"localInstallRequired": "ติดตั้งเอนจินนี้บนคอมพิวเตอร์แบ็กเอนด์ผ่าน localhost หรือคู่มือการตั้งค่า การติดตั้งจากระยะไกลถูกปิดใช้งาน",
"argosRuntimeUnavailable": "ไม่สามารถใช้งานรันไทม์ Argos ได้ โปรดติดตั้งแบ็กเอนด์ใหม่หรือเลือก NLLB"
},
"performanceProfile": {
"requiresEngine": "ติดตั้งและเลือก {{engine}} เพื่อปรับความเร็วและคุณภาพ",
@@ -2519,6 +2522,7 @@
"recorder_unsupported": "ระบบนี้ไม่สามารถเข้ารหัสเสียงจากไมโครโฟนได้"
},
"tts_errors": {
"profile_language_rejected": "อัตโนมัติใช้ภาษาของโปรไฟล์เสียง ({{language}}) ซึ่งเอนจินนี้ไม่รองรับ เปลี่ยนภาษาโปรไฟล์ เลือกภาษาที่รองรับ หรือเปลี่ยนเอนจิน",
"generation_in_progress": "มีการสร้างเสียงกำลังทำงานอยู่ — รอให้เสร็จก่อนเริ่มงานใหม่",
"error_prefix": "ข้อผิดพลาด: {{message}}",
"ignored_duplicate": "ละเว้น (หมวดหมู่ที่ตั้งไว้แล้ว): {{items}}",
@@ -2237,6 +2237,7 @@
"tailscale_qr_alt": "Kuyruk Ölçeği URL'si için QR kodu"
},
"convert": {
"cloning_required": "Ses klonlamayı destekleyen, kullanıma hazır bir metinden konuşmaya modeli seçin.",
"source_kicker": "Kaynak klip",
"drop_audio": "Dönüştürülecek klibi bırakın — veya tıklayın. WAV, MP3, M4A…",
"target_voice": "Hedef ses",
@@ -2484,7 +2485,9 @@
"none_ready_title": "Konuşma sentezi hazır değil",
"none_ready_body": "VoiceStudio bu cihaz için en iyi uyumlu seçeneği otomatik olarak ayarlayabilir.",
"active": "Motor: {{name}}",
"none": "Etkin motor yok"
"none": "Etkin motor yok",
"localInstallRequired": "Bu motoru arka uç bilgisayarında localhost veya kurulum kılavuzu üzerinden yükleyin. Uzaktan kurulum devre dışıdır.",
"argosRuntimeUnavailable": "Argos çalışma ortamı kullanılamıyor. Arka ucu yeniden yükleyin veya NLLB seçin."
},
"performanceProfile": {
"requiresEngine": "Hız ve kaliteyi ayarlamak için {{engine}} yükleyip seçin.",
@@ -2519,6 +2522,7 @@
"channels_stereo": "Stereo"
},
"tts_errors": {
"profile_language_rejected": "Otomatik, bu motorun desteklemediği ses profili dilini ({{language}}) kullanır. Profil dilini değiştirin, desteklenen bir dil seçin veya motoru değiştirin.",
"generation_in_progress": "Bir oluşturma işlemi zaten çalışıyor — yenisini başlatmadan önce bitmesini bekleyin.",
"error_prefix": "Hata: {{message}}",
"ignored_duplicate": "Yoksayıldı (kategori zaten ayarlandı): {{items}}",
@@ -2241,6 +2241,7 @@
"tailscale_qr_alt": "QR-код для URL-адреси Tailscale"
},
"convert": {
"cloning_required": "Виберіть готову до роботи модель синтезу мовлення з підтримкою клонування голосу.",
"source_kicker": "Вихідний кліп",
"drop_audio": "Перетягніть кліп для перетворення — або натисніть. WAV, MP3, M4A…",
"target_voice": "Цільовий голос",
@@ -2488,7 +2489,9 @@
"none_ready_title": "Синтез мовлення не готовий",
"none_ready_body": "VoiceStudio може автоматично налаштувати найкращий сумісний варіант для цього пристрою.",
"active": "Рушій: {{name}}",
"none": "Немає активного рушія"
"none": "Немає активного рушія",
"localInstallRequired": "Установіть цей рушій на комп’ютері сервера через localhost або за інструкцією. Віддалене встановлення вимкнено.",
"argosRuntimeUnavailable": "Середовище виконання Argos недоступне. Перевстановіть серверну частину або виберіть NLLB."
},
"performanceProfile": {
"requiresEngine": "Установіть і виберіть {{engine}}, щоб налаштувати швидкість і якість.",
@@ -2523,6 +2526,7 @@
"recorder_unsupported": "Ця система не може кодувати звук із мікрофона."
},
"tts_errors": {
"profile_language_rejected": "Авто використовує мову голосового профілю ({{language}}), яку цей рушій не підтримує. Змініть мову профілю, виберіть підтримувану мову або інший рушій.",
"generation_in_progress": "Генерація вже виконується — дочекайтеся її завершення, перш ніж запускати наступну.",
"error_prefix": "Помилка: {{message}}",
"ignored_duplicate": "Проігноровано (категорію вже встановлено): {{items}}",
@@ -2237,6 +2237,7 @@
"tailscale_qr_alt": "Mã QR cho URL Tailscale"
},
"convert": {
"cloning_required": "Chọn mô hình chuyển văn bản thành giọng nói sẵn sàng sử dụng và hỗ trợ nhân bản giọng nói.",
"source_kicker": "Clip nguồn",
"drop_audio": "Thả clip cần chuyển đổi vào đây — hoặc nhấp. WAV, MP3, M4A…",
"target_voice": "Giọng đích",
@@ -2484,7 +2485,9 @@
"none_ready_title": "Tổng hợp giọng nói chưa sẵn sàng",
"none_ready_body": "VoiceStudio có thể tự động thiết lập tùy chọn tương thích tốt nhất cho thiết bị này.",
"active": "Bộ máy: {{name}}",
"none": "Không có bộ máy đang hoạt động"
"none": "Không có bộ máy đang hoạt động",
"localInstallRequired": "Cài đặt công cụ này trên máy chủ qua localhost hoặc hướng dẫn thiết lập. Cài đặt từ xa đã bị tắt.",
"argosRuntimeUnavailable": "Không thể sử dụng môi trường chạy Argos. Hãy cài lại backend hoặc chọn NLLB."
},
"performanceProfile": {
"requiresEngine": "Cài đặt và chọn {{engine}} để điều chỉnh tốc độ và chất lượng.",
@@ -2519,6 +2522,7 @@
"recorder_unsupported": "Hệ thống này không thể mã hóa âm thanh từ micrô."
},
"tts_errors": {
"profile_language_rejected": "Tự động sử dụng ngôn ngữ của hồ sơ giọng nói ({{language}}), nhưng bộ máy này không hỗ trợ. Đổi ngôn ngữ hồ sơ, chọn ngôn ngữ được hỗ trợ hoặc đổi bộ máy.",
"generation_in_progress": "Một tác vụ tạo âm thanh đang chạy — hãy đợi hoàn tất trước khi bắt đầu tác vụ khác.",
"error_prefix": "Lỗi: {{message}}",
"ignored_duplicate": "Đã bỏ qua (danh mục đã được đặt): {{items}}",
@@ -2241,6 +2241,7 @@
"tailscale_qr_alt": "Tailscale URL 的二维码"
},
"convert": {
"cloning_required": "请选择已就绪且支持声音克隆的语音合成模型。",
"source_kicker": "源音频",
"drop_audio": "拖入要转换的音频 — 或点击。WAV、MP3、M4A…",
"target_voice": "目标声音",
@@ -2488,7 +2489,9 @@
"none_ready_title": "语音合成尚未就绪",
"none_ready_body": "VoiceStudio 可以自动为此设备设置最佳兼容选项。",
"active": "引擎:{{name}}",
"none": "没有启用的引擎"
"none": "没有启用的引擎",
"localInstallRequired": "请在后端计算机上通过 localhost 或安装指南安装此引擎。远程安装已禁用。",
"argosRuntimeUnavailable": "Argos 运行时不可用。请重新安装后端或选择 NLLB。"
},
"performanceProfile": {
"requiresEngine": "安装并选择 {{engine}} 以调整速度和质量。",
@@ -2523,6 +2526,7 @@
"recorder_unsupported": "此系统无法编码麦克风音频。"
},
"tts_errors": {
"profile_language_rejected": "自动模式使用声音档案的语言({{language}}),但此引擎不支持该语言。请修改档案语言、选择支持的语言或更换引擎。",
"generation_in_progress": "已有生成任务正在运行,请等待其完成后再启动新任务。",
"error_prefix": "错误:{{message}}",
"ignored_duplicate": "忽略(类别已设置):{{items}}",
@@ -2237,6 +2237,7 @@
"tailscale_qr_alt": "Tailscale URL 的二維碼"
},
"convert": {
"cloning_required": "請選擇已就緒且支援聲音複製的語音合成模型。",
"source_kicker": "來源音訊",
"drop_audio": "拖入要轉換的音訊 — 或點擊。WAV、MP3、M4A…",
"target_voice": "目標聲音",
@@ -2484,7 +2485,9 @@
"none_ready_title": "語音合成尚未就緒",
"none_ready_body": "VoiceStudio 可以自動為此裝置設定最佳相容選項。",
"active": "引擎:{{name}}",
"none": "沒有啟用的引擎"
"none": "沒有啟用的引擎",
"localInstallRequired": "請在後端電腦上透過 localhost 或安裝指南安裝此引擎。遠端安裝已停用。",
"argosRuntimeUnavailable": "Argos 執行環境無法使用。請重新安裝後端或選擇 NLLB。"
},
"performanceProfile": {
"requiresEngine": "安裝並選取 {{engine}} 以調整速度與品質。",
@@ -2519,6 +2522,7 @@
"recorder_unsupported": "此系統無法編碼麥克風音訊。"
},
"tts_errors": {
"profile_language_rejected": "自動模式使用聲音設定檔的語言({{language}}),但此引擎不支援該語言。請修改設定檔語言、選擇支援的語言或更換引擎。",
"generation_in_progress": "已有產生工作正在執行,請等待完成後再啟動新工作。",
"error_prefix": "錯誤:{{message}}",
"ignored_duplicate": "忽略(類別已設定):{{items}}",
@@ -1,3 +1,4 @@
import i18next from 'i18next';
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest';
import {
_resetBackendContactForTests,
@@ -18,11 +19,20 @@ beforeEach(() => {
});
afterEach(() => {
vi.restoreAllMocks();
vi.unstubAllGlobals();
_resetBackendContactForTests();
});
describe('errorFromResponse', () => {
it('localizes Argos runtime errors and retains diagnostics', async () => {
const translate = vi.spyOn(i18next, 't').mockReturnValue('Localized recovery guidance');
const detail = { code: 'argos_runtime_unavailable', message: 'Raw native diagnostic' };
const err = await errorFromResponse(new Response(JSON.stringify({ detail }), { status: 400 }));
expect(err.detail).toBe('Localized recovery guidance');
expect(translate).toHaveBeenCalledWith('engines.argosRuntimeUnavailable');
expect(err.payload?.detail).toEqual(detail);
});
it('localizes background preservation errors and retains diagnostics', async () => {
const detail = { code: 'dub_background_unavailable', message: 'Raw diagnostic' };
const err = await errorFromResponse(new Response(JSON.stringify({ detail }), { status: 409 }));
@@ -159,3 +169,23 @@ it('uses a structured error message without losing recovery metadata', async ()
expect(error.message).toBe(detail.message);
expect(error.payload?.detail).toEqual(detail);
});
it('localizes structured profile language failures', async () => {
const translate = vi.spyOn(i18next, 't').mockReturnValue('Localized profile guidance');
try {
const detail = {
code: 'profile_language_rejected',
language: 'Persian',
message: 'English fallback',
};
const error = await errorFromResponse(
new Response(JSON.stringify({ detail }), { status: 400 }),
);
expect(error.detail).toBe('Localized profile guidance');
expect(translate).toHaveBeenCalledWith('tts_errors.profile_language_rejected', {
language: 'Persian',
});
} finally {
translate.mockRestore();
}
});
+16 -1
View File
@@ -1,3 +1,4 @@
import { languageRejectionMessage } from '../../../../../../frontend/src/utils/languageRejection.ts';
/**
* Same-origin API client. The renderer never talks to 127.0.0.1:<port>
* directly (CORS); `/api/*` is proxied by the dev server / the app:// protocol
@@ -44,7 +45,21 @@ export function describeError(err: unknown): string {
}
function detailToString(detail: unknown): string {
if (detail && typeof detail === 'object' && 'code' in detail && detail.code === 'dub_background_unavailable')
const localized = languageRejectionMessage(detail, tr);
if (localized) return localized;
if (
detail &&
typeof detail === 'object' &&
'code' in detail &&
detail.code === 'argos_runtime_unavailable'
)
return tr('engines.argosRuntimeUnavailable');
if (
detail &&
typeof detail === 'object' &&
'code' in detail &&
detail.code === 'dub_background_unavailable'
)
return tr('dubIntegrity.backgroundUnavailable');
if (typeof detail === 'string') return detail;
if (detail == null) return '';
@@ -1,8 +1,11 @@
import i18n from 'i18next';
import { afterEach, describe, expect, it, vi } from 'vitest';
import {
CLONE_MAX_SECONDS,
REF_HARD_MAX_SECONDS,
generateClone,
generateCloneStreaming,
shouldFallbackToClassic,
parseGenerateHeaders,
sanitizeInstruct,
toGenerateForm,
@@ -277,3 +280,31 @@ describe('generateClone', () => {
await assertion;
});
});
it('localizes streamed profile language refusals and prevents classic fallback', async () => {
const translate = vi.spyOn(i18n, 't').mockReturnValue('Localized profile guidance');
try {
vi.stubGlobal(
'fetch',
vi.fn(
async () =>
new Response(
JSON.stringify({
type: 'error',
code: 'profile_language_rejected',
language: 'Persian',
detail: 'English fallback',
retryable: false,
terminal: true,
}) + '\n',
{ headers: { 'content-type': 'application/x-ndjson' } },
),
),
);
const error = await generateCloneStreaming(BASE_INPUT).catch((e) => e);
expect(error.message).toBe('Localized profile guidance');
expect(shouldFallbackToClassic(error)).toBe(false);
} finally {
translate.mockRestore();
}
});
+14 -6
View File
@@ -1,3 +1,5 @@
import i18next from 'i18next';
import { languageRejectionMessage } from '../../../../../../frontend/src/utils/languageRejection.ts';
import { ApiError, apiFetch, isAbortError } from './client';
import type { CloneGenerateInput, GenerateResult } from './types';
import { beginAppActivity } from '@/lib/app-activity';
@@ -255,6 +257,9 @@ interface StreamEvent {
count?: number;
text?: string[];
detail?: string;
code?: string;
language?: string;
terminal?: boolean;
retryable?: boolean;
percent?: number;
}
@@ -300,12 +305,15 @@ export async function generateCloneStreaming(
} else if (event.type === 'done') {
meta = event;
} else if (event.type === 'error') {
const message = event.detail || 'TTS stream reported an error';
const terminal = [
'[clone_ref_unusable]',
'[clone_ref_too_long]',
'[clone_ref_no_speech]',
].some((marker) => message.includes(marker));
const message =
languageRejectionMessage(event, i18next.t) ||
event.detail ||
'TTS stream reported an error';
const terminal =
event.terminal === true ||
['[clone_ref_unusable]', '[clone_ref_too_long]', '[clone_ref_no_speech]'].some((marker) =>
message.includes(marker),
);
throw new StreamingPreviewError(message, { retryable: event.retryable, terminal });
}
};
@@ -66,6 +66,7 @@ export interface EngineBackend {
setup_snippet?: string | null;
docs_url?: string | null;
one_click_install?: boolean;
local_install_required?: boolean;
license_required?: boolean;
license_accepted?: boolean;
effective_device?: string;
+1
View File
@@ -27,6 +27,7 @@
"src/preload/index.d.ts",
"src/shared/**/*",
"src/renderer/env.d.ts",
"../frontend/src/utils/languageRejection.ts",
"../frontend/src/utils/coalescedJsonStorage.ts",
"../frontend/src/utils/indexedDbLongformStore.ts",
"../frontend/src/utils/cookieExport.ts",
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "omnivoice-studio",
"version": "0.5.3",
"version": "0.5.4",
"private": true,
"license": "AGPL-3.0-only",
"type": "module",
+1 -1
View File
@@ -3093,7 +3093,7 @@ dependencies = [
[[package]]
name = "omnivoice-studio"
version = "0.5.3"
version = "0.5.4"
dependencies = [
"arboard",
"cap-std",

Some files were not shown because too many files have changed in this diff Show More