Compare commits

...
Author SHA1 Message Date
Palash Debnath 215c4fed26 test: reproduce resource churn without per-user hook override 2026-09-07 16:44:20 +05:30
Palash Debnath 92e6a80472 test: run generated asset hook inside MSI fixture 2026-09-07 16:43:48 +05:30
Palash Debnath 990705fe0e fix: preserve frontend resources during per-user MSI build 2026-09-07 16:41:41 +05:30
Palash Debnath 4dd8a42d3a test: exercise MSI build hooks with changing resource filenames 2026-09-07 16:39:33 +05:30
Palash Debnath 33ce88e465 Merge pull request #1874 from debpalash/codex/pr-queue-integration
chore(integration): land reviewed app, lifecycle and installer fixes
2026-09-07 15:51:52 +05:30
Palash Debnath 10a0fc27f0 Merge commit '63ade08b' into codex/pr-queue-integration 2026-09-07 15:30:30 +05:30
Palash Debnath 63ade08b6f fix(i18n): reuse translated token cleanup error 2026-09-07 15:30:09 +05:30
Palash Debnath fc7021e2b4 Merge commit '1038a487' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 15:26:02 +05:30
Palash Debnath 1038a487a8 fix(auth): gate onboarding replacement on known token state 2026-09-07 15:25:01 +05:30
Palash Debnath 19b9b4a136 Merge commit 'd9bf2516' into codex/pr-queue-integration 2026-09-07 15:09:06 +05:30
Palash Debnath d9bf251665 test(auth): exercise token paths on native backend hosts 2026-09-07 15:08:24 +05:30
Palash Debnath 5bc9f7856a Merge commit '8428517d' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 15:04:50 +05:30
Palash Debnath 8428517d95 fix(auth): preserve local Hugging Face token paths and cleanup 2026-09-07 15:03:12 +05:30
Palash Debnath 6caabf3b0b Merge commit '6b802bea' into codex/pr-queue-integration
# Conflicts:
#	tests/test_locale_parity.py
2026-09-07 14:29:30 +05:30
Palash Debnath 6b802beaf9 fix: complete compact engine family locale labels 2026-09-07 14:28:45 +05:30
Palash Debnath 2ed38c476e Merge commit '99f93164' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
#	tests/test_locale_parity.py
2026-09-07 14:16:01 +05:30
Palash Debnath 078bc06428 Merge commit 'af245a45' into codex/pr-queue-integration 2026-09-07 14:15:11 +05:30
Palash Debnath 99f93164be fix: translate Hugging Face token source labels 2026-09-07 14:14:29 +05:30
Palash Debnath af245a45ee fix: complete workspace playback and engine translations 2026-09-07 14:14:28 +05:30
Palash Debnath dbbad3d58e Merge commit 'bcd2e531' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
#	docs/install/macos.md
2026-09-07 13:48:19 +05:30
Palash Debnath 8667b34ca5 Merge commit 'f18a7add' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
#	docs/install/macos.md
2026-09-07 13:47:50 +05:30
Palash Debnath bcd2e53175 fix(lifecycle): verify root exit after Darwin signal permission errors 2026-09-07 13:46:59 +05:30
Palash Debnath f18a7adddf fix(lifecycle): verify root exit after Darwin signal permission errors 2026-09-07 13:46:06 +05:30
Palash Debnath a968bc57b3 Merge commit 'f2cfdaa4' into codex/pr-queue-integration 2026-09-07 13:31:45 +05:30
Palash Debnath 7ac5e5ab3d Merge commit 'a768e84f' into codex/pr-queue-integration 2026-09-07 13:31:35 +05:30
Palash Debnath f2cfdaa430 test: assert first-sound Chinese taxonomy normalization 2026-09-07 13:31:12 +05:30
Palash Debnath a768e84ffe test: enforce device-neutral Confucius catalog label 2026-09-07 13:31:11 +05:30
Palash Debnath 4648c23256 Merge commit '24252712' into codex/pr-queue-integration 2026-09-07 13:19:19 +05:30
Palash Debnath 1e96748402 Merge commit '30307ae8' into codex/pr-queue-integration
# Conflicts:
#	.gitignore
2026-09-07 13:19:19 +05:30
Palash Debnath 169e36d5b6 Merge commit '0e7ee3d9' into codex/pr-queue-integration 2026-09-07 13:18:56 +05:30
Palash Debnath 135dd0a3a7 Merge commit '9dbf45ca' into codex/pr-queue-integration 2026-09-07 13:18:49 +05:30
Palash Debnath ef51c6b75d Merge commit 'e32bc942' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 13:18:49 +05:30
Palash Debnath 24252712d2 fix(moss): use declared device identifiers in install hint 2026-09-07 13:18:27 +05:30
Palash Debnath b49d2954d5 Merge commit '4008499e' into codex/pr-queue-integration 2026-09-07 13:18:20 +05:30
Palash Debnath bf8da941d1 Merge commit '3b64692e' into codex/pr-queue-integration 2026-09-07 13:18:11 +05:30
Palash Debnath ca3bf8367f Merge commit 'fc8db259' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 13:18:11 +05:30
Palash Debnath 0c255dfabd Merge commit '9afc9db2' into codex/pr-queue-integration 2026-09-07 13:17:45 +05:30
Palash Debnath 4008499ea5 fix(confucius): list supported device codes in the install hint 2026-09-07 13:17:37 +05:30
Palash Debnath e32bc94248 fix: preserve conversion state and finish workspace labels 2026-09-07 13:17:00 +05:30
Palash Debnath fc8db259a9 fix(bootstrap): prevent stale timeout after retry invalidation 2026-09-07 13:16:43 +05:30
Palash Debnath 0921292319 fix(confucius): keep catalogue hardware metadata accurate 2026-09-07 13:15:07 +05:30
Palash Debnath 9afc9db287 test(budget): resolve application modules at test runtime 2026-09-07 13:14:14 +05:30
Palash Debnath 3b64692eed fix(moss): align catalog install hint with accelerator routing 2026-09-07 13:13:49 +05:30
Palash Debnath 30307ae883 chore: ignore generated Windows MSI diagnostic artifacts 2026-09-07 13:13:36 +05:30
Palash Debnath 9dbf45ca2c test(onboarding): resolve the current runtime validator 2026-09-07 13:12:52 +05:30
Palash Debnath 0e7ee3d912 test(release): isolate cleanup module instances per test 2026-09-07 13:12:52 +05:30
Palash Debnath 140e9262b9 Merge commit '18a0c11c' into codex/pr-queue-integration 2026-09-07 12:47:40 +05:30
Palash Debnath 18a0c11c23 test(devices): clear mocked live probe cache 2026-09-07 12:44:32 +05:30
Palash Debnath a0b84a6b61 Merge commit '8c67b109' into codex/pr-queue-integration 2026-09-07 12:42:34 +05:30
Palash Debnath 8c67b10942 test(devices): patch the active capability module after reloads 2026-09-07 12:42:13 +05:30
Palash Debnath 760e9e5ac7 Merge commit '25498dd1' into codex/pr-queue-integration 2026-09-07 12:31:58 +05:30
Palash Debnath 25498dd17b docs(devices): explain optional DirectML fallback 2026-09-07 12:31:33 +05:30
Palash Debnath 16020c2178 Merge commit 'a41e66e0' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 12:28:46 +05:30
Palash Debnath e4471008b4 Merge commit 'df46b71b' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 12:28:27 +05:30
Palash Debnath 2823d6a8ea Merge commit '57150ede' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 12:28:06 +05:30
Palash Debnath 57150ede64 fix(moss): recover from accelerator probe failures 2026-09-07 12:26:04 +05:30
Palash Debnath df46b71be2 fix: tolerate Confucius and DOTS accelerator probe failures 2026-09-07 12:24:42 +05:30
Palash Debnath a41e66e0f0 docs: note capture widget hide permission repair 2026-09-07 12:23:48 +05:30
Palash Debnath 2b2bfe1a7f fix(capture): allow the widget window to hide after recording 2026-09-07 12:23:32 +05:30
Palash Debnath f728071819 Merge commit '59a487e7' into codex/pr-queue-integration 2026-09-07 12:13:44 +05:30
Palash Debnath 59a487e742 test: start upload watchdog at the blocked write 2026-09-07 12:13:29 +05:30
Palash Debnath ef9bda2e17 Merge commit 'e8a69508' into codex/pr-queue-integration 2026-09-07 12:09:10 +05:30
Palash Debnath e8a6950898 fix: allow exact nonsecret dubbing pane storage key 2026-09-07 12:08:47 +05:30
Palash Debnath 43d47f4d36 Merge commit '93bbfd64' into codex/pr-queue-integration 2026-09-07 12:04:25 +05:30
Palash Debnath 93bbfd64ea docs(device): explain optional accelerator probe fallbacks 2026-09-07 12:03:38 +05:30
Palash Debnath c973b1c983 Merge commit 'b4b10c14' into codex/pr-queue-integration 2026-09-07 12:03:21 +05:30
Palash Debnath c4530b4919 Merge commit '99218e2c' into codex/pr-queue-integration
# Conflicts:
#	frontend/src/components/Header.jsx
2026-09-07 12:03:21 +05:30
Palash Debnath b4b10c1431 fix(header): satisfy native Mac detection lint gate 2026-09-07 12:02:35 +05:30
Palash Debnath 99218e2c34 fix(header): satisfy native Mac detection lint gate 2026-09-07 12:02:15 +05:30
Palash Debnath b759ae7749 Merge commit 'ee3e1474' into codex/pr-queue-integration 2026-09-07 11:58:24 +05:30
Palash Debnath b2f5de328e Merge commit 'af7f1ceb' into codex/pr-queue-integration 2026-09-07 11:58:16 +05:30
Palash Debnath f2dd6c95db Merge branch 'codex/fix-per-user-msi' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 11:58:16 +05:30
Palash Debnath ee3e147401 test: prove upload revocation without scheduling threshold 2026-09-07 11:57:04 +05:30
Palash Debnath af7f1ceb3d test: construct heartbeat backend with lifecycle state 2026-09-07 11:56:16 +05:30
Palash Debnath 0cc22ad803 test: reject unresolved MSI authoring and invalid component identity 2026-09-07 11:56:13 +05:30
Palash Debnath 62b0e05c1d fix(windows): preserve registry separators during template expansion 2026-09-07 11:55:42 +05:30
Palash Debnath aed50121b4 docs: reference per-user MSI repair PR (#1873) 2026-09-07 11:54:25 +05:30
Palash Debnath 079be2fbe5 docs: describe automatic Windows MSI authoring validation 2026-09-07 11:52:41 +05:30
Palash Debnath f463276ca5 ci: validate both Windows MSI scopes with a tiny payload 2026-09-07 11:52:22 +05:30
Palash Debnath 5a1ada0757 fix(windows): give binary registry keypaths explicit stable GUIDs 2026-09-07 11:51:55 +05:30
Palash Debnath 6396cc2b3f style: format Windows installer authoring and regression tests 2026-09-07 11:49:37 +05:30
Palash Debnath 997ba9e673 fix(windows): author per-user MSI file keypaths and folder cleanup 2026-09-07 11:48:20 +05:30
Palash Debnath a438f5db0e Merge branch 'codex/fix-sidecar-timeout-reap' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 11:45:27 +05:30
Palash Debnath f72ab3a3b2 fix(sidecar): quarantine timeout owners until bounded cleanup succeeds 2026-09-07 11:41:18 +05:30
Palash Debnath 1d30a50c1c docs: reference sidecar recovery PR (#1872) 2026-09-07 11:30:21 +05:30
Palash Debnath c36df0ddf0 Merge branch 'codex/fix-sidecar-timeout-reap' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 11:29:26 +05:30
Palash Debnath 85cb7242ea fix(sidecars): finish timeout cleanup before allowing recovery 2026-09-07 11:28:13 +05:30
Palash Debnath c547da351c Merge branch 'codex/consolidate-1809' into codex/pr-queue-integration 2026-09-07 11:22:48 +05:30
Palash Debnath 602be02a17 Merge branch 'codex/review-1862' into codex/pr-queue-integration 2026-09-07 11:22:48 +05:30
Palash Debnath f395c90aa1 Merge branch 'codex/review-dialog-resolution' into codex/pr-queue-integration 2026-09-07 11:22:48 +05:30
Palash Debnath 01cb810c65 test(bootstrap): release launch gate before timeout assertion 2026-09-07 11:22:18 +05:30
Palash Debnath f4e5d7b9ab Merge branch 'codex/review-1806' into codex/pr-queue-integration 2026-09-07 11:22:11 +05:30
Palash Debnath caedcde0c6 Merge branch 'codex/review-logs-state' into codex/pr-queue-integration 2026-09-07 11:22:11 +05:30
Palash Debnath 2a36a24612 Merge branch 'codex/consolidate-1831' into codex/pr-queue-integration 2026-09-07 11:22:11 +05:30
Palash Debnath 610303ec11 Merge branch 'codex/consolidate-1830' into codex/pr-queue-integration 2026-09-07 11:22:11 +05:30
Palash Debnath 3a114d62dd test(header): isolate visual fixture system polling 2026-09-07 11:21:55 +05:30
Palash Debnath 65e6c275c6 fix(workers): retain conservative budgets for legacy missing workers 2026-09-07 11:21:39 +05:30
Palash Debnath 9a90fe41be test(vite): exercise conditional dialog alias configuration 2026-09-07 11:21:32 +05:30
Palash Debnath d55838542e fix(confucius): preserve legacy torch device selection 2026-09-07 11:21:13 +05:30
Palash Debnath 9c0a5469cc fix(moss): retain older manually provisioned torch environments 2026-09-07 11:21:12 +05:30
Palash Debnath c7a8d24af6 fix(logs): report failed refresh after clearing logs 2026-09-07 11:19:51 +05:30
Palash Debnath 7d963cf297 Merge branch 'fix/release-rerun-asset-collisions' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 11:19:07 +05:30
Palash Debnath b9cb6c204f Merge branch 'codex/review-1865' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
#	frontend/src/components/Header.jsx
2026-09-07 11:18:50 +05:30
Palash Debnath 6157c1717b Merge branch 'codex/review-1863' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 11:18:28 +05:30
Palash Debnath c3b54393c4 docs: record release retry recovery fix (#1871) 2026-09-07 11:17:30 +05:30
Palash Debnath f0bb3fb8b2 Merge branch 'codex/consolidate-1810' into codex/pr-queue-integration 2026-09-07 11:17:24 +05:30
Palash Debnath 157f2987dc fix(api): retain Node ESM compatibility for abortable delay imports 2026-09-07 11:17:24 +05:30
Palash Debnath 426bd1ecea fix(desktop): preserve macOS window behavior with native controls 2026-09-07 11:17:17 +05:30
Palash Debnath d34490fd15 docs: explain retrying partially published releases 2026-09-07 11:16:29 +05:30
Palash Debnath 9b9d4d6579 fix(header): reserve traffic-light space only in native Mac windows 2026-09-07 11:16:24 +05:30
Palash Debnath eee35ab771 Merge branch 'codex/consolidate-1831' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 11:16:00 +05:30
Palash Debnath 0095c4a089 Merge branch 'codex/consolidate-1830' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 11:15:51 +05:30
Palash Debnath ef30dbece7 fix(release): clear target asset collisions before retry uploads 2026-09-07 11:15:40 +05:30
Palash Debnath a9062c2fa1 Merge branch 'codex/review-logs-state' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
#	frontend/src/test/LogsFooterNotifications.test.jsx
2026-09-07 11:15:38 +05:30
Palash Debnath b7ebd6e54d Merge branch 'codex/review-hf-onboarding' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 11:15:16 +05:30
Palash Debnath cd58bbded0 fix(moss): align accelerator detection and routing status 2026-09-07 11:13:33 +05:30
Palash Debnath b9183bb2a6 fix(engines): align accelerator metadata and DOTS runtime precision 2026-09-07 11:12:38 +05:30
Palash Debnath e9495beb8b fix(auth): inspect token presence locally until explicit validation 2026-09-07 11:12:26 +05:30
Palash Debnath 409dd016ce fix(logs): require successful current snapshots before all-clear 2026-09-07 11:11:46 +05:30
Palash Debnath db10e4b70f Merge branch 'codex/review-1862' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
#	frontend/src/test/visual/specs.jsx
2026-09-07 11:10:42 +05:30
Palash Debnath 3613b3d39b Merge branch 'codex/review-1861' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 11:10:11 +05:30
Palash Debnath 4277859a44 Merge branch 'codex/review-ui-1841' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 11:10:11 +05:30
Palash Debnath eefc45ec49 Merge branch 'codex/review-1821' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 11:10:11 +05:30
Palash Debnath 2f5a52f9eb Merge branch 'codex/review-1819' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 11:10:11 +05:30
Palash Debnath 729cad8912 Merge branch 'codex/review-1806' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 11:10:11 +05:30
Palash Debnath 86f9eda65f Merge branch 'codex/consolidate-1810' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 11:10:11 +05:30
Palash Debnath 32c96eeee4 Merge branch 'codex/consolidate-1809' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 11:10:11 +05:30
Palash Debnath f5d52e479e Merge branch 'codex/review-dialog-resolution' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 11:10:11 +05:30
Palash Debnath eafd94d0a6 Merge branch 'codex/review-oom-order' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 11:10:11 +05:30
Palash Debnath 2b5e3cdbc6 Merge branch 'codex/consolidate-1799' into codex/pr-queue-integration
# Conflicts:
#	CHANGELOG.md
2026-09-07 11:10:11 +05:30
Palash Debnath 2229a68cfc fix(gallery): calibrate preview guard against shipped speech fixtures 2026-09-07 11:07:16 +05:30
Palash Debnath 0581fb69fd Merge current main into PR #1819 2026-09-07 11:07:16 +05:30
Palash Debnath 85a368a08b test(header): verify reduced motion and responsive status visibility 2026-09-07 11:06:37 +05:30
Palash Debnath a8c57c816b test(onboarding): validate first-sound request against engine taxonomy 2026-09-07 11:05:30 +05:30
Palash Debnath 360d1d29c8 docs: record connection diagnostics fix under Unreleased 2026-09-07 11:04:51 +05:30
Palash Debnath 7366555c95 docs: record startup fix under Unreleased 2026-09-07 11:04:50 +05:30
Palash Debnath b8ff2721a5 fix(tts): normalize complete signed ranges without rewriting chains 2026-09-07 11:04:44 +05:30
Palash Debnath da26a240e2 Merge current main into PR #1821 2026-09-07 11:04:44 +05:30
Palash Debnath fb3d9b7139 fix(bootstrap): preserve retry cancellation across lifecycle acquisition 2026-09-07 11:03:42 +05:30
Palash Debnath 6c47ae7b21 fix(api): honor cancellation throughout transport diagnostic waits 2026-09-07 11:03:05 +05:30
Palash Debnath fda73384f0 fix(workers): retain granted deadlines across disconnects and restart 2026-09-07 11:02:45 +05:30
Palash Debnath 84add8b279 Merge current main into PR #1806 2026-09-07 11:02:27 +05:30
Palash Debnath 8b510d73db fix(ui): address workspace review findings and restore CI 2026-09-07 11:02:18 +05:30
Palash Debnath b1194c6223 test: cover nested and hoisted dialog dependency resolution 2026-09-07 11:02:06 +05:30
Palash Debnath 75f8924222 Merge remote-tracking branch 'origin/main' into codex/review-dialog-resolution
# Conflicts:
#	CHANGELOG.md
2026-09-07 11:00:31 +05:30
Palash Debnath 2a6d089f4a Merge remote-tracking branch 'origin/main' into codex/consolidate-1810 2026-09-07 11:00:17 +05:30
Palash Debnath 5d134f22ec test: enforce VRAM reclaim before clone prompt retry 2026-09-07 10:59:09 +05:30
Palash Debnath 08af0a971e Merge remote-tracking branch 'origin/main' into codex/review-oom-order 2026-09-07 10:58:32 +05:30
Palash Debnath ff0ce4d37d docs: keep pending transcription fix under Unreleased 2026-09-07 10:58:08 +05:30
Palash Debnath 1ab03067cf Merge remote-tracking branch 'origin/main' into codex/consolidate-1809 2026-09-07 10:58:01 +05:30
Palash Debnath 7a2f86066b fix(transcriptions): consolidate safe clipboard handling and regression tests 2026-09-07 10:57:08 +05:30
Palash Debnath d91beef0fd docs: credit Windows console help fix (#1815) 2026-09-07 10:56:19 +05:30
Palash Debnath 574b634688 Merge remote-tracking branch 'origin/main' into codex/review-cp1252 2026-09-07 10:55:45 +05:30
Palash Debnath 62ab62fc57 Merge remote-tracking branch 'origin/main' into codex/consolidate-1799 2026-09-07 10:55:05 +05:30
电车司机小李 ecbd152c43 fix(logs): avoid false all-clear state
Signed-off-by: 电车司机小李 <39351936+motodriver@users.noreply.github.com>
2026-09-07 11:51:02 +08:00
psiberfunkandClaude Sonnet 5 fec2e7b5f3 docs(changelog): credit the contributor for #1864
Greptile flagged the Unreleased entry as missing the contributor
credit the changelog convention requires for community PRs.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-09-06 21:02:33 -04:00
psiberfunkandClaude Sonnet 5 04bc9cedac fix(header): hide the custom window controls on macOS
macOS draws its own native traffic-light cluster even with
decorations:false (tauri.conf.json's titleBarStyle:"Overlay" still
overlays it), but Header.jsx's showWindowControls only checked
whether the app was running under Tauri, not which OS — so the
custom Windows-style minimize/maximize/close row rendered on macOS
too, duplicating the native controls.

Gate it on platform using the same navigator.platform check already
used in HotkeyTab.jsx / SettingsSearch.jsx. Windows/Linux keep the
custom row since decorations:false gives them no chrome otherwise.

Fixes #1864.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-09-06 20:51:38 -04:00
psiberfunkandClaude Opus 5 e397e10d64 test(header): restore navigator.platform after mac-inset tests
CodeRabbit flagged that setPlatform() redefined navigator.platform as
an own property but nothing ever restored it, so after this file's
tests run the global stays pinned to whichever platform ran last
('Linux x86_64') — order-dependent and able to leak into any later
test in the same environment that reads navigator.platform. Capture
the original descriptor (undefined, since it's an inherited jsdom
getter) and restore it — or delete the own-property override — in
afterEach.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-06 20:33:58 -04:00
psiberfunkandClaude Opus 5 8e1da3c22f fix(firstrun): send a voice-design-safe instruct instead of omitting it
Greptile flagged that omitting `instruct` from the first-sound request
fixes OmniVoice (whose `_resolve_instruct` rejected the old free-text
prose) but breaks a different engine: mlx-audio's Qwen3 VoiceDesign
backend requires a truthy `instruct` and raises ValueError without one
(`_is_voice_design()` in backend/services/tts_backend.py), so a user who
picked that engine during onboarding would still get silent first-sound
failure — same bug class, different engine.

Send 'middle-aged, low pitch' instead of omitting the field: it's the
exact taxonomy string the backend's own "Narrator" personality preset
uses (backend/core/personalities.py), so it's valid vocabulary for
OmniVoice's `_resolve_instruct` and a non-empty description for any
voice-design engine. Rewrote firstSoundInstruct.test.js, which
previously asserted instruct was absent entirely (passing for the wrong
reason); it now asserts a non-empty, taxonomy-only instruct is sent and
cross-checks its value against personalities.py's narrator preset so the
two can't silently drift apart.

Also credited the community contributor in CHANGELOG.md per the repo's
own convention (Greptile P2) and promoted the entry to Highlights,
matching every other credited entry in the file.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-06 20:32:25 -04:00
psiberfunkandClaude Opus 5 b92cc3bb79 fix(setup): keep the overwrite warning visible on narrow screens
CodeRabbit flagged that the hf_token_replace_warning text shared the
same `max-[560px]:hidden` class as the dismissable "add a token" pitch,
so a user replacing an already-active token on a narrow viewport (mobile
width, or a small first-run window) never saw the warning that doing so
clobbers the working token. Only the pitch should hide at that width —
the overwrite warning is safety copy and must always render. Added a
regression test asserting the warning's className never carries the
responsive-hide class.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-06 20:26:33 -04:00
psiberfunkandClaude Opus 5 d757f160fa fix(header): inset the breadcrumb clear of macOS traffic lights
tauri.conf.json sets decorations:false + titleBarStyle:"Overlay" on
every platform, so on macOS the native traffic-light cluster is drawn
on top of the web content instead of getting its own row; Windows and
Linux draw nothing there. The header's left block (status dot +
kicker) had no inset at all for that zone, so on macOS the traffic
lights sat on top of it.

Fix, macOS-only: detect macOS the same way HotkeyTab.jsx /
SettingsSearch.jsx already do (navigator.platform), and apply a new
.header-area__left--mac-inset class that completes header-area's own
16px left padding to the same flat 64px-from-window-edge total that
.header-area--tabs already reserves for the identical cluster. Windows
and Linux get no inset, so no space is wasted where nothing is
overlaid.

Fixes #1860.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-06 19:53:18 -04:00
psiberfunkandClaude Opus 5 a9ac8b1b46 fix(header): stop the status dot pulsing under Reduce Motion
Header.jsx's status dot animates via the hqPulse keyframes, applied as
a Tailwind arbitrary [animation:...] utility. None of index.css's
twelve @media (prefers-reduced-motion: reduce) blocks named hqPulse
(it isn't a stable CSS class, so those selector-based blocks can't
reach it), so the purely decorative pulse kept running with OS Reduce
Motion on.

Fix: append motion-reduce:[animation:none] to the dot's className -
the same mechanism LogsFooter.jsx already uses for its own
arbitrary-utility pulses (heart-glow, donate-pop-in).

Part B only, per the issue split - Part A, an in-app motion toggle, is
a product decision and stays open.

Fixes #1857 (Part B).

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-06 19:49:32 -04:00
psiberfunkandClaude Opus 5 f88bfecabc fix(firstrun): stop sending free-text instruct prose on first sound
App.jsx's post-onboarding "first sound" request appended a hardcoded
narrator prose string as `instruct`. Every engine's instruct is a
controlled vocabulary (OmniVoice's `_resolve_instruct` rejects
anything outside a fixed token list), so this 400ed on every first
run — silently, since the surrounding catch is deliberately silent
by design (a first impression must never surface an error).

Omit `instruct` entirely instead of swapping in valid vocabulary:
it matches every other call site in the app (`if (instruct)
fd.append('instruct', ...)`), matches the seeded demo profile's
empty stored instruct, and every engine backend already treats a
missing/empty instruct as "no styling" rather than a required field
— so this can't regress no matter which TTS engine is active.

Fixes #1853.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-06 19:46:15 -04:00
psiberfunkandClaude Opus 5 753891adc9 fix(setup): stop pitching an HF token when one is already active
HfTokenCard.jsx unconditionally rendered the "add a free Hugging Face
token" pitch in first-run's Models & engines step, even when the
backend had already resolved and validated one (app/env/hf-cli). Since
Save persists via huggingface_hub.login(), which overwrites the
canonical $HF_HOME/token file outright, complying with the unnecessary
prompt could silently clobber an already-working token.

The card now checks GET /system/hf-token/state (the same resolver the
Settings -> API Keys panel already consumes) before rendering:
- an active, validated token shows the source + masked value instead
  of the pitch
- replacing it requires an explicit "Replace..." click plus an inline
  overwrite warning, rather than one blind paste-and-Save
- a still-loading check shows a neutral placeholder
- a failed check falls back to the pre-fix pitch rather than hiding
  the card

Fixes #1851.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-06 16:54:26 -04:00
Palash Debnath 4efde9ce4e feat(ui): polish dubbing workspace layout and controls 2026-09-06 20:26:43 +05:30
li-lizhe ae2fa83b02 fix: handle None from current_accelerator(); exclude MPS in dots_tts
current_accelerator() returns None on CPU-only builds (no accelerator
compiled in), so .type would crash. Use check_available=True and fall back
to 'cpu' when None. For dots_tts, also select fp32 on MPS since
DotsTtsRuntime is untested on MPS.

Addresses greptile P1 + coderabbit Functional Correctness review comments.
2026-09-06 21:06:33 +08:00
li-lizhe a671004cb7 fix: handle None from current_accelerator() on CPU-only builds
current_accelerator() returns None on CPU-only PyTorch builds (no
accelerator compiled in), so accel.type would crash. Use check_available=True
and fall back to 'cpu' when None.

Addresses greptile P1 + coderabbit Stability review comments.
2026-09-06 21:06:19 +08:00
li-lizhe ed6fca46a9 fix(tts-engines): select device via torch.accelerator in confucius4 and dots_tts
Both Confucius4 and DOTS-TTS engine sidecars hardcode device selection to
`torch.cuda.is_available()`, which returns False on Ascend NPU, Intel XPU,
and other non-CUDA accelerators — causing the models to silently run on CPU
(in fp32) instead of the available accelerator.

Replace with `torch.accelerator.current_accelerator().type`, the
device-agnostic API that auto-detects CUDA, NPU, XPU, MPS, and CPU. MPS is
excluded for Confucius4 (upstream untested on Apple Silicon). dtype stays
bf16 for any GPU-class accelerator and fp32 on CPU.

Also verified on Ascend 910B (torch 2.14, torch_npu, 4 NPU):
  before: cuda_available=False → device "cpu", precision "float32"
  after:  accelerator → confucius4 device="npu", dots_tts precision="bfloat16"
2026-09-06 09:24:39 +08:00
li-lizhe 65d37288ce fix(moss_tts_v15): select device via torch.accelerator instead of CUDA hardcode
MOSS-TTS-v1.5 engine hardcoded device selection to
`device = "cuda" if torch.cuda.is_available() else "cpu"`. On an Ascend
NPU (torch_npu) host, `torch.cuda.is_available()` is False, so the whole
model silently runs on CPU in fp32 — never using the accelerator — even
though torch.accelerator reports `npu` and bf16 is supported.

Replace with the device-agnostic `torch.accelerator.current_accelerator()`
so any backend (CUDA / NPU / XPU / MPS) is picked up automatically. MPS is
still excluded (MOSS's upstream trust_remote_code modelling code is untested
on Apple Silicon); dtype is bf16 for any GPU-class accelerator and fp32 on
CPU.

Verified on Ascend 910B (torch 2.14, torch_npu, 4 NPU):
  before: torch.cuda.is_available()==False -> device "cpu", dtype float32
  after:  accelerator -> device "npu", dtype bfloat16
2026-09-06 09:23:22 +08:00
Palash Debnath 293b0812f4 feat(ui): organize dubbing controls and export drawer 2026-09-05 23:17:50 +05:30
Palash Debnath b441a486cc feat(ui): refine voice controls and expandable navigation 2026-09-05 22:37:02 +05:30
Palash Debnath d4bf1fe9a9 feat(ui): polish voice interactions and fix notification badge 2026-09-05 20:43:42 +05:30
Palash Debnath 5574cbbc16 fix(ui): name OmniVoice correctly and cycle active engine labels 2026-09-05 20:06:24 +05:30
Palash Debnath 13eae6ff02 fix(ui): show selected engine on title bar 2026-09-05 19:44:55 +05:30
Palash Debnath fd5e78cdab refactor(ui): simplify engine quick access with family tabs 2026-09-05 19:38:19 +05:30
Palash Debnath 67c316e81b feat(studio): consolidate engine controls and pin mode actions 2026-09-05 18:52:43 +05:30
Palash Debnath 81fb585557 fix(studio): anchor engine menu to header trigger 2026-09-05 17:51:57 +05:30
Palash Debnath 3675d750d6 fix(studio): reuse rich language picker for cloning 2026-09-05 17:46:08 +05:30
Palash Debnath bbbf185cf3 fix(studio): keep sticky language picker inside viewport 2026-09-05 17:11:24 +05:30
Palash Debnath 0c4ed0546f fix(studio): constrain sticky controls on short screens 2026-09-05 16:38:43 +05:30
Palash Debnath 17df9210cd fix(studio): pin synthesis controls below scrolling form 2026-09-05 16:26:13 +05:30
Palash Debnath aedca15f7e docs: record voice workspace tabs 2026-09-05 15:23:59 +05:30
Palash Debnath 7a14c31a8e feat(studio): promote voice modes to workspace tabs 2026-09-05 15:22:44 +05:30
Palash Debnath 53ff367c1f fix(studio): simplify voice cloning setup (#1817)
* fix(studio): simplify voice cloning setup

* docs: link cloning redesign changelog

* fix(studio): keep clone recording controls available

* fix(studio): lock clone capture transitions
2026-09-05 14:12:43 +05:30
flutterkage2kandClaude Opus 5 ed746bab57 fix(tts): speak the tilde in digit ranges instead of mashing the numbers
"20~30초" is read aloud as a single number — OmniVoice says "이십삼" (23).
The separator never reaches the listener, so any written range is heard as
the wrong figure.

`normalize_text` only ran its number pass behind `_num2words_lang`, which
returns None for ko/ja/zh/th/vi (those scripts read digits natively and are
deliberately outside num2words). Nothing else looked at the range mark, so
the tilde went to the engine untouched and the two numbers ran together.

Rewrite `N~M` into the spoken form before the engine sees it, outside the
num2words gate so the CJK languages are covered too. Verified by rendering
each candidate and transcribing it back (ko, OmniVoice, cloned voice):

    "대략 20~30초짜리"      heard "23초"           WRONG
    "대략 20-30초짜리"      heard "23초"           WRONG (reproduces it)
    "대략 20에서 30초짜리"   heard "20에서 30초짜리"  correct
    "20〜30分ぐらい" → "20から30分" heard "20〜30分くらい"  correct

Deliberately narrow:

* Only the tilde family (U+007E, U+301C, U+FF5E). Japanese and Korean IMEs
  emit the latter two. An ASCII hyphen is left alone — between digits it
  also spells dates, phone numbers and product codes, where "to" is wrong
  (`tests` already pin "pages 3-5" as unchanged).
* Only languages with a verified spoken form (ko/ja/zh/en). Anything else
  keeps its tilde, matching how `_PERCENT_WORD` is scoped.
* Spacing belongs to the form, not the caller: a Korean postposition binds
  to its numeral ("20에서 30"), Japanese and Chinese set no spaces, English
  needs them on both sides.
* Neighbour guards block digits and ASCII letters but allow CJK, because
  CJK writes the unit hard against the digits ("20~30초"); a `\w` guard
  rejects exactly the cases the rule exists for.

`ko`/`ja`/`zh` join `_FULL_NAME_TO_CODE` so the new resolver can see them.
They stay out of `_NUM2WORDS_LANGS`, so this does not open a num2words path
for them — the same inert-entry pattern the file already documents for
"vietnamese".

`backend/services/text_normalization.py` joins the functional-CJK allowlist
in tests/test_no_hardcoded_cjk.py, under the text-processing group and by
the procedure that file documents: the range words are engine input, not
user-facing UI strings.

Tests: 7 new change-cases and 8 new leave-unchanged cases (hyphen, date,
phone number, product code, decimals, a non-numeric tilde, an unverified
language, and no language at all). All 7 change-cases fail against the
previous implementation.

Full suites before and after: the same 17 failures, none of them touched by
this change.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-05 17:06:15 +09:00
flutterkage2kandClaude Opus 5 847ca6d44c fix(desktop): resolve plugin-dialog wherever the package manager put it
`bun run desktop` dies before the window opens on a fresh clone:

    Error: ENOENT: no such file or directory, open
    '.../frontend/node_modules/@tauri-apps/plugin-dialog/dist-js/index.js'

The alias hardcoded `frontend/node_modules/...`, but this is a bun
workspace: bun hoists the package to the workspace root and leaves
`frontend/node_modules` empty, so the path the alias names does not
exist. Vite's dep optimizer reads it directly and throws, taking
`beforeDevCommand` — and the whole desktop shell — down with it.

Probe both layouts and fall through to Vite's own resolution when
neither is present, so a missing package degrades to normal resolution
instead of crashing the dev server.

Verified on macOS 26.6 (Apple Silicon), bun 1.2.22, fresh clone: the
window now opens and the backend serves.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-05 17:06:12 +09:00
flutterkage2kandClaude Opus 5 784494dc3c fix(gallery): measure spectral flatness per frame so real speech passes
Most gallery previews fail with "the voice engine returned no audible
audio for this archetype". The renders are fine — the guard is not.

Two problems, both in the degenerate-buzz check:

1. `_spectral_flatness` took ONE FFT of the whole clip. Spectral
   flatness is defined over short frames; a full-length transform gets
   finer frequency resolution the longer the clip is, so voiced
   harmonics carve deeper and deeper nulls and the geometric mean
   collapses. The number tracked clip length, not timbre.

2. `_DEGENERATE_FLATNESS = 0.015` was calibrated against
   `_speech_like()` in the unit test — a synthetic harmonics+noise
   stand-in that is far flatter than real speech. Real renders measure
   well below it, so the threshold sat inside the speech range.

Measured on this engine's own output (framed, per this patch):

    pure tone 80 Hz        2.6e-10    two-tone buzz    3.3e-09
    quietest real speech   2.0e-04    (VoxCPM2 ko)

Frame the measurement (1024/512, skipping inter-word frames at the
noise floor) and move the threshold to 1e-5 — ~3000x above the tonal
cases, ~20x below the quietest real render.

Before: 6 of 8 renders rejected; ml_japanese_explainer,
ml_japanese_companion and feat_23_the_explainer all 503 through
GET /archetypes/{id}/preview.
After: 0 false positives across 27 real clips (Japanese, Korean and
English archetypes, cloned voices, human reference recordings), and
those three previews return 200. Every accepted clip was confirmed as
real speech by transcribing it with the app's own ASR.

Not addressed: a render that collapses toward NOISE rather than a tone
still passes (one observed at flatness 0.073, ASR returns a
hallucination). The old threshold missed it too, so this is not a
regression — calibrating an upper bound needs more than one sample.

Tests: frame-based measurement must be clip-length invariant, and the
threshold must sit between the measured tonal ceiling and the measured
real-speech floor. Both fail against the previous implementation.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-05 17:05:59 +09:00
dajiaohuang 6c42b50fe3 test(ci): document cp1252 help regression 2026-09-05 05:21:45 +08:00
dajiaohuang 2594abebf8 fix(ci): keep install-docs help cp1252-safe 2026-09-05 05:17:34 +08:00
Palash DebnathandClaude Opus 5 e4f00bc564 fix(desktop): let Retry preempt the readiness wait
Greptile's P1 on #1809, and it is right. `launch_backend_and_wait` holds
`BackendState::lifecycle` around the entire launch, including the readiness
wait — which this branch just made unbounded for as long as the backend
answers `/startup/progress`. Retry, Clean & Retry, reset and uninstall all
need that same lock, so on a slow start the user's own escape hatch would
block behind the wait instead of interrupting it: an app with no way out,
which is worse than the early kill the branch set out to remove.

Every flow that is about to take lifecycle ownership now bumps a generation
counter first, before reaching for the lock. The waiting loop snapshots that
counter once its caller holds ownership — so a bump that predates it is not
mistaken for a preemption — and stands down within one 500 ms poll when it
changes, releasing the lock for whoever asked.

That also settles what happens at the splash's six-minute stall budget: it
flips to failed and offers Retry and the logs, and Retry now actually works,
while its /health recovery poll still walks straight into the app if the slow
start finishes first. Either way the user gets out.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HR6J9zKQop9TGGVUwjypnF
2026-09-04 18:15:30 +05:30
Palash DebnathandClaude Opus 5 6a2ada2f11 fix(tts): don't answer a GPU OOM by repeating the same allocation
`_get_clone_prompt` catches everything and returns None so synthesis falls
back to `generate()`'s inline reference path. For a device OOM that is not a
fallback at all: the inline path runs the SAME encode on the SAME device —
producing identical output is the entire point of the precompute — so it is
guaranteed to hit the same wall moments later, on a GPU with even less
headroom than the first attempt found. Two reporters' backends died with a
Windows access violation (exit code -1073741819) seconds after this fallback
logged, mid-generation, on a card that had just refused an 86 MiB
allocation.

An OOM here is also the most recoverable kind. The allocator is typically
sitting on reserved-but-unallocated blocks — #1790's own log reports 90 MiB
reserved against that 86 MiB request — so drop them and try once more. If it
still will not fit, raise: the failure layer turns a device OOM into "close
other GPU-heavy apps or unload models, then retry", which is a far better
answer than walking into a native fault.

Every other failure still falls back silently, since for a non-memory fault
the inline path may genuinely succeed.

Fixes #1790.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HR6J9zKQop9TGGVUwjypnF
2026-09-04 18:11:20 +05:30
Palash DebnathandClaude Opus 5 9ac1045917 test: reject a backslash in a tracked path too
CodeRabbit's catch on #1799: git stores paths with `/` separators, so a `\`
that survives into a path component is part of a NAME. It is legal to commit
one from Linux or macOS and impossible to check out on Windows, where git
refuses it under `core.protectNTFS` — the same checkout-time failure, before
any test runs, that the stray `:memory:.ses` caused.

The rule moves into a pure `windows_hostile_reason` so it can be exercised
directly: the repo cannot carry a fixture for each hostile shape without
becoming the very thing the test rejects. Both directions are pinned — every
shape Windows refuses, and ordinary paths that merely resemble one (a file
called `console.md`, `com10.py`, a component containing but not ending in a
dot).

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HR6J9zKQop9TGGVUwjypnF
2026-09-04 18:04:38 +05:30
Palash DebnathandClaude Opus 5 31d90def83 fix(errors): don't claim the backend crashed with no evidence that it did
Two Apple Silicon reporters were told "it most likely crashed or was killed
mid-request" while generating. Neither bug report carried a crash marker,
because none had been recorded — the app had no evidence for the one thing
it asserted, and the advice that follows that sentence is Retry and Clean &
Retry, which rebuilds the whole Python environment to fix a backend that had
not died.

Two causes, both fixed here.

The desktop shell learns the backend died from a ~2 s poll: it has to notice
the child exit before it can write the marker. `apiFetch` asked for that
marker exactly once, at the instant the transport gave up, so it raced the
poll and lost either way round — a backend that really died was reported
with the vague sentence instead of its exit code and crash notice, and one
that never died was reported as dead anyway. `streamDropError` already waits
that poll out (#1119); the request path never did. The loop is now a shared
`awaitBackendCrashMarker`, used by both, with a shorter budget here because
the transport cascade has already cost the user a few seconds.

And the copy itself overshot what it could know. By construction it is
reached only once a crash has been looked for and not found, so it no longer
names one: it says the backend stopped answering with no crash recorded, and
that a heavy job holding the engine is the likelier story — which on a
memory-pressured Mac mid-generation it is. Updated in all 21 locales, since
a translation still asserting a crash would be the same bug in another
language.

The #1337 test that required the crash wording is updated with it: #1337
established that the backend had answered seconds earlier, not what silenced
it, and requiring the stronger claim is what pinned this in place.

Fixes #1802.
Fixes #1805.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HR6J9zKQop9TGGVUwjypnF
2026-09-04 18:03:23 +05:30
Palash DebnathandClaude Opus 5 d5c33eda27 fix(desktop): don't kill a backend that is still starting
The launcher waited a flat five minutes from spawn for the backend to
report ready, then killed it and tried again. On a host where the cold
start genuinely takes longer — the reporter's project lived on a mapped
network drive, and `import torch` off one is slow the first time, as is a
first CUDA load or a cold spinning disk — that deadline expired *while the
backend was still importing*. The respawn threw away the warm page cache
and raced the same clock, so the app could never start, and it blamed the
backend: "the backend never reported ready". Launching that same backend by
hand reached ready in well under a minute once the cache was warm.

A backend answering `/startup/progress` with `status: "starting"` is not
one we have to guess about: it bound its socket, it is serving HTTP, and it
is naming the step it is on. Killing it cannot make the retry faster, and
the launcher knows nothing the user doesn't. So keep waiting while it
answers, and keep narrating each step. The budget still governs silence —
nothing answering, or a self-reported `failed` — where a slow backend and a
wedged one really are indistinguishable and the existing stderr-tail
failure is the right answer.

The splash needed the same correction. Its stall watchdog keys on
`bootstrap_status`, which sits on `starting_backend` for the whole of a slow
start, so it would have called the launch stuck at six minutes anyway; the
proof of life arrives on the separate `bootstrap-log` stream. Output now
counts as activity, and a genuinely silent backend still trips the watchdog
so the info-less spinner of #879 stays fixed.

Fixes #1791.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HR6J9zKQop9TGGVUwjypnF
2026-09-04 17:50:23 +05:30
Palash DebnathandClaude Opus 5 36a38bdbb9 fix(ci): don't ship a tracked path Windows cannot check out
A stray sqlite session artifact named `:memory:.ses` was committed by
accident on this branch. Git on Windows rejects a path containing `:` with
`error: invalid path` and exits 128 during **checkout** — so both Windows
jobs went red before a single build or test step ran, pointing at a file
nobody had edited, while Linux and macOS stayed green.

Drop the file, ignore the `*.ses` artifact class, and add a guard that scans
the index on every platform for paths Windows cannot represent: illegal
characters, components ending in a space or dot, and reserved DOS device
names. The failure now surfaces as a named test on every runner instead of
as a checkout crash on one leg of the matrix.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HR6J9zKQop9TGGVUwjypnF
2026-09-04 17:41:02 +05:30
VishvakR 02c3a651ac test(generate): drive the scheduler itself, and restore env before reloading
Third review round on the PR, both findings in the new test file.

CodeRabbit: the disconnect regression called deadlines.for_task directly, so it
would have passed even if Scheduler._budget_for stopped coercing a missing
worker to the CPU budget -- the very thing it exists to pin. It now builds a
real WorkerPool and Scheduler, assigns the task to the 4 GB worker, asserts the
bound budget is the CPU one, disconnects the worker and asserts the
recomputation is not shorter. Forcing under_provisioned=False in _budget_for
fails it with `assert 300 == 600`.

CodeRabbit: the two env-var tests deleted the variables and reloaded
model_manager inside a finally, which runs BEFORE pytest restores them -- so on
a machine that already exports either var, the module constants would describe
an environment pytest was about to put back, and every later test would read
the mismatch. Both use monkeypatch.context() now, so the environment is restored
before the reload.
2026-09-04 15:36:18 +05:30
VishvakR fa148ecb55 fix(generate): tighten the VRAM-floor tests and document budget precedence
Second review round on the PR.

CodeRabbit: the repo-wide dispatch assertion accepted any nested min_vram_gb
keyword, so a budget computed with 0 or another engine's floor would pass while
the guard used the right one. It now compares the two expressions.

CodeRabbit: the awaiting-side deadline test restated gpu_gateway's formula
instead of calling it, so it would not have noticed that function starting to
select a shorter ceiling. It calls _default_deadline now, on cuda and rocm.

CodeRabbit: the docs said an explicit OMNIVOICE_GENERATE_TIMEOUT_S is honoured
"everywhere" while also saying the CPU var governs under-provisioned cards --
the two cannot both be true. Verified against the code (both vars set, 4 GB
cuda, engine floor 6 GB -> 200s, the accelerated value) and documented as a
precedence table rather than prose. The accelerated var deliberately wins on an
under-provisioned host: that is what keeps "lower it to fail fast everywhere"
working. Pinned by a test so the table cannot drift from the behaviour.

CodeRabbit also flagged that Scheduler._budget_for recomputes with no worker
after a disconnect, dropping under_provisioned to False. That cannot shorten
anything: no worker means no execution_device, which _base_execution_seconds
already coerces to "cpu" -- the same budget the floor raises an
under-provisioned card to. Added a test pinning that rather than persisting a
dispatch-time budget on the attempt. The residual case it describes -- an
operator who raised the accelerated budget ABOVE the CPU one sees a shorter
recomputation once the worker is gone -- predates this change and applies to
every GPU worker, not just under-provisioned ones, so it belongs in its own fix.
2026-09-04 13:52:23 +05:30
VishvakR f172d0c3be fix(generate): apply the VRAM-floor budget to remote workers and /convert
Review findings on the PR, fixed here rather than left for a fourth report.

Greptile (P1): the control plane sets a remote attempt's deadline, so the same
inversion reached remote workers. Its suggested fix -- thread the engine floor
into generate_timeout_s() -- would read the wrong machine: that function probes
THIS host, so a Mac control plane dispatching to a 4 GB Windows worker learns
nothing (MPS is excluded by design), and a 4 GB box dispatching to a 24 GB
worker would wrongly get the longer budget. The worker already advertises both
figures it takes -- free_memory_bytes and min_memory_bytes, both set in
worker/capabilities.py -- so ConnectedWorker.under_provisioned() decides from
those, and deadlines.for_task() floors the execution budget at what the same job
would get on a CPU. The task-level ceiling in gpu_gateway._default_deadline is
computed before a worker is bound and already asks for the CPU budget, so it
still covers the raised lease; a test pins that.

CodeRabbit (major): /convert had the identical split -- min_vram_gb to the
guard so a timeout could name the card, and a budget computed without it.

CodeRabbit (minor): the docs promised the CPU-class floor for any GPU, while
the code scopes it to dedicated-VRAM families. Reworded to say CUDA/ROCm and to
say why MPS is excluded.

CodeRabbit (minor): the call-site assertion compared global occurrence counts,
so one dispatch could drop both arguments while another gained an extra and the
total still matched. It now walks the AST and checks each dispatch on its own,
and the pairing is additionally enforced repo-wide across backend/api/routers:
a dispatch that knows the engine's floor well enough to explain a timeout must
know it well enough to set the budget.

Three inline capability-selection loops in ConnectedWorker collapse into one
_capability_for(), so the new predicate cannot select a different capability
than execution_device() does.
2026-09-04 13:33:00 +05:30
VishvakR fcac8e1bae fix(generate): budget an under-provisioned GPU like the CPU it performs like
A GPU with less VRAM than the engine declares it needs pages to system RAM
over PCIe, so it renders slower than the same machine's CPU. The compute-time
budget picked its value from the device family alone, so that card was treated
as fast hardware and given 300s -- half the 600s a plain CPU host gets. It is
the slowest configuration the app supports and it had the shortest watchdog.

Everything else already acted on the verdict. resolve_routing() raises the
caveat, the synth preflight warns before the user waits, and _timeout_guidance()
names the card in the failure. Each TTS generate dispatch even hands the guard
the engine's floor on the line above the timeout that ignored it. #1226 and
#1222 were the same 4 GB cards on the same engine; both were closed by making
the app explain the timeout better, never by correcting the budget behind it.

generate_timeout_s() now floors an under-provisioned accelerator at the CPU
budget. The length scaling is unchanged, and an explicitly configured
OMNIVOICE_GENERATE_TIMEOUT_S is still honoured verbatim, so an operator who
lowered the watchdog to fail fast keeps that. The floor is a max(), never an
assignment, so a raised accelerated budget is never cut down. Engines that
declare no floor, a failed VRAM probe, and MPS (whose vram_gb is a unified-
memory heuristic, not a dedicated pool) are all untouched.

The three-clause "is this host under-provisioned" test was written out inline
in the caveat and in the timeout message, which is how the budget came to
disagree with the warning printed beside it; it is now one predicate,
under_provisioned_vram(), that all three read.

Reported on a GTX 1650 (4 GB) running the omnivoice engine, whose breadcrumbs
show the budget ending the job on the dot: 372s and 301s are exactly
300 + max(0, len - 1200) / 40 for the two takes.

Fixes #1804.
2026-09-04 13:02:14 +05:30
Palash DebnathandClaude Opus 5 4e6c36848c fix(transcriptions): render segments that have no timings
An OpenAI-compatible ASR answering in json/text format returns no
timestamps, and services/asr_backend.py records that honestly as
`end: None` rather than inventing a number. The segment list called
`.toFixed()` on it unconditionally, so the render threw and the whole
Transcriptions view went blank — a transcript that merely lacked timings
became one the user could not read at all.

Show whichever bound is known and nothing when neither is, so the text
stays readable either way. Non-finite values are treated as unknown too,
so a bad timing prints nothing rather than NaN.

Fixes #1798.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HR6J9zKQop9TGGVUwjypnF
2026-09-04 03:26:51 +05:30
Palash Debnath f2302e8c95 fix(desktop): don't adopt a backend running stale code (#1796)
Exports failed with a 422 naming a field the current app never sends — twice, from different users. The cause was the attach handshake: if something already answers on the backend port and reports a matching version, the app adopts it and skips the source sync a normal launch performs. A version string holds steady for a whole release cycle, so a same-version process can still be running weeks-old code, and that code then serves a current UI.

The handshake now compares a fingerprint of the shipped Python sources, read from the same response as the version so a dropped probe can't masquerade as a missing field. A backend predating the mechanism is treated as stale; one that is current but started outside the app is still accepted. Refusals are logged with a greppable marker, since this class previously took two reports and a code audit to identify.

Fixes #1770. Closes the duplicate report tracked in #1792.
2026-09-04 03:22:41 +05:30
Palash Debnath 8b4e4ebf56 fix(windows): start the backend when the install path has non-English characters (#1795)
On Windows with a non-UTF-8 system code page, a venv path containing non-English characters (commonly a CJK username) killed the interpreter during startup, before any VoiceStudio code ran — Python reads .pth files in the active code page, and uv's editable install writes the project path there in UTF-8. The backend could never start, and the setup screen only said it had stalled.

A new or genuinely broken environment now builds at an ASCII-safe short path; an existing environment that starts cleanly is never relocated. When no safe path can be produced, the app says so before downloading rather than after. The failure message names the cause and a remedy that works for the install mode in use, since portable installs ignore the setting managed installs use.

Fixes #1783.
2026-09-04 03:21:42 +05:30
Palash Debnath 7f7a4c5f83 fix(settings): make the generation budget reachable and honest (#1797)
The compute-time error told users to raise a generation timeout that had no control anywhere in the app — the only knob was an environment variable, and on Windows the docs explicitly warn against the usual way of setting one. Both budgets are now editable in Settings under Performance & Device, persisted and applied on the next start.

Two defects found in review and fixed here rather than shipped: an explicit universal budget silently overrode a separately saved CPU budget, so the CPU row would have looked like it worked and done nothing; and a value already set in the environment shadowed the saved preference while the panel still reported success. A shadowed row now says so instead. Long-input warnings also fire on Apple Silicon, which gets the accelerated budget and was the device in one of the duplicate reports.

Fixes #1787. Closes the reports tracked in #1774 and #1778.
2026-09-04 01:55:06 +05:30
Palash Debnath e5916acc01 feat(design): simplify the Voice Design panel (#1793)
The panel printed every chosen value three times and held twelve control rows open before anyone touched it. Details now collapse to a single recipe line that expands on request, gender/age/pitch/style become selects, and English accent and Chinese dialect merge into one grouped field so the combination the engine rejects cannot be selected at all. Starting points show five with an overflow, the bottom-bar slider is labelled Steps, and the panel ends where its content ends.

Also fixes two races found in review: a manual pick or a cleared description now beats an in-flight describe response, and arrow-key chip navigation moves focus without resetting the design.
2026-09-04 00:20:36 +05:30
Palash Debnath 06e69d6d6b fix(desktop): resolve the capability-store dir from the running backend (#1789)
Exports and every other native-picker action 403'd with "Invalid or expired desktop authorization" whenever Tauri and the backend resolved different data directories — a dev backend spawned without the OMNIVOICE_* env, a custom data folder, or portable mode. Tauri now takes the directory the running backend advertises, so the two processes cannot disagree, falling back to its own resolution when the backend is unreachable. One resolver covers all six capability kinds. Also decodes JSON escapes when reading that field, which Windows paths depend on, requires an absolute path so a relative data dir cannot recreate the same split, and keeps the 403 and its log line free of filesystem paths. Fixes #1781.
2026-09-03 22:25:08 +05:30
Palash Debnath e4ce065864 fix(design): enforce dialect/accent exclusivity in the voice-design picker (#1788)
Voice Design rendered EnglishAccent and ChineseDialect as two unlinked controls, so a user could select both and only learn they conflict from a 400 after a round trip. A shared exclusive-groups map now mirrors the engine's rule across every path that builds or restores instruct state: the live picker, free-text entry, saved-profile and imported-session restore, plus a message-matching backstop for a conflict arriving by any other route. Picking one clears the other with a visible reason instead of a silent reset. Fixes #1771.
2026-09-03 21:26:07 +05:30
Jaesik Lee f95e73b710 fix(i18n): ja "Cleaning…" is denoising, not housekeeping (#1775)
The Japanese clone.cleaning status read 掃除中 (tidying up a room) instead of ノイズ除去中, which is what the step actually does: denoising the reference audio. Thanks @j30231!
2026-09-03 20:39:42 +05:30
Jaesik Lee b20cd6cb95 fix(i18n): overhaul the ko locale (#1776)
Corrects 231 machine-translation defects in the Korean locale and translates all 493 previously missing keys, dropping ko to zero in the missing-key baseline. Includes terminology consistency (Cinematic, export, UI scale) and fixes copy that named the wrong control. Maintainer commits merged current main and folded in the five live-preview keys added by #1769. Thanks @j30231!
2026-09-03 20:19:42 +05:30
Matt Van Horn 1515b46adb feat(batch): watch-folder auto-ingest (#1768)
Opt-in watch folder on the batch queue: pick a directory once and new videos are auto-enqueued with the last Add-to-queue settings, with pause/stop controls and copy-in-progress protection. Files stream to the loopback backend as bytes; paths never leave the app. Also gives the batch queue a reachable UI entry point and streams multipart uploads to disk. Maintainer fix: the watched directory handle is opened with full share mode on Windows so users can rename or delete the folder while it is watched, matching macOS/Linux behaviour, with a cross-platform regression test. Thanks @mvanhorn!
2026-09-03 18:47:32 +05:30
Matt Van Horn 999345de41 feat(dub): realtime dub preview (#1769)
Opt-in live preview for dub segments: edits debounce into a streamed /ws/tts synthesis played through the chunk player, with cancellation preserved through buffered playback. Maintainer fixes: /ws/tts added to the backend ticket allowlist (feature was dead off-loopback), handshake failures surface a toast, loopback-only plaintext refusal reverted to keep the documented remote-GPU setup working, PCM16 decode hardened. Thanks @mvanhorn!
2026-09-03 18:27:07 +05:30
Palash Debnath ac287c612f docs(readme): enrich with audio samples, hardware guide, and doc links (#1785)
* docs: polish README hero hierarchy

* docs: enrich README with audio samples, hardware guide, and doc links

* docs: address review findings on Docker port binding, MCP transport, and privacy

* docs: address CodeRabbit review feedback on cURL format and Colab links

* docs: align Docker quick-start with stable tag and named volume mount
2026-09-03 18:00:14 +05:30
Palash Debnath d80286562e docs: polish README hero hierarchy (#1780) 2026-09-03 14:28:42 +05:30
9d30774ee2 feat(audiobook): synced lyrics playback (#1766)
* feat(audiobook): synced-lyrics player

Replace the bare <audio controls> in the audiobook result with a player
that renders the chapter text and highlights the word under
audio.currentTime, karaoke-style. Timing reuses what the render stream
already emits — per-chapter duration_s on the chapter SSE events — with
words even-split inside each chapter (the karaoke burn-in's old-job
fallback, ported from services/karaoke_ass.py); after a reload the whole
book even-splits over the file's own duration. No ASR pass, no new
backend surface, nothing leaves the machine. Download keeps going
through the Tauri-safe downloadMedia util.

New pure helper utils/audiobookLyrics.js mirrors the longform parser's
chapter drop rules so cue indices line up with the stream's chapter
list, and degrades to the proportional split on any drift (script
edited after the render, stopped mid-book). Words are buttons —
click-to-seek, keyboard reachable — restyled as prose in index.css.
audiobook.lyrics translated in all 21 locales.

Co-authored-by: Matt Van Horn <mvanhorn@users.noreply.github.com>

* docs(changelog): synced-lyrics audiobook player entry (#1766)

Co-authored-by: Matt Van Horn <mvanhorn@users.noreply.github.com>

* style(audiobook): apply current formatter

* fix(audiobook): preserve synced render cues

---------

Co-authored-by: Cursor Agent <cursoragent@cursor.com>
Co-authored-by: Palash Debnath <4178343+debpalash@users.noreply.github.com>
2026-09-02 16:32:23 +05:30
65236774aa feat(dub): project-level drag casting board (#1767)
* feat(dub): project-level drag casting board

Adds an expandable Casting Board to the dub editor's CAST strip: speaker
rows (with the auto-clone chip when the extractor found a usable passage)
and draggable voice chips — Default, clone profiles, design presets.
Dropping a chip on a speaker writes the exact fields the CAST <select>
always has (profile_id + merge_parts/merge_parts_original attribution),
via a shared assignSpeakerProfile helper both views now call, so the
dropdowns stay in sync and job persistence is unchanged. Keyboard path:
focus a speaker row, pick from a listbox (arrows/Enter/Escape).

The pre-existing CAST dropdown strip moves verbatim into the new
CastingBoard.jsx (DubLeftColumn shrinks below 800 lines; the new file
holds the 300-line soft cap). Styles extend the .dub-cast-* cluster in
index.css. Six new i18n keys translated in all 21 locales.

Co-authored-by: Matt Van Horn <mvanhorn@users.noreply.github.com>

* docs: changelog + roadmap entries for the casting board (#1767)

Co-authored-by: Matt Van Horn <mvanhorn@users.noreply.github.com>

* fix(casting): validate and preserve speaker assignments

* fix(casting): recover cleared merged assignments

---------

Co-authored-by: Cursor Agent <cursoragent@cursor.com>
Co-authored-by: Palash Debnath <4178343+debpalash@users.noreply.github.com>
2026-09-02 15:37:57 +05:30
Palash Debnath 0ed7d2ec22 chore(release): prepare v0.5.2 (#1761)
Synchronize VoiceStudio release metadata, lockfiles, installers, container references, documentation, and the dated v0.5.2 changelog after all planned fixes landed.
2026-09-02 10:26:14 +05:30
Matt Van Horn a95041f1e6 feat(studio): speech-to-speech voice changer (#1765)
Add a bounded, local-first speech-to-speech Convert workflow with shared ASR/TTS admission, duration matching, stale-request cancellation, profile conditioning, watermarking, persistence, localization, and regression coverage.\n\nCo-authored-by: Matt Van Horn <mvanhorn@users.noreply.github.com>
2026-09-02 09:45:40 +05:30
Matt Van Horn 4053397921 feat(dub): karaoke word-highlight caption burn-in (#1764)
Adds opt-in word-timed ASS karaoke captions while preserving the existing line-caption default.\n\nCo-authored-by: Matt Van Horn <mvanhorn@users.noreply.github.com>
2026-09-02 08:32:43 +05:30
Palash Debnath e2446c3e61 fix(release): keep Preview ahead of Stable (#1763)
Closes #1762.
2026-09-02 07:49:22 +05:30
Palash Debnath ccd984f324 Merge pull request #1760 from agudmund/feat/mcp-output-mode-files
feat(mcp): output mode + base-path file lane so agents never carry audio in context
2026-09-02 06:27:46 +05:30
Palash Debnath 89f0a25082 fix(mcp): bound encoded audio before decode 2026-09-02 06:13:11 +05:30
Palash Debnath b351dc5b1a Merge remote-tracking branch 'origin/main' into review/pr-1760
# Conflicts:
#	CHANGELOG.md
2026-09-02 06:07:57 +05:30
Palash Debnath 026410fbc6 test(worker): make loop responsiveness checks deterministic (#1759)
Reviewed by CodeRabbit and Greptile. Required Tests (backend + frontend) gate passed on the current, mergeable head.
2026-09-02 05:43:08 +05:30
Palash Debnath b6a1ee50ff Merge remote-tracking branch 'origin/main' into fix/deterministic-inbound-loop-tests
# Conflicts:
#	tests/test_worker_inbound_transport.py
2026-09-02 05:29:25 +05:30
Ævar GuðmundssonandClaude Fable 5.1 3c3c37615c feat(mcp): output mode + base-path file lane so agents never carry audio in context
An LLM agent pays for every byte it receives, and generate_speech returned
each WAV as base64 inline - a short clip already brushed per-result limits.
This adds two knobs in the OMNIVOICE_* family, the pattern the ElevenLabs
MCP settled on (OUTPUT_MODE + a BASE_PATH security boundary):

- OMNIVOICE_MCP_OUTPUT_MODE = resources (default, the original contract) |
  files | both. In files mode generate_speech returns audio_url (the render
  the backend already keeps, served at /audio/<id>.wav) and, when a base
  path is set, output_path - the WAV written into that directory.
- OMNIVOICE_MCP_BASE_PATH: the one directory agents may read from and
  receive files in. transcribe(audio_path=) and clone_voice(ref_audio_path=)
  read only inside it (relative paths resolve against it, absolute ones must
  lie within it, symlinks resolved before the check); with no base path,
  path arguments are refused with a reason.
- OMNIVOICE_MCP_TIMEOUT_S (default 120): the tools' backend timeout, since a
  CPU host serializes generations and an agent queued behind another render
  outlasted the fixed budget with an empty-message ToolError.

Also: transcribe and clone_voice share one input helper (data-URI tolerance
now covers transcribe too), the upload filename carries the sniffed
extension, and the reply is built with json.dumps instead of hand-rolled
JSON. Tests cover the mode parsing, the boundary (escape and missing-base
refusals), both input lanes, all four reply shapes, and the timeout knob.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
2026-09-01 23:46:44 +00:00
Palash Debnath c44f2fc042 test: make key revocation ordering deterministic (#1758)
Replaces a scheduler-sensitive 200 ms assertion with a deterministic ordering check against the blocked-write release barrier. Repairs the red post-merge main run from #1751.
2026-09-02 05:09:21 +05:30
Palash Debnath 62604ec9bb test(worker): make loop responsiveness checks deterministic 2026-09-02 04:56:29 +05:30
Palash Debnath 08569397d3 fix(asr): secure configured endpoints and refresh guidance (#1751)
Refreshes README and linked docs with accurate installation, platform, privacy, API, and model-license guidance; documents local gigastt; and pins OpenAI-compatible ASR traffic to the configured secure origin. Closes #1736.
2026-09-02 04:41:36 +05:30
Palash Debnath c9a468e500 Merge pull request #1756 from debpalash/fix/windows-direct-job-owner-1734
fix(windows): remove the sidecar supervisor hop
2026-09-02 04:07:36 +05:30
Palash Debnath aec4694628 test(windows): cover delayed Job exit 2026-09-02 03:52:17 +05:30
Palash Debnath 5df1dc973c Merge remote-tracking branch 'origin/main' into fix/windows-direct-job-owner-1734
# Conflicts:
#	CHANGELOG.md
2026-09-02 03:48:02 +05:30
Palash Debnath 8cc978588d Merge pull request #1754 from debpalash/fix/backend-startup-budget-1749
fix(frontend): align backend startup stall budget
2026-09-02 03:28:13 +05:30
Palash Debnath 6f80110a42 fix(windows): await timed-out Job teardown 2026-09-02 03:15:16 +05:30
Palash Debnath c5ac548100 Merge remote-tracking branch 'origin/main' into fix/windows-direct-job-owner-1734
# Conflicts:
#	CHANGELOG.md
2026-09-02 03:14:09 +05:30
Palash Debnath 26d7a333e8 Merge remote-tracking branch 'origin/main' into fix/backend-startup-budget-1749
# Conflicts:
#	CHANGELOG.md
2026-09-02 03:12:28 +05:30
Palash Debnath 75eb7c6099 test: make repository ID bound deterministic (#1757)
Replaces a runner-speed-dependent security assertion with a deterministic proof that oversized repository IDs are rejected before library validation. Repairs the red post-merge main run from #1755.
2026-09-02 02:51:20 +05:30
Palash Debnath 1d06c6f079 fix: accept ASR-detected dub source codes (#1755)
Accepts every Whisper language code persisted by ASR, including three-letter Cantonese yue, so subsequent dubbing uploads no longer fail validation. Closes #1737.
2026-09-02 02:26:56 +05:30
Palash Debnath 35d2e27b5e Classify RX 6700 XT WSL2 path as unverified (#1752)
Closes #1716.\n\nDefines supported, best-effort override, unverified, and unsupported architecture evidence; marks RX 6700 XT/gfx1031 over WSL2 ROCDXG as unverified; and requires routing, utilization, and CPU-fallback evidence before claiming acceleration.
2026-09-02 01:45:56 +05:30
Palash Debnath 267ded0e79 fix(windows): simplify nested job cleanup 2026-09-02 01:44:52 +05:30
Palash Debnath 00d923b4fa fix: repair incomplete Sherpa model caches (#1753)
Repairs missing and zero-byte Sherpa ONNX cache assets before model loading, with offline regression coverage. Closes #1733.
2026-09-02 01:42:03 +05:30
Palash Debnath 549a56fc2d fix: remove Windows sidecar supervisor hop 2026-09-02 01:38:05 +05:30
Palash Debnath 6e47b15e82 test: pin backend stall timeout boundary 2026-09-02 01:29:22 +05:30
Palash Debnath 5e08dde5e0 fix: align backend startup stall budget 2026-09-02 01:24:27 +05:30
Palash Debnath 3c488f7189 Add Trendshift repository badge (#1742)
Adds the repository owner-requested live Trendshift badge at the top of the README.
2026-09-02 00:55:29 +05:30
Palash Debnath 4e5e8d1f89 Allow OmniVoice slow sidecar startup (#1743)
Fixes #1711.\n\nGives only the OmniVoice subprocess a 120-second readiness budget while retaining the shared 30-second default for all other sidecars, with regression coverage.
2026-09-02 00:22:05 +05:30
Palash Debnath 497d57ee62 Show complete engine disk costs before install (#1728)
Closes #1718.

Adds structured pre-install and measured post-install disk costs, complete sidecar preflight accounting, strict authorization for recursive scans, and localized catalogue details with confidence and deduplication context.
2026-09-01 23:49:05 +05:30
Palash Debnath a4f68e4ea0 Preserve cloned voices during queued SRT imports (#1745)
Closes #1709.

Queues SRT selection until speaker analysis and clone extraction finish, then applies the newest selected subtitle file with stale-result, failure, replacement, retry, and abort guards.
2026-09-01 23:15:43 +05:30
Palash Debnath 51bbf50ce3 Ship a supported per-user Windows installer (#1730)
Closes #1713

Adds a separately identified per-user MSI and updater channel, non-administrator install/uninstall verification, and fail-closed WebView2 handling for current-user installs.
2026-09-01 22:03:33 +05:30
Palash Debnath 0c005abd07 Merge pull request #1731 from debpalash/fix/dictation-listener-registration-1707
Keep dictation events queued across listener reloads
2026-09-01 21:30:02 +05:30
Palash Debnath 67d7236edd fix: retain dictation events until listener acknowledgement 2026-09-01 21:03:35 +05:30
Palash Debnath 8f96692537 Merge remote-tracking branch 'origin/main' into fix/dictation-listener-registration-1707 2026-09-01 20:59:24 +05:30
Palash Debnath 60fe6cb814 Merge pull request #1727 from debpalash/fix/msi-webview2-control-1714
fix: make WebView2 MSI bootstrap controllable
2026-09-01 20:38:53 +05:30
Palash Debnath 88e2d4cb00 Merge remote-tracking branch 'origin/main' into fix/dictation-listener-registration-1707 2026-09-01 20:00:42 +05:30
Palash Debnath 05a8badf86 Merge remote-tracking branch 'origin/main' into fix/msi-webview2-control-1714 2026-09-01 19:43:47 +05:30
Palash Debnath 7d49faccdc Merge pull request #1724 from debpalash/fix/engine-execution-evidence-1717
Expose per-engine execution evidence
2026-09-01 19:09:44 +05:30
Palash Debnath 3b6a05c1a8 Merge remote-tracking branch 'origin/main' into fix/msi-webview2-control-1714
# Conflicts:
#	CHANGELOG.md
2026-09-01 18:58:13 +05:30
Palash Debnath ea4dcb6c61 Merge remote-tracking branch 'origin/main' into fix/dictation-listener-registration-1707 2026-09-01 18:56:06 +05:30
Palash Debnath 88e3e71fd7 Merge remote-tracking branch 'origin/main' into fix/engine-execution-evidence-1717 2026-09-01 18:47:30 +05:30
Palash Debnath a30b71666c Merge pull request #1725 from debpalash/fix/subtitle-timing-controls-1710
fix: make subtitle timing directly adjustable
2026-08-30 22:57:39 +05:30
Palash Debnath fb724b8633 fix: isolate diagnostic engine imports 2026-08-30 22:57:12 +05:30
Palash Debnath 0643965583 style: format dictation listener test 2026-08-30 22:53:49 +05:30
Palash Debnath 958c9de7b9 fix: lease dictation listener readiness 2026-08-30 22:44:11 +05:30
Palash Debnath 6dbfeda2e5 Merge remote-tracking branch 'origin/main' into fix/msi-webview2-control-1714 2026-08-30 22:32:25 +05:30
Palash Debnath 9447ace2c3 fix: invalidate unloaded ASR evidence 2026-08-30 22:21:18 +05:30
Palash Debnath 07cff18dfc Merge pull request #1726 from debpalash/fix/backend-parent-liveness-1707
fix: reap backend when desktop owner exits
2026-08-30 22:12:44 +05:30
Palash Debnath 7e777a3a0f fix(dubbing): accept exact minimum timing boundary 2026-08-30 22:12:28 +05:30
Palash Debnath 9fe68ddb3f docs(installer): qualify fail-closed WebView behavior 2026-08-30 22:10:10 +05:30
Palash Debnath cbc89e2d15 fix(diagnostics): track live execution evidence lifecycle 2026-08-30 22:09:05 +05:30
Palash Debnath 87b052489a Merge remote-tracking branch 'origin/main' into fix/backend-parent-liveness-1707 2026-08-30 21:53:15 +05:30
Palash Debnath e35c6ad987 fix(backend): document parent pipe closure 2026-08-30 21:53:15 +05:30
Palash Debnath 1ceebe5aa2 Merge remote-tracking branch 'origin/main' into fix/subtitle-timing-controls-1710 2026-08-30 21:52:56 +05:30
Palash Debnath 7658865735 fix(i18n): clarify Dutch timing labels 2026-08-30 21:52:56 +05:30
Palash Debnath 2c232ac863 Merge remote-tracking branch 'origin/main' into fix/msi-webview2-control-1714 2026-08-30 21:52:15 +05:30
Palash Debnath 2af47dbd4b fix(installer): make WebView2 download opt-in 2026-08-30 21:51:46 +05:30
Palash Debnath 6b5825c5a5 Merge remote-tracking branch 'origin/main' into fix/engine-execution-evidence-1717 2026-08-30 21:48:02 +05:30
Palash Debnath bfdb8dc7f5 docs: add private production API deployment (#1723)
* docs: add private production API deployment

* docs: harden private API deployment guidance

* docs: make proxy trust configuration executable

* docs: clarify private API trust boundaries

* docs: persist private service application data
2026-08-30 21:29:14 +05:30
Palash Debnath e9c2c451e7 test: resolve parent watchdog at execution time 2026-08-30 21:28:53 +05:30
Palash Debnath d4f89ecdc7 fix: clamp and localize subtitle steppers 2026-08-30 21:28:18 +05:30
Palash Debnath e1101255bd fix: make WebView2 MSI bootstrap controllable 2026-08-30 21:25:12 +05:30
Palash Debnath 0a20aeb0c9 fix: reap backend when desktop owner exits 2026-08-30 21:18:36 +05:30
Palash Debnath 7e50640f97 fix: make subtitle timing directly adjustable 2026-08-30 21:15:52 +05:30
Palash Debnath bab794e13b fix: report only observed engine execution 2026-08-30 21:12:54 +05:30
Palash Debnath 503e4ffde8 Merge remote-tracking branch 'origin/main' into fix/engine-execution-evidence-1717 2026-08-30 21:10:29 +05:30
Palash Debnath 80aafa3a53 fix: preserve startup and media failure evidence (#1722)
* fix: preserve startup and media failure evidence

* fix(dictation): bind readiness to installed revision

* test(dictation): align cache resolution contract

* fix: keep retry failures actionable
2026-08-30 20:49:53 +05:30
Palash Debnath 0ec9c074e8 fix: distinguish opaque loaded sidecars 2026-08-30 20:49:26 +05:30
Palash Debnath 2be73b43e3 feat: expose engine execution evidence 2026-08-30 20:40:35 +05:30
Palash Debnath 4e61c2782d fix(engines): harden skill discovery and CosyVoice setup 2026-08-29 05:10:35 +05:30
Palash Debnath d6a2277e6a Merge pull request #1700 from debpalash/chore/finalize-v0.5.1
chore(release): finalize v0.5.1
2026-08-28 20:21:25 +05:30
Palash Debnath a59fdee094 docs(changelog): preserve unreleased structure 2026-08-28 20:06:32 +05:30
Palash Debnath 3fe2018589 chore(release): finalize v0.5.1 2026-08-28 20:02:55 +05:30
Palash Debnath 00bf55c96f Merge pull request #1699 from debpalash/fix/mps-omnivoice-isolation-1697-1698
fix(macos): isolate OmniVoice MPS generation
2026-08-28 19:46:31 +05:30
Palash Debnath 1e4fdd12e0 fix(worker): preserve isolated OmniVoice synthesis settings 2026-08-28 19:32:20 +05:30
Palash Debnath 3b6e15dad5 fix(macos): isolate OmniVoice MPS generation 2026-08-28 19:21:06 +05:30
Palash Debnath de51120d6a fix(models): handle load OOMs safely (#1696)
* fix(models): handle load OOMs safely (#1695)

* fix(dub): sanitize streamed generation failures

* style(ui): format readiness checklist
2026-08-28 16:00:24 +05:30
Palash Debnath 3e5aa90e3f fix(dev): recover backend worker crashes (#1694)
Run uvicorn directly under the dev wrapper so a worker crash cannot hide behind a live reload parent. Preserve Python source reloads, restart isolated crashes with bounded diagnostics, and keep persistent crash loops loud.

Closes #1690.
2026-08-28 14:31:54 +05:30
Palash Debnath 8cc7c88692 fix(dub): restore audio when the media preview falls back (#1693)
Preserve audible, language-matched playback when WebView video decoding falls back, keep waveform timing aligned with the selected dub, prevent recursive recovery failures, and disable stale audio caching.

Closes #1692.
2026-08-28 13:37:34 +05:30
Palash Debnath e7e97e4b4e fix(ui): give Model Catalogue engine rows room (#1691)
Closes #1689.

Uses the Catalogue workspace width for readable engine rows, keeps status badges out of engine names, and collapses by the named Tauri-scaled container. Compact Settings rows remain unchanged.
2026-08-28 12:38:16 +05:30
Palash Debnath 97bfbfecd1 chore(release): prepare v0.5.1 (#1688)
Prepare the tested main branch for the v0.5.1 patch release with synchronized version sources, mirrors, lockfiles, release notes, and install guidance.

Closes #1687.
2026-08-28 11:27:59 +05:30
Palash Debnath 0c8a40774f feat(ui): give Model Catalogue room to breathe (#1686)
Redesign Model Catalogue as one responsive workspace canvas with a compact wide-screen header, simplified navigation, and a single primary scroll plane.

Closes #1685.
2026-08-28 10:50:45 +05:30
Palash Debnath f45c885171 fix(ui): reflow workspaces after bootstrap (#1684)
* fix(ui): reflow shell after bootstrap

* fix(ui): serialize native scale updates

* test(ui): mount responsive shell after bootstrap
2026-08-28 09:54:01 +05:30
Palash Debnath da8358916c fix(desktop): preflight Linux runtime dependencies (#1681)
Fail before backend/window startup when Linux source hosts lack Enigo’s libxdo linker input or WebKit’s GStreamer audio sink. Print an exact distro package command, sync source-build docs, and lock the probes with deterministic tests.

Closes #1680
Closes #1682
2026-08-28 07:19:36 +05:30
Palash Debnath 0268f43e7f fix(dub): keep language and media tools ready (#1679)
Fixes #1677 and #1678.

Publishes first-run media tools to the live backend, provides precise cross-platform missing-process guidance, and keeps localized source-language selection available before transcription. Includes regression coverage and deterministic model-store test isolation.
2026-08-28 06:32:49 +05:30
Palash Debnath 6d42d2255e Merge pull request #1675 from debpalash/fix/runtime-stability-1652
fix(runtime): prevent overlapping native work and false crashes
2026-08-27 22:45:11 +05:30
Palash Debnath 1103898d7b Merge remote-tracking branch 'origin/main' into fix/runtime-stability-1652
# Conflicts:
#	CHANGELOG.md
2026-08-27 22:30:58 +05:30
Palash Debnath 28220ff2dd Merge pull request #1672 from debpalash/fix/dub-srt-voices-1660
fix(dub): preserve workflow state and browser compatibility
2026-08-27 22:12:52 +05:30
Palash Debnath fa9bfd41be fix(stories): serialize marker preview synthesis 2026-08-27 21:57:41 +05:30
Palash Debnath 0e47053823 fix(dub): scope imported speaker clones to matched cues 2026-08-27 21:57:37 +05:30
Palash Debnath bea72cadbe Merge remote-tracking branch 'origin/main' into fix/runtime-stability-1652
# Conflicts:
#	CHANGELOG.md
2026-08-27 21:50:51 +05:30
Palash Debnath 6624b3fb19 Merge remote-tracking branch 'origin/main' into fix/dub-srt-voices-1660
# Conflicts:
#	CHANGELOG.md
2026-08-27 21:50:46 +05:30
Palash Debnath fc603bcc92 Merge pull request #1673 from debpalash/fix/docker-admin-auth-1651
fix(docker): secure admin access and document WSL ROCm
2026-08-27 21:28:11 +05:30
Palash Debnath 2300d4d5f7 fix(tts): localize shared admission failures 2026-08-27 21:27:40 +05:30
Palash Debnath c7890d2a5a fix(dub): close final media and retry review gaps 2026-08-27 21:25:56 +05:30
Palash Debnath 633e991edc Merge remote-tracking branch 'origin/main' into fix/runtime-stability-1652
# Conflicts:
#	CHANGELOG.md
2026-08-27 21:09:34 +05:30
Palash Debnath 64b3785b0b Merge remote-tracking branch 'origin/main' into fix/docker-admin-auth-1651
# Conflicts:
#	CHANGELOG.md
2026-08-27 21:09:30 +05:30
Palash Debnath 4b3c62d783 Merge remote-tracking branch 'origin/main' into fix/dub-srt-voices-1660
# Conflicts:
#	CHANGELOG.md
2026-08-27 21:09:27 +05:30
Palash Debnath 103a5dbbe4 Merge pull request #1674 from debpalash/fix/voice-clone-reference-lease-1668
fix(generate): lease ad-hoc references across abandoned jobs
2026-08-27 20:47:00 +05:30
Palash Debnath 86f6c9ec8e fix(auth): reject blank server admin keys 2026-08-27 20:46:24 +05:30
Palash Debnath e1a76f9aca fix(dub): close shared workflow review gaps 2026-08-27 20:43:21 +05:30
Palash Debnath 57263edad7 fix(tts): centralize single-generation admission 2026-08-27 20:38:29 +05:30
Palash Debnath 4c0a0a26b7 fix(runtime): bound external debugger restarts 2026-08-27 20:26:52 +05:30
Palash Debnath 634936eb10 fix(rocm): diagnose disabled WSL DXG detection 2026-08-27 20:26:09 +05:30
Palash Debnath 422dbd1313 fix(dub): address cast, language, and cancellation review 2026-08-27 20:25:16 +05:30
Palash Debnath d23f22d56b test(generation): assert cancelled reference workers 2026-08-27 20:22:19 +05:30
Palash Debnath eaa87ff041 fix(docker): require the Studio administrator key 2026-08-27 20:21:42 +05:30
Palash Debnath 0ee9bc35d0 fix(runtime): prevent overlapping native work and false crashes 2026-08-27 20:20:58 +05:30
Palash Debnath a1cd15964c Merge remote-tracking branch 'origin/main' into fix/voice-clone-reference-lease-1668 2026-08-27 20:10:51 +05:30
Palash Debnath 895e62df7c Merge remote-tracking branch 'origin/main' into fix/docker-admin-auth-1651 2026-08-27 20:10:51 +05:30
Palash Debnath 49c175301a Merge remote-tracking branch 'origin/main' into fix/dub-srt-voices-1660 2026-08-27 20:10:07 +05:30
Palash Debnath b92d35ac5d feat: add local speech platform (#1671)
Fixes #1646
2026-08-27 19:51:07 +05:30
Palash Debnath b6bd125f23 fix(dub): keep preview extraction off event loop 2026-08-27 19:45:31 +05:30
Palash Debnath 53331a6f5a fix(dub): normalize uploaded video for preview 2026-08-27 19:45:10 +05:30
Palash Debnath 1d445855d2 fix(dub): make language workflow reliable 2026-08-27 19:43:49 +05:30
Palash Debnath 9dc01d4b8b fix(docker): require admin key in quick starts 2026-08-27 19:31:55 +05:30
Palash Debnath e93b6366e6 fix(dub): preserve voices across SRT imports 2026-08-27 19:14:56 +05:30
Palash Debnath 89cee3f824 fix(generate): lease ad-hoc references across abandoned jobs 2026-08-27 19:06:51 +05:30
Parker Hill 078bad8e0f fix(setup): repair fresh-clone desktop and ROCm source installs (#1664, #1665)
Fix fresh-clone desktop development by creating the required dist placeholder before Tauri starts, and make source setup install the selected CUDA or ROCm PyTorch stack consistently. Adds behavior-level cross-platform regression coverage.\n\nFixes #1664.\nFixes #1665.\n\nThanks @uberclokr for the contribution.
2026-08-27 18:33:55 +05:30
Paolo Antinori 032c5aba86 fix(core): macOS fallback for os.waitid ownership probe (#1656)
Fixes #1656. Adds the safe waitpid fallback, proves child ownership before process-group signals, and credits @paoloantinori.
2026-08-27 17:23:25 +05:30
Palash Debnath a8371baaa8 fix(desktop): serialize backend lifecycle (#1635)
Closes #1635.
2026-08-24 18:11:37 +05:30
Palash Debnath 5a615d2c66 feat(workers): package headless GPU nodes (#1638) (#1648)
Closes #1638.\n\nPackages headless GPU workers with durable enrollment, bounded artifact handling, cross-platform lifecycle cleanup, and regression coverage. Incorporates CodeRabbit, Greptile, CodeQL, and platform-CI findings before merge.
2026-08-24 16:32:56 +05:30
Palash Debnath 12480b81b1 fix(persistence): move longform projects to IndexedDB (#1636) (#1650)
Move unbounded Stories and Audiobook project data to revisioned IndexedDB storage with bounded local fallback, durable clear tombstones, migration/recovery safeguards, and deadline-safe persistence before exits and relaunches.

Closes #1636.
2026-08-24 12:59:24 +05:30
Palash Debnath ef1cb57944 fix: route OmniVoice to ROCm GPUs (#1647) 2026-08-24 03:25:56 +05:30
Palash Debnath 6447788fdf fix(dictation): cancel widget work on unmount (#1645)
Fix the post-merge Vitest failure by invalidating CaptureWidget startup continuations and releasing timers, sockets, recorder/worklet state, and microphone streams on teardown. Adds fail-before/pass-after coverage for active capture, pending fallback, delayed microphone permission, and delayed WebSocket authentication.
2026-08-24 02:44:00 +05:30
Palash Debnath afa361913c fix(generate): surface the real cause on streaming failures (#1633)
Classify and journal local and remote streaming generation failures, return actionable scrubbed guidance, and keep exception details, tokens, and user paths out of logs and NDJSON responses.
2026-08-23 15:55:05 +05:30
Palash Debnath 0d3c81b596 feat(gguf): add Linux ARM64 runtime support (#1641)
Add linux-aarch64 platform detection, Vulkan-preferred source builds with CPU fallback, ARM64-safe PyTorch dependency markers, native artifact CI, regression coverage, and synchronized architecture documentation.
2026-08-23 15:07:52 +05:30
Palash Debnath 98c9e68aae test(worker): carry the capacity limit through the handshake, not a post-connect mutation (#1630)
The at-capacity test set client.config.max_concurrent_tasks = 1 after
connect_worker, racing the server's stream-open ConfigUpdate — which
carries the REGISTERED capacity (2, from the hello). When the frame
lands after the mutation (slow CI runners), the override is clobbered,
the worker honours 2 slots, accepts the second assignment, and the test
reports over-concurrency that never existed. Failed CI twice on
2026-08-21 (#1627 first run + main).

Fix: pass the limit through WorkerConfig so hello -> registration ->
ConfigUpdate all agree from the start; assert the registered capacity
as a regression guard. Demonstrated fail-before: with the old pattern
the server record stays 2 and a delivered config frame reverts the
client gate to 2.
2026-08-21 20:36:12 +00:00
debpalash 772e3e82b4 feat(uninstall): --app/-RemoveApp removes the prebuilt binary install
The default curl|sh and irm|iex installs now put a real app on disk
(/Applications or ~/Applications, ~/.local/bin/VoiceStudio, MSI product).
Both uninstallers gain an opt-in flag that targets exactly those:
- uninstall.sh --app: adds the app bundle / AppImage to the dry-run plan
- uninstall.ps1 -RemoveApp: resolves the MSI product across HKLM/HKCU/
  WOW6432Node and uninstalls it silently under -Yes
2026-08-22 01:58:39 +05:30
debpalash 228019c9a8 fix(install): usage text works through a curl pipe ($0 is not a file) 2026-08-21 21:08:06 +05:30
Palash Debnath e38b5f5741 feat(install): binary-first installer with version picking and --source opt-in (#1628)
* feat(install): prebuilt-app installs by default, --source opt-in, --version picker

- install.sh: default mode now downloads the verified release asset
  (dmg/AppImage + SHA256SUMS check) instead of cloning and building;
  --source keeps the previous clone-and-build flow; --version pins a release
- install.ps1: same split — msi download with checksum verification and
  setup wizard by default; -Source (or VOICESTUDIO_INSTALL_MODE) for the
  source build; VOICESTUDIO_VERSION picks a release
- worker landing page documents the modes

* fix(install): CI smoke covers binary + source modes; hdiutil output parse

- drop -quiet from hdiutil attach (it suppresses the mount-point line the
  script parses — caught by the macOS smoke)
- install.ps1 runs msiexec silently under CI, wizard interactively
- smoke verifies binary installs per OS (app bundle / AppImage / MSI
  registry entry) and keeps full source coverage behind --source

* ci(install): check HKCU/WOW6432Node too — Tauri MSI registers per-user

* fix(install): rename $version — collides with bun installer's $Version under iex
2026-08-21 14:30:33 +00:00
debpalash f60f5f1f59 chore(install): route voicestudio.sh/install* to the installer worker 2026-08-21 17:31:50 +05:30
Palash Debnath 7640f42dce feat(install): one-command installer URL (.sh + .ps1) + 3-OS install smoke (#1627)
* feat(install): one-command installer URL (.sh + .ps1) + 3-OS install smoke

- scripts/install.ps1: Windows source installer (winget deps, uv, bun,
  clone, uv sync, frontend build); honors OMNIVOICE_PYTHON/OMNIVOICE_REGION
- infra/install-redirect: Cloudflare Worker serving /install with
  User-Agent sniffing (curl -> sh, PowerShell -> ps1, browser -> landing
  page); proxies live from main; /install.sh + /install.ps1 aliases
- scripts/install.sh: fix stale advertised URL (main/install.sh never
  existed) and repo-root resolution so a local run no longer clones a
  duplicate repo into ~/VoiceStudio (verified on macOS arm64)
- .github/workflows/install-smoke.yml: run both installers end-to-end on
  ubuntu/macos/windows when they change
- docs-sync: install one-liners lead each platform guide; STRUCTURE.md

(#1626)

* fix(install): don't let a failed bun download pass silently

curl | sh runs an empty script and exits 0 when the download fails, so
a bun.sh hiccup surfaced much later as 'bun: command not found' (seen
on the macos-latest smoke runner). Fetch to a temp file, verify, fall
back to npm -g bun when node exists, and die with the manual command.
Same post-install verification for uv.

* fix(install): UTF-8 BOM for install.ps1 + quiet-style changelog entry

- tests/scripts/test_uninstall_ping.py requires shipped PowerShell
  scripts with non-ASCII text to carry a UTF-8 BOM (Windows PowerShell
  5.1 mis-decodes otherwise); same treatment uninstall.ps1 already gets
- test_changelog_style caps entries at ~400 chars
2026-08-21 11:47:39 +00:00
Palash Debnath 3eeed0cfe9 chore: refresh application dependencies (#1625)
Refreshes frontend, tooling, and Python dependencies; regenerates the frozen lockfiles and worker protocol stubs. Keeps compatibility caps for FastAPI, Oxlint, and Vitest where newer releases break repository contracts. Updates the Uvicorn bind-failure regression test for its new nonzero exit code. Fully tested after merging current main.
2026-08-21 10:30:49 +00:00
Muhammad Ahmad Ali b946fded12 fix(dub): preserve speaker attribution across segment edits (#1624)
Adds per-line subtitle timing, insertion, and bidirectional merge controls while preserving each speaker's attribution across merge/split, restore, translation, and cast edits. Includes regression coverage and localization updates. Closes #1612.
2026-08-21 09:55:04 +00:00
Palash Debnath 7718a7a10b fix: cross-platform dictation delivery (#1610)
Makes dictation delivery, capture, recovery, model fallback, AEC, and localized status behavior reliable across macOS, Windows, and Linux.
2026-08-21 03:33:50 +00:00
Palash Debnath 0687e13b57 test(perf): performance regression budgets as operation-count guards (#1622)
Adds deterministic operation-count regression budgets for streaming TTS, cached dub re-mixes, and native batch dubbing.
2026-08-21 02:53:15 +00:00
Palash Debnath 89d585a36e perf(dub,stream): reuse cached segments, batch the default engine, report real TTFA (#1620)
Reuses verified cached segments, safely batches default-engine dubbing, and reports synthesis-only TTFA/RTF.
2026-08-20 21:12:04 +00:00
Paolo Antinori 3f5114923b feat(pockettts): opt-in 24-layer checkpoints via OMNIVOICE_POCKETTTS_24L (#1613)
Adds an opt-in 24-layer PocketTTS checkpoint path, with French correctly using its required 24-layer model.
2026-08-20 20:42:40 +00:00
Palash Debnath 43f1d46fe6 fix(indextts): accept the config name upstream ships, and keep long text alive (#1619)
* fix(indextts): accept the config name upstream ships, and keep long text alive

Two independent defects, both reported on a working IndexTTS 2.5 install.

Install always failed. IndexTeam/IndexTTS-2.5 ships the model config as
config.yaml — at the pinned revision d0aa86e7 and at HEAD; config_v2_5.yaml
exists in no upstream revision. VoiceStudio demanded that name, so
_weights_floor_ok never found it and the install died claiming 'the download
was likely interrupted' when the download had been perfect. The only way
through was to hand-rename the file. Both names are accepted now, in the
installer and on the load path, so installs created with the workaround keep
working without a reinstall.

Long text was killed at 60s. infer() is one blocking upstream call that puts
nothing on the wire, and IndexTTS was the only sidecar still on the 60s
recv_timeout_s class default while pockettts and omnivoice-subprocess had both
raised theirs. Raising the default alone does not fix it — which is why the
reporter's RECV_TIMEOUT_S=3600 edit didn't help: progress frames are also what
report activity to the GPU pool's execution clock (#1367), so a silent sidecar
still trips the outer generate budget. The sidecar now heartbeats every 5s
while infer() runs (and during the cold model construction), _send takes a
lock so the beat thread can't interleave framing, and the deadline rises to
900s via OMNIVOICE_INDEXTTS_RECV_TIMEOUT_S.

test_indextts25_health_requires_25_config_name asserted the bug — that a
checkout holding only config.yaml is unhealthy — so it is rewritten to the
corrected contract, including that a genuinely truncated download is still
caught.

Fixes #1611

* test(indextts): follow the installed config name in the sidecar loader tests

Two more tests encoded the config_v2_5.yaml assumption, both asserting
cfg_path against a directory where no config existed at all — so they were
pinning the literal name rather than the resolution. They now lay down a real
checkpoints/ tree and assert the resolved path, including that a checkout
carrying the pre-fix hand-renamed config still resolves.

Caught by the full suite; the targeted runs during development did not reach
tests/backend/services/.

* test(indextts): event-driven heartbeat tests, real interleaving proof, precedence pin

Review round on #1619 — all four findings taken.

- The docs line naming 0.5.1 is version-neutral now ('Earlier installs') —
  version labels are the owner's call.
- The heartbeat tests waited on wall-clock sleeps; they now block on a
  per-write Event with a bounded deadline, so scheduler load can't flake them.
- The _send test asserted the lock EXISTS — a tautology. It now drives four
  concurrent writers through a stream that yields between every byte and
  asserts every frame decodes; verified fail-before by removing the lock
  (torn frame) and pass-after.
- The precedence test deleted config.yaml before creating the renamed one, so
  reversed precedence still passed. Both files now coexist for the assertion;
  verified fail-before by reversing _CFG_NAMES.
2026-08-20 19:43:52 +00:00
Paolo Antinorianddebpalash 3223a20f88 fix(openai-compat): reuse cached engine instances in _resolve_engine (#1614)
* fix(openai-compat): reuse cached engine instances in _resolve_engine

The direct engine-ID path in /v1/audio/speech constructed a fresh
backend per request (return cls()). For SubprocessBackend engines that
meant: a new sidecar process, a full torch import and an engine model
reload on EVERY request (measured ~28s floor per pockettts request on
an M3 Pro), plus another atexit hook registration each time — exactly
what get_engine_instance_for()'s docstring warns against.

Route the explicit-ID path through the same cached-singleton seam the
active-engine path already uses. Unknown/unavailable IDs keep their
400s; tts-1/tts-1-hd and the OmniVoiceBackend special case are
unchanged.

* fix(openai-compat): unload the outgoing engine on explicit-ID switches

Review follow-up (Greptile/CodeRabbit on #1614): caching instances without
a switch rule would let each distinct explicit engine ID stay resident,
accumulating sidecars / multi-GB in-process models. Mirror
get_active_tts_backend's MM2-01 switch rule: a different explicit ID
(omnivoice included, which resolves to the active engine) unloads the
outgoing instance first, best-effort.

* fix(openai-compat): evict via the shared single-engine-resident seam, not a router-local cache

The explicit-ID unload cache (13c14e2c) kept its own instance ref keyed by
model id. The shared engine cache is deliberately keyed by CLASS (registry
rebinds, idle sweeps and engine_memory eviction all mutate it), so the
router's id-keyed ref could go stale and keep serving an instance the
lifecycle system no longer tracked — caught by
test_openai_speech_toggle_off_sends_raw_text in full-suite order, and it
also introduced a novel unload path that ignored the
OMNIVOICE_SINGLE_ENGINE_RESIDENT opt-out.

Drop the router-local cache entirely: _resolve_engine returns the shared
cached singleton (get_engine_instance_for), and create_speech calls
evict_other_tts_engines(backend.id) before warming the engine — the exact
seam /generate uses. That covers every transition (explicit id → explicit
id, explicit id → tts-1/omnivoice aliases), honors the policy opt-out, and
leaves no per-router state to drift. Regression pinned at the route level in
test_speech_request_evicts_other_resident_engines.

* chore(changelog): trim the #1614 entry to the one-liner limit

415 chars against the 400 the style test allows — CI would have failed on it.

---------

Co-authored-by: debpalash <4178343+debpalash@users.noreply.github.com>
2026-08-20 19:14:10 +00:00
Palash Debnath 2d37627ab2 fix(setup): tolerate reserved memory in the RAM preflight, add OMNIVOICE_RAM_PREFLIGHT=0 escape hatch (#1621)
* fix(setup): tolerate reserved memory in the RAM preflight, add OMNIVOICE_RAM_PREFLIGHT=0 escape hatch (#1618)

An "8 GB" machine reports ~7.8 GB usable (firmware/iGPU/kernel
reservations), so comparing OS-reported RAM against the marketing-size
8 GB threshold hard-blocked exactly the boundary hardware the minimum is
meant to admit — with no way past the wizard. Both thresholds are now
compared with a 7% reserved-memory allowance, and
OMNIVOICE_RAM_PREFLIGHT=0 downgrades a genuine fail to a warning for
users who accept the OOM risk (same opt-out shape as
OMNIVOICE_ASR_VRAM_PREFLIGHT).

Regression tests: backend/tests/test_ram_preflight_1618.py.
Docs: troubleshooting §1c.

* review: hermetic preflight stubs in tests; correct the escape-hatch doc

Greptile P1: the Settings panel can't set OMNIVOICE_RAM_PREFLIGHT (and the
blocker appears before setup completes anyway) — the doc now points at
PowerShell / shell env only.
CodeRabbit: stub _network_check and media_tools.summary so each RAM
assertion stays fast and offline (26s -> 6s locally).
2026-08-20 19:03:50 +00:00
Palash Debnath 54a88f694b fix(watermark): run AudioSeal eagerly instead of through torch.compile (#1617)
* fix(watermark): run AudioSeal eagerly instead of through torch.compile

AudioSeal vendors moshi's @torch_compile_lazy on SEANetEncoder.forward, so
the first embed of a session — not the model load, which #1576's prefetch
already warms — called torch.compile and dropped into Inductor's C++ codegen.
On a macOS arm64 deployment that compile raised CppCompileError on 10/10
takes: the embed fail-opened and the audio shipped UNMARKED, an EU AI Act
Art. 50(2) provenance gap, after burning 30-40s on the first take and 5-8s on
each later one.

The compile is pure cost even where it succeeds. Measured on an M3 (5s of
24kHz audio, three consecutive embeds): compiled 9.70/0.26/0.23s vs eager
0.30/0.28/0.27s — a ~10s first-embed tax to save ~0.03s afterwards, on CPU
work already bounded by the 30s chunk loop. Both embed and detect now run
inside audioseal's own no_compile() switch, restored on the way out (it is a
process global, and other models are entitled to compile).

Verified end-to-end: first embed 9.70s -> 0.26s, watermark still round-trips
at confidence 1.0 with the OmniVoice message intact.

Fixes #1615

* fix(watermark): collapse the eager-guard globals into one lock-guarded state

CodeQL flagged _eager_saved's module-level initializer as dead, and it was
right: depth 0->1 always writes the field before depth 1->0 reads it, so the
None at import was never observed. Depth and saved-value are only meaningful
together and only under _eager_lock, so they become one dict rather than two
module scalars — which also drops the global statement.

Also splits three semicolon-joined statements in the regression test (Ruff
E702, CodeRabbit).

Mutation re-checked after the refactor: a naive no_compile() body still fails
with 'compile was handed back mid-embed'.
2026-08-20 18:16:00 +00:00
Palash Debnath 3441201be0 fix(dictation): clear granted accessibility blocker (#1609)
* fix(dictation): refresh accessibility blocker

* docs(changelog): note accessibility refresh

* test(dictation): assert the native widget hide on accessibility grant

The recheck regression asserted only that the Accessibility pill text left
the DOM, so it still passed with hideWidgetWindow() removed and the native
capsule stranded on screen. Hold one stable getCurrentWindow().hide spy and
assert it after the poll (fails before the fix, passes after).
2026-08-20 17:10:56 +00:00
Palash Debnath de5d848189 Merge pull request #1598 from debpalash/integrate/open-repairs-20260820
merge: land reviewed repair train
2026-08-20 05:16:03 +00:00
debpalash aa7c2f5801 fix: fail open on watermark dispatch deadlines 2026-08-20 10:32:02 +05:30
debpalash b7f14ce4ad fix: bind deadlines to selected capability device 2026-08-20 10:17:31 +05:30
debpalash 7a928da1a0 fix: size remote deadlines for worker device 2026-08-20 10:14:50 +05:30
debpalash 0961a5e512 docs: explain deadline capability fallback 2026-08-20 10:06:40 +05:30
debpalash 0619df8dff fix(cloning): retain selected passage volume 2026-08-20 10:04:54 +05:30
debpalash a91b27b518 test: make cross-loop event delivery deterministic 2026-08-20 10:01:41 +05:30
debpalash 605236566c fix: align generation routing and CPU deadlines 2026-08-20 09:59:47 +05:30
debpalash 918c400f29 fix: preserve audio on watermark teardown cancellation 2026-08-20 09:48:43 +05:30
debpalash 76d16ac1bd fix: fail open on watermark shutdown submission race 2026-08-20 09:43:58 +05:30
debpalash f22606f3ad Merge PR #1600 follow-up: enforce reference bound without preprocessing 2026-08-20 09:36:57 +05:30
debpalash f8492dd676 fix(cloning): cover the fifteen-second boundary 2026-08-20 09:36:31 +05:30
debpalash f6afa43d07 Merge PR #1600: bound long cloning references
# Conflicts:
#	CHANGELOG.md
2026-08-20 09:31:02 +05:30
debpalash 835a889326 fix(cloning): bound exhaustive reference selection 2026-08-20 09:30:16 +05:30
debpalash 2f3888549b Merge PR #1606: isolate workspace DOM during rapid navigation
# Conflicts:
#	CHANGELOG.md
2026-08-20 09:18:01 +05:30
debpalash e9e4d95d06 perf(cloning): bound reference ASR candidates 2026-08-20 09:17:43 +05:30
debpalash 2048d2793a Merge PR #1604: keep healthy CPU synthesis past five minutes
# Conflicts:
#	CHANGELOG.md
2026-08-20 09:17:30 +05:30
debpalash 52ce462396 test: exercise delayed workspace cleanup 2026-08-20 09:14:09 +05:30
debpalash e300739d78 fix(cloning): select bounded speech with ASR 2026-08-20 09:13:42 +05:30
debpalash 7efae54cf8 fix: budget the selected engine device 2026-08-20 09:13:18 +05:30
debpalash 0257bfcfec fix: isolate workspace DOM lifecycles 2026-08-20 09:10:07 +05:30
debpalash 933c1a2cf1 Merge PR #1605: add resizable Dub editor columns 2026-08-20 09:09:59 +05:30
debpalash c59a787a10 fix(dub): follow RTL splitter direction 2026-08-20 09:07:15 +05:30
debpalash ed6d7a9652 fix: preserve explicit generation watchdog 2026-08-20 09:06:50 +05:30
debpalash 3a1013527f Merge PR #1601 follow-up: normalize original-only exports 2026-08-20 09:04:00 +05:30
debpalash 243220fc3a feat(dub): resize editor columns 2026-08-20 09:03:14 +05:30
debpalash d57b4babc5 fix(cloning): compare trimmed reference activity 2026-08-20 09:02:05 +05:30
debpalash c198a8349a fix: allow bounded CPU synthesis time 2026-08-20 09:01:43 +05:30
debpalash c4f3ca457d Merge latest PR #1599 original-only normalization 2026-08-20 09:00:43 +05:30
debpalash e1e3a477a7 fix: normalize original-only dub exports 2026-08-20 09:00:34 +05:30
debpalash e20add344c Merge PR #1601: close integration review findings 2026-08-20 09:00:01 +05:30
debpalash 0a07202634 Merge PR #1603: dispatch foreign loops through serving loop 2026-08-20 08:57:25 +05:30
debpalash 3cf1007f28 fix(cloning): recover speech after silent trim 2026-08-20 08:57:08 +05:30
debpalash 1044483edf Merge latest PR #1599 trusted output paths 2026-08-20 08:56:29 +05:30
debpalash a04c972b71 fix: keep dub export paths trusted 2026-08-20 08:56:20 +05:30
debpalash 8c1afe6d9d Merge PR #1602: harden packaged frontend recovery 2026-08-20 08:56:16 +05:30
debpalash 809314a459 fix(cloning): select densest bounded speech passage 2026-08-20 08:56:09 +05:30
debpalash 2dcfd0bb55 Merge remote-tracking branch 'contributor/fix/watermark-prefetch-cold-start' into fix/integration-eventbus-loop-clean 2026-08-20 08:53:51 +05:30
debpalash 93616a9c2a fix(watermark): fail open while pool drains 2026-08-20 08:53:42 +05:30
debpalash 4072ec3db4 fix(desktop): retain live shell during backup cleanup 2026-08-20 08:52:36 +05:30
debpalash 0548386cb3 Merge latest PR #1599 effective default fix 2026-08-20 08:52:15 +05:30
debpalash 45ec840ead fix(dubbing): resolve effective export track once 2026-08-20 08:52:07 +05:30
debpalash fcc6e4a843 fix(cloning): retain active speech in bounded references 2026-08-20 08:51:04 +05:30
debpalash 99a98eaefe Merge trusted #1599 download labels 2026-08-20 08:50:07 +05:30
debpalash f72439d6cf fix(security): use trusted audio download labels 2026-08-20 08:49:57 +05:30
debpalash a355ad4ab6 Merge remote-tracking branch 'contributor/fix/event-bus-threadpool-emit' into fix/integration-eventbus-loop-clean 2026-08-20 08:49:38 +05:30
debpalash daefad8769 Merge remote-tracking branch 'contributor/fix/watermark-prefetch-cold-start' into fix/integration-eventbus-loop-clean
# Conflicts:
#	CHANGELOG.md
2026-08-20 08:49:38 +05:30
debpalash afe013a6bc fix(events): dispatch foreign loops through serving loop 2026-08-20 08:48:13 +05:30
debpalash f0764532e2 Merge PR #1601 follow-up: keep display labels outside path analysis 2026-08-20 08:47:11 +05:30
debpalash d11d608c2d Merge latest PR #1599 CodeQL fix 2026-08-20 08:46:58 +05:30
debpalash 109199e024 fix(security): keep download labels out of path boundary 2026-08-20 08:46:46 +05:30
debpalash 9d133870e5 fix(watermark): isolate lifecycle state between tests 2026-08-20 08:46:29 +05:30
debpalash 0d9a392e8d test(desktop): resolve Bash portably 2026-08-20 08:45:27 +05:30
debpalash e6284a5d5f fix(desktop): recover interrupted frontend swap 2026-08-20 08:45:27 +05:30
debpalash 69567e2e56 Merge PR #1601: fix integration migration and initial-load findings 2026-08-20 08:45:26 +05:30
debpalash 8e98e7a1be fix(frontend): preserve migration and retry semantics 2026-08-20 08:44:48 +05:30
debpalash b0785c4e6d Merge updated PR #1599 into integration findings 2026-08-20 08:43:59 +05:30
debpalash 5baf82bfb9 fix(security): validate audio export labels 2026-08-20 08:43:30 +05:30
debpalash f5d33aad8c Merge PR #1599: default exported video to dubbed audio
# Conflicts:
#	CHANGELOG.md
2026-08-20 08:35:57 +05:30
debpalash 539309ea84 fix(cloning): bound long reference audio 2026-08-20 08:35:21 +05:30
debpalash e450a37d4b fix(security): separate export labels from paths 2026-08-20 08:34:38 +05:30
debpalash 77b66abd94 Merge PR #1577 follow-up: serialize watermark pool restarts 2026-08-20 08:34:33 +05:30
debpalash 1a39061849 test(watermark): isolate executor lifecycle 2026-08-20 08:33:39 +05:30
debpalash dd1aa3654d fix(dubbing): default exports to dubbed audio 2026-08-20 08:26:57 +05:30
debpalash 8430f9843c Merge PR #1597: ship LAN web UI in desktop installs
# Conflicts:
#	CHANGELOG.md
2026-08-20 08:21:46 +05:30
debpalash 3299d5986b Merge PR #1596: accept omnivoice ASR alias on ROCm
# Conflicts:
#	CHANGELOG.md
2026-08-20 08:21:28 +05:30
debpalash a53ddc35a8 Merge PR #1595: isolate desktop cleanup script tests 2026-08-20 08:21:11 +05:30
debpalash 62eddfad41 Merge PR #1584: improve dubbing detection, timing, and layout
# Conflicts:
#	CHANGELOG.md
2026-08-20 08:21:08 +05:30
debpalash 5462eeacaa Merge PR #1562: deliver sync endpoint WebSocket events 2026-08-20 08:20:52 +05:30
debpalash 9aa01d4e72 Merge PR #1577: background-prefetch AudioSeal watermark generator 2026-08-20 08:20:48 +05:30
debpalash 17d181e427 fix(watermark): reopen pool per app lifespan 2026-08-20 08:18:24 +05:30
debpalash 1762c57355 fix(watermark): block replacement during shutdown 2026-08-20 08:14:03 +05:30
Palash Debnath e3eda1af2c Merge branch 'main' into fix/issue-1582-omnivoice-asr-alias 2026-08-20 02:42:29 +00:00
debpalash 40b3c4f460 Merge remote-tracking branch 'origin/main' into pr-1584 2026-08-20 08:11:51 +05:30
debpalash eed841a8ca fix(bootstrap): replace packaged frontend safely 2026-08-20 08:09:15 +05:30
debpalash 49ab178db7 Merge remote-tracking branch 'origin/main' into codex/pr1577 2026-08-20 08:04:29 +05:30
debpalash 3482399197 fix(watermark): bound executor shutdown 2026-08-20 08:04:28 +05:30
debpalash df4d016a7d Merge remote-tracking branch 'origin/main' into fix/1589-packaged-lan-ui 2026-08-20 08:03:24 +05:30
debpalash 128b07c923 Merge remote-tracking branch 'origin/main' into pr-1562 2026-08-20 08:03:13 +05:30
debpalash fd6d21401b Merge remote-tracking branch 'origin/main' into fix/1589-packaged-lan-ui
# Conflicts:
#	CHANGELOG.md
2026-08-20 07:51:49 +05:30
debpalash c9adcb2647 fix(sharing): bundle LAN frontend in desktop installs 2026-08-20 07:50:50 +05:30
debpalash e6d3103ba8 Merge remote-tracking branch 'origin/main' into codex/pr1577
# Conflicts:
#	CHANGELOG.md
2026-08-20 07:50:37 +05:30
debpalash 1e6de9155b Merge remote-tracking branch 'origin/main' into pr-1562 2026-08-20 07:50:24 +05:30
debpalash a41dc8bcac fix(asr): accept omnivoice alias on ROCm 2026-08-20 07:49:43 +05:30
debpalash a8c5ce5c31 Merge remote-tracking branch 'origin/main' into pr-1584
# Conflicts:
#	CHANGELOG.md
2026-08-20 07:49:10 +05:30
debpalash e0e19f3dc9 Merge remote-tracking branch 'origin/main' into codex/pr1562 2026-08-20 07:36:37 +05:30
debpalash b73f31b237 test(dub): align persistence and review coverage 2026-08-20 07:33:18 +05:30
debpalash 6bcd3429ac Merge remote-tracking branch 'origin/main' into pr-1584 2026-08-20 07:30:04 +05:30
debpalash 81b6bbc4d3 Merge remote-tracking branch 'origin/main' into codex/pr1577 2026-08-20 07:29:44 +05:30
debpalash def15b8423 fix(dub): address review findings for responsive dubbing 2026-08-20 06:23:40 +05:30
debpalash 4df7d4e97e fix(watermark): make preload local-only and drain shutdown 2026-08-20 06:14:27 +05:30
debpalash 42b63488e9 Merge remote-tracking branch 'origin/main' into pr-1584 2026-08-20 06:12:15 +05:30
debpalash ee3e87c0a7 Merge remote-tracking branch 'origin/main' into codex/pr1562
# Conflicts:
#	CHANGELOG.md
2026-08-20 06:09:23 +05:30
debpalash b8f1d7f19d Merge remote-tracking branch 'origin/main' into codex/pr1577
# Conflicts:
#	CHANGELOG.md
2026-08-20 06:07:47 +05:30
victordonat0 155b9345b9 Improve dubbing speaker detection, timing, and responsive UI 2026-08-19 02:57:18 +00:00
Paolo Antinori 37c5df6f3a refactor(watermark): /simplify round — grace in one critical section, _env_float
Four-angle /simplify on the cumulative branch diff:

- The idle-reaper grace flag now lives entirely inside _get_generator's
  lock: the prefetch claims it only when THAT call builds the model, and
  every other getter call consumes it. This deletes the duplicated
  call-site clears in embed/detect (detect no longer touches the
  generator's grace at all — it was clearing a flag for a model it never
  uses), and closes the lock-gap window where the prefetch's claim could
  land on an already-used model, which the old comment claimed was
  impossible.

- Shared _env_float(name, default) for main.py's three inline float-env
  parsers (capture delay, watermark delay, MCP start timeout): one
  NaN/negative-rejecting implementation instead of three drifting
  copies; the older two lacked the isfinite guard entirely.

- Test cleanups: dead isinstance-Future assert half removed, the
  fake-audioseal Event-wait simplified to sleep, the reaper-diversion
  guard simplified to a plain no-op lambda, stale setdefault sentence
  dropped from the conftest comment.

Skipped with reason: merging the double will_mark() gate (they guard
different invariants — pool creation vs model load, both tested) and
hoisting the reaper guard to conftest (an autouse module-attr patch
would break tests that verify release_idle_models directly).
2026-08-18 07:33:27 +02:00
Paolo Antinori 3be001f3fd fix(watermark): close the get_watermark_pool None race + assert the grace
CodeRabbit on 28c7bace:

1. (Major) get_watermark_pool's double-checked pattern re-read the
   global after an unlocked null-check, so shutdown_watermark_pool's
   reset could land in between and the caller received None. The
   executor is now captured and returned under _watermark_pool_lock.

2. (Minor) the idle-grace test overwrote _prefetched_unused after the
   embed call, making the embed's clearing unobservable — a failing
   embed would have passed unnoticed. It now asserts the flag directly,
   and a guard diverts any leaked idle reaper (idle_worker resolves
   release_idle_models per call) to a no-op for the test's duration.
2026-08-18 07:12:31 +02:00
Paolo Antinori 28c7bacefb test(watermark): make the idle-grace test immune to a leaked idle reaper
Second CI red on the same test, different assert: the conftest fix killed
the leaked PRELOAD task, but a test lifespan that exits without shutdown
also leaves idle_worker running, and idle_worker calls
release_idle_models on these same module globals from another thread —
re-stamping _last_used mid-test. Each phase of the test now re-
establishes its preconditions immediately before its release call and
pins now= to a far-future monotonic, so an interleaved reaper tick
cannot change the outcome. Verified against the full 5801-test suite
run in one process.
2026-08-17 21:04:47 +02:00
Paolo Antinori fbb258d2e2 fix(watermark): rebuild the pool after the shutdown drain + review round
The shutdown drain killed the module singleton with no replacement, so
any process that keeps running after a lifespan shutdown — the CI suite
does exactly this — dead-submitted on the next watermark op: "cannot
schedule new futures after shutdown" (CI red; independently confirmed
by Greptile P1, CodeRabbit Major, and the plugin code review at 95/100
confidence). shutdown_watermark_pool() now resets the singleton under
its build lock before draining, so the next get_watermark_pool() hands
out a live replacement. Regression test covers
drained-pool-refuses + replacement-accepts.

Same round, minor findings: the drain's except now logs with exc_info
instead of a bare pass (GHAS CodeQL empty-except); the watermark delay
knob rejects negative/non-finite overrides (CodeRabbit); conftest sets
OMNIVOICE_PRELOAD_WATERMARK=0 unconditionally so a stray export from
the runner shell cannot re-enable background warm-ups mid-suite
(CodeRabbit).
2026-08-17 19:56:32 +02:00
Paolo Antinori 6837ba25ac fix(watermark): review follow-ups for #1577 — CI red + bot findings
Two CI failures, both understood:

1. test_shutdown_preload_race_1000 pins the production _cancel_and_await
  _tasks call site by regex; the new fifth handle broke the pattern. The
   guard now pins all FIVE handles (its property — every preload handle
   awaited under one generous bound — is unchanged).

2. test_prefetched_model_gets_one_extra_idle_window flaked only in the
   full suite: many tests boot the app lifespan, and any that exits
   without a lifespan shutdown leaves the deferred watermark-preload
   task pending — 35s later it fires mid-suite in another thread and
   re-stamps _last_used under whatever test is running. conftest now
   defaults OMNIVOICE_PRELOAD_WATERMARK=0 for the test session (a test
   can still opt in), and the grace test neutralizes will_mark so a
   leaked warm-up can't touch it.

Bot findings: Greptile P1 + CodeRabbit — cancelling the preload task
doesn't stop a watermark-pool thread already inside the ~42s cold
import, and nothing drained that pool at shutdown (only the GPU pool
was reset). Shutdown now drains the watermark pool's queue
(shutdown(wait=False, cancel_futures=True)) — bounded abandon, same
documented reality that Python can't kill a running thread. CodeRabbit
Major: the warm-up reads its own delay knob
(OMNIVOICE_PRELOAD_WATERMARK_DELAY, default 35s) instead of reusing the
capture-ASR delay, so a capture env override no longer retimes it.
CodeRabbit Minor: the _prefetched_unused claim/clear transitions now
happen under _generator_lock, so the retention grace can't be granted
to a model that has actually been used; the test fixture resets all
lifecycle globals.

Skipped with reason: gating prefetch on local-checkpoint presence — the
warm-up downloads only what the first embed would download anyway;
time-shifting that download is the feature, not a new network call.
2026-08-17 15:26:49 +02:00
Paolo Antinori 366b55d9d1 perf(watermark): background-prefetch the AudioSeal generator at startup
The first mark_synthetic serialized the audioseal import plus the
generator load INSIDE the first synthesis — measured at ~42s inline on
a cold filesystem (macOS, 2026-08-17 report), pushing a cold first
synthesis to ~87s and 3s past a 90s client timeout. The generator now
warms on a background task ~35s after boot (+5s past the capture-ASR
warm so the two cold imports don't contend), on the watermark pool,
cancellable at shutdown (OMNIVOICE_PRELOAD_WATERMARK=0 opts out; the
pool is only created when will_mark() says watermarking is active, and
setup-half failures log immediately instead of surfacing at shutdown).

Because the prefetch thread races the first embed, the lazy builds now
hold per-model locks — one build per model, no cross-blocking: a
detector load no longer queues behind a ~42s generator build, and
release_idle_models takes both locks in a fixed order. A
prefetch-warmed, never-used generator survives ONE extra idle-reaper
window so a first synthesis shortly after boot still finds it warm;
real embed/detect use clears the grace.

Also: embed/detect failures now log the full traceback (exc_info). The
catch-all printed only the message, which today left a
ModuleNotFoundError('getopt') inside AudioSeal's forward undiagnosable
from the log — audio silently ships unmarked when this fires.
2026-08-17 14:54:13 +02:00
Paolo Antinori be007e9d77 style(frontend): oxfmt useAppData (CI format check) 2026-08-15 15:19:49 +02:00
Paolo Antinori 030bc47515 fix(lint): makeLoader can't call useCallback (rules-of-hooks) — generation state in a ref instead 2026-08-15 15:08:20 +02:00
Paolo Antinori 31d4db65e8 Merge remote-tracking branch 'upstream/main' into fix/event-bus-threadpool-emit
# Conflicts:
#	CHANGELOG.md
#	frontend/src/hooks/useAppData.js
2026-08-15 14:49:11 +02:00
Paolo Antinori fd30a6c4ad fix(review): last-write-wins loaders + no sleep-polling in the emit test
CodeRabbit #1562 findings, both real:

- makeLoader is now generation-guarded: the initial retry loop overlaps
  freely with WS-triggered reloads, and a slow in-flight response could
  resolve AFTER a fresher reload and overwrite its list with stale data.
  Each invocation bumps a generation; only the newest may setState.
- The regression test awaited the queue via sleep-polling; it now uses
  asyncio.wait_for(q.get()) so a failure surfaces as TimeoutError instead
  of depending on 10ms poll timing (repo rule: no sleeps as sync).
2026-08-15 14:10:38 +02:00
Paolo Antinori dcaed7cbf4 docs(changelog): one-line Unreleased entry with issue ref + credit (Greptile P2) 2026-08-15 13:54:13 +02:00
Paolo Antinori 9615cd5294 fix(events): sync endpoints dropped their WS events — rename/delete left every open tab stale
PUT/DELETE /profiles (rename, delete, revoke consent) and the history/export
mutators are sync FastAPI endpoints: their bodies run in threadpool workers
where asyncio.get_running_loop() raises, so event_bus.emit() hit the
RuntimeError branch and silently dropped the "profiles" event. The UI only
refetches the voice list on that event, so after a rename the list kept stale
state, and a reload during that window could land on an empty panel (no
retry on the initial load either) — which reads to a user as "all my voices
are gone" even though nothing was deleted.

emit() now captures the serving loop in subscribe() and hands off from
foreign threads via call_soon_threadsafe (async callers are unchanged).
Also: the initial list loads in useAppData retry until FIRST success via
retryInitialLoad — a WS-triggered reload failure still keeps the previous
list, but the first load has nothing to keep. Loaders gained {rethrow: true}
for the initial path so the retry actually engages (they swallow errors by
design elsewhere); an integration test pins that wiring.

Tests: tests/test_event_bus_thread_emit.py fails on the old emit (verified
by stashing the fix) and passes with it; a live two-instance probe confirmed
PUT rename → WS event arrives on the fixed build and never on the original.
2026-08-15 13:49:47 +02:00
Paolo Antinori e4c1ef0de6 Merge remote-tracking branch 'upstream/main'
# Conflicts:
#	.gitignore
2026-08-13 22:06:19 +02:00
Paolo Antinori 5229a9504c chore: untrack local backlog/ task tracker (gitignored) 2026-07-30 16:10:36 +02:00
Paolo Antinori 214a859344 Merge remote-tracking branch 'upstream/main' 2026-07-30 06:34:28 +02:00
Paolo Antinori b72436a4e5 Merge remote-tracking branch 'upstream/main' 2026-07-29 15:49:36 +02:00
Paolo Antinori c2955dbe92 chore(backlog): initialize Backlog.md project structure 2026-07-28 17:58:56 +02:00
Paolo Antinori a6f008ec38 docs(backlog): investigate VRAM/lifecycle bug + durable-fix exploration
TASK-1: stuck model loads (>1200s), unkillable abandoned workers holding device
TASK-2: exploration of durable fixes (flush caches, CPU engine, timeout, shorter text)
2026-07-28 17:46:16 +02:00
647 changed files with 85335 additions and 10770 deletions
+4 -1
View File
@@ -5,6 +5,9 @@ description: "Local TTS, voice cloning, voice design, and video dubbing via the
# VoiceStudio
The canonical cross-agent package lives at `skills/omnivoice/SKILL.md`. This
Claude-specific package retains the MCP lifecycle helpers and references.
## Overview
Generate audio locally via the VoiceStudio MCP server. Tools: `generate_speech`, `list_voices`, `list_personalities`, `list_languages`, `check_health`. Resources: `voice://{id}`, `history://recent`.
@@ -166,4 +169,4 @@ The MCP server does not expose the dubbing endpoint. The full transcribe → tra
Backend Swagger / OpenAPI: `http://127.0.0.1:3900/docs` (when backend is up).
Upstream: github.com/debpalash/VoiceStudio — FSL-1.1-ALv2 (free for personal/internal/non-commercial; auto-converts to Apache-2.0 two years after each release).
Upstream: github.com/debpalash/VoiceStudio. The app uses AGPL-3.0-only; optional engines and downloaded models retain their own licenses. See `LICENSE-NOTICE.md` in the repository.
@@ -1,6 +1,6 @@
#!/usr/bin/env bash
# Start the OmniVoice FastAPI backend on 127.0.0.1:3900, detached, idempotent.
# Honors $OMNIVOICE_HOME (default ~/OmniVoice-Studio).
# Honors $OMNIVOICE_HOME (default ~/VoiceStudio).
#
# Exit codes:
# 0 success (already running, or freshly started + healthy within 60s)
@@ -11,7 +11,7 @@
set -euo pipefail
HOME_DIR="${OMNIVOICE_HOME:-$HOME/OmniVoice-Studio}"
HOME_DIR="${OMNIVOICE_HOME:-$HOME/VoiceStudio}"
URL="${OMNIVOICE_API_URL:-http://127.0.0.1:3900}"
LOG="$HOME_DIR/backend.log"
+1
View File
@@ -40,6 +40,7 @@ sudo apt-get install -y \
libwebkit2gtk-4.1-dev libgtk-3-dev libpango1.0-dev libcairo2-dev \
libsoup-3.0-dev libgdk-pixbuf-2.0-dev \
libayatana-appindicator3-dev librsvg2-dev libssl-dev libxdo-dev \
gstreamer1.0-plugins-good \
libasound2-dev build-essential curl wget file
```
+1 -1
View File
@@ -4,7 +4,7 @@
| Version | Supported |
|---------|-----------|
| 0.3.x (latest release + `main` previews) | ✅ Current — all fixes land here |
| 0.5.x (latest release + `main` previews) | ✅ Current — all fixes land here |
| 0.2.7 | ⚠️ Legacy stable — security fixes only, upgrade recommended |
| < 0.2.7 | ❌ No longer supported |
+15 -1
View File
@@ -51,6 +51,13 @@ jobs:
- os: ubuntu-latest
platform: linux-x86_64
experimental: false
- os: ubuntu-24.04-arm
platform: linux-aarch64
# Apple Silicon under Asahi Linux. Experimental: the Vulkan
# (Honeykrisp GPU) build path is new and the hosted arm64
# runner has no GPU — it validates that the binary builds;
# on-host Vulkan acceleration is exercised by users.
experimental: true
- os: windows-latest
platform: windows-x86_64
experimental: false
@@ -80,11 +87,18 @@ jobs:
# Linux-only: upstream `buildcpu.sh` enables `-DGGML_BLAS=ON` which
# requires a system BLAS implementation at cmake configure time.
- name: Linux system deps (BLAS for ggml-blas backend)
if: matrix.platform == 'linux-x86_64'
if: startsWith(matrix.platform, 'linux')
run: |
sudo apt-get update
sudo apt-get install -y libopenblas-dev pkg-config
# linux-aarch64: let the build script's Vulkan path (Honeykrisp GPU
# on Asahi) engage instead of silently falling back to CPU.
- name: Vulkan dev deps (linux-aarch64 GPU backend)
if: matrix.platform == 'linux-aarch64'
run: |
sudo apt-get install -y glslc libvulkan-dev spirv-headers
- name: Build omnivoice-tts
shell: bash
# Pass values through env (quoted) rather than ${{ }} interpolation
+31 -1
View File
@@ -11,6 +11,11 @@ on:
push:
branches: [main]
workflow_dispatch:
inputs:
windows_wix_diagnostic:
description: Run only the tiny nonpublishing Windows MSI authoring diagnostic
type: boolean
default: false
permissions:
contents: read
@@ -21,6 +26,7 @@ env:
jobs:
test:
if: ${{ !inputs.windows_wix_diagnostic }}
name: Tests (backend + frontend)
runs-on: ubuntu-22.04
env:
@@ -189,6 +195,7 @@ jobs:
# and `cargo test --lib` runs the shell's unit tests natively on each OS.
# Full bundling stays in release.yml on tag push.
tauri-cross-platform:
if: ${{ !inputs.windows_wix_diagnostic }}
name: Tauri shell check (${{ matrix.label }})
needs: test
strategy:
@@ -289,6 +296,7 @@ jobs:
# job above misses. Narrow scope (tests/smoke/ only) — full pytest stays
# on Linux until Phase 1's INST-01 lands setuptools for WhisperX.
smoke-matrix:
if: ${{ !inputs.windows_wix_diagnostic }}
name: Smoke (${{ matrix.label }})
needs: test
strategy:
@@ -428,8 +436,9 @@ jobs:
PY
- name: Run smoke tests
# Exercise credential paths on native Windows as well as POSIX hosts.
if: matrix.backend_supported
run: uv run --no-sync pytest tests/smoke/ -q --tb=short
run: uv run --no-sync pytest tests/smoke/ tests/test_hf_token_cache_paths.py -q --tb=short
env:
HF_HUB_OFFLINE: "1" # same no-silent-downloads guard as the main pytest job
HF_HUB_CACHE: ${{ runner.temp }}/pockettts-empty-hf-cache
@@ -442,3 +451,24 @@ jobs:
env:
HF_HUB_OFFLINE: "1"
HF_HUB_CACHE: ${{ runner.temp }}/worker-artifact-empty-hf-cache
windows-wix-diagnostic:
name: Windows MSI authoring (no publishing)
needs: test
if: ${{ !cancelled() && (inputs.windows_wix_diagnostic || needs.test.result == 'success') }}
runs-on: windows-2022
timeout-minutes: 15
steps:
- uses: actions/checkout@v4
- uses: oven-sh/setup-bun@v1
- name: Bundle canonical system and per-user templates with a tiny payload
shell: pwsh
run: ./scripts/diagnose-windows-wix.ps1
- name: Preserve verbose linker output and rendered authoring
if: always()
uses: actions/upload-artifact@v4
with:
name: windows-wix-diagnostic
path: wix-diagnostic-artifacts/
if-no-files-found: warn
retention-days: 3
+127
View File
@@ -0,0 +1,127 @@
# Installer smoke — runs scripts/install.sh / scripts/install.ps1 end-to-end
# on all three desktop platforms so the one-liner installers can't rot.
#
# Gated by `paths` because a cold run downloads multi-GB wheels (torch) and
# takes ~15-30 min per OS; it only needs to fire when an installer or this
# workflow changes. The heavy Tauri bundles stay in release.yml (tag push).
name: Install smoke
on:
pull_request:
paths:
- "scripts/install.sh"
- "scripts/install.ps1"
- ".github/workflows/install-smoke.yml"
push:
branches: [main]
paths:
- "scripts/install.sh"
- "scripts/install.ps1"
- ".github/workflows/install-smoke.yml"
workflow_dispatch:
permissions:
contents: read
env:
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: true
jobs:
install:
name: Install (${{ matrix.os }})
runs-on: ${{ matrix.os }}
timeout-minutes: 60
strategy:
fail-fast: false
matrix:
os: [ubuntu-22.04, macos-latest, windows-latest]
steps:
- uses: actions/checkout@v4
# Running `sh scripts/install.sh` from the repo root exercises the
# repo-root resolution (script dir is scripts/, project root one level
# up) — the exact bug that made a local run clone a duplicate repo.
# Binary mode is the default: prebuilt release asset, checksum verified.
- name: Run installer — binary (macOS/Linux)
if: runner.os != 'Windows'
run: sh scripts/install.sh
- name: Verify install — binary (macOS/Linux)
if: runner.os != 'Windows'
run: |
if [ "$(uname)" = "Darwin" ]; then
test -d "/Applications/VoiceStudio.app" || { echo "::error::VoiceStudio.app missing from /Applications"; exit 1; }
echo "✓ VoiceStudio.app installed in /Applications"
else
test -x "$HOME/.local/bin/VoiceStudio" || { echo "::error::AppImage missing from ~/.local/bin"; exit 1; }
"$HOME/.local/bin/VoiceStudio" --appimage-help >/dev/null 2>&1 || true
echo "✓ AppImage installed and executable"
fi
# Source mode stays covered end-to-end behind --source.
- name: Run installer — source (macOS/Linux)
if: runner.os != 'Windows'
run: sh scripts/install.sh --source
- name: Verify install — source (macOS/Linux)
if: runner.os != 'Windows'
working-directory: ${{ github.workspace }}
run: |
test -d .venv || { echo "::error::.venv missing"; exit 1; }
test -f frontend/dist/index.html || { echo "::error::frontend build missing"; exit 1; }
echo "✓ venv + frontend bundle present"
# Binary mode is the default; CI runs msiexec silently.
- name: Run installer — binary (Windows)
if: runner.os == 'Windows'
env:
CI: true
shell: pwsh
run: '& { $ErrorActionPreference = "Stop"; & "${{ github.workspace }}\scripts\install.ps1" }'
- name: Verify install — binary (Windows)
if: runner.os == 'Windows'
shell: pwsh
run: |
$paths = @(
"HKLM:\Software\Microsoft\Windows\CurrentVersion\Uninstall\*",
"HKLM:\Software\WOW6432Node\Microsoft\Windows\CurrentVersion\Uninstall\*",
"HKCU:\Software\Microsoft\Windows\CurrentVersion\Uninstall\*"
)
$key = Get-ItemProperty $paths -ErrorAction SilentlyContinue |
Where-Object { $_.DisplayName -match "VoiceStudio|OmniVoice" } |
Select-Object -First 1
if (-not $key) {
Get-ItemProperty $paths -ErrorAction SilentlyContinue |
Where-Object DisplayName | ForEach-Object { Write-Host " installed: $($_.DisplayName)" }
Write-Host "::error::MSI product not registered"; exit 1
}
Write-Host "✓ MSI product registered: $($key.DisplayName)"
# Source mode stays covered end-to-end behind -Source.
- name: Run installer — source (Windows)
if: runner.os == 'Windows'
env:
VOICESTUDIO_INSTALL_MODE: source
shell: pwsh
run: '& { $ErrorActionPreference = "Stop"; & "${{ github.workspace }}\scripts\install.ps1" }'
- name: Verify install — source (Windows)
if: runner.os == 'Windows'
shell: pwsh
run: |
if (-not (Test-Path .venv)) { Write-Host "::error::.venv missing"; exit 1 }
if (-not (Test-Path frontend\dist\index.html)) { Write-Host "::error::frontend build missing"; exit 1 }
Write-Host "✓ venv + frontend bundle present"
- name: Upload install log on failure
if: failure()
uses: actions/upload-artifact@v4
with:
name: install-log-${{ matrix.os }}
path: |
/Users/runner/Library/Application Support/OmniVoice/*.log
/home/runner/.local/share/VoiceStudio/*.log
${{ runner.temp }}/VoiceStudio/**/*.log
if-no-files-found: ignore
+91 -25
View File
@@ -148,15 +148,20 @@ jobs:
preview-gate:
name: Preview gate
runs-on: ubuntu-22.04
permissions:
contents: read
outputs:
is_preview: ${{ steps.decide.outputs.is_preview }}
proceed: ${{ steps.decide.outputs.proceed }}
stable_tag: ${{ steps.decide.outputs.stable_tag }}
steps:
- uses: actions/checkout@v4
with:
fetch-depth: 50
- id: decide
shell: bash
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
set -euo pipefail
event="${{ github.event_name }}"
@@ -171,6 +176,13 @@ jobs:
exit 1
fi
echo "is_preview=true" >> "$GITHUB_OUTPUT"
# Resolve once before the matrix starts so every platform stamps
# against the same immutable Stable-channel snapshot.
STABLE_TAG=$(gh release view --repo "$GITHUB_REPOSITORY" --json tagName --jq .tagName)
[[ "$STABLE_TAG" =~ ^v[0-9]+\.[0-9]+\.[0-9]+$ ]] || {
echo "::error::latest stable release has an invalid tag"; exit 1;
}
echo "stable_tag=$STABLE_TAG" >> "$GITHUB_OUTPUT"
else
echo "is_preview=false" >> "$GITHUB_OUTPUT"
fi
@@ -497,35 +509,22 @@ jobs:
echo "APPLE_TEAM_ID=$TID"
} >> "$GITHUB_ENV"
# Stamp each preview build with a unique, monotonically increasing semver
# PRERELEASE so the updater actually offers it (a rolling preview that
# always reported the static 0.3.0 never looked "newer", so no update was
# ever delivered). Ephemeral, CI-only — never committed. Tauri reads the
# bundle + updater version from tauri.conf.json, so rewriting it here
# stamps the artifacts + latest.json. Under the versioning hard rule
# (owner-set 2026-06-11) main is always last-release + 1, so BASE-N is a
# prerelease of the NEXT version and semver-sorts ABOVE the last stable
# (0.3.6-N > 0.3.5) — preview users naturally upgrade past stable, and
# the Windows MSI ProductVersion (which strips the prerelease → 0.3.6)
# is also correctly above the last stable.
# Stamp each preview with a numeric prerelease that is strictly above the
# latest stable release. Main may intentionally retain the released
# version while AUTO_VERSION_BUMP is disabled; in that case the helper
# advances the preview base by one patch so stable users can still opt in
# and receive it. The edit is ephemeral and never committed.
- name: Stamp preview version
if: needs.preview-gate.outputs.is_preview == 'true'
shell: bash
env:
STABLE_TAG: ${{ needs.preview-gate.outputs.stable_tag }}
run: |
set -euo pipefail
# package.json is the single source of truth; tauri.conf.json reads its
# version from it ("version": "../package.json"), so stamping
# package.json restamps the whole bundle.
CONF=frontend/package.json
BASE=$(jq -r .version "$CONF")
# MSI/WiX requires the semver pre-release identifier to be numeric-only
# (and <= 65535). "preview.N" hard-fails the Windows bundler, so the
# preview stamp is BASE-N — still sorts below the stable BASE for the
# updater, still unique per run.
PREVIEW_VERSION="${BASE}-${{ github.run_number }}"
tmp=$(mktemp)
jq --arg v "$PREVIEW_VERSION" '.version = $v' "$CONF" > "$tmp"
mv "$tmp" "$CONF"
PREVIEW_VERSION=$(python scripts/stamp-preview-version.py \
--package-json frontend/package.json \
--stable-tag "$STABLE_TAG" \
--run-number "${{ github.run_number }}")
echo "Stamped preview version: $PREVIEW_VERSION"
# The rolling `preview` release is REUSED every night, and macOS updater
@@ -605,6 +604,21 @@ jobs:
fi
done < /tmp/stale.txt
# A retried job reuses its version and can collide with installers it
# uploaded before a later step failed. Keep other versions/arches intact;
# macOS versionless updater archives are scoped by release tag and arch.
- name: Clear this target's installer assets on retry
if: github.run_attempt > 1
shell: bash
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
RELEASE_TAG: ${{ (needs.preview-gate.outputs.is_preview == 'true') && 'preview' || github.ref_name }}
RELEASE_TARGET: ${{ matrix.rust_target }}
run: |
VERSION=$(python -c 'import json; print(json.load(open("frontend/package.json"))["version"])')
python scripts/clear-release-rerun-assets.py \
--tag "$RELEASE_TAG" --version "$VERSION" --target "$RELEASE_TARGET"
- name: Build + release (Tauri)
uses: tauri-apps/tauri-action@v0
env:
@@ -649,6 +663,47 @@ jobs:
updaterJsonPreferNsis: false
includeUpdaterJson: true
- name: Build per-user Windows MSI
if: runner.os == 'Windows'
shell: bash
working-directory: frontend
env:
TAURI_SIGNING_PRIVATE_KEY: ${{ secrets.TAURI_SIGNING_PRIVATE_KEY }}
TAURI_SIGNING_PRIVATE_KEY_PASSWORD: ${{ secrets.TAURI_SIGNING_PRIVATE_KEY_PASSWORD }}
run: |
set -euo pipefail
python ../scripts/render-per-user-wix.py \
--source src-tauri/wix/main.wxs \
--system-wxs src-tauri/target/${{ matrix.rust_target }}/release/wix/x64/main.wxs \
--output src-tauri/target/wix-per-user/main.wxs
bunx tauri build --target ${{ matrix.rust_target }} --bundles msi \
--config src-tauri/tauri.per-user.conf.json
DIR="src-tauri/target/${{ matrix.rust_target }}/release/bundle/msi"
while IFS= read -r artifact; do
safe=${artifact// (Current User)/_Current_User}
[ "$safe" = "$artifact" ] || mv "$artifact" "$safe"
done < <(find "$DIR" -maxdepth 1 -type f -name '*Current*User*.msi*')
- name: Publish per-user Windows updater channel
if: runner.os == 'Windows'
shell: bash
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
RELEASE_TAG: ${{ (needs.preview-gate.outputs.is_preview == 'true') && 'preview' || github.ref_name }}
run: |
set -euo pipefail
DIR="frontend/src-tauri/target/${{ matrix.rust_target }}/release/bundle/msi"
MSI=$(find "$DIR" -name '*Current*User*.msi' -type f | head -1)
[ -n "$MSI" ] || { echo "per-user MSI missing"; find "$DIR" -type f; exit 1; }
[ -f "$MSI.sig" ] || { echo "per-user MSI signature missing"; exit 1; }
VERSION=$(jq -r .version frontend/package.json)
python scripts/build_windows_user_manifest.py \
--repo "$GITHUB_REPOSITORY" --tag "$RELEASE_TAG" --version "$VERSION" \
--asset "$(basename "$MSI")" --signature-file "$MSI.sig" \
--output latest-user.json
gh release upload "$RELEASE_TAG" "$MSI" "$MSI.sig" latest-user.json \
--clobber --repo "$GITHUB_REPOSITORY"
# ── Installer smoke (Phase 0 GATE-03) ─────────────────────────────
# Structural verification of the installed/extracted bundle. The thin
# uv-venv installer ships NO frozen backend binary (the venv is built on
@@ -717,8 +772,9 @@ jobs:
shell: bash
run: |
set -euo pipefail
MSI=$(find frontend/src-tauri/target/${{ matrix.rust_target }}/release/bundle/msi -name "*.msi" | head -1)
MSI=$(find frontend/src-tauri/target/${{ matrix.rust_target }}/release/bundle/msi -name "*.msi" ! -name '*Current*User*' | head -1)
echo "Smoke-testing MSI: $MSI"
powershell.exe -NoProfile -ExecutionPolicy Bypass -File scripts/verify-windows-msi.ps1 -MsiPath "$(cygpath -w "$MSI")"
# /quiet = no UI, /norestart = don't reboot the runner if a dep asks
msiexec.exe //i "$(cygpath -w "$MSI")" //quiet //norestart
INSTALL="/c/Program Files/VoiceStudio"
@@ -731,6 +787,16 @@ jobs:
find "$INSTALL" -type f -path '*backend*main.py' | grep -q . || fail "backend source main.py missing"
echo "OK — MSI installed shell + uv + backend resources"
- name: Per-user installer smoke (Windows, non-admin account)
if: runner.os == 'Windows'
timeout-minutes: 8
shell: bash
run: |
set -euo pipefail
MSI=$(find frontend/src-tauri/target/${{ matrix.rust_target }}/release/bundle/msi -name '*Current*User*.msi' | head -1)
powershell.exe -NoProfile -ExecutionPolicy Bypass \
-File scripts/smoke-per-user-msi.ps1 -MsiPath "$(cygpath -w "$MSI")"
# linuxdeploy re-links .DirIcon as an ABSOLUTE symlink into the build
# machine AFTER tauri's files-map has placed the real icon bytes — the
# exact bug #1518 guarded against, resurfacing on the first real tag
+17
View File
@@ -159,7 +159,24 @@ tests/probe/reports/
# reports). Working notes for whoever is driving a change, not a repo artifact.
/remote/
# OmniVoice GGUF runtime build artifacts (scripts/build-omnivoice-tts.sh).
# Only 0-byte placeholders of omnivoice-tts-* are tracked; real binaries,
# the checksums manifest and the copied libggml shared libs ship via CI.
bin/libggml*
bin/checksums.sha256
bin/omnivoice-tts-linux-aarch64
# Dubbing-demo intermediates. The .mp4/.srt/manifest.json in this directory ARE
# committed (they ship with the app); the per-language source WAVs are just the
# inputs scripts/render_dub_demo_audio.py hands to scripts/build_dub_demo.sh.
backend/assets/samples/demo/dubbing/*.src.wav
# Stray sqlite session artifacts (`<db-path>.ses`). An in-memory DB yields the
# literal name `:memory:.ses`, and a path containing `:` cannot be checked out
# on Windows at all — committing one fails every Windows CI job at the git
# checkout step, before a single test runs. Guarded by
# tests/test_no_windows_hostile_paths.py.
*.ses
# Generated Windows MSI diagnostic logs and installer payloads
/wix-diagnostic-artifacts/
+2
View File
@@ -25,6 +25,8 @@ regexes = [
'''^hf_QWERTYUIOPasdfghjklZXCVBNM0123456789xyzAB$''',
# NLLB generation length argument, not the value of a credential.
'''^max_length=400$''',
# Dubbing pane split-position localStorage key, not a credential.
'''^omnivoice\.dubSplit\.v1$''',
# cryptography's Ed25519 private-key type name, not key material.
'''^Ed25519PrivateKey$''',
]
+221 -3
View File
@@ -10,20 +10,200 @@ the frozen-backend fallback mirror it for their toolchains.
**Highlights**
- The backend now answers within a second of launch and narrates its startup step by step
- Reporting a bug from an outdated build now offers the latest release first
- The backend is only announced ready once it can actually serve, and crash-loop restarts now pace themselves
- The desktop app builds and opens from a fresh clone again (#1818) — thanks @flutterkage2k!
- GPUs with less VRAM than the engine needs no longer get half the compute-time budget a CPU gets (#1806) — thanks @VishvakR!
- Gallery voice previews play again — the quality guard was rejecting good renders as silent (#1819) — thanks @flutterkage2k!
- Tilde-separated number ranges are spoken clearly without running their endpoints together (#1821) — thanks @flutterkage2k!
- Voice modes use themed tabs, with Synthesize and Convert pinned below their scrolling forms (#1823)
- Fix current-user Windows installer validation and nested resource cleanup (#1873)
- Keep generated frontend assets available while building the current-user Windows installer (#1873)
- Voice cloning now starts with a clear upload-or-record choice, reveals recording and reference details only when needed, and keeps sampling controls under Production Overrides (#1817)
- The first-run welcome line uses an instruction accepted by OmniVoice and VoiceDesign engines (#1861) — thanks @psiberfunk!
### Changed
- Casting uses responsive SVG voice cards and searchable speaker menus that stay above surrounding panels (#1823)
- Dubbing aligns output settings, brings review status forward, and simplifies transcript and glossary editing; Launchpad files and voices reflow into responsive grids (#1823)
- Transcript segments use three readable rows for text, timing/status and voice controls, with heights that adapt to wrapping (#1823)
- Dragging the waveform pans horizontally while a click still seeks, keeping the timed transcript aligned (#1823)
- Bulk segment editing uses searchable voice and language menus, readable language names and a responsive selection toolbar (#1823)
- Dubbing overlays playback controls on video, combines waveform and transcript in a compact timeline, and removes header/action background fills (#1823)
- Dubbing uses compact casting, translation and output controls with responsive rows to leave more room for editing (#1823)
- Export uses grouped format settings, themed track menus and switches, with a pinned filename summary and download action (#1823)
- Dubbing output settings use icon-labelled switches, themed track and speaker menus, and clearer timing/transcript controls (#1823)
- Casting voice menus use searchable themed options with SVG preset icons instead of native dropdowns (#1823)
- Dubbing groups casting and translation controls with readable labels, SVG icons, searchable menus, and compact timeline spacing (#1823)
- Production Overrides use readable icon-labelled controls and accessible Denoise/Postprocess switches (#1823)
- Expanded navigation uses a theme-accent tint with subtle static wave gradients (#1823)
- Convert groups source audio, target voice, and timing options into clearer controls; design choices include theme-matched SVG icons (#1823)
- The expandable sidebar reveals workspace labels with restrained active states; language menus adapt to multiple columns on wider screens (#1823)
- Voice design and recording use themed, keyboard-accessible selectors with clearer spacing and labels (#1823)
- Voice tabs and upload/record controls have subtle SVG motion; Text adds clipboard paste and the upload area fills available height (#1823)
- The title-bar label cycles through active speech, transcription, and LLM engines; bundled model labels correctly say OmniVoice (#1823)
- The top-bar Engines panel groups Speech, Transcription, and LLM choices into tabs, with compact memory controls and no duplicate pickers (#1823)
### Added
### Docs
### Fixed
- Install documentation help now prints correctly on Windows consoles using legacy encodings (#1815) — thanks @dajiaohuang!
- Saved transcriptions with missing or invalid timestamps now remain readable (#1799) — thanks @yunaremaia and @tvbht!
- Copying a saved transcription now uses the shared clipboard helper and reports failed copies accurately (#1803) — thanks @tvbht!
- Voice reference preparation reclaims allocator memory before one bounded retry, then reports persistent GPU out-of-memory failures (#1811)
- `bun run desktop` now opens on a fresh clone: the Vite alias for `@tauri-apps/plugin-dialog` no longer assumes a nested `frontend/node_modules`, which bun's workspace hoisting leaves empty (#1818) — thanks @flutterkage2k!
- Slow backend startups remain running with progress updates, and Retry interrupts startup without stale timeout failures (#1809)
- Backend connection errors report crashes only when recorded evidence exists, and diagnostic waits honor cancellation (#1810)
- A CUDA or ROCm GPU with less VRAM than the engine needs now gets the CPU compute-time budget instead of the shorter accelerated one, since it pages to system RAM and renders slower than the CPU would — applied to local generation, voice conversion, and remote worker deadlines alike (#1806) — thanks @VishvakR!
- Gallery previews no longer fail with "the voice engine returned no audible audio" on perfectly good renders: the degenerate-buzz guard measured spectral flatness over the whole clip (so the value tracked clip length) against a threshold calibrated on a synthetic signal, and rejected real speech in every language tested (#1819) — thanks @flutterkage2k!
- Speak tilde separators in integer, signed, and decimal ranges in English, Korean, Japanese, and Chinese (#1821) — thanks @flutterkage2k!
- Keep recording and conversion work safe while switching methods, synchronize dubbing language controls, and localize timeline controls and timing warnings (#1841)
- Dubbing playback starts before waveform decoding, automatic cast names are readable, and transcript timestamps have more room (#1823)
- The title-bar engine button stays compact and stable while cycling labels, with engine names aligned right (#1823)
- Long dubbing segment errors wrap in a bounded scrollable notice instead of widening the editor (#1823)
- Voice dropdowns match their field width, use theme accents, and show recent voices only once (#1823)
- Language menus no longer show a pale frame around their search header (#1823)
- The notification count stays inside the title bar instead of clipping above the bell (#1823)
- The workspace engine menu opens beside its button instead of at the opposite edge of the page (#1823)
- Cloning reuses the dubbing language picker with flags, search, and single selection, opening above the pinned synthesis controls (#1823)
- The first-run welcome line uses an instruction accepted by OmniVoice and VoiceDesign engines (#1861) — thanks @psiberfunk!
- The header status dot now honors OS Reduce Motion instead of pulsing regardless (#1862) — thanks @psiberfunk!
- Onboarding reads Hugging Face tokens locally, preserves Windows CLI logins, and requires successful discovery before replacing saved credentials (#1852) — thanks @psiberfunk!
- The logs panel no longer reports “All clear” before log retrieval succeeds or while logs contain warnings or errors (#1870) — thanks @motodriver!
- MOSS accelerator routing and status match runtime selection, with CPU fallback when device probing fails (#1830) — thanks @li-lizhe!
- Confucius accelerator routing tolerates failed device probes, and dots.tts keeps safe default precision on non-CUDA hosts (#1831) — thanks @li-lizhe!
- On macOS, the header status dot and kicker no longer render underneath the overlaid traffic lights (#1863) — thanks @psiberfunk!
- The capture widget can hide after recording and recover from being left visible while idle (#1865) — thanks @psiberfunk!
- macOS retains the shared desktop window sizing, resize limits, and file-drop behavior when native chrome is applied (#1865) — thanks @psiberfunk!
- On macOS, the header no longer shows Windows-style minimize/maximize/close buttons alongside the native traffic lights (#1865) — thanks @psiberfunk!
- Release retries replace their own partially uploaded installers without colliding with existing assets (#1871)
- Timed-out voice engines finish process cleanup before retrying, and old timeout callbacks cannot kill replacement engines (#1872)
- Fast macOS process exits no longer turn a completed shutdown into a permission error (#1809)
## [0.5.2] — 2026-09-02
**Highlights**
- Show estimated and measured model, dependency, cache, and temporary disk costs in the engine catalogue (#1718)
- Preview builds now stay newer than Stable even when automatic post-release version bumps are disabled (#1762)
- CosyVoice setup guidance now separates downloaded model files from the runtime that makes the engine available (#1761)
- MCP tools can now keep audio out of agent context by returning files and accepting base-path-confined file inputs (#1760) — thanks @agudmund!
- Hear a dub line as you type it — an opt-in live preview streams TTS for the edited segment (#1769) — thanks @mvanhorn!
- Studio gains a Convert method: re-say any clip in one of your saved voices, speech to speech, fully local (#1765) — thanks @mvanhorn!
- Hardsub video export gains an opt-in karaoke word-highlight caption style (#1764) — thanks @mvanhorn!
- The batch queue can now watch a folder: new videos dropped into it are dubbed automatically (#1768) — thanks @mvanhorn!
- The audiobook player now shows the chapter text and highlights the word being narrated (#1766) — thanks @mvanhorn!
- The dub editor gains a casting board: drag voice chips onto speakers, dropdowns stay in sync (#1767) — thanks @mvanhorn!
### Changed
- Voice Design simplified: the 12-row fine-grained block collapses to one summary line with a five-field editor, English accent and Chinese dialect merge into a single field, and the starting-point chips now show 5 with an overflow toggle (#1793)
### Added
- The audiobook result is now a synced-lyrics player: chapter text follows playback with the current word highlighted and click-to-seek, timed from the render's own chapter durations with a karaoke-style even split — no ASR pass, fully local (#1766) — thanks @mvanhorn!
- The dub CAST strip expands into a project-level casting board: drag voice chips (clone profiles, design presets, Default) onto speaker rows — or pick from a keyboard listbox — writing the same per-speaker cast fields as the existing dropdowns (#1767) — thanks @mvanhorn!
- Studio's new Convert method turns a dropped or recorded clip into an existing voice profile's voice, with optional source-duration matching (#1765) — thanks @mvanhorn!
- Opt-in watch folder on the batch queue: pick a directory once and new videos are auto-enqueued with your last Add-to-queue settings, with pause/stop controls and copy-in-progress protection — files upload as bytes, paths never leave the app (#1768) — thanks @mvanhorn!
- Hardsub export can now burn karaoke word-highlight captions: an opt-in Line | Karaoke control renders a word-timed ASS sweep from timings persisted at transcription, with an even-split fallback for older jobs and translated tracks, plus a `GET /dub/ass/{job_id}` sidecar (#1764) — thanks @mvanhorn!
- Windows releases now include an independently updatable per-user MSI that installs and uninstalls without elevation (#1713)
- Dub segments can now stream live TTS while you edit a translated line — opt-in toggle, existing `/ws/tts` socket, shared generation admission, exports still render at full quality (#1769) — thanks @mvanhorn!
- Engine status and diagnostic bundles now record loaded execution provider, device, precision, fallback stage, accelerator identity, runtime versions, and parent-process memory visibility (#1717)
### Docs
- Local gigastt is now documented as a supported OpenAI-compatible ASR endpoint, with loopback privacy distinguished from remote servers (#1736) — thanks @ekhodzitsky!
- The CosyVoice guide now states that packaged builds have no one-click runtime installer and records the exact readiness checks exposed by [Discussion 1631](https://github.com/debpalash/VoiceStudio/discussions/1631) (#1761)
- A production private-API guide now covers pinned containers, root credentials, network isolation, streaming proxies, health checks, upgrades, and benchmark evidence (#1720)
- RX 6700 XT/gfx1031 over WSL2 ROCDXG is now explicitly unverified until a published end-to-end GPU workload proves the mapped path (#1716)
### Fixed
- The generation compute-time budget is now a Settings control (Performance & Device) instead of an env-var-only setting the timeout error recommended with no UI path — the error copy points there too, and long CPU/MPS renders get an upfront heads-up before they start (#1787)
- Windows: the backend can now start when the install path contains non-English characters (e.g. a CJK username) on a non-UTF-8 system code page — a new or broken Python environment now builds at an ASCII-safe path automatically (a healthy existing one is never relocated), and a specific error message names the cause and a working fix if the interpreter still crashes in `site` (#1783)
- Exports and other native-picker actions no longer 403 with "Invalid or expired desktop authorization" when the desktop app and backend resolve different data directories, e.g. dev mode or a custom data folder (#1781)
- Voice Design no longer lets you pick a Chinese dialect and an English accent together — the picker keeps them mutually exclusive instead of round-tripping a 400 (#1771)
- The desktop app no longer attaches to an already-running backend on version string alone: it now verifies the backend's actual code fingerprint too, so an orphaned or manually started backend reporting the current version but running older code (e.g. a stale `destination_path` export 422) gets replaced instead of adopted (#1770)
- Korean locale overhauled: 231 mistranslations corrected and all 493 missing keys translated (#1776) — thanks @j30231!
- Japanese "Cleaning…" clone status now reads as denoising instead of housekeeping (#1775) — thanks @j30231!
- The batch dubbing queue now has a UI entry point — a quiet link on the Dub landing (it was previously unreachable: the app switched on a mode nothing ever set) (#1768) — thanks @mvanhorn!
- OpenAI-compatible ASR now requires HTTPS outside loopback and refuses redirects so audio stays on the configured origin (#1736)
- Windows isolated engines now retain direct Job ownership without an extra Python supervisor process that can deadlock the child loader (#1734)
- The setup splash now waits through the backend's full startup budget instead of reporting slow Windows CUDA initialization as stuck after two minutes (#1749)
- Dubbing jobs can now reuse every source-language code produced by automatic ASR detection without a 400 error on the next upload (#1737)
- Incomplete Sherpa-ONNX model snapshots now self-repair before recognizer startup instead of failing on a missing ONNX file (#1733)
- OmniVoice subprocess startup now allows slow packaged Windows Python runtimes to signal readiness before termination (#1711)
- SRT files selected during source analysis now wait for speaker cloning, then replace transcript text without losing voices (#1709)
- Windows MSI deployments can now prohibit WebView2 bootstrap with `DISABLEWEBVIEW2BOOTSTRAP=1`, and `AUTOLAUNCHAPP=0` reliably suppresses first launch (#1714)
- Subtitle rows now provide 100 ms timing steppers and flag adjacent overlaps without requiring precise timeline dragging (#1710)
- Repair-sync failures now retain uv's final dependency error instead of reporting only an opaque exit status (#1705)
- YouTube ingest now retries yt-dlp's transient “page needs to be reloaded” response (#1706)
- Dictation model readiness now follows the live Hugging Face cache selected in Settings (#1707)
- Dictation capture now queues native events whenever its webview listener unmounts or reloads instead of emitting them to nobody (#1707)
- Desktop-contained backends now exit when their owning app disappears instead of surviving as stale port-3900 processes (#1707)
## [0.5.1] — 2026-08-28
**Highlights**
- OmniVoice generation on Apple Silicon now runs in a crash-isolated child, so fatal MPS memory exits no longer take down the local backend (#1697, #1698) — thanks @ndntran14!
- Model-load GPU exhaustion now returns a sanitized, actionable dubbing error, and readiness correctly attributes the shared model status to TTS (#1695)
- Source-mode development now restarts an isolated backend crash without tearing down the UI, while repeated crash loops still stop loudly with diagnostics (#1690)
- Dubbing playback now keeps an audible companion source when a WebView can render the preview picture but cannot decode its audio (#1692)
- Model Catalogue engine rows now use the available desktop width and keep identity, runtime state, and actions from crowding one another (#1689)
- VoiceStudio now acts as a local speech platform: other apps can trigger its native dictation or connect through versioned HTTP, WebSocket, JSON-RPC, CLI, and MCP transports (#1646)
- A timed-out in-process dub transcription no longer starts a second WhisperX/CTranslate2 call over the abandoned native worker, preventing the overlapping access that preceded Windows `0xC0000005` exits (#1669)
- Windows debugger termination code `0x40010004` is no longer misreported as a backend crash or charged against automatic restart recovery (#1663)
- Studio now keeps one generation reservation across page changes, preventing a remount from stacking native jobs until the backend reports capacity busy or is killed under memory pressure (#1670)
- Uploaded dubbing videos are normalized to browser-safe H.264/AAC before preview, preventing valid VP9, AV1, or Opus media from failing with “no supported sources” (#1644)
- Dubbing now separates spoken and target languages, preserves translations through segment cleanup, and lets failed translations be retried or skipped without restarting the batch (#1654) — thanks @Number16BusShelter!
- Importing replacement SRT subtitles now keeps each cue bound to the best-overlapping source speaker and clone instead of resetting every line to a random default voice (#1660) — thanks @invio-a11y!
- Uploading a Dub preview no longer blocks every backend request while ffmpeg extracts its audio (#1667) — thanks @tfreyd!
- Docker quick starts now require the administrator key needed through container NAT instead of starting a UI whose protected actions return 403 (#1651) — thanks @wd357dui!
- WSL2 AMD containers now use the `/dev/dxg` ROCDXG bridge with actionable GPU diagnostics instead of silently falling back to CPU (#1655) — thanks @wd357dui!
- Ad-hoc voice-clone references now stay alive until cancelled or timed-out GPU work actually stops reading them, so prompt caching can finish instead of failing on a deleted temp file (#1668) — thanks @tfreyd!
- Dictation now stays bound to the app where it started and recovers locally from silent recognizer output (#1175)
- The backend now answers within a second of launch and narrates its startup step by step (#1550)
- Reporting a bug from an outdated build now offers the latest release first (#1547)
- The backend is only announced ready once it can actually serve, and crash-loop restarts now pace themselves (#1548)
- Invisible watermarking no longer stalls — or silently skips — the first take of a session (#1615)
- Dub subtitles can be retimed, inserted, and merged in either direction from the segment table (#1612) — thanks @invio-a11y!
### Changed
- Model Catalogue now uses one breathable workspace canvas with simpler pane and engine-family navigation instead of nested cards and scroll regions (#1685)
- Linux source launchers now catch missing libxdo and GStreamer audio plugins before they can cause a linker error or an aborted, blank WebKit renderer (#1680, #1682)
- Dictation now carries one native output session from shortcut-down through final delivery, restores text, HTML, image, or file-list clipboards only when untouched, keeps Wayland copy-safe unless current-focus insertion is explicitly enabled, and retries silent Sherpa speech only through an already-installed local ASR model (#1175)
- The backend binds its port immediately and reports startup progress live — `/health` answers 503-with-step and a new `/startup/progress` endpoint lists every step while PyTorch, API routes, and database migrations load in the background, so "starting at step X" is never mistakable for "dead"; the desktop splash narrates each step (#1550)
### Added
- A bundled Rust loopback sidecar exposes dictation start/stop/toggle, focused-output sessions, discovery, and JSON-RPC; the backend adds versioned streaming events and a dependency-free CLI bridge for Herdr, coding agents, editors, desktop apps, and TUIs (#1646)
- Headless NVIDIA and ROCm machines can now join as worker-only Docker Compose services with no published UI and durable protocol-v2 enrollment; update both machines together before reconnecting (#1638) — thanks @jkrogers9862!
- Linux ARM64 (Asahi Apple Silicon) support for the OmniVoice GGUF engine — a `linux-aarch64` binary built with GGML Vulkan where the toolchain allows it, so Apple GPUs accelerate generation through the open-source Honeykrisp driver instead of falling back to CPU-only (#1641)
- One-command install on every desktop OS: `curl -fsSL https://voicestudio.sh/install | sh` (macOS/Linux/WSL) or `irm https://voicestudio.sh/install | iex` (Windows) — the URL serves the right script per platform, and Windows gains a source installer (`scripts/install.ps1`) with a 3-OS CI smoke (#1626)
- Per-line subtitle management in the dub table: a line's end time is editable alongside its start (typing a time and dragging its timeline edge now take the same path), lines merge with the previous row as well as the next (`Ctrl/Cmd+Shift+M`), and a new line can be inserted into the gap after any row (#1612) — thanks @invio-a11y!
- CI now enforces performance regression budgets on the hot paths — operation-count tests pin streaming TTS to one synthesis per sentence and cached dub re-mixes to zero re-synthesis; fast-path guards cover zero re-decoding and ⌈N/W⌉ native batch calls when enabled (#1594)
- Default-engine dubbing now synthesizes several segments per forward pass instead of one call per line — the width follows the host's device headroom (1 on CPU and low-VRAM cards, up to 8), `OMNIVOICE_DUB_BATCH_WIDTH` overrides it, and engines without native batching keep the single-segment path (#1594)
- `/ws/tts` now reports real time-to-first-audio, and its RTF measures synthesis alone so a slow client can't inflate it (#1594)
- The locally cached AudioSeal watermark generator warms on a background thread ~35s after boot (`OMNIVOICE_PRELOAD_WATERMARK=0` opts out; explicitly setting `=1` may download it), so the first synthesis no longer serializes the audioseal import + model load inline — measured at ~42s on a cold filesystem, 3s short of a 90s client timeout (#1576) — thanks @paoloantinori!
- Voices you've cloned stay "warm" across restarts — encoded references now persist to disk (~10 KB each), so the first generation of a session skips the re-encode and any transcription pass; `OMNIVOICE_PROMPT_DISK_CACHE=0` opts out (#1565)
- Optional FlashInfer acceleration for the default engine on CUDA (`OMNIVOICE_FLASHINFER=1`, ~2.2x measured) — needs the optional `flashinfer-python` package; missing package or kernel failure logs why and falls back to the standard path (#1565)
- The bug reporter notices when you're on an outdated build and offers the latest release before filing — with a "File anyway" escape hatch — and stamps a `Build status` line into every report so up-to-date reports are tellable from stale ones (#1547)
- Settings → Performance & Device gains a compute-device override (Auto / CUDA / ROCm / XPU / MPS / CPU, or `OMNIVOICE_DEVICE`) — pin the device when auto-detect picks wrong; only devices your machine actually has are offered (#1557)
- Opt-in 24-layer PocketTTS checkpoints via `OMNIVOICE_POCKETTTS_24L` — better prosody for it/de/es/pt at roughly 2x render time (still faster than real-time); the fast 6-layer model stays the default (#1613) — thanks @paoloantinori!
### Docs
- Supported-version and install guidance now identifies 0.5.1 as the stable desktop and container release (#1687)
- The Docker Hub overview now shows the current engine-switching demo, Model Catalogue, and gallery voice workflow (#1593)
- The Docker Hub overview and install guide now show the v0.5 tags and the built-in API-key/share-PIN security model instead of obsolete v0.4 and no-authentication guidance (#1592)
- The READMEs now lead with download buttons and a three-step first-clone walkthrough, and a new benchmarks page anchors measured per-engine/per-device numbers on the in-repo harness (#1555)
@@ -31,6 +211,42 @@ the frozen-backend fallback mirror it for their toolchains.
- The OmniVoice guide now covers combining style attributes with a reference clip (consistent instruct stabilizes cloning; the reference wins conflicts), inline pronunciation control (pinyin / CMU phonemes), and corrects the claim that the default engine can't do voice design — it can, from attributes (#1565)
### Fixed
- Workspaces now measure their responsive width when the post-bootstrap shell actually mounts, so native UI scaling reflows Projects and History instead of crushing the Dubbing demo into unreadable columns (#1683)
- Dubbing keeps the source-language selector visible after a local file is chosen, so ASR can be pinned before transcription starts (#1678) — thanks @Lonki-lomki-cloud!
- First-run media-engine downloads become available to TTS immediately without a restart, and missing media-process failures now point to repair controls (#1677) — thanks @farhataligpt-dev!
- Source installs on AMD GPUs honour `OMNIVOICE_TORCH_VARIANT=rocm`: `bun run desktop` now swaps in the ROCm torch wheel after `uv sync` and launches the backend without re-syncing, instead of silently reverting to the CPU-only CUDA build on every start (#1665) — thanks @uberclokr!
- `bun run desktop` on a fresh clone no longer fails with "resource path `../../frontend/dist` doesn't exist" — the dev launcher creates the placeholder Tauri resource directory before compiling (#1664) — thanks @uberclokr!
- macOS no longer loses TTS after the first request when Python lacks `os.waitid`; subprocess ownership now uses a safe `waitpid` fallback without risking reused process groups (#1656) — thanks @paoloantinori!
- Desktop startup, Retry, reset, uninstall, shutdown, and crash recovery now share one backend lifecycle owner; quitting interrupts first-run installers and gracefully drains then force-cleans the full backend process tree, so overlaps cannot duplicate or orphan it (#1635) — thanks @Xohaibxobi!
- Large Stories and Audiobook projects now persist in IndexedDB instead of overflowing the `omnivoice.app` localStorage envelope, with quota-safe migration and orderly exit/reload flushing (#1636) — thanks @leodzai!
- OmniVoice and its crash-isolated subprocess now route to AMD ROCm GPUs instead of warning and falling back to CPU (#1629) — thanks @j4r3kb!
- Dictation now cancels pending startup work, capture resources, sockets, and timers when the capture widget closes, preventing late work against a destroyed webview (#1645)
- Streaming generation failures now show recognized recovery guidance and appear in Diagnostics instead of only returning a generic error (#1607)
- The worker-capacity transport test no longer races its own setup: the 1-slot limit now goes through the enrollment handshake instead of mutating client config after connect, where the server's stream-open ConfigUpdate (carrying the registered capacity of 2) could overwrite it and fake an over-accept; failed CI twice on 2026-08-21 (#1630)
- Moving words across a speaker boundary in a dub — merging two lines and splitting them again — no longer dubs the second half in the first speaker's voice; each half now keeps the speaker, voice, direction, gain, and language of whoever actually says it (#1612) — thanks @invio-a11y!
- Dictation on a WebView that refuses a 16 kHz audio context (WKWebView) now low-passes before downsampling, so frequencies above 8 kHz stop folding into the speech the recognizer is fed (#1610)
- A microphone context that cannot be resumed now reports a mic error instead of leaving the dictation pill on "Listening" while capturing nothing (#1610)
- Dictation no longer retains a whole session's audio for silent-model recovery — an open mic grew that buffer by ~115 MB an hour; the recent two minutes are kept instead (#1610)
- The clipboard-delivery status is now translated in all 21 languages, so Wayland users — where clipboard delivery is the default — no longer see an English string (#1610)
- A native sherpa-onnx load failure of any exception type now degrades to "engine unavailable" instead of taking the dictation WebSocket down (#1610)
- Dictation now ships Whisper Tiny as its one cross-platform default, avoiding Parakeet's measured empty decoding on Windows while keeping Parakeet selectable behind runtime fallback (#1175)
- Re-mixing a dub no longer decodes, rewrites, and re-reads every cached segment — same-rate cached audio is reused directly (and rejected if truncated), switching timing modes can't reuse slot-truncated audio as natural-rate, and RVC respects natural-rate modes (#1594)
- PocketTTS French works again — pocket-tts only ships a 24-layer French model and rejected the name the sidecar asked for, so every French request failed at model load; French now always loads `french_24l` (#1613) — thanks @paoloantinori!
- Installing IndexTTS 2.5 no longer fails claiming an interrupted download — the weights repo ships `config.yaml` and VoiceStudio demanded a `config_v2_5.yaml` that exists in no upstream release; both names are accepted, so a hand-renamed checkout keeps working (#1611) — thanks @zuiaiyutu!
- IndexTTS 2.5 no longer has long-text generation killed at 60 seconds — the sidecar now proves it is alive every 5 seconds while `infer()` runs, and its deadline rises to 900s (`OMNIVOICE_INDEXTTS_RECV_TIMEOUT_S`) (#1611) — thanks @zuiaiyutu!
- The OpenAI-compatible `/v1/audio/speech` route now reuses the shared cached engine for explicit `model` ids instead of constructing a fresh engine — and its sidecar/model load, a ~28s floor per call for subprocess engines — on every request, with the same single-engine-resident discipline `/generate` applies (#1614) — thanks @paoloantinori!
- The setup wizard's RAM check no longer blocks 8 GB machines whose OS reports ~7.8 GB usable — the thresholds now tolerate reserved memory, and `OMNIVOICE_RAM_PREFLIGHT=0` turns a genuine block into a warning for those who accept the OOM risk (#1618)
- Invisible watermarking now runs eagerly instead of through `torch.compile` — AudioSeal's lazy compile sent the first embed of every session into Inductor's C++ codegen, which failed outright on macOS hosts whose toolchain couldn't serve it and shipped the audio unmarked after a 30-40s wait; first embed drops from 9.70s to 0.26s (#1615) — thanks @paoloantinori!
- The macOS Accessibility blocker now rechecks while visible and closes as soon as the grant is enabled instead of keeping a stale permission prompt on screen (#1609)
- The dubbing editor's video and transcript columns can now be resized by pointer or keyboard, and the chosen split persists across launches (#1571) — thanks @invio-a11y!
- CPU-only synthesis now gets a bounded ten-minute execution budget, and a render that exhausts it is reported as a compute timeout instead of misleading "generation capacity is busy" queue pressure (#1588) — thanks @ChienNguyen1111!
- Rapid Launchpad ↔ Dub navigation now replaces the workspace DOM owner cleanly, so late media/waveform cleanup cannot trigger React's `insertBefore` crash (#1590) — thanks @nicolas-jacques!
- Watermark embedding failures now log the full traceback instead of just the exception message, so a silently-unmarked-audio incident (audio passes through unmarked by design) is diagnosable from the log alone (#1576) — thanks @paoloantinori!
- Dubbing now recovers rapid two-speaker exchanges when diarization collapses them, defaults new projects to lip sync without overwriting saved timing choices, and keeps the editor usable on narrow screens (#1584) — thanks @victordonat0!
- `OMNIVOICE_ASR_BACKEND=omnivoice` now selects the PyTorch-native Whisper path, so the documented ROCm escape hatch no longer fails as an unknown engine (#1582) — thanks @patmansk!
- Network Sharing from Windows MSI/portable installs now serves the bundled web interface to LAN devices instead of redirecting them to their own `localhost` (#1589) — thanks @TWIISTED-STUDIOS!
- Exported dubbed videos now mark the dubbed language as the default audio stream while keeping Original available as an explicit choice (#1575) — thanks @invio-a11y!
- Cloning references can no longer exhaust system memory: transcript-free clips up to 75 seconds are searched in five bounded passages, longer clips ask to be trimmed, and supplied transcripts remain capped at 20 seconds to preserve alignment (#1578) — thanks @ACKAPOB!
- Stored artifact subpaths now resolve after moving a data directory between Windows, macOS, Linux, and Docker, while traversal and symlink escapes remain blocked (#1559) — thanks @Eman-Yousaf!
- A remote browser hitting an API-key-configured server's admin 403 now gets the API-key login form instead of endless console 403s, while desktop and PIN-only/no-key servers keep the plain loopback error so guests are never offered a login no key can satisfy (#1568) — thanks @paoloantinori!
- The crash-isolated ASR sidecar and its download preflight now agree on which model to load — setting the shared faster-whisper model variable applies to both variants instead of the sidecar quietly using a different one (#1556)
@@ -38,6 +254,8 @@ the frozen-backend fallback mirror it for their toolchains.
- Supervisor restarts after repeat crashes now back off (immediate, then 5s, then 15s) instead of respawning back-to-back, so a tight crash loop can't burn the whole restart budget in seconds (#1548)
- The Linux desktop cleanup regression test now isolates build artifacts, so an existing developer build can no longer change its result (#1566)
- Renaming, deleting, or revoking consent on a voice (and starring/clearing history, recording exports) now live-updates every open tab again — the sync routes' WebSocket events were silently dropped, which could look like "all my voices are gone" (#1561) — thanks @paoloantinori!
### CI
- Project agents now share pinned Vite and FastAPI skills from skills.sh (#1594)
- Weekly full-history secret scans no longer mistake the Ed25519 private-key type name for committed key material (#1591)
+10 -4
View File
@@ -10,10 +10,10 @@ Copyright 2024-present Palash Debnath and VoiceStudio contributors.
VoiceStudio is **free and open-source software, licensed under the GNU
Affero General Public License, Version 3 (AGPL-3.0)**. You are free to use,
copy, modify, and redistribute it — and that **includes commercial and internal
business use**: run the app, use its outputs commercially, sell the audio you
produce with it, provide professional/client services with it, and deploy it
within your organization.
copy, modify, and redistribute it. That **includes commercial and internal
business use** of the application itself. Model weights, tokenizers, and other
third-party assets retain their own terms; this application license does not
grant or summarize rights under those separate terms.
Because this is the **Affero** GPL, one additional obligation applies: if you
modify VoiceStudio and make that modified version available to others over
@@ -41,6 +41,12 @@ is **separately licensed under Apache License 2.0** by its upstream authors and
is not relicensed here. Apache License 2.0 is compatible with, and may be
combined under, the GNU AGPL-3.0. See `pyproject.toml`.
Downloaded model weights are not relicensed by VoiceStudio. The default
`k2-fsa/OmniVoice` model card identifies its code as Apache-2.0 and pretrained
weights as CC-BY-NC. Its `audio_tokenizer/LICENSE` contains separate Boson
Higgs Audio 2 and Meta Llama community terms. A commercial license for
VoiceStudio-owned code does not replace any of those terms.
Third-party dependencies retain their own licenses. See `Cargo.lock`,
`bun.lock`, and `uv.lock` for the resolved set.
+172 -65
View File
@@ -1,24 +1,30 @@
<div align="center">
<img src="docs/logo.png" alt="VoiceStudio logo" width="120" height="120" />
<p><img src="docs/logo.png" alt="VoiceStudio logo" width="120" height="120" /></p>
<h1>VoiceStudio</h1>
<p>
<a href="https://trendshift.io/repositories/28176?utm_source=repository-badge&amp;utm_medium=badge&amp;utm_campaign=badge-repository-28176" target="_blank" rel="noopener noreferrer"><img src="https://trendshift.io/api/badge/repositories/28176" alt="VoiceStudio ranking on Trendshift" width="220" height="48" /></a>
</p>
<p><sub>Previously OmniVoice-Studio</sub></p>
<h3>Local voice cloning, dubbing, dictation, and long-form audio.</h3>
<p>16 TTS engines · 11 ASR engines · 646-language catalogue · macOS, Windows, and Linux</p>
<p><strong>Local-first.</strong> No account, API key, subscription, or usage meter for the core workflow.</p>
<h3>Clone voices, dub video, dictate, and produce long-form audio on your own hardware.</h3>
<p>16 TTS engines · 11 ASR engines · 646-language catalogue · macOS, Windows, Linux, and Docker</p>
<p>No account, API key, subscription, or usage meter for the local workflow.</p>
<p>
<a href="#install">Install</a> ·
<a href="#features">Features</a> ·
<a href="#comparison">Compare</a> ·
<a href="#requirements">Requirements</a> ·
<a href="#hardware-recommendations">Hardware</a> ·
<a href="#engines">Engines</a> ·
<a href="#architecture">Architecture</a> ·
<a href="#api">API</a> ·
<a href="#documentation">Docs</a> ·
<a href="#faq">FAQ</a> ·
<a href="README_CN.md"><strong>简体中文</strong></a>
</p>
<p>
<a href="https://github.com/debpalash/VoiceStudio/actions/workflows/ci.yml"><img src="https://img.shields.io/github/actions/workflow/status/debpalash/VoiceStudio/ci.yml?branch=main&style=flat-square&label=CI" alt="CI status" /></a>
<a href="https://github.com/debpalash/VoiceStudio/stargazers"><img src="https://img.shields.io/github/stars/debpalash/VoiceStudio?style=flat-square&color=f59e0b" alt="GitHub stars" /></a>
<a href="https://github.com/debpalash/VoiceStudio/releases"><img src="https://img.shields.io/github/downloads/debpalash/VoiceStudio/total?style=flat-square&color=8b5cf6&label=downloads" alt="Total downloads" /></a>
<a href="https://github.com/debpalash/VoiceStudio/releases/latest"><img src="https://img.shields.io/github/v/release/debpalash/VoiceStudio?style=flat-square&color=10b981" alt="Latest release" /></a>
@@ -36,7 +42,7 @@
</div>
> [!WARNING]
> **Active beta.** Use the [latest release](https://github.com/debpalash/VoiceStudio/releases/latest) for stable work or `main` for current fixes. Report problems through [GitHub Issues](https://github.com/debpalash/VoiceStudio/issues).
> **Active beta.** Use the [latest release](https://github.com/debpalash/VoiceStudio/releases/latest) for stable work. `main` contains the newest fixes and may change between releases. Report problems through [GitHub Issues](https://github.com/debpalash/VoiceStudio/issues).
## At a glance
@@ -49,33 +55,63 @@
| **Compute** | CUDA · Apple Silicon MPS/MLX · ROCm on Linux · CPU · optional remote workers |
| **Interfaces** | Desktop app · local REST/SSE/WebSocket API · OpenAI-compatible audio API · MCP Server |
| **Storage** | Voices, projects, settings, and outputs stay on the machine by default |
| **License** | AGPL-3.0; optional engines keep their own model licenses |
| **License** | AGPL-3.0 application; downloaded models keep their upstream terms |
The Voice workspace starts with three tabs: **From audio** for cloning, **By design** for creating a voice, and **Convert** for speech-to-speech conversion. Each tab displays its own workflow, with Synthesize Audio or Convert pinned below the scrolling form. The top-bar **Engines** panel combines engine selection, loaded models, and unload/flush controls; <kbd>Ctrl</kbd>/<kbd>Cmd</kbd>+<kbd>E</kbd> opens it. The searchable language picker shares Dubbings flags and language list layout, selects one output language, and retains Auto and the full cloning catalogue. Language options flow into multiple columns when space allows. Expand **Workspaces** in the sidebar to reveal navigation labels; Escape collapses it.
Dubbing places playback controls over the video with background blur and combines the waveform and timed transcript in one compact editing surface. Drag the zoomed waveform left or right to pan; click to seek. Translation language and ISO-code controls stay synchronized; Auto clears any previous language code and dialect. Transcript items group editable text, timing and status, and voice controls into three readable rows that wrap with the panel width. Output Options stays compact with the active settings shown in its summary; expand it to change output, timing, or voice matching. Transcript, glossary, and paste controls share a toolbar above the segment editor. Project details, workflow steps, and Generate/Verify/Export actions use an unfilled header.
Output settings use aligned rows; review status appears before the collapsible transcript and glossary. Glossary terms have labelled entry fields and an explicit edit action. Launchpad arranges recent files and saved voices side by side when space allows, with responsive card grids and visible Open actions.
The casting board shows icon-based voice cards and searchable selectors for each speaker. Drag a card onto a speaker or choose a voice from that speakers menu.
<a id="install"></a>
## Install
Download a package from the [latest release](https://github.com/debpalash/VoiceStudio/releases/latest), then follow the platform guide.
| Platform | Package | Guide |
|---|---|---|
| macOS 13.3+ | DMG, Apple Silicon | [Install on macOS](docs/install/macos.md) |
| Windows 10/11 | MSI, x64 | [Install on Windows](docs/install/windows.md) |
| macOS 13.3+ | Apple Silicon DMG | [Install on macOS](docs/install/macos.md) |
| Windows 10/11 | x64 MSI; choose the current-user build when listed to install without admin access | [Install on Windows](docs/install/windows.md#install-pre-built-msi) |
| Linux | AppImage, x86_64 with glibc 2.39+ | [Install on Linux](docs/install/linux.md) |
| Docker | CUDA, ROCm, or CPU | [Run with Docker](docs/install/docker.md) |
| Docker | CUDA, ROCm, CPU, and worker-only GPU profiles | [Run with Docker](docs/install/docker.md) |
Download packages from the [latest release](https://github.com/debpalash/VoiceStudio/releases/latest). First launch creates a managed Python environment and downloads the default model. Later launches reuse both.
First launch creates a managed Python environment and downloads the default model. Later launches reuse both.
> [!NOTE]
> On macOS, first launch needs a one-time right-click **Open** approval. Intel Macs cannot run the local Python backend; use a [remote backend](docs/install/macos.md) instead.
> On macOS, first launch needs a one-time right-click, then **Open** approval. Intel Macs cannot run the local Python backend; use a [remote backend](docs/install/macos.md) instead.
### Quick Docker run
```bash
docker run -d -p 127.0.0.1:3900:3900 -v omnivoice-data:/app/omnivoice_data --name voicestudio palashdeb/omnivoice-studio:stable
```
### First voice
1. Launch VoiceStudio and open **Voice Cloning**.
2. Add a clean voice sample. Three seconds works; 515 seconds usually gives a better prompt.
2. Add a clean voice sample. Three seconds works; 5 to 15 seconds usually gives a better prompt.
3. Enter text, choose a language, then select **Generate**.
> [!TIP]
> **Try without installing:** Run VoiceStudio in the cloud via the [Google Colab notebook](https://colab.research.google.com/github/debpalash/VoiceStudio/blob/main/notebooks/OmniVoice_Studio_Colab.ipynb). Explore audio quality comparisons in [benchmarks](docs/benchmarks.md) and prompt design tips in [expressive speech](docs/expressive-speech.md).
### Audio samples
Listen to sample outputs produced locally with VoiceStudio:
| Workflow | Prompt / Reference Audio | Generated Audio |
|---|---|---|
| **Voice Cloning** | [demo_voice.wav](backend/assets/samples/demo_voice.wav) | [demo_clone_output.wav](backend/assets/samples/demo_clone_output.wav) |
| **Voice Design** (US News Anchor) | *"Clear, authoritative American broadcast tone"* | [demo_voice_design_us_news_anchor.wav](backend/assets/samples/voice_design/demo_voice_design_us_news_anchor.wav) |
| **Voice Design** (UK Audiobook) | *"Warm, expressive British storytelling voice"* | [demo_voice_design_audiobook_uk_narrator.wav](backend/assets/samples/voice_design/demo_voice_design_audiobook_uk_narrator.wav) |
| **Video Dubbing** (Multilingual) | [source.src.wav](backend/assets/samples/demo/dubbing/source.src.wav) | [Spanish](backend/assets/samples/demo/dubbing/dubbed_es.src.wav) · [French](backend/assets/samples/demo/dubbing/dubbed_fr.src.wav) · [Japanese](backend/assets/samples/demo/dubbing/dubbed_ja.src.wav) · [Chinese](backend/assets/samples/demo/dubbing/dubbed_zh.src.wav) |
### Run from source
Install the [development prerequisites](.github/CONTRIBUTING.md#development-setup), then:
Install the [development prerequisites](.github/CONTRIBUTING.md#development-setup) (Node 20+/Bun and Python 3.11+), then:
```bash
git clone https://github.com/debpalash/VoiceStudio.git
@@ -84,7 +120,7 @@ bun install
bun run desktop
```
Use `bun run dev` for the browser UI. See [Contributing](.github/CONTRIBUTING.md) for services, tests, and platform packages.
The desktop launcher configures Python dependencies on first run via `uv` automatically. Use `bun run dev` for the browser UI. See [Contributing](.github/CONTRIBUTING.md) for services, tests, and platform packages.
### If setup fails
@@ -99,22 +135,22 @@ Use `bun run dev` for the browser UI. See [Contributing](.github/CONTRIBUTING.md
| Area | Included |
|---|---|
| **Voice Cloning** | Zero-shot synthesis from a short reference clip |
| **Voice Design** | Create a voice from age, accent, pitch, style, and delivery instructions |
| **Video Dubbing** | Transcribe, translate, preserve speakers, synthesize, and export video |
| **Voice Cloning** | Zero-shot synthesis from a short reference clip ([guide](docs/engines/README.md)) |
| **Voice Design** | Create a voice from age, accent, pitch, style, and delivery instructions ([expressive speech](docs/expressive-speech.md)) |
| **Video Dubbing** | Transcribe, translate, preserve speakers, synthesize, and export video; compact translation settings include track selection, and completed dubs flag timing issues for review ([export guide](docs/dubbing/export.md)) |
| **Stories and audiobooks** | Multi-voice scripts · EPUB/PDF import · chapter rendering · `.m4b` export |
| **Dictation Widget** | System-wide shortcut, live transcription, optional local-LLM cleanup |
| **[Dictation Widget](docs/features/dictation.md)** | System-wide shortcut, live transcription, optional local-LLM cleanup |
| **Vocal Isolation** | Demucs speech/background separation |
| **Speaker Diarization** | Pyannote and WhisperX speaker assignment |
| **Batch Queue** | Queue large sets of audio and video jobs with per-job progress |
| **Model Catalogue** | Install, remove, select, and route TTS, ASR, and LLM models |
| **Remote Model Downloads** | Install models on enrolled remote workers with live progress |
| **GPU Auto-Detect** | CUDA, MPS, ROCm, and CPU routing with per-engine checks |
| **Speaker Diarization** | Pyannote and WhisperX speaker assignment ([guide](docs/features/diarization.md)) |
| **Batch Queue** | Queue large sets of audio and video jobs with per-job progress, or watch a local folder for new videos |
| **Model Catalogue** | Install, remove, select, and route TTS, ASR, and LLM models ([catalogue](docs/engines/README.md)) |
| **Remote Model Downloads** | Install models on enrolled remote workers with live progress ([guide](docs/downloading-models.md)) |
| **GPU Auto-Detect** | CUDA, MPS, ROCm, and CPU routing with per-engine checks ([performance](docs/performance.md)) |
| **AI Watermark** | AudioSeal embedding and detection |
| **MCP Server** | Synthesis and transcription tools for MCP clients |
| **Diagnostics** | Self-checks, error journal, logs, and scrubbed support bundles |
| **MCP Server** | Synthesis and transcription tools for MCP clients ([guide](docs/mcp.md)) |
| **Diagnostics** | Self-checks, error journal, logs, and scrubbed support bundles ([troubleshooting](docs/install/troubleshooting.md)) |
| **Local-first** | Core creation stays local; network-backed features are explicit opt-ins |
| **Extensible** | Registry-based TTS, ASR, and plugin interfaces |
| **Extensible** | Registry-based TTS, ASR, and plugin interfaces ([acceptance](docs/engine-acceptance.md)) |
<table>
<tr>
@@ -157,10 +193,20 @@ Requirements vary by engine. These values cover the default local workflow.
| **Disk** | 10 GB free | 20 GB+ SSD |
| **GPU** | Optional; CPU mode is supported | NVIDIA CUDA or Apple Silicon |
| **VRAM** | 4 GB when using a GPU | 8 GB+; large optional engines need more |
| **Python from source** | 3.11+ | 3.113.12 |
| **Python from source** | 3.11+ | 3.11 or 3.12 |
ROCm is Linux-only and opt-in. Windows AMD/Ryzen AI uses CPU. Systems with limited VRAM offload work to CPU when required. See [performance](docs/performance.md), [benchmarks](docs/benchmarks.md), and [engine disk usage](docs/engines/disk-usage.md).
<a id="hardware-recommendations"></a>
### Recommended stack by hardware
| Hardware | Recommended TTS | Recommended ASR | Why |
|---|---|---|---|
| **Apple Silicon (M1M4)** | [MLX-Audio](docs/engines/mlx-audio.md) · [OmniVoice](docs/engines/omnivoice.md) (MPS) | [MLX Whisper](docs/engines/mlx-whisper.md) · [Parakeet MLX](docs/engines/parakeet-mlx.md) | Native unified memory, lowest latency on macOS |
| **NVIDIA GPU (8 GB+ VRAM)** | [OmniVoice](docs/engines/omnivoice.md) · [CosyVoice 3](docs/engines/cosyvoice.md) | [WhisperX](docs/engines/whisperx.md) | High-fidelity zero-shot cloning, word timestamps, diarization |
| **Low VRAM / CPU-only** | [PocketTTS](docs/engines/pockettts.md) · [Sherpa-ONNX](docs/engines/sherpa-onnx.md) · [KittenTTS](docs/engines/kittentts.md) | [Moonshine](docs/engines/moonshine.md) · [Faster-Whisper](docs/engines/faster-whisper.md) (`int8`) | Low memory footprint, optimized CPU inference |
<a id="engines"></a>
## Engines
@@ -173,22 +219,22 @@ Engine support is capability-specific. Check cloning, language, platform, memory
| Engine | Languages | Clone | Instruct | Linux | macOS ARM | Windows | License |
|---|:---:|:---:|:---:|:---:|:---:|:---:|---|
| **VoiceStudio** (default, powered by k2-fsa/OmniVoice) | 600+ | Yes | Yes | CUDA/CPU | MPS | CUDA/CPU | [AGPL-3.0](LICENSE) app · [Apache-2.0](LICENSE-NOTICE.md) model |
| **CosyVoice 3** | 9 + 18 dialects | Yes | Yes | CUDA/CPU | CPU | CUDA/CPU | Apache-2.0 |
| **GPT-SoVITS** | 5 | Yes | | CUDA/CPU | | CUDA/CPU | MIT |
| **VoxCPM2** | 30 | Yes | Yes | CUDA/CPU | MPS | CUDA/CPU | Apache-2.0 |
| **MOSS-TTS-Nano** | 20 | Yes | | CUDA/CPU | CPU | CUDA/CPU | Apache-2.0 |
| **KittenTTS** | English | | | CPU | CPU | CPU | MIT |
| **MLX-Audio** | Model-dependent | Varies | Varies | | MLX | | Varies |
| **Sherpa-ONNX** | 20+ | | | CUDA/CPU | CPU | CUDA/CPU | Apache-2.0 |
| **IndexTTS 2.5** ⚡ | ZH · EN · JA · ES · AR | Yes | | CUDA/CPU | CPU | CUDA/CPU | Bilibili model license¹ |
| **OmniVoice GGUF** ⚡ | 600+ | Yes | Yes | CUDA/CPU | MPS/CPU | CUDA/CPU | [AGPL-3.0](LICENSE) app · [Apache-2.0](LICENSE-NOTICE.md) model |
| **OmniVoice (subprocess)** ⚡ | 600+ | Yes | Yes | CUDA/CPU | MPS | CUDA/CPU | [AGPL-3.0](LICENSE) app · [Apache-2.0](LICENSE-NOTICE.md) model |
| **PocketTTS** ⚡ | EN · FR · DE · PT · IT · ES | Yes | | CPU | CPU | CPU | CC-BY-4.0, gated² |
| **Supertonic 3** ⚡ | 31 | | | CPU | CPU | CPU | OpenRAIL-M |
| **MOSS-TTS-v1.5** ⚡ | 31 | Yes | | CUDA/CPU | CPU | CUDA/CPU | Apache-2.0 |
| **dots.tts** ⚡ | 24 | Yes | | CUDA/CPU | CPU | | Apache-2.0 |
| **Confucius4-TTS** ⚡ | 14 | Yes | | CUDA/CPU | CPU | CUDA/CPU | Apache-2.0 |
| [**VoiceStudio** (default, powered by k2-fsa/OmniVoice)](docs/engines/omnivoice.md) | 600+ | Yes | Yes | CUDA/CPU | MPS | CUDA/CPU | [AGPL-3.0](LICENSE) app · [Apache-2.0 code, CC-BY-NC weights](https://huggingface.co/k2-fsa/OmniVoice#license |
| [**CosyVoice 3**](docs/engines/cosyvoice.md) | 9 + 18 dialects | Yes | Yes | CUDA/CPU | CPU | CUDA/CPU | Apache-2.0 |
| [**GPT-SoVITS**](docs/engines/gpt-sovits.md) | 5 | Yes | No | CUDA/CPU | No | CUDA/CPU | MIT |
| [**VoxCPM2**](docs/engines/voxcpm2.md) | 30 | Yes | Yes | CUDA/CPU | MPS | CUDA/CPU | Apache-2.0 |
| [**MOSS-TTS-Nano**](docs/engines/moss-tts-nano.md) | 20 | Yes | No | CUDA/CPU | CPU | CUDA/CPU | Apache-2.0 |
| [**KittenTTS**](docs/engines/kittentts.md) | English | No | No | CPU | CPU | CPU | MIT |
| [**MLX-Audio**](docs/engines/mlx-audio.md) | Model-dependent | Varies | Varies | No | MLX | No | Varies |
| [**Sherpa-ONNX**](docs/engines/sherpa-onnx.md) | 20+ | No | No | CUDA/CPU | CPU | CUDA/CPU | Apache-2.0 |
| [**IndexTTS 2.5** ⚡](docs/engines/indextts.md) | ZH · EN · JA · ES · AR | Yes | No | CUDA/CPU | CPU | CUDA/CPU | Bilibili model license¹ |
| [**OmniVoice GGUF** ⚡](docs/engines/omnivoice-gguf.md) | 600+ | Yes | Yes | CUDA/CPU | MPS/CPU | CUDA/CPU | [AGPL-3.0](LICENSE) app · [review the derivative model terms](https://huggingface.co/Serveurperso/OmniVoice-GGUF#license |
| [**OmniVoice (subprocess)** ⚡](docs/engines/omnivoice-subprocess.md) | 600+ | Yes | Yes | CUDA/CPU | MPS | CUDA/CPU | [AGPL-3.0](LICENSE) app · [Apache-2.0 code, CC-BY-NC weights](https://huggingface.co/k2-fsa/OmniVoice#license |
| [**PocketTTS** ⚡](docs/engines/pockettts.md) | EN · FR · DE · PT · IT · ES | Yes | No | CPU | CPU | CPU | CC-BY-4.0, gated² |
| [**Supertonic 3** ⚡](docs/engines/supertonic3.md) | 31 | No | No | CPU | CPU | CPU | OpenRAIL-M |
| [**MOSS-TTS-v1.5** ⚡](docs/engines/moss-tts-v15.md) | 31 | Yes | No | CUDA/CPU | CPU | CUDA/CPU | Apache-2.0 |
| [**dots.tts** ⚡](docs/engines/dots-tts.md) | 24 | Yes | No | CUDA/CPU | CPU | No | Apache-2.0 |
| [**Confucius4-TTS** ⚡](docs/engines/confucius4-tts.md) | 14 | Yes | No | CUDA/CPU | CPU | CUDA/CPU | Apache-2.0 |
⚡ Installed or registered on demand.
@@ -196,6 +242,8 @@ Engine support is capability-specific. Check cloning, language, platform, memory
² PocketTTS shows its gated-access and CC-BY-4.0 terms before first use.
³ The OmniVoice snapshot also includes an audio tokenizer under separate [Boson Higgs Audio 2 and Meta Llama community terms](https://huggingface.co/k2-fsa/OmniVoice/blob/main/audio_tokenizer/LICENSE). VoiceStudio's application license does not replace model or tokenizer terms.
Clone-less engines cannot preserve a reference speaker in dubbing or pinned-voice batch jobs. VoiceStudio rejects those jobs instead of silently changing engines. Heavy engines have separate memory and platform limits; check their engine guide first.
<a id="asr-engines"></a>
@@ -204,17 +252,17 @@ Clone-less engines cannot preserve a reference speaker in dubbing or pinned-voic
| Engine | ID | Languages | Best fit |
|---|---|:---:|---|
| **WhisperX** (default) | `whisperx` | ~100 | Dubbing, subtitles, word-level timing |
| **Faster-Whisper** | `faster-whisper` | ~100 | General cross-platform transcription |
| **Faster-Whisper (isolated)** | `faster-whisper-isolated` | ~100 | Crash-isolated batch transcription |
| **MLX Whisper** | `mlx-whisper` | ~100 | Apple Silicon |
| **PyTorch Whisper** | `pytorch-whisper` | ~100 | CUDA, MPS, and CPU fallback |
| **Parakeet TDT** | `nemo-parakeet` | English + 25 EU | Fast CPU/CUDA transcription |
| **Parakeet TDT v3 (MLX)** | `parakeet-mlx` | 25 EU | Apple Silicon dictation and word timestamps |
| **Moonshine** | `moonshine` | English | Low-power, low-latency ONNX |
| **FunASR** | `funasr` | 50+ | VAD and inline diarization |
| **sherpa-onnx** (live dictation) | `sherpa-onnx-asr` | Model-dependent | Streaming CPU dictation |
| **OpenAI-compatible** ⚠️ remote | `openai-compat-asr` | Server-dependent | Qwen3-ASR or another compatible endpoint; audio leaves the machine |
| [**WhisperX** (default)](docs/engines/whisperx.md) | `whisperx` | ~100 | Dubbing, subtitles, word-level timing |
| [**Faster-Whisper**](docs/engines/faster-whisper.md) | `faster-whisper` | ~100 | General cross-platform transcription |
| [**Faster-Whisper (isolated)**](docs/engines/faster-whisper-isolated.md) | `faster-whisper-isolated` | ~100 | Crash-isolated batch transcription |
| [**MLX Whisper**](docs/engines/mlx-whisper.md) | `mlx-whisper` | ~100 | Apple Silicon |
| [**PyTorch Whisper**](docs/engines/pytorch-whisper.md) | `pytorch-whisper` | ~100 | CUDA, MPS, and CPU fallback |
| [**Parakeet TDT**](docs/engines/nemo-parakeet.md) | `nemo-parakeet` | English + 25 EU | Fast CPU/CUDA transcription |
| [**Parakeet TDT v3 (MLX)**](docs/engines/parakeet-mlx.md) | `parakeet-mlx` | 25 EU | Apple Silicon dictation and word timestamps |
| [**Moonshine**](docs/engines/moonshine.md) | `moonshine` | English | Low-power, low-latency ONNX |
| [**FunASR**](docs/engines/funasr.md) | `funasr` | 50+ | VAD and inline diarization |
| [**sherpa-onnx** (live dictation)](docs/engines/sherpa-onnx-asr.md) | `sherpa-onnx-asr` | Model-dependent | Streaming CPU dictation |
| [**OpenAI-compatible** ⚠️ configured server](docs/engines/openai-compatible-asr.md) | `openai-compat-asr` | Server-dependent | Local gigastt/Qwen3-ASR or a remote endpoint; audio goes only to that server |
WhisperX and Faster-Whisper retry with `int8` when efficient `float16` is unavailable. Pin `ASR_COMPUTE_TYPE=int8` or `float32` only if automatic selection still fails.
@@ -249,12 +297,12 @@ FastAPI backend
- The desktop talks to a loopback-only backend on `localhost:3900`.
- Loopback API calls need no server key. Remote access requires a share PIN or API key.
- Remote workers and OpenAI-compatible ASR are opt-in. The UI identifies when audio leaves the machine.
- Analytics is off until consent. If enabled, it sends allowlisted, content-free usage metadata—not text, audio, file names, or projects.
- Remote workers and OpenAI-compatible ASR are opt-in. Loopback ASR may use HTTP and keeps audio on the machine; non-loopback endpoints require HTTPS, and redirects are not followed.
- Analytics is off until consent. If enabled, it sends allowlisted, content-free usage metadata. It never sends text, audio, file names, or projects.
<a id="api"></a>
## OpenAI-compatible API
## Local speech platform and OpenAI-compatible API
Point an OpenAI-compatible audio client at the local backend:
@@ -267,6 +315,8 @@ Point an OpenAI-compatible audio client at the local backend:
|---|---|
| `POST /v1/audio/speech` | TTS to `mp3`, `opus`, `aac`, `flac`, `wav`, or `pcm`; select a profile with `voice` and an engine with `model` |
| `POST /v1/audio/transcriptions` | STT to `json`, `text`, `verbose_json`, `srt`, or `vtt` |
| `WS /v1/audio/transcriptions/stream` | Live PCM/WebM transcription with partial, utterance, and session-final events |
| `GET /.well-known/voicestudio-speech` | Discover HTTP, WebSocket, MCP, and native dictation-control transports |
| `GET /v1/audio/voices` | List local voice profiles and engines |
```python
@@ -283,19 +333,61 @@ with client.audio.speech.with_streaming_response.create(
response.stream_to_file("speech.wav")
```
The full API reference is in **Settings → OpenAPI Reference**. For LAN, Tailscale, or proxy access, read [API authentication](docs/api-auth.md) before exposing the backend.
```bash
# Quick test via cURL
curl http://localhost:3900/v1/audio/speech \
-H "Content-Type: application/json" \
-d '{"model": "tts-1", "input": "Made on my own hardware.", "voice": "default", "response_format": "wav"}' \
--output speech.wav
```
The bundled Rust control sidecar lets Herdr, coding agents, VS Code, desktop apps,
and TUIs trigger the system-wide dictation flow or reuse its native text
insertion. See the [speech platform guide](docs/speech-platform.md). The full API
reference is in **Settings → OpenAPI Reference**. For LAN, Tailscale, or proxy
access, read [API authentication](docs/api-auth.md) before exposing the backend.
### Agent skills
Install the VoiceStudio skills for Claude Code, Codex, Cursor, and other [skills.sh](https://skills.sh)-compatible agents:
```bash
npx skills add debpalash/omnivoice-studio
npx skills add debpalash/VoiceStudio
```
- `omnivoice`: synthesize speech and transcribe audio through local VoiceStudio.
- `oss-maintainer`: the repository's open-source maintenance workflow.
### Model Context Protocol (MCP)
VoiceStudio mounts an MCP server at `http://localhost:3900/mcp` for Claude Desktop, Cursor, and AI agents:
```json
{
"mcpServers": {
"voicestudio": {
"url": "http://localhost:3900/mcp"
}
}
}
```
For clients requiring stdio transport, use the bundled local shim (`docs/mcp.json`):
```json
{
"mcpServers": {
"voicestudio": {
"command": "python",
"args": ["-m", "backend.mcp_shim"],
"cwd": "/path/to/VoiceStudio"
}
}
}
```
See the [MCP guide](docs/mcp.md) for tools (`generate_speech`, `clone_voice`, `transcribe`), file streaming modes, and client bindings.
### Google Colab
[![Open in Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/debpalash/VoiceStudio/blob/main/notebooks/OmniVoice_Studio_Colab.ipynb)
@@ -312,11 +404,13 @@ The [notebook](notebooks/OmniVoice_Studio_Colab.ipynb) runs the app and web UI o
| Fix setup | [Troubleshooting](docs/install/troubleshooting.md) · [model downloads](docs/downloading-models.md) · [Hugging Face token](docs/setup/huggingface-token.md) |
| Choose an engine | [Engine guides](docs/engines/README.md) · [benchmarks](docs/benchmarks.md) · [expressive speech](docs/expressive-speech.md) |
| Tune hardware | [Performance](docs/performance.md) · [remote workers](docs/remote-workers.md) |
| Build integrations | [API auth](docs/api-auth.md) · [MCP](docs/mcp.md) · [examples](examples/README.md) |
| Build integrations | [Speech platform](docs/speech-platform.md) · [Private production API](docs/production-private-api.md) · [API auth](docs/api-auth.md) · [MCP](docs/mcp.md) · [examples](examples/README.md) |
| Build VoiceStudio | [Contributing](.github/CONTRIBUTING.md) · [engine acceptance](docs/engine-acceptance.md) |
| Track changes | [Changelog](CHANGELOG.md) · [roadmap](docs/ROADMAP.md) · [latest release](https://github.com/debpalash/VoiceStudio/releases/latest) |
| Remove everything | [Uninstall guide](docs/install/uninstall.md) |
<a id="faq"></a>
## FAQ
<details>
@@ -328,19 +422,19 @@ Apple Silicon is supported with MPS and MLX options. Intel Macs cannot run the l
<details>
<summary><strong>How much VRAM do I need?</strong></summary>
A GPU is optional. Use 4 GB VRAM as the minimum for accelerated work and 8 GB+ for the default multi-stage workflow. Large optional engines can require 1216 GB or more. Check the [benchmarks](docs/benchmarks.md) and engine guide.
A GPU is optional. Use 4 GB VRAM as the minimum for accelerated work and 8 GB+ for the default multi-stage workflow. Large optional engines can require 12 to 16 GB or more. Check the [benchmarks](docs/benchmarks.md) and engine guide.
</details>
<details>
<summary><strong>Why does a longer reference clip not always improve the clone?</strong></summary>
Cloning is zero-shot: the clip is a prompt, not training data. Use 515 seconds of one speaker, close to the microphone, without music, noise, or reverb. Match the tone and pace you want in the output. For training, see [data preparation](docs/data_preparation.md) and [training](docs/training.md).
Cloning is zero-shot: the clip is a prompt, not training data. Use 5 to 15 seconds of one speaker, close to the microphone, without music, noise, or reverb. Match the tone and pace you want in the output. For training, see [data preparation](docs/data_preparation.md) and [training](docs/training.md).
</details>
<details>
<summary><strong>Can I use generated audio commercially?</strong></summary>
Yes under VoiceStudio's AGPL-3.0 terms. Optional engines and model weights may use different licenses; review the selected engine's license before commercial use.
VoiceStudio's application license does not restrict generated audio, but it does not grant rights under a model's separate terms. The default OmniVoice repository labels its pretrained weights CC-BY-NC and includes a tokenizer under separate community terms. Review the selected model terms before commercial use.
</details>
<details>
@@ -362,17 +456,30 @@ Use `scripts/uninstall.sh` on macOS/Linux or `scripts\uninstall.ps1` on Windows.
- [Good first issues](https://github.com/debpalash/VoiceStudio/labels/good%20first%20issue) for a scoped starting point.
- [Contributing guide](.github/CONTRIBUTING.md) for setup, tests, and pull requests.
<p align="center">
<a href="https://star-history.com/#debpalash/VoiceStudio&Date">
<img src="https://api.star-history.com/svg?repos=debpalash/VoiceStudio&type=Date" alt="Star History Chart" width="100%" />
</a>
</p>
## Support development
VoiceStudio is free and has no paid tier. Donations fund development and infrastructure.
[Ko-fi](https://ko-fi.com/debpalash) · [PayPal](https://paypal.me/palashCoder) · [Sponsorship details](SPONSORS.md)
## Responsible use and safety
VoiceStudio enables zero-shot voice cloning and speech generation on personal hardware. Please use it responsibly:
- **Consent:** Only clone or synthesize voices with explicit permission from the speaker.
- **Audio provenance:** VoiceStudio integrates [AudioSeal](https://github.com/facebookresearch/audioseal) imperceptible watermarking by default to detect and identify synthetic speech without altering sound quality.
- **Local privacy:** For the default local workflow, audio recordings, transcripts, voices, and projects remain strictly on your local disk; data leaves your device only when you explicitly configure remote workers or external ASR endpoints.
## License
VoiceStudio is licensed under [AGPL-3.0](LICENSE). You may run it, modify it, use it internally, and sell generated audio. If you modify VoiceStudio and provide that modified version as a network service, AGPL requires you to offer the corresponding source under the same license. A commercial license is available for proprietary embedding; contact **VoiceStudio@palash.dev**. See [LICENSE-NOTICE.md](LICENSE-NOTICE.md) for the plain-language scope.
VoiceStudio is licensed under [AGPL-3.0](LICENSE). You may run it, modify it, and use it internally. The application license itself does not restrict selling generated audio, but downloaded model and tokenizer terms may. If you modify VoiceStudio and provide that modified version as a network service, AGPL requires you to offer the corresponding source under the same license. A commercial license for VoiceStudio-owned code is available for proprietary embedding; it does not relicense third-party models. Contact **VoiceStudio@palash.dev**. See [LICENSE-NOTICE.md](LICENSE-NOTICE.md) for the plain-language scope.
Optional engines and downloaded models retain their own licenses. The bundled `omnivoice/` model remains Apache-2.0 upstream.
Optional engines and downloaded models retain their own licenses. The bundled `omnivoice/` Python code is Apache-2.0 upstream; the default downloaded weights and audio tokenizer use separate terms.
## Acknowledgments
+67 -3
View File
@@ -20,6 +20,7 @@
</p>
<p>
<a href="https://github.com/debpalash/VoiceStudio/actions/workflows/ci.yml"><img src="https://img.shields.io/github/actions/workflow/status/debpalash/VoiceStudio/ci.yml?branch=main&style=flat-square&label=CI" alt="CI 状态" /></a>
<a href="https://github.com/debpalash/VoiceStudio/stargazers"><img src="https://img.shields.io/github/stars/debpalash/VoiceStudio?style=flat-square&color=f59e0b" alt="Star 数" /></a>
<a href="https://github.com/debpalash/VoiceStudio/releases/latest"><img src="https://img.shields.io/github/v/release/debpalash/VoiceStudio?style=flat-square&color=10b981" alt="版本" /></a>
<a href="LICENSE"><img src="https://img.shields.io/badge/license-AGPL--3.0-blue?style=flat-square" alt="许可证" /></a>
@@ -65,11 +66,27 @@
- 🐧 **Linux** — [docs/install/linux.md](docs/install/linux.md)
- 🐳 **Docker** — [docs/install/docker.md](docs/install/docker.md) · [Docker Hub: `palashdeb/omnivoice-studio`](https://hub.docker.com/r/palashdeb/omnivoice-studio)
```bash
# Docker 快速运行 (CPU / 本地环回模式)
docker run -d -p 127.0.0.1:3900:3900 -v omnivoice-data:/app/omnivoice_data --name voicestudio palashdeb/omnivoice-studio:stable
```
**三步克隆出你的第一个声音:**
1. **安装并启动。** 首次启动会自动搭建 Python 运行环境并下载模型权重——启动画面会逐步显示进度(仅首次,需要几分钟;之后即开即用)。
2. 从启动台打开**语音克隆**,拖入任意声音的 **3 秒音频**
3. **输入一句话,点击生成。** 音频完全属于你——在你的设备上生成保存,支持 646 种语言。
3. **输入一句话,点击生成。** 音频在你的设备上生成保存,支持 646 种语言(商业使用前请审阅所选模型与分词器的许可条款)
### 🎧 音频示例
在线试听 VoiceStudio 本地生成的实际音频样例:
| 工作流 | 提示词 / 参考音频 | 生成音频 |
|---|---|---|
| **声音克隆** | [demo_voice.wav](backend/assets/samples/demo_voice.wav) | [demo_clone_output.wav](backend/assets/samples/demo_clone_output.wav) |
| **声音设计** (美语新闻主播) | *"清晰、权威的美国广播级音色"* | [demo_voice_design_us_news_anchor.wav](backend/assets/samples/voice_design/demo_voice_design_us_news_anchor.wav) |
| **声音设计** (英式有声书) | *"温暖生动的英式故事讲述音色"* | [demo_voice_design_audiobook_uk_narrator.wav](backend/assets/samples/voice_design/demo_voice_design_audiobook_uk_narrator.wav) |
| **视频配音** (多语种) | [source.src.wav](backend/assets/samples/demo/dubbing/source.src.wav) | [西班牙语](backend/assets/samples/demo/dubbing/dubbed_es.src.wav) · [法语](backend/assets/samples/demo/dubbing/dubbed_fr.src.wav) · [日语](backend/assets/samples/demo/dubbing/dubbed_ja.src.wav) · [中文](backend/assets/samples/demo/dubbing/dubbed_zh.src.wav) |
觉得慢?[docs/performance.md](docs/performance.md) 讲清了生成时间到底花在哪里、有哪些调优开关,以及“它变慢了”的三个经典原因。各引擎/设备的实测数据见 [docs/benchmarks.md](docs/benchmarks.md)。
@@ -217,6 +234,16 @@ Hugging Face Token 的配置见
> [!IMPORTANT]
> **macOS Intelx86_64)不支持本地后端:** 应用 UI 可以安装,但 Python 后端无法运行,因为 PyTorch 已不再发布 Intel Mac 轮子([#889](https://github.com/debpalash/VoiceStudio/issues/889))。Intel Mac 用户仍可让 UI 指向另一台机器上的远程后端——参见 [docs/install/macos.md](docs/install/macos.md)。
<a id="hardware-recommendations"></a>
### 💡 按硬件推荐引擎配置
| 硬件配置 | 推荐 TTS 引擎 | 推荐 ASR 语音识别 | 优势 |
|---|---|---|---|
| **Apple Silicon (M1M4)** | [MLX-Audio](docs/engines/mlx-audio.md) · [OmniVoice](docs/engines/omnivoice.md) (MPS) | [MLX Whisper](docs/engines/mlx-whisper.md) · [Parakeet MLX](docs/engines/parakeet-mlx.md) | 原生统一内存,macOS 上延迟最低、性能最强 |
| **NVIDIA 显卡 (8 GB+ 显存)** | [OmniVoice](docs/engines/omnivoice.md) · [CosyVoice 3](docs/engines/cosyvoice.md) | [WhisperX](docs/engines/whisperx.md) | 极致零样本克隆品质、字级时间戳对齐与说话人分离 |
| **低显存 / 仅 CPU 设备** | [PocketTTS](docs/engines/pockettts.md) · [Sherpa-ONNX](docs/engines/sherpa-onnx.md) · [KittenTTS](docs/engines/kittentts.md) | [Moonshine](docs/engines/moonshine.md) · [Faster-Whisper](docs/engines/faster-whisper.md) (`int8`) | 超低内存占用,针对 CPU 指令集深度优化 |
<a id="tts-engines"></a>
### 🗣️ TTS 引擎
@@ -338,9 +365,9 @@ print(result.text)
### 📓 在 Google Colab 上运行
[![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/debpalash/VoiceStudio/blob/main/notebooks/VoiceStudio_Studio_Colab.ipynb)
[![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/debpalash/VoiceStudio/blob/main/notebooks/OmniVoice_Studio_Colab.ipynb)
没有本地 GPU?官方笔记本([notebooks/VoiceStudio_Studio_Colab.ipynb](notebooks/VoiceStudio_Studio_Colab.ipynb))可在免费的 Colab T4 上启动完整应用(包含 Web 界面):在笔记本内直接构建前端,用 uv 安装后端(复用 Colab 预装的 CUDA PyTorch),并通过 Colab 内置端口代理打开界面。无需第三方隧道,也无需任何 API 密钥。随后还有一套覆盖全部主要功能的 API 导览,全部可在笔记本内直接播放:多语言 TTS、声音克隆与声音设计、已保存的声音档案、语音转写、AI 水印检测、OpenAI 兼容 API、多角色故事、带章节的 m4b 有声书,以及一个附带人声分离音轨的迷你视频配音。
没有本地 GPU?官方笔记本([notebooks/OmniVoice_Studio_Colab.ipynb](notebooks/OmniVoice_Studio_Colab.ipynb))可在免费的 Colab T4 上启动完整应用(包含 Web 界面):在笔记本内直接构建前端,用 uv 安装后端(复用 Colab 预装的 CUDA PyTorch),并通过 Colab 内置端口代理打开界面。无需第三方隧道,也无需任何 API 密钥。随后还有一套覆盖全部主要功能的 API 导览,全部可在笔记本内直接播放:多语言 TTS、声音克隆与声音设计、已保存的声音档案、语音转写、AI 水印检测、OpenAI 兼容 API、多角色故事、带章节的 m4b 有声书,以及一个附带人声分离音轨的迷你视频配音。
### 🤝 智能体技能(Agent Skills
@@ -352,6 +379,36 @@ npx skills add debpalash/omnivoice-studio
内含两个 [skills](https://skills.sh)**`omnivoice`**——让任何智能体通过你的本地安装进行语音合成与转录(包括你克隆的声音),免费且离线;以及 **`oss-maintainer`**——本项目所遵循的维护者方法论,适合任何用智能体运营自己开源项目的人。
### 🔌 模型上下文协议(MCP 服务器)
VoiceStudio 在 `http://localhost:3900/mcp` 挂载了 MCP 服务,可供 Claude Desktop、Cursor 与自主智能体调用:
```json
{
"mcpServers": {
"voicestudio": {
"url": "http://localhost:3900/mcp"
}
}
}
```
对于需要 stdio 管道传输的客户端,请使用内置的本地桥接脚本(`docs/mcp.json`):
```json
{
"mcpServers": {
"voicestudio": {
"command": "python",
"args": ["-m", "backend.mcp_shim"],
"cwd": "/path/to/VoiceStudio"
}
}
}
```
支持 `generate_speech``clone_voice``transcribe` 等工具与流式文件输出模式,详见 [docs/mcp.md](docs/mcp.md)。
---
## 🗺️ 路线图
@@ -542,6 +599,13 @@ VoiceStudio **免费**且采用 **AGPL-3.0** 许可——没有付费版,没
VoiceStudio 完全本地运行——卸载就是删除应用及其写入的文件夹(模型缓存、Python 环境、你的声音/项目、配置)。运行 <code>scripts/uninstall.sh</code>macOS/Linux)或 <code>scripts\uninstall.ps1</code>Windows)——它会先以干跑方式列出每个文件夹及其大小,加 <code>--yes</code> 才会真正删除。完整的各平台路径列表和应用移除步骤见 <a href="docs/install/uninstall.md"><b>docs/install/uninstall.md</b></a>。
</details>
## 🛡️ 负责任使用与安全
VoiceStudio 在个人硬件上提供零样本语音克隆与语音创作能力。我们提倡负责任的技术使用:
- **明确授权:** 严禁在未经说话人本人知情并明确授权的情况下克隆其声音。
- **AI 溯源:** VoiceStudio 默认集成 [AudioSeal](https://github.com/facebookresearch/audioseal) 不可见神经音频水印,在完全不影响听感音质的前提下精准标记合成语音。
- **本地隐私:** 默认本地工作流下,所有音频、声音档案、项目与转录文本始终保存在你的本地设备上;仅当你主动配置远程工作节点或第三方 ASR 端点时,相应数据才会传输到对应服务。
---
<a id="license"></a>
+9
View File
@@ -54,6 +54,15 @@ def _server_mode() -> bool:
return os.environ.get("OMNIVOICE_SERVER_MODE", "").strip().lower() in _TRUTHY
def validate_server_admin_key() -> None:
"""Reject an explicitly blank key before a server-mode app starts."""
raw_key = os.environ.get("OMNIVOICE_API_KEY")
if _server_mode() and raw_key is not None and not raw_key.strip():
raise RuntimeError(
"OMNIVOICE_API_KEY is blank; configure a non-whitespace administrator key"
)
def _configured_pin(request) -> str | None:
"""The active share PIN (``app.state.network_share.pin``) or None. Read via
getattr so a bare Request stub (or a request that hit before lifespan set
+7
View File
@@ -49,6 +49,13 @@ def public_backends(entries: list[dict]) -> list[dict]:
item["routing_reason"] = _public_routing_reason(
item.get("routing_status"), item["routing_reason"]
)
evidence = item.get("execution_evidence")
if isinstance(evidence, dict) and evidence.get("cpu_fallback_reason") is not None:
evidence = dict(evidence)
evidence["cpu_fallback_reason"] = _public_routing_reason(
"cpu_fallback", evidence["cpu_fallback_reason"]
)
item["execution_evidence"] = evidence
safe.append(item)
return safe
+42 -23
View File
@@ -57,11 +57,12 @@ _PREVIEW_SEED = 42
# 32 reliably converges to speech across the gallery's instruct/script space
# at a one-time (cached) render cost.
_PREVIEW_NUM_STEP = 32
# Spectral-flatness floor below which a render is a degenerate tonal artifact
# rather than speech. Real, mastered speech sits ~0.040.07; a tonal buzz
# collapses to <0.005. 0.015 separates the two with wide margin and sits well
# below even breathy/whisper voices (which are broadband → high flatness).
_DEGENERATE_FLATNESS = 0.015
# Reject near-pure tonal artifacts using mean framed spectral flatness.
# Calibrated against the tracked speech demos exercised by
# test_archetype_preview_quality.py: the quietest (Mandarin dubbing, 44.1 kHz)
# measures ~7.7e-6, while the worst tested tonal buzz measures ~3.3e-9.
# 1e-7 leaves >10x margin on both sides without rejecting low-flatness speech.
_DEGENERATE_FLATNESS = 1e-7
def _preview_key(a: dict) -> str:
@@ -248,24 +249,46 @@ def _is_blank_audio(audio_tensor) -> bool:
return False
_FLATNESS_FRAME = 1024
_FLATNESS_HOP = 512
#: Frames quieter than this fraction of the loudest frame's energy are the gaps
#: between words, not speech; their spectrum is the noise floor and averaging it
#: in drags the measurement toward the value of whatever silence sounds like.
_FLATNESS_FRAME_FLOOR = 1e-4
def _spectral_flatness(audio_tensor) -> Optional[float]:
"""Geometric-mean / arithmetic-mean of the power spectrum.
"""Mean per-frame geometric-mean / arithmetic-mean of the power spectrum.
~1.0 for broadband noise, →0 for a pure tone. The degenerate diffusion
renders this guards against are near-pure tonal buzzes (flatness <0.005),
distinct from both silence (caught by ``_is_blank_audio``) and real speech
(~0.04+). Returns ``None`` if it can't be computed so callers don't act on
a bad measurement.
renders this guards against are near-pure tonal buzzes, distinct from both
silence (caught by ``_is_blank_audio``) and real speech. Returns ``None``
if it can't be computed so callers don't act on a bad measurement.
Measured over short frames and averaged — the standard definition. A single
FFT of the whole clip (what this used to do) is not the same quantity: its
frequency resolution grows with clip length, so speech harmonics carve
ever-deeper nulls into the spectrum and the geometric mean collapses. That
made the result depend on how long the clip was rather than on what it
sounded like, and put real speech below the rejection threshold.
"""
try:
import torch
t = audio_tensor if isinstance(audio_tensor, torch.Tensor) else torch.as_tensor(audio_tensor)
t = t.detach().to("cpu", dtype=torch.float32).flatten()
if t.numel() < 1024 or not torch.isfinite(t).all():
t = t.detach().to("cpu", dtype=torch.float32)
if t.ndim > 1:
t = t.mean(dim=0)
t = t.flatten()
if t.numel() < _FLATNESS_FRAME or not torch.isfinite(t).all():
return None
spec = torch.fft.rfft(t * torch.hann_window(t.numel())).abs().pow(2) + 1e-12
return float(torch.exp(torch.mean(torch.log(spec))) / torch.mean(spec))
frames = t.unfold(0, _FLATNESS_FRAME, _FLATNESS_HOP)
spec = torch.fft.rfft(frames * torch.hann_window(_FLATNESS_FRAME)).abs().pow(2) + 1e-12
energy = spec.sum(dim=1)
spec = spec[energy > energy.max() * _FLATNESS_FRAME_FLOOR]
if spec.shape[0] == 0:
return None
return float((torch.exp(spec.log().mean(dim=1)) / spec.mean(dim=1)).mean())
except Exception: # never let the checker itself block a render
return None
@@ -330,7 +353,7 @@ async def _render_archetype_wav(a: dict, out_path: Path) -> None:
# GPU pool and brick the backend (#730 class). Budget comes from the shared
# length-scaled helper (#1190) instead of the flat 300s default.
from services.model_manager import generate_timeout_s
_budget = generate_timeout_s(text)
_budget = generate_timeout_s(text, engine=model)
audio_tensor = await run_on_gpu_pool_guarded(
lambda: _infer(_PREVIEW_SEED), what="Archetype preview generate",
timeout=_budget)
@@ -357,15 +380,11 @@ async def _render_archetype_wav(a: dict, out_path: Path) -> None:
# Runs on the dedicated watermark pool (#1190): AudioSeal embedding is CPU
# work that holds no VRAM, so it must not occupy a GPU worker ahead of the
# next generate on 1-worker hosts.
from services.watermark import mark_synthetic
from services.model_manager import get_watermark_pool
import functools
audio_tensor = await run_on_gpu_pool_guarded(
functools.partial(mark_synthetic, audio_tensor, model.sampling_rate,
context="archetypes.render"),
what="Archetype watermark",
from services.watermark import mark_synthetic_async
audio_tensor = await mark_synthetic_async(
audio_tensor, model.sampling_rate,
context="archetypes.render",
timeout=generate_timeout_s(""),
executor=get_watermark_pool(),
)
out_path.parent.mkdir(parents=True, exist_ok=True)
+13 -6
View File
@@ -361,7 +361,7 @@ LONGFORM_NUM_STEP = 32
LONGFORM_GUIDANCE_SCALE = 2.0
def _seed_segment_rng(base_seed, text: str, nonce: int = 0) -> None:
def _seed_segment_rng(base_seed, text: str, nonce: int = 0) -> int | None:
"""Apply a profile's pinned seed to this synth call (#1139).
``_resolve_voice`` has always fetched the profile ``seed`` — but only the
@@ -380,11 +380,13 @@ def _seed_segment_rng(base_seed, text: str, nonce: int = 0) -> None:
must cover /generate and here together, not one path.
"""
if base_seed is None:
return
return None
import torch
from services.audiobook import segment_seed
torch.manual_seed(segment_seed(base_seed, text, nonce))
seed = segment_seed(base_seed, text, nonce)
torch.manual_seed(seed)
return seed
def _base_seed(opts: ExpressiveOptions, voice: dict):
@@ -508,16 +510,21 @@ def _build_synth(
"get_model": get_model, "language": language, "opts": opts}
backend = cls()
extra = _generic_extra_kwargs(opts)
native_proxy = bool(getattr(cls, "supports_native_omnivoice_controls", False))
extra = (_omnivoice_sampling_kwargs(opts) if native_proxy
else _generic_extra_kwargs(opts))
next_nonce = _make_occ_counter(opts)
def synth(text, voice_id, speed=None):
v = resolve(voice_id)
_seed_segment_rng(_base_seed(opts, v), text, next_nonce())
seed = _seed_segment_rng(_base_seed(opts, v), text, next_nonce())
call_extra = dict(extra)
if native_proxy and seed is not None:
call_extra["seed"] = seed
return backend.generate(
text, language=language, ref_audio=v["ref_audio"],
ref_text=v["ref_text"], instruct=v["instruct"], duration=None,
speed=float(speed) if speed else 1.0, **extra,
speed=float(speed) if speed else 1.0, **call_extra,
)
return {"mode": "generic", "resolve": resolve, "engine_id": engine_id,
"synth": synth, "sample_rate": backend.sample_rate}
+207 -16
View File
@@ -103,6 +103,93 @@ def _set_progress(job, stage, percent=0, **extra):
job["progress"] = {"stage": stage, "percent": percent, **extra}
#: Override for the native dub batch width. Set to 1 to disable batching.
BATCH_WIDTH_ENV = "OMNIVOICE_DUB_BATCH_WIDTH"
#: Hard ceiling on the override — a batch this wide is already amortizing
#: almost all of the per-call setup, and beyond it the failure mode is an OOM
#: that costs more than the saving.
_MAX_BATCH_WIDTH = 16
# Bound each allocation while persisting multipart uploads. Video inputs can
# be many gigabytes; `await UploadFile.read()` with no size used to mirror the
# entire file in process memory before writing it back out.
_UPLOAD_CHUNK_BYTES = 1024 * 1024
async def _save_upload(upload: UploadFile, destination: str) -> None:
try:
with open(destination, "wb") as output:
while chunk := await upload.read(_UPLOAD_CHUNK_BYTES):
output.write(chunk)
except BaseException:
try:
unlink_if_present(destination)
except FileCleanupError:
logger.warning("Could not remove incomplete batch upload", exc_info=True)
raise
def _native_batch_width(backend) -> int:
"""How many segments to render in one native batch on THIS host.
A native batch widens the forward pass, so the width cannot be a constant.
The default engine declares ``min_vram_gb = 6.0`` for a SINGLE job; an
unconditional 8-wide batch would OOM the 4-8 GB CUDA cards and the MPS
Macs where the per-segment path succeeds today turning a throughput
optimization into a regression on exactly the hardware that already
struggles (#1616 is a 4 GB card reporting capacity failures). Default
behaviour must not get riskier on a host, so the width is derived from
measured headroom and falls back to 1 (no batching) when unknown.
CPU hosts get 1: batching there buys no kernel amortization and only
multiplies peak RAM.
"""
override = os.environ.get(BATCH_WIDTH_ENV, "").strip()
if override:
try:
return max(1, min(_MAX_BATCH_WIDTH, int(override)))
except (TypeError, ValueError):
logger.warning(
"%s=%r is not an integer — deriving the batch width from the host instead.",
BATCH_WIDTH_ENV, override,
)
try:
from core.device_caps import detect_host_caps
caps = detect_host_caps()
except Exception: # noqa: BLE001 — an unprobeable host takes the safe path
return 1
if caps.family == "cpu" or not caps.vram_gb:
return 1
headroom = caps.vram_gb - float(getattr(backend, "min_vram_gb", 0.0) or 0.0)
if headroom < 2.0:
return 1
if headroom < 6.0:
return 2
if headroom < 12.0:
return 4
return 8
def _batch_timeout_s(texts: list[str], backend) -> float:
"""Execution budget for one native batch.
Not the sum of the per-item budgets: ``generate_timeout_s`` returns a
floor (300s GPU / 600s CPU) plus per-length overage, so summing it across
eight items yields a ~2400s budget and a wedged batch would hold a
GPU-pool worker for forty minutes before the reset this file depends on
(#730). One floor covers wedge detection for the whole call; only the
length-driven overage is genuinely additive.
"""
from services.model_manager import generate_timeout_s
floor = generate_timeout_s("", engine=backend)
overage = sum(
max(0.0, generate_timeout_s(text, engine=backend) - floor) for text in texts
)
return floor + overage
async def _run_batch_pipeline(job_id: str, job: dict):
"""Full batch dub pipeline: extract → transcribe → translate → generate → mix → export."""
import subprocess
@@ -279,6 +366,111 @@ async def _run_batch_pipeline(job_id: str, job: dict):
full_audio = torch.zeros(1, total_samples)
total_segs = len(translated_segments)
# Native engines can amortize encoder/decoder setup across a small
# batch. Keep the adapter seam optional: engines without a real batch
# implementation inherit TTSBackend.generate_batch(), which preserves
# the established one-segment behavior below.
from services.tts_backend import TTSBackend
batched_audio: dict[int, torch.Tensor] = {}
has_native_batch = type(backend).generate_batch is not TTSBackend.generate_batch
if has_native_batch:
from services.text_normalization import normalize_for_tts
batch_ref_audio = None
batch_ref_text = None
if job.get("voice_id"):
from core.db import db_conn
from core.config import VOICES_DIR as _VD
with db_conn() as conn:
row = conn.execute(
"SELECT * FROM voice_profiles WHERE id=?",
(job["voice_id"],),
).fetchone()
if row:
if row["is_locked"] and row["locked_audio_path"]:
batch_ref_audio = os.path.join(_VD, row["locked_audio_path"])
elif row["ref_audio_path"]:
batch_ref_audio = os.path.join(_VD, row["ref_audio_path"])
batch_ref_text = row["ref_text"]
batch_width = _native_batch_width(backend)
async def _prefetch_batch(first_index: int) -> None:
"""Render the batch beginning at ``first_index`` into
``batched_audio``.
Rendered on demand rather than prerendering the whole track:
the tensors are popped as they are placed, so peak host memory
is one batch instead of every segment of the language and
the progress bar tracks placement instead of running to the
end and restarting at segment 1.
"""
if job["status"] == "cancelled":
return
batch_rows = []
index = first_index
while index < total_segs and len(batch_rows) < batch_width:
seg = translated_segments[index]
if (seg.get("end", 0) - seg.get("start", 0) > 0.05
and seg.get("text", "").strip()):
batch_rows.append((index, seg))
index += 1
if len(batch_rows) < 2:
return # nothing to amortize — the per-segment path is equal
batch_indices = [index for index, _ in batch_rows]
batch_texts = [
normalize_for_tts(row.get("text", "").strip(), target_lang)
for _, row in batch_rows
]
batch_durations = [
row.get("end", 0) - row.get("start", 0)
for _, row in batch_rows
]
def _render_native_batch():
generated = backend.generate_batch(
batch_texts,
language=target_lang,
ref_audio=batch_ref_audio,
ref_text=batch_ref_text,
duration=batch_durations,
num_step=16,
guidance_scale=2.0,
speed=1.0,
denoise=True,
postprocess_output=True,
)
if len(generated) != len(batch_indices):
raise RuntimeError(
f"native batch returned {len(generated)} outputs for "
f"{len(batch_indices)} segments"
)
rendered = []
for audio_out in generated:
if not getattr(backend, "applies_own_mastering", False):
audio_out = apply_mastering(audio_out, sample_rate=sr)
rendered.append(normalize_audio(audio_out, target_dBFS=-2.0))
return rendered
try:
rendered = await run_on_gpu_pool_guarded(
_render_native_batch,
what="Batch generate",
timeout=_batch_timeout_s(batch_texts, backend),
)
batched_audio.update(zip(batch_indices, rendered))
except TimeoutError:
# Do not immediately queue the same expensive work again:
# the timed-out pool task may still be holding the device.
raise
except Exception as e:
logger.warning(
"Native TTS batch failed for segments %s-%s; falling back per segment: %s",
batch_indices[0] + 1,
batch_indices[-1] + 1,
e,
)
for i, seg in enumerate(translated_segments):
if job["status"] == "cancelled":
return
@@ -356,10 +548,15 @@ async def _run_batch_pipeline(job_id: str, job: dict):
# Budget is the shared length-scaled one (#1190): a long segment
# on CPU-class hardware no longer dies on the flat 300s.
from services.model_manager import generate_timeout_s
audio_tensor = await run_on_gpu_pool_guarded(
_gen, what="Batch generate",
timeout=generate_timeout_s(seg_text),
)
if has_native_batch and i not in batched_audio:
await _prefetch_batch(i)
if i in batched_audio:
audio_tensor = batched_audio.pop(i)
else:
audio_tensor = await run_on_gpu_pool_guarded(
_gen, what="Batch generate",
timeout=generate_timeout_s(seg_text, engine=backend),
)
# Fit to slot
target_samples_seg = int(seg_duration * sr)
@@ -413,19 +610,15 @@ async def _run_batch_pipeline(job_id: str, job: dict):
# unmarked while the interactive dub pipeline marked every segment.
# One whole-track embed (chunked internally, #1045) is equivalent to
# dub_generate's per-segment marks: the 16-bit message repeats
# throughout. Runs in the GPU pool like generate's finalize; never
# raises (degrades to unmarked on failure, same as every producer).
# throughout. Never raises (degrades to unmarked on failure, same as
# every producer).
# Dispatched to the dedicated watermark pool, not the GPU pool (#1190):
# AudioSeal embedding is CPU work that holds no VRAM, and a whole-track
# embed is long enough that occupying a GPU worker with it stalled the
# next language's segments on 1-worker hosts.
from services.watermark import mark_synthetic
from services.model_manager import get_watermark_pool
import functools
full_audio = await loop.run_in_executor(
get_watermark_pool(),
functools.partial(mark_synthetic, full_audio, sr,
context="batch.dub_track"),
from services.watermark import mark_synthetic_async
full_audio = await mark_synthetic_async(
full_audio, sr, context="batch.dub_track",
)
# Same assembly pattern as dub_generate.py:390 — `full_audio` is a
@@ -515,9 +708,7 @@ async def enqueue_batch_job(
ext = os.path.splitext(video.filename or "video.mp4")[1] or ".mp4"
video_path = os.path.join(batch_dir, f"{job_id}{ext}")
with open(video_path, "wb") as f:
content = await video.read()
f.write(content)
await _save_upload(video, video_path)
job = {
"id": job_id,
+320 -73
View File
@@ -27,6 +27,10 @@ Protocol:
"detail": "..."} error ("detail"
kept for legacy)
Sherpa ``final`` frames additionally carry
``"final_kind": "utterance"|"summary"``. Utterances are mid-session
commits; the summary is the authoritative whole-session result at EOF.
Every ``final`` text is normalised by services.text_polish (leading
capital for Latin scripts, terminal punctuation, single-spaced) so the
pasted result reads like typed text. Partials are raw.
@@ -34,10 +38,14 @@ Protocol:
from __future__ import annotations
import asyncio
import json
import logging
import math
import os
import tempfile
import time
import uuid
from typing import Any
from fastapi import APIRouter, WebSocket, WebSocketDisconnect
@@ -47,6 +55,9 @@ from services.text_polish import polish_text
router = APIRouter()
logger = logging.getLogger("omnivoice.capture_ws")
SPEECH_PROTOCOL = "voicestudio.speech.v1"
PLATFORM_STREAM_PATH = "/v1/audio/transcriptions/stream"
# How often (seconds) to run transcription on the accumulated buffer.
# Shorter = more responsive but more GPU load.
PARTIAL_INTERVAL_S = float(os.environ.get("OMNIVOICE_STREAM_INTERVAL", "2.0"))
@@ -70,17 +81,79 @@ _AEC_NEAR = 0x00 # microphone frame (clean it, then buffer for ASR)
_AEC_FAR = 0x01 # playback reference frame (feed the echo model only)
def _requested_pcm_sample_rate(query_params) -> int | None:
"""Return a bounded PCM rate for ``?pcm=1``/``?aec=1`` sessions."""
raw_pcm = query_params.get("pcm") in ("1", "true", "on")
aec = query_params.get("aec") in ("1", "true", "on")
if not raw_pcm and not aec:
return None
# Client-supplied ``?sr=`` values outside the range real capture devices use
# are replaced with 16 kHz. The rate sizes server-side state — RecoveryTail
# multiplies it by RECOVERY_TAIL_SECONDS to compute its byte ceiling — so an
# absurd rate must never be believed: it would re-open the unbounded-memory
# path the recovery-tail cap closed.
SR_MIN, SR_MAX = 8000, 96000
def _is_end_control(text: str | None) -> bool:
"""Accept the versioned JSON control frame and the legacy ``EOF`` frame."""
if text == "EOF":
return True
if not text:
return False
try:
message = json.loads(text)
except (TypeError, json.JSONDecodeError):
return False
return isinstance(message, dict) and message.get("type") == "input_audio.end"
class _PlatformWebSocket:
"""Add v1 session metadata without changing the legacy WebSocket contract."""
def __init__(self, websocket: WebSocket):
self._websocket = websocket
self.session_id = uuid.uuid4().hex
def __getattr__(self, name: str) -> Any:
return getattr(self._websocket, name)
async def send_json(self, data: Any, mode: str = "text") -> None:
if isinstance(data, dict):
data = dict(data)
data.setdefault("protocol", SPEECH_PROTOCOL)
data.setdefault("session_id", self.session_id)
if data.get("type") == "final":
data.setdefault("final_kind", "summary")
await self._websocket.send_json(data, mode=mode)
def _bounded_sample_rate(query_params) -> int:
try:
sample_rate = int(query_params.get("sr", "16000"))
except (TypeError, ValueError):
return 16000
return sample_rate if 8000 <= sample_rate <= 96000 else 16000
return sample_rate if SR_MIN <= sample_rate <= SR_MAX else 16000
def _requested_pcm_sample_rate(query_params) -> int | None:
"""Return the bounded rate when the client transport is raw PCM.
Sherpa clients omit ``pcm=1`` because the selected model already defines
that transport. If the model is demoted or its runtime is unavailable, the
legacy recognizer fallback must still decode those same bytes as PCM.
"""
raw_pcm = query_params.get("pcm") in ("1", "true", "on")
aec = query_params.get("aec") in ("1", "true", "on")
sherpa_pcm = False
requested_model = query_params.get("model")
if requested_model:
try:
from services.sherpa_dictation import is_sherpa_model
sherpa_pcm = is_sherpa_model(requested_model)
except Exception: # noqa: BLE001
# A broken sherpa install must not decide the framing question —
# sherpa_pcm stays False and the session negotiates the
# MediaRecorder path; availability is re-probed (and reported)
# when the model is actually selected.
sherpa_pcm = False
if not raw_pcm and not aec and not sherpa_pcm:
return None
return _bounded_sample_rate(query_params)
def _demux_aec_frame(data: bytes) -> tuple[str, bytes]:
@@ -137,21 +210,47 @@ def _select_sherpa_spec(websocket: WebSocket):
from services import sherpa_dictation as sd
except Exception:
return None
def _usable_spec(model_id):
spec = sd.get_spec(model_id)
if spec is not None and sd.is_demoted(spec.id):
logger.warning(
"dictation model %s is demoted — using the capture ASR fallback",
spec.id,
)
return None
return spec
requested = websocket.query_params.get("model")
if requested:
return sd.get_spec(requested) # explicit selection (may be None if bad)
return _usable_spec(requested) # explicit selection (may be unavailable)
# Fall back to the persisted dictation pref.
try:
from services.asr_backend import dictation_model_id
mid = dictation_model_id()
except Exception:
mid = None
return sd.get_spec(mid) if mid else None
return _usable_spec(mid) if mid else None
@router.websocket(PLATFORM_STREAM_PATH)
@router.websocket("/ws/transcribe")
async def ws_transcribe(websocket: WebSocket):
"""Stream audio in, get partial + final transcription out."""
is_platform_stream = websocket.url.path == PLATFORM_STREAM_PATH
if is_platform_stream:
websocket = _PlatformWebSocket(websocket)
# A browser can reach localhost regardless of the page's own origin.
# Reject ambient cross-site WebSocket handshakes before the loopback-host
# shortcut or accept(), while keeping native clients (no Origin header)
# and configured/same-origin browser UIs working (#1646 review).
origin = websocket.headers.get("origin")
if origin:
from core.csrf import origin_allowed
if not origin_allowed(websocket):
await websocket.close(code=1008, reason="browser origin not allowed")
return
# Loopback origin guard — refuse anything not from 127.0.0.1, ::1, or
# localhost. Privileged HTTP routers use Depends(require_admin) at router
# level; WebSocket dependency injection differs across FastAPI versions, so we
@@ -166,6 +265,16 @@ async def ws_transcribe(websocket: WebSocket):
return
await websocket.accept()
if is_platform_stream:
await websocket.send_json({
"type": "session.started",
"input_format": (
"audio/pcm;encoding=s16le;channels=1"
if _requested_pcm_sample_rate(websocket.query_params) is not None
else "audio/webm;codecs=opus"
),
"sample_rate": _bounded_sample_rate(websocket.query_params),
})
# Live-dictation engine selection. When a sherpa-onnx model is selected
# (via ?model= or the dictation.model_id pref) AND sherpa is installed,
@@ -288,7 +397,7 @@ async def ws_transcribe(websocket: WebSocket):
total_bytes += len(data)
last_audio_time = time.monotonic()
continue
if msg.get("text") == "EOF":
if _is_end_control(msg.get("text")):
# Client signals end-of-audio but stays connected for `final`.
running = False
break
@@ -422,6 +531,64 @@ SHERPA_OFFLINE_SILENCE_S = float(os.environ.get("OMNIVOICE_SHERPA_OFFLINE_SILENC
SHERPA_OFFLINE_RMS_FLOOR = float(os.environ.get("OMNIVOICE_SHERPA_OFFLINE_RMS", "0.01"))
#: Seconds of audio retained for silent-model recovery. Recovery only needs
#: enough speech to prove the model is broken and to re-transcribe what was
#: said; retaining the whole session grew ~115 MB/hour at 16 kHz on an open
#: mic, unbounded, and only ever got read when the fallback fired.
RECOVERY_TAIL_DEFAULT_SECONDS = 120.0
RECOVERY_TAIL_MAX_SECONDS = 300.0
def _bounded_recovery_tail_seconds(value: str | None) -> float:
"""Parse the recovery tail override without allowing unbounded buffers."""
try:
seconds = float(value) if value is not None else RECOVERY_TAIL_DEFAULT_SECONDS
except (TypeError, ValueError):
return RECOVERY_TAIL_DEFAULT_SECONDS
if not math.isfinite(seconds) or seconds <= 0:
return RECOVERY_TAIL_DEFAULT_SECONDS
return min(seconds, RECOVERY_TAIL_MAX_SECONDS)
RECOVERY_TAIL_SECONDS = _bounded_recovery_tail_seconds(
os.environ.get("OMNIVOICE_DICTATION_RECOVERY_TAIL_S")
)
class RecoveryTail:
"""The most recent ``RECOVERY_TAIL_SECONDS`` of session audio.
Keeps the *tail* rather than the head: a long dictation's useful speech is
what the user just said, and the silent-model check cares about how much
audio the session carried overall which ``total_bytes`` still reports
truthfully after trimming.
"""
__slots__ = ("_buf", "_max", "total_bytes")
def __init__(self, sample_rate: int, seconds: float = RECOVERY_TAIL_SECONDS):
# int16 mono → 2 bytes/sample. Floor of one frame so a nonsense rate
# or seconds value can't produce a zero-length buffer.
self._max = max(2, int(seconds * max(1, sample_rate)) * 2)
self._buf = bytearray()
self.total_bytes = 0
def extend(self, pcm: bytes) -> None:
self._buf.extend(pcm)
self.total_bytes += len(pcm)
excess = len(self._buf) - self._max
if excess > 0:
# int16 mono: trim whole samples only. A split frame can carry an
# odd byte count, and an odd trim would leave the tail starting
# mid-sample — every later sample byte-shifted, and the recovery
# transcription fed noise.
excess += excess % 2
del self._buf[:excess]
def tail(self) -> bytes:
return bytes(self._buf)
def is_model_silent(text: str, heard_speech: bool, pcm_bytes: int) -> bool:
"""True when the dictation model produced NO text despite real speech.
@@ -448,19 +615,74 @@ def _pcm16_to_f32(pcm: bytes):
return np.frombuffer(pcm, dtype=np.int16).astype(np.float32) / 32768.0
async def _sherpa_session(websocket: WebSocket):
"""Shared WS receive setup for the sherpa handlers.
def _pcm16_rms(pcm: bytes) -> float:
samples = _pcm16_to_f32(pcm)
if not len(samples):
return 0.0
return float((samples * samples).mean() ** 0.5)
Returns ``(get_frame, state)`` where ``get_frame`` is an async callable
that yields the next near-end (mic) PCM bytes, ``b""`` for a keepalive/ref
frame, or ``None`` on EOF/disconnect. ``state`` carries sample rate, AEC,
and the disconnect flag for the caller's finaliser.
"""
pcm_sr = 16000
async def _recover_silent_sherpa(
spec, pcm: bytes, pcm_sr: int,
) -> tuple[str, list[dict]]:
"""Retry a token-silent Sherpa session through an installed local ASR."""
logger.warning(
"dictation model %s decoded NOTHING from %.1fs of speech-level audio "
"— falling back to the capture ASR engine for this session",
spec.id, len(pcm) / float(max(1, pcm_sr) * 2),
)
try:
pcm_sr = int(websocket.query_params.get("sr", "16000"))
except (TypeError, ValueError):
pcm_sr = 16000
from services.asr_backend import asr_model_missing_error
fallback_missing = await asyncio.to_thread(
asr_model_missing_error,
purpose="dictation",
skip_sherpa=True,
require_installed=True,
)
if fallback_missing is not None:
logger.warning(
"dictation silent-model fallback is not installed (%s); "
"skipping recovery to avoid an automatic download",
fallback_missing.get("missing_repo_id", "unknown"),
)
return "", []
result = await _transcribe_buffer_full(
[pcm], pcm_sr=pcm_sr, skip_sherpa=True,
)
text = polish_text(_result_text(result))
if not text:
return "", []
# The RMS gate can fire on fan/keyboard noise. Only another recognizer
# producing words proves the audio held speech and makes persistent
# demotion safe.
try:
from services.sherpa_dictation import demote_model
if await asyncio.to_thread(demote_model, spec.id):
logger.error(
"dictation model %s demoted on this machine — it will no longer be "
"auto-selected. Pick it again in Settings to give it another chance.",
spec.id,
)
except Exception:
logger.exception("silent-model demotion failed")
segments = (result or {}).get("segments") or [
{"start": 0.0, "end": None, "text": text}
]
return text, segments
except Exception:
logger.exception("dictation silent-model fallback failed")
return "", []
async def _sherpa_session(websocket: WebSocket):
"""Shared WS setup for the sherpa handlers.
Returns ``(pcm_sr, aec)``: the bounded PCM sample rate for the session
and the echo canceller when ``?aec=1`` requested one (``None`` otherwise
or when AEC setup fails).
"""
pcm_sr = _bounded_sample_rate(websocket.query_params)
aec = None
if websocket.query_params.get("aec") in ("1", "true", "on"):
try:
@@ -498,7 +720,7 @@ async def _recv_pcm_frame(websocket: WebSocket, aec):
return "skip", b""
return "near", aec.process_near_end(payload)
return "near", data
if msg.get("text") == "EOF":
if _is_end_control(msg.get("text")):
return "eof", b""
return "skip", b""
@@ -569,6 +791,8 @@ async def _run_sherpa_streaming(websocket: WebSocket, spec):
last_partial = ""
committed: list[str] = [] # finalized utterances this session
session_pcm = RecoveryTail(pcm_sr) # bounded audio for silent-model recovery
heard_speech = False
client_disconnected = False
async def _send(payload) -> bool:
@@ -610,6 +834,9 @@ async def _run_sherpa_streaming(websocket: WebSocket, spec):
break
if kind == "skip":
continue
session_pcm.extend(pcm)
if not heard_speech and _pcm16_rms(pcm) >= SHERPA_OFFLINE_RMS_FLOOR:
heard_speech = True
text, endpoint = await asyncio.to_thread(_decode_after_feed, pcm)
if endpoint:
# Commit this utterance (polished — it gets pasted); reset
@@ -618,6 +845,7 @@ async def _run_sherpa_streaming(websocket: WebSocket, spec):
if text:
committed.append(text)
await _send({"type": "final", "text": text,
"final_kind": "utterance",
"segments": [{"start": 0.0, "end": None, "text": text}],
"language": "auto", "engine": backend.id})
rec.reset(stream)
@@ -644,7 +872,28 @@ async def _run_sherpa_streaming(websocket: WebSocket, spec):
# Pieces are already polished; the join is too (polish is idempotent).
full = " ".join(t for t in committed if t).strip()
segments = [{"start": 0.0, "end": None, "text": t} for t in committed if t]
model_silent = is_model_silent(full, heard_speech, session_pcm.total_bytes)
if model_silent:
recovered, recovered_segments = await _recover_silent_sherpa(
spec, session_pcm.tail(), pcm_sr,
)
if recovered:
full = recovered
segments = recovered_segments
if not client_disconnected:
payload = {"type": "final", "text": full, "final_kind": "summary",
"segments": segments,
"language": "auto", "engine": backend.id}
if model_silent:
payload["engine"] = "capture-asr-fallback" if full else backend.id
payload["model_silent"] = spec.id
payload["warning"] = (
f"The selected dictation model ({spec.id}) produced no text from your "
"speech. Switched to the fallback engine for this session — pick a "
"different model in Settings → Dictation."
)
if full:
# Hard-bounded refinement (~4s): never delays this summary `final`
# beyond OMNIVOICE_REFINE_TIMEOUT_S even with a dead LLM endpoint.
@@ -653,14 +902,9 @@ async def _run_sherpa_streaming(websocket: WebSocket, spec):
refined = await maybe_refine_async(full)
except Exception:
refined = None
payload = {"type": "final", "text": full, "segments": segments,
"language": "auto", "engine": backend.id}
if refined and refined != full:
payload["refined_text"] = refined
await _send(payload)
else:
await _send({"type": "final", "text": "", "segments": [],
"language": "auto", "engine": backend.id})
await _send(payload)
try:
await websocket.close()
except Exception:
@@ -697,7 +941,7 @@ async def _run_sherpa_offline(websocket: WebSocket, spec):
# whisper/zipformer transcribe the same bytes). Keep the whole session's
# audio and whether any of it was speech-level, so the finaliser can tell
# "user said nothing" (fine) from "model produced nothing" (broken).
session_pcm = bytearray()
session_pcm = RecoveryTail(pcm_sr)
heard_speech = False
running = True
client_disconnected = False
@@ -716,12 +960,6 @@ async def _run_sherpa_offline(websocket: WebSocket, spec):
client_disconnected = True
return False
def _rms(pcm: bytes) -> float:
samples = _pcm16_to_f32(pcm)
if not len(samples):
return 0.0
return float((samples * samples).mean() ** 0.5)
def _decode_window(pcm: bytes) -> str:
samples = _pcm16_to_f32(pcm)
if not len(samples):
@@ -740,7 +978,7 @@ async def _run_sherpa_offline(websocket: WebSocket, spec):
continue
buf.extend(pcm)
session_pcm.extend(pcm)
if not heard_speech and _rms(pcm) >= SHERPA_OFFLINE_RMS_FLOOR:
if not heard_speech and _pcm16_rms(pcm) >= SHERPA_OFFLINE_RMS_FLOOR:
heard_speech = True
last_audio = time.monotonic()
except WebSocketDisconnect:
@@ -766,6 +1004,7 @@ async def _run_sherpa_offline(websocket: WebSocket, spec):
if text:
committed.append(text)
await _send({"type": "final", "text": text,
"final_kind": "utterance",
"segments": [{"start": 0.0, "end": None, "text": text}],
"language": "auto", "engine": backend.id})
@@ -777,8 +1016,8 @@ async def _run_sherpa_offline(websocket: WebSocket, spec):
continue
snapshot = bytes(buf)
if len(snapshot) > sil_bytes and \
_rms(snapshot[-sil_bytes:]) < SHERPA_OFFLINE_RMS_FLOOR:
if _rms(snapshot[:-sil_bytes]) >= SHERPA_OFFLINE_RMS_FLOOR:
_pcm16_rms(snapshot[-sil_bytes:]) < SHERPA_OFFLINE_RMS_FLOOR:
if _pcm16_rms(snapshot[:-sil_bytes]) >= SHERPA_OFFLINE_RMS_FLOOR:
await _commit(snapshot)
else:
# Pure silence — drop it (keep the gate window for
@@ -824,39 +1063,18 @@ async def _run_sherpa_offline(websocket: WebSocket, spec):
# quiet user — hand the session to the capture ASR backend so the user
# still gets their words, and say which model let them down. Bounded to
# this session; the pref is left alone so the user stays in control.
model_silent = is_model_silent(full, heard_speech, len(session_pcm))
model_silent = is_model_silent(full, heard_speech, session_pcm.total_bytes)
if model_silent:
logger.warning(
"dictation model %s decoded NOTHING from %.1fs of speech-level audio "
"— falling back to the capture ASR engine for this session",
spec.id, len(session_pcm) / float(max(1, pcm_sr) * 2),
recovered, recovered_segments = await _recover_silent_sherpa(
spec, session_pcm.tail(), pcm_sr,
)
# Demote it so the NEXT session doesn't repeat this round trip. The
# curated default can be broken on a platform we never tested (the
# NeMo-TDT decoder is, on Windows), and observing it beats guessing.
try:
from services.sherpa_dictation import demote_model
if demote_model(spec.id):
logger.error(
"dictation model %s demoted on this machine — it will no longer be "
"auto-selected. Pick it again in Settings to give it another chance.",
spec.id,
)
except Exception:
logger.exception("silent-model demotion failed")
try:
result = await _transcribe_buffer_full([bytes(session_pcm)], pcm_sr=pcm_sr)
fb_text = polish_text((result or {}).get("text", "") or "")
if fb_text:
full = fb_text
segments = (result or {}).get("segments") or [
{"start": 0.0, "end": None, "text": fb_text}
]
except Exception:
logger.exception("dictation silent-model fallback failed")
if recovered:
full = recovered
segments = recovered_segments
if not client_disconnected:
payload = {"type": "final", "text": full, "segments": segments,
payload = {"type": "final", "text": full, "final_kind": "summary",
"segments": segments,
"language": "auto", "engine": backend.id}
if model_silent:
# The client surfaces this so a silently-broken model can't look
@@ -884,6 +1102,35 @@ async def _run_sherpa_offline(websocket: WebSocket, spec):
pass
def _result_text(result: dict | None) -> str:
"""Normalize text from every ASR backend result shape.
Some backends return a top-level ``text`` value, while WhisperX, Faster
Whisper, Moonshine, and OpenAI-compatible ASR expose only ``segments`` and
``chunks``. Dictation partials and finals must interpret both contracts the
same way.
"""
if not isinstance(result, dict):
return ""
text = result.get("text")
if isinstance(text, str) and text.strip():
return text.strip()
for key in ("segments", "chunks"):
items = result.get(key)
if not isinstance(items, (list, tuple)):
continue
text = " ".join(
str(item.get("text", "")).strip()
for item in items
if isinstance(item, dict) and item.get("text")
).strip()
if text:
return text
return ""
async def _transcribe_buffer(chunks: list[bytes], *, pcm_sr: int | None = None) -> str:
"""Quick partial transcription of the current audio buffer."""
@@ -898,7 +1145,7 @@ async def _transcribe_buffer(chunks: list[bytes], *, pcm_sr: int | None = None)
def _run():
backend = get_capture_asr_backend()
result = backend.transcribe(tmp, word_timestamps=False)
return result.get("text", "")
return _result_text(result)
# Bound dictation transcribes (#730): a wedged whisperx/CTranslate2 call
# must not hold its GPU-pool worker forever and starve TTS / other ASR
@@ -912,7 +1159,9 @@ async def _transcribe_buffer(chunks: list[bytes], *, pcm_sr: int | None = None)
pass
async def _transcribe_buffer_full(chunks: list[bytes], *, pcm_sr: int | None = None) -> dict:
async def _transcribe_buffer_full(
chunks: list[bytes], *, pcm_sr: int | None = None, skip_sherpa: bool = False,
) -> dict:
"""Full transcription with timing info for the final result."""
tmp = _pcm16_to_wav(b"".join(chunks), pcm_sr) if pcm_sr else _chunks_to_wav(chunks)
if tmp is None:
@@ -924,15 +1173,13 @@ async def _transcribe_buffer_full(chunks: list[bytes], *, pcm_sr: int | None = N
from services.asr_backend import get_capture_asr_backend, run_transcribe_guarded
def _run():
backend = get_capture_asr_backend()
backend = get_capture_asr_backend(skip_sherpa=skip_sherpa)
t0 = time.perf_counter()
result = backend.transcribe(tmp, word_timestamps=False)
elapsed = round(time.perf_counter() - t0, 2)
segments = result.get("segments", [])
full_text = result.get("text", "")
if not full_text and segments:
full_text = " ".join(s.get("text", "") for s in segments).strip()
full_text = _result_text(result)
# Wave 1.1: strip Whisper hallucination loops from the final
# text (the string that gets auto-pasted). Segments keep the
+340 -34
View File
@@ -134,6 +134,82 @@ _save_job = dub_pipeline.save_job
# paste (or a mis-aimed binary) burn CPU in the parser.
_MAX_SUBTITLE_PASTE_CHARS = 2_000_000
_SRT_REPLACED_FIELDS = {
"id",
"start",
"end",
"text",
"text_original",
"translations",
"translate_error",
"translate_degraded",
}
def _best_overlapping_segment(cue: dict, existing: list[dict]) -> dict | None:
"""Return the prior segment with the strongest temporal overlap."""
cue_start = float(cue.get("start") or 0.0)
cue_end = float(cue.get("end") or cue_start)
cue_mid = (cue_start + cue_end) / 2.0
best = None
best_key = None
for index, segment in enumerate(existing):
start = float(segment.get("start") or 0.0)
end = float(segment.get("end") or start)
overlap = min(cue_end, end) - max(cue_start, start)
if overlap <= 0:
continue
midpoint_distance = abs(cue_mid - ((start + end) / 2.0))
key = (overlap, -midpoint_distance, -index)
if best_key is None or key > best_key:
best = segment
best_key = key
return best
def _carry_srt_voice_metadata(
cues: list[dict],
existing: list[dict],
segment_clones: dict | None,
speaker_clones: dict | None = None,
) -> tuple[list[dict], dict]:
"""Replace subtitle content while retaining the source cast assignment."""
source_clones = dict(segment_clones or {})
source_speaker_clones = dict(speaker_clones or {})
# Replacement cues get new positional ids. Starting from the old map would
# let an unmatched cue whose new id happens to equal an old id inherit an
# unrelated reference. Only explicitly overlap-matched references survive.
clones = {}
merged_segments = []
for new_id, cue in enumerate(cues):
prior = _best_overlapping_segment(cue, existing)
metadata = {
key: value
for key, value in (prior or {}).items()
if key not in _SRT_REPLACED_FIELDS
}
merged = {
**metadata,
"id": new_id,
"start": cue.get("start", 0.0),
"end": cue.get("end", 0.0),
"text": cue.get("text", ""),
"text_original": cue.get("text", ""),
}
if not merged.get("speaker_id"):
merged["speaker_id"] = cue.get("speaker_id") or "Speaker 1"
if prior is not None:
prior_id = str(prior.get("id", ""))
clone = source_clones.get(prior_id)
if clone is None:
clone = source_speaker_clones.get(prior.get("speaker_id"))
if clone is not None:
clones[str(new_id)] = clone
if merged.get("profile_id") == f"auto-seg:{prior_id}":
merged["profile_id"] = f"auto-seg:{new_id}"
merged_segments.append(merged)
return merged_segments, clones
@router.post("/dub/parse-subtitle-text")
def dub_parse_subtitle_text(req: ParseSubtitleTextRequest):
@@ -234,7 +310,32 @@ async def dub_import_srt(job_id: str, file: UploadFile = File(...)):
else:
segments = result.segments
prior_segments = [
segment for segment in (job.get("segments") or []) if isinstance(segment, dict)
]
segments, segment_clones = _carry_srt_voice_metadata(
segments,
prior_segments,
job.get("segment_clones"),
job.get("speaker_clones"),
)
job["segments"] = segments
job["segment_clones"] = segment_clones
# A pooled speaker clone is keyed only by a display label. Replacement
# cues can reuse that label without overlapping the original speaker, so
# retain matched pooled references as segment-specific clones above and
# drop the global map before rebuilding the cast.
job["speaker_clones"] = {}
if segment_clones:
from services.speaker_clone import build_cast_sources
job["cast_sources"] = build_cast_sources(
segments,
None,
segment_clones,
)
else:
job.pop("cast_sources", None)
# `source_lang` stays whatever the user (or the upload step) set; we
# don't try to language-detect off the cue text — that's noisy and the
# user usually knows what their .srt is.
@@ -351,12 +452,13 @@ async def preview_upload(video: UploadFile = File(...)):
safe_name = f"{uuid.uuid4().hex[:12]}"
vid_path = os.path.join(PREVIEW_DIR, f"{safe_name}{ext}")
wav_path = os.path.join(PREVIEW_DIR, f"{safe_name}.wav")
with open(vid_path, "wb") as f:
f.write(await video.read())
has_audio = False
if ext not in [".wav", ".mp3", ".m4a", ".aac"]:
payload = await video.read()
def _write_and_extract() -> bool:
with open(vid_path, "wb") as f:
f.write(payload)
if ext in {".wav", ".mp3", ".m4a", ".aac"}:
return False
try:
ffmpeg_cmd = [
find_ffmpeg(), "-y", "-i", vid_path,
@@ -368,10 +470,16 @@ async def preview_upload(video: UploadFile = File(...)):
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
timeout=300,
)
has_audio = True
return True
except Exception as e:
logger.warning("FFmpeg extraction failed: %s", log_safe(e))
pass
return False
# File writes and ffmpeg are blocking operations. Keep them on the bounded
# CPU pool so a large preview cannot stall unrelated API requests (#1667).
has_audio = await asyncio.get_running_loop().run_in_executor(
_cpu_pool, _write_and_extract
)
return {
"url": f"/preview/{safe_name}{ext}",
@@ -410,12 +518,52 @@ _ingest_gen = dub_pipeline.ingest_pipeline
#: container so a mislabelled video can't slip past the video-skipping branch.
_AUDIO_EXTS = {".wav", ".mp3", ".m4a", ".aac", ".flac", ".ogg", ".opus", ".wma"}
# Source-language choices exposed by the first-party dub UI, plus every
# language code Whisper can write back after auto-detection. A restored job
# may reuse that detected value as the next upload's override, so rejecting our
# own persisted codes strands otherwise valid dubbing sessions (#1737).
# Keeping this an allow-list still rejects language names and private-use
# BCP-47 tags. Values are normalized to lowercase below.
_DUB_SOURCE_LANG_CODES = frozenset({
"af", "sq", "am", "ar", "hy", "az", "eu", "be", "bn", "bs", "bg",
"my", "ca", "cmn-hans", "cmn-hant", "hr", "cs", "da", "nl", "en",
"et", "fi", "fr", "gl", "ka", "de", "el", "gu", "ht", "ha", "haw",
"he", "hi", "hu", "is", "id", "it", "ja", "jw", "kn", "kk", "km",
"ko", "ku", "ky", "lo", "la", "lv", "lt", "mk", "ms", "ml", "mt",
"mi", "mr", "mn", "ne", "no", "ps", "fa", "pl", "pt", "pa", "ro",
"ru", "sm", "gd", "sr", "sn", "sd", "si", "sk", "sl", "so", "es",
"su", "sw", "sv", "tg", "ta", "te", "th", "tr", "uk", "ur", "uz",
"vi", "cy", "xh", "yi", "yo", "zu",
"as", "ba", "bo", "br", "fo", "lb", "ln", "mg", "nn", "oc", "sa",
"tk", "tl", "tt", "yue", "zh",
})
def _source_lang_override(value: str | None) -> str | None:
"""Normalize a user-selected source language; auto/und means detect."""
code = (value or "").strip().lower()
if code in {"", "auto", "und"}:
return None
if code not in _DUB_SOURCE_LANG_CODES:
raise HTTPException(status_code=400, detail="Invalid source language code")
return code
def _detected_source_lang(value: str | None) -> str:
"""Normalize an ASR language without truncating valid three-letter codes."""
code = (value or "en").split("_", 1)[0].strip().lower()
if code in _DUB_SOURCE_LANG_CODES:
return code
short = code[:2]
return short if short in _DUB_SOURCE_LANG_CODES else "en"
@router.post("/dub/upload")
async def dub_upload(
video: UploadFile = File(...),
job_id: Optional[str] = Form(None),
input_type: str = Form("video"),
source_lang: Optional[str] = Form(None),
):
"""Accept a media upload, write to disk, queue background prep task.
@@ -445,6 +593,7 @@ async def dub_upload(
detail=f"Audio-only dubbing needs an audio file ({', '.join(sorted(_AUDIO_EXTS))}); got '{ext or 'no extension'}'.",
)
source_lang_override = _source_lang_override(source_lang)
os.makedirs(job_dir, exist_ok=True)
video_path = os.path.join(job_dir, f"original{ext}")
@@ -456,7 +605,13 @@ async def dub_upload(
await task_manager.add_task(
task_id, "prep",
_ingest_gen, job_id, job_dir,
{"kind": "file", "path": video_path, "input_type": input_type}, filename,
{
"kind": "file",
"path": video_path,
"input_type": input_type,
"source_lang": source_lang_override,
},
filename,
)
return JSONResponse(
status_code=202,
@@ -478,6 +633,7 @@ async def dub_ingest_url(req: DubIngestUrlRequest, request: Request):
status_code=400,
detail="URL must start with http:// or https://. Paste a full video link (e.g. https://youtube.com/watch?v=…) or drop a local file instead.",
)
source_lang_override = _source_lang_override(req.source_lang)
try:
import yt_dlp # noqa: F401
@@ -513,6 +669,7 @@ async def dub_ingest_url(req: DubIngestUrlRequest, request: Request):
"fetch_subs": bool(req.fetch_subs),
"sub_langs": req.sub_langs or None,
"cookie_file": cookie_path,
"source_lang": source_lang_override,
}
try:
await task_manager.add_task(
@@ -577,6 +734,118 @@ def _clamp_num_speakers(value) -> Optional[int]:
return value if 1 <= value <= 20 else None
def _recover_from_phrase_embeddings(
diar_pipe,
diarized_segments: list[dict],
*,
phrases: list[dict],
requested_speakers: int | None,
audio_target: str,
segments: list[dict],
words: list,
):
"""Recover rapid turns when pyannote collapses a two-speaker exchange.
Uses ASR phrase boundaries and the embedding/audio components already
loaded by speaker-diarization-3.1. Weak or imbalanced clusters are rejected
so ordinary single-speaker recordings remain untouched. Returns
``(segments, separation)`` or ``None``.
"""
present = {
str(seg.get("speaker_id")) for seg in diarized_segments
if seg.get("speaker_id")
}
if len(present) > 1:
return None
usable_phrases = [
phrase for phrase in phrases
if phrase.get("text")
and float(phrase.get("end", 0.0)) - float(phrase.get("start", 0.0)) >= 0.75
]
if len(usable_phrases) < 4:
return None
requested = int(requested_speakers) if requested_speakers else 2
if requested != 2:
return None
embedding = getattr(diar_pipe, "_embedding", None)
audio = getattr(diar_pipe, "_audio", None)
if embedding is None or audio is None:
return None
try:
import numpy as np
from pyannote.core import Segment as _PyannoteSegment
from sklearn.cluster import AgglomerativeClustering
vectors = []
durations = []
for phrase in usable_phrases:
start, end = float(phrase["start"]), float(phrase["end"])
duration = end - start
waveform, _ = audio.crop(
audio_target, _PyannoteSegment(start, end),
duration=duration, mode="pad",
)
vector = np.asarray(embedding(waveform[None])).reshape(-1)
if not np.isfinite(vector).all():
return None
vectors.append(vector)
durations.append(duration)
matrix = np.vstack(vectors)
labels = np.asarray(AgglomerativeClustering(
n_clusters=2, metric="cosine", linkage="average",
).fit_predict(matrix))
if len(set(labels.tolist())) != 2:
return None
counts = [int(np.sum(labels == cluster)) for cluster in (0, 1)]
cluster_durations = [
float(sum(duration for duration, label in zip(durations, labels) if label == cluster))
for cluster in (0, 1)
]
if min(counts) < 2 or min(cluster_durations) < 1.5:
return None
normalized = matrix / np.maximum(np.linalg.norm(matrix, axis=1, keepdims=True), 1e-8)
similarities = normalized @ normalized.T
within, cross = [], []
for left in range(len(labels)):
for right in range(left + 1, len(labels)):
target = within if labels[left] == labels[right] else cross
target.append(float(similarities[left, right]))
if not within or not cross:
return None
separation = float(np.mean(within) - np.mean(cross))
min_separation = 0.12 if requested_speakers == 2 else 0.18
if separation < min_separation:
logger.info(
"phrase-embedding speaker recovery rejected (separation=%.3f < %.3f)",
separation, min_separation,
)
return None
speaker_map = {}
turns = []
for phrase, label in zip(usable_phrases, labels.tolist()):
if label not in speaker_map:
speaker_map[label] = f"Speaker {len(speaker_map) + 1}"
turns.append({
"start": float(phrase["start"]),
"end": float(phrase["end"]),
"speaker": speaker_map[label],
})
# Assignment mutates segment dictionaries. Work on copies so a recovery
# rejected by the final two-speaker check cannot leak partial labels
# into the ordinary pyannote result.
assigned = assign_speakers_from_turns([dict(item) for item in segments], turns)
recovered = resplit_segments_by_turns(assigned, words, turns)
if len({item.get("speaker_id") for item in recovered if item.get("speaker_id")}) < 2:
return None
return recovered, separation
except Exception:
logger.exception("phrase-embedding speaker recovery failed")
return None
@router.get("/dub/transcribe-stream/{job_id}")
async def dub_transcribe_stream(
job_id: str,
@@ -912,6 +1181,12 @@ async def dub_transcribe_stream(
# Words (global-timeline) retained so diarization can re-split a segment
# that spans two speakers' turns at the word boundary (#486).
all_words: list = []
# Preserve the ASR backend's natural phrase boundaries before
# segment_transcript merges short neighboring phrases. Pyannote 3.1
# occasionally collapses rapid exchanges into one dominant speaker; in
# that narrow case these phrase spans give its own WeSpeaker embedding
# model clean candidate utterances for a conservative recovery pass.
asr_phrase_segments: list[dict] = []
detected_lang = None
next_seg_id = 0
chunk_errors: list[str] = []
@@ -981,21 +1256,13 @@ async def dub_transcribe_stream(
"error_code": failure["code"],
}
# Retry a failed/timed-out chunk once on a fresh pool before giving
# up. Otherwise a transient wedge on the FIRST chunk (whisperx often
# cold-loads its model there, the #730 hang) drops that whole window
# and the transcript is "missing the beginning, only middle+end".
# The retry reuses the same audio window, so a recovered chunk fills
# the hole instead of leaving silent gaps.
# Retry an ordinary completed failure once. A timed-out native call
# is different: its thread is still executing and must not overlap
# a retry against the same backend (#1669).
part = None
timed_out = False
for _attempt in range(1, _CHUNK_TRANSCRIBE_ATTEMPTS + 1):
# A wedged chunk gets the SAME guarded-timeout + pool-reset
# semantics as the whole-file paths (#730/#851):
# run_transcribe_guarded bounds the call, abandons the poisoned
# pool so the retry (and any concurrent TTS work) gets a fresh
# worker, and raises the actionable ASRTimeoutError. Run it as
# a task and poll so we can keep yielding pings — the
# EventSource connection drops without them.
# Run as a task and poll so pings keep the EventSource alive.
task = asyncio.ensure_future(run_transcribe_guarded(
_gpu_pool, _transcribe_chunk,
what=f"Dub chunk {i + 1}/{chunks_n}",
@@ -1010,9 +1277,12 @@ async def dub_transcribe_stream(
try:
part = task.result()
except ASRTimeoutError:
# The guard already reset the pool; keep the actionable
# message (it names the durable fixes, and — after repeated
# timeouts — the crash-isolated engine escape hatch).
# Python cannot kill an in-process native transcribe. Do
# not swap pools and retry over the still-running call:
# concurrent whisperx/CTranslate2 access caused the native
# Windows access violation in #1669. Stop this transcript;
# the worker remains honestly occupied until it exits.
timed_out = True
logger.error(
"Transcribe chunk %d/%d timed out after %.0fs (attempt %d/%d, job=%s)",
i + 1, chunks_n, transcribe_timeout_s, _attempt,
@@ -1031,23 +1301,36 @@ async def dub_transcribe_stream(
# error-part; the timeout path already reset the pool).
if part is not None and not part.get("error"):
break
if _attempt < _CHUNK_TRANSCRIBE_ATTEMPTS:
if timed_out:
break
if not timed_out and _attempt < _CHUNK_TRANSCRIBE_ATTEMPTS:
logger.warning(
"Retrying transcribe chunk %d/%d after failure/timeout (next attempt %d/%d, job=%s)",
i + 1, chunks_n, _attempt + 1, _CHUNK_TRANSCRIBE_ATTEMPTS, log_safe(job_id),
)
# A completed exception did not wedge the worker. Resetting
# the pool here leaked a healthy executor on every ordinary
# decode failure; run_transcribe_guarded already resets the
# pool on the only case that needs it: a real timeout.
# A completed exception did not leave native work behind,
# so retrying this same audio window is safe.
if part.get("error"):
chunk_errors.append(part["error"])
if part.get("error_code"):
chunk_error_codes.append(part["error_code"])
logger.warning("Chunk %d/%d error: %s", i + 1, chunks_n, log_safe(part["error"]))
if timed_out:
break
if detected_lang is None and part.get("language"):
detected_lang = part["language"]
asr_speaker_turns.extend(part.get("speaker_turns") or [])
for _phrase in part.get("chunks", []) or []:
_pts = _phrase.get("timestamp") or (None, None)
_ptext = (_phrase.get("text") or "").strip()
try:
_ps, _pe = float(_pts[0]), float(_pts[1])
except (TypeError, ValueError, IndexError):
continue
if _ptext and _pe > _ps:
asr_phrase_segments.append({
"start": _ps, "end": _pe, "text": _ptext,
})
chunk_segs = segment_transcript(part, duration=t1, scene_cuts=scene_cuts)
# Same word source segment_transcript used (already global-timeline),
# kept for the post-diarization speaker re-split (#486).
@@ -1313,7 +1596,25 @@ async def dub_transcribe_stream(
assigned = assign_speakers_from_diarization(all_segments, diar)
# #486: split any segment that spans two speakers' turns at the
# word boundary (single-speaker segments pass through unchanged).
return resplit_segments_by_diarization(assigned, all_words, diar), None, "pyannote"
resplit = resplit_segments_by_diarization(assigned, all_words, diar)
recovered = _recover_from_phrase_embeddings(
diar_pipe,
resplit,
phrases=asr_phrase_segments,
requested_speakers=num_speakers,
audio_target=asr_audio_target,
segments=all_segments,
words=all_words,
)
if recovered is not None:
recovered_segments, separation = recovered
logger.info(
"Recovered rapid two-speaker exchange from ASR phrase embeddings "
"(phrases=%d, separation=%.3f).",
len(asr_phrase_segments), separation,
)
return recovered_segments, None, "phrase_embeddings"
return resplit, None, "pyannote"
except Exception as e:
logger.exception("Diarization failed")
# Inline ASR turns beat the silence-gap heuristic as a crash
@@ -1522,7 +1823,9 @@ async def dub_transcribe_stream(
except Exception as e:
logger.warning("speaker_clone extraction skipped: %s", e)
job["source_lang"] = ((detected_lang or "en").split("_")[0][:2] or "en").lower()
job["source_lang"] = job.get("source_lang_override") or _detected_source_lang(
detected_lang
)
job["full_transcript"] = " ".join(s.get("text", "") for s in final_segs)
_save_job(job_id, job)
@@ -1719,7 +2022,9 @@ async def dub_transcribe(job_id: str, num_speakers: Optional[int] = None):
except Exception as e:
logger.warning("Failed to unload ASR backend: %s", e)
job["source_lang"] = (detected_lang or "en").split("_")[0][:2].lower()
job["source_lang"] = job.get("source_lang_override") or _detected_source_lang(
detected_lang
)
scene_cuts = job.get("scene_cuts") or []
segments = segment_transcript(result, duration=job.get("duration", 0.0), scene_cuts=scene_cuts)
@@ -1775,7 +2080,8 @@ async def dub_transcribe(job_id: str, num_speakers: Optional[int] = None):
# Bound the whole-file transcribe (#730): a wedged whisperx/CTranslate2
# call would otherwise hold its GPU-pool worker forever and starve
# every other request into a "can't reach backend". run_transcribe_guarded
# also resets the pool on timeout so capacity is restored.
# leaves an unkillable native worker accounted for on timeout so a
# retry cannot overlap it (#1669).
segments_result = await run_transcribe_guarded(_gpu_pool, _transcribe, what="Dub")
except asyncio.CancelledError:
job["aborted"] = True
+147 -17
View File
@@ -23,6 +23,7 @@ from services.ffmpeg_utils import (
find_ffmpeg,
run_ffmpeg,
)
from services.karaoke_ass import build_ass, scale_words
from services.video_retime import (
DRIFT_TOLERANCE_S,
RetimeError,
@@ -403,6 +404,27 @@ def _write_burn_srt(job: dict, exports_dir: str, stamp: str, dual: bool,
return sub_path
def _write_burn_ass(job: dict, exports_dir: str, stamp: str,
fitted_segments: "list[dict] | None" = None,
lang: "str | None" = None) -> str | None:
"""Karaoke variant of ``_write_burn_srt``: word-timed ASS via ``build_ass``.
Same text/timing resolution (``_segments_for_lang`` + fitted-cue overlay,
which also scales per-word times onto the fitted timeline); the basename
is plain ASCII under exports_dir so it is ffmpeg-filter-safe. Returns
None if there are no segments to render.
"""
segments = _segments_for_lang(job, lang)
if not segments:
return None
if fitted_segments:
segments = _apply_fitted_times(segments, fitted_segments)
sub_path = os.path.join(exports_dir, f"burn_subs_{stamp}.ass")
with open(sub_path, "w", encoding="utf-8") as f:
f.write(build_ass(segments))
return sub_path
def _ffmpeg_filter_escape(path: str) -> str:
"""Escape a path for use inside an ffmpeg filter value (subtitles=...).
@@ -515,6 +537,20 @@ def _apply_fitted_times(segments: list[dict], fitted: list[dict]) -> list[dict]:
patched = dict(seg)
patched["start"] = float(cue["start"])
patched["end"] = float(cue["end"])
# Karaoke burn-in: persisted word times live on the original timeline;
# scale them linearly onto the fitted cue span so the highlight sweep
# follows the retimed audio. Degenerate spans drop the words — export
# then falls back to an even split over the fitted span. Inert for
# SRT/VTT, which never read ``words``.
if isinstance(seg.get("words"), list) and seg.get("words"):
scaled = scale_words(
seg["words"], seg.get("start", 0.0), seg.get("end", 0.0),
patched["start"], patched["end"],
)
if scaled is not None:
patched["words"] = scaled
else:
patched.pop("words", None)
out.append(patched)
return out
@@ -572,11 +608,12 @@ def _build_audio_export_cmd(
async def dub_download(
job_id: str,
preserve_bg: bool = Query(True, description="Mix background noise into dubbed tracks"),
default_track: str = Query("original"),
default_track: str = Query("", description="Default audio track; omitted selects the first dubbed track"),
include_tracks: str = Query("", description="Comma-separated list of tracks to include (e.g. 'original,de,es'). Empty = include all."),
save_authorization: str = Header("", alias="X-VoiceStudio-Path-Authorization"),
burn_subs: bool = Query(False, description="Burn subtitles into the video stream (forces re-encode). Uses dual-subtitle layout when dual=1."),
dual: bool = Query(False, description="When burn_subs=1, render translated on top of italicised original."),
karaoke: bool = Query(False, description="When burn_subs=1, burn a word-timed karaoke highlight (ASS) instead of line subtitles. Ignored when dual=1 (dual karaoke is unsupported — the line burn renders instead)."),
out_format: str = Query("m4a", description="Audio-only jobs (#119): output container — wav, m4a, mp3, or flac. Ignored for video jobs."),
):
# Strict allowlist on the path param BEFORE it reaches any filesystem
@@ -607,6 +644,18 @@ async def dub_download(
for key, value in filtered_tracks.items()
}
# A dub export should play the dub without requiring player-specific track
# selection. Keep ``original`` as an explicit opt-in, but when callers omit
# the preference choose the first generated dub consistently (#1575).
if (
filtered_tracks
and not (default_track == "original" and include_original)
and default_track not in filtered_tracks
):
default_track = next(iter(filtered_tracks))
elif not filtered_tracks and include_original:
default_track = "original"
if not filtered_tracks and not include_original:
raise HTTPException(status_code=400, detail="No tracks selected for export")
@@ -631,12 +680,17 @@ async def dub_download(
fmt = (out_format or "m4a").lower()
if fmt not in _AUDIO_FORMAT_CODECS:
fmt = "m4a"
# lang_code is already constrained to an existing track key, but
# allowlist-sanitize it before it reaches the output path so a path
# component can never carry separators/traversal (same pattern as
# safe_name below).
safe_lang = "".join(c for c in lang_code if c.isalnum() or c in "-_") or "track"
out_path = os.path.join(exports_dir, f"dubbed_audio_{safe_lang}_{stamp}.{fmt}")
# Keep route/job data out of the filesystem and logging trust boundary.
# The selected format reaches the path only through literal branches.
if fmt == "wav":
output_name = f"dubbed_audio_{stamp}.wav"
elif fmt == "mp3":
output_name = f"dubbed_audio_{stamp}.mp3"
elif fmt == "flac":
output_name = f"dubbed_audio_{stamp}.flac"
else:
output_name = f"dubbed_audio_{stamp}.m4a"
out_path = os.path.join(exports_dir, output_name)
bg = _optional_dub_artifact(job.get("no_vocals_path"), job_id) if preserve_bg else None
cmd = _build_audio_export_cmd(ffmpeg, track_info["path"], bg, out_path, fmt)
try:
@@ -654,15 +708,28 @@ async def dub_download(
)
if not os.path.exists(out_path) or os.path.getsize(out_path) == 0:
raise HTTPException(status_code=500, detail="ffmpeg audio export produced no output file")
logger.info("Dub audio export wrote %s (%d bytes)", out_path, os.path.getsize(out_path))
logger.info("Dub audio export completed (%d bytes)", os.path.getsize(out_path))
base_name = os.path.splitext(job.get("filename", "output"))[0]
safe_name = "".join(c for c in base_name if c.isalnum() or c in "-_ ").strip() or "output"
dl_name = f"dubbed_{safe_name}_{safe_lang}_{stamp}.{fmt}"
# Response metadata must not become a second path-like sink for job or
# request data. Keep the user-selected format through explicit literal
# branches; source names and language keys never enter the label.
if fmt == "wav":
dl_name = f"dubbed_audio_{stamp}.wav"
elif fmt == "mp3":
dl_name = f"dubbed_audio_{stamp}.mp3"
elif fmt == "flac":
dl_name = f"dubbed_audio_{stamp}.flac"
else:
dl_name = f"dubbed_audio_{stamp}.m4a"
media_type = _MEDIA_TYPES.get(f".{fmt}", "audio/mp4")
save_path = _consume_native_save(save_authorization)
if save_path:
return _native_save(out_path, save_path, dl_name, media_type=media_type)
# Keep the request-derived download label out of the filesystem
# trust boundary. It is response metadata, not a source or
# destination path (CodeQL, #1575).
result = _native_save(out_path, save_path, "dubbed_audio", media_type=media_type)
result["display_name"] = dl_name
return result
return FileResponse(
out_path, media_type=media_type,
headers={"Content-Disposition": content_disposition(dl_name)},
@@ -699,7 +766,18 @@ async def dub_download(
fitted_segments = _fitted_segments_for(job, default_track) if default_track and default_track != "original" else None
# Burn the DEFAULT track's text (P1.2) — it's the audio the viewer hears.
_burn_lang = default_track if default_track and default_track != "original" else None
sub_path = _write_burn_srt(job, exports_dir, stamp, dual, fitted_segments=fitted_segments, lang=_burn_lang) if burn_subs else None
# Karaoke (word-highlight) burn writes an ASS instead of the line SRT.
# Dual layout keeps the line burn — dual karaoke is out of scope, matching
# the disabled control in the Export drawer. The default (karaoke off)
# takes exactly the legacy SRT path.
sub_path = None
sub_is_ass = False
if burn_subs:
if karaoke and not dual:
sub_path = _write_burn_ass(job, exports_dir, stamp, fitted_segments=fitted_segments, lang=_burn_lang)
sub_is_ass = sub_path is not None
if sub_path is None:
sub_path = _write_burn_srt(job, exports_dir, stamp, dual, fitted_segments=fitted_segments, lang=_burn_lang)
# ── Smart Fit video retime (two-tier) ─────────────────────────────────
# Tier 1 (≤48 chunks): single filter_complex graph inlined into the mux
@@ -799,14 +877,16 @@ async def dub_download(
esc = _ffmpeg_filter_escape(sub_path)
# Burn AFTER any retime so cues (already on the fitted timeline for
# Smart Fit) land on the retimed video. Without retime this reduces
# to the legacy `[0:v]subtitles=…[vsub]` graph.
# to the legacy `[0:v]subtitles=…[vsub]` graph. Karaoke burns the
# word-timed ASS through the ass filter at the same graph position.
if video_map.startswith("["):
sub_src = video_map
elif retimed_idx is not None:
sub_src = f"[{retimed_idx}:v]"
else:
sub_src = "[0:v]"
filter_parts.append(f"{sub_src}subtitles='{esc}'[vsub]")
_sub_filter = "ass" if sub_is_ass else "subtitles"
filter_parts.append(f"{sub_src}{_sub_filter}='{esc}'[vsub]")
video_map = "[vsub]"
if stretch_entry:
orig_dur = float(stretch_entry.get("orig_duration") or job.get("duration") or 0.0)
@@ -887,7 +967,10 @@ async def dub_download(
if default_track == "original" and include_original:
cmd += ["-disposition:a:0", "default"]
else:
target_idx = 0
# A stale/missing language preference still means "play a dub", not
# "silently fall back to the source". The first processed dub is the
# deterministic fallback; ``original`` above remains explicit.
target_idx = tracks_to_process[0]["stream_idx"] if tracks_to_process else 0
for t in tracks_to_process:
if t['lang_code'] == default_track:
target_idx = t["stream_idx"]
@@ -1566,7 +1649,10 @@ async def dub_download_audio(
return _native_save(wav_path, save_path, dl_name, media_type="audio/wav")
return FileResponse(
wav_path, media_type="audio/wav",
headers={"Content-Disposition": content_disposition(dl_name)},
headers={
"Cache-Control": "no-store",
"Content-Disposition": content_disposition(dl_name),
},
)
@@ -1708,6 +1794,50 @@ async def dub_export_vtt(
)
@router.get("/dub/ass/{job_id}")
@router.get("/dub/ass/{job_id}/{filename}")
async def dub_export_ass(
job_id: str,
lang: str = Query(None, description="Track language code. Same text/timing resolution as /dub/srt, rendered as a karaoke (word-highlight) ASS sidecar."),
):
"""Karaoke ASS sidecar — the same script the karaoke burn-in renders.
Raw text body like /dub/srt and /dub/vtt (the Tauri side writes the file
itself; no ?save_path= variant see the comment above /dub/srt).
"""
_job_dir_or_400(job_id)
lang = _safe_lang_or_400(lang)
job = _get_job(job_id)
if not job:
raise HTTPException(status_code=404, detail="Job not found")
segments = _segments_for_lang(job, lang)
if not segments:
raise HTTPException(status_code=400, detail="No transcript segments available")
# Same strategy-aware cue timing as /dub/srt. The fitted overlay also
# scales word times; the stretch_video cue path has no per-word record,
# so words are dropped and build_ass even-splits over the new spans.
fitted = _fitted_segments_for(job, lang)
if fitted:
segments = _apply_fitted_times(segments, fitted)
else:
cues = _fitted_cue_times(job, lang)
if cues:
segments = [
{**{k: v for k, v in seg.items() if k != "words"}, "start": s, "end": e}
for seg, (s, e) in zip(segments, cues)
]
base_name = os.path.splitext(job.get('filename', 'video'))[0]
dl_name = f"subtitles_{base_name}_karaoke.ass"
return Response(
content=build_ass(segments),
media_type="text/plain",
headers={"Content-Disposition": content_disposition(dl_name)},
)
@router.get("/dub/export-segments/{job_id}")
async def dub_export_segments_zip(job_id: str, lang: str = Query(None)):
import zipfile
+138 -21
View File
@@ -1,6 +1,7 @@
import os
import re
import json
import struct
import logging
import time
import asyncio
@@ -80,6 +81,62 @@ def _prepare_oom_retry(error: Exception, *, execution_target: str) -> bool:
return True
def _cached_payload_intact(path: str, info) -> bool:
"""Cheap truth check on a cached WAV whose header we are about to trust.
The natural-rate fast path hands the mixer a PATH instead of decoded
audio, so a cache whose header reads fine but whose payload is truncated
would only fail later, during assembly after the timing plan (Smart Fit,
video stretch) had been computed from the header's frame count. The plan
would then describe audio that no longer exists and the segment would be
replaced by slot-length silence, leaving the persisted video plan and the
rendered track disagreeing.
Comparing the declared frame count against the physical ``data`` chunk
catches that without decoding: a truncated file cannot hold the samples
its header claims. Anything failing here falls through to the decoding path, which
already degrades to a warning plus silence. Formats with no fixed
bits-per-sample (compressed caches) are left to the decoder as before.
"""
try:
bits = int(getattr(info, "bits_per_sample", 0) or 0)
frames = int(getattr(info, "num_frames", 0) or 0)
channels = int(getattr(info, "num_channels", 0) or 0)
if bits <= 0 or frames <= 0 or channels <= 0:
# Undecidable metadata fails CLOSED (review on #1620): these caches
# are PCM WAVs this module wrote itself, so anything else is
# unexpected — and the decode path this falls through to handles
# every format the fast path would have.
return False
payload = frames * channels * (bits // 8)
if payload <= 0:
return False
# A WAV may carry JUNK/LIST metadata before data, so its header is not
# necessarily 44 bytes. Locate the data chunk instead of counting
# metadata as audio; otherwise an extended header can mask truncation.
file_size = os.path.getsize(path)
with open(path, "rb") as wav:
header = wav.read(12)
if len(header) != 12 or header[:4] != b"RIFF" or header[8:12] != b"WAVE":
return False
offset = 12
while offset + 8 <= file_size:
wav.seek(offset)
chunk_id = wav.read(4)
chunk_size_raw = wav.read(4)
if len(chunk_id) != 4 or len(chunk_size_raw) != 4:
return False
chunk_size = struct.unpack("<I", chunk_size_raw)[0]
data_offset = offset + 8
if chunk_id == b"data":
return chunk_size >= payload and file_size >= data_offset + payload
offset = data_offset + chunk_size + (chunk_size % 2)
return False
except Exception: # noqa: BLE001 — an unstattable cache is the decoder's problem
return False
def _underrun_min_rate() -> float:
"""Floor for the underrun fill (audio slowed toward its slot, never below
this rate). Default 0.85 stays natural-sounding; OMNIVOICE_UNDERRUN_MIN_RATE=1.0
@@ -446,6 +503,18 @@ async def dub_generate(job_id: str, req: DubRequest):
backend = await resolve_generation_backend(require_cloning=True)
except ValueError as e:
raise HTTPException(status_code=400, detail=str(e))
except Exception as e:
from core.failure import is_gpu_oom
if not is_gpu_oom(e):
raise
from core.public_errors import public_exception_response
payload = public_exception_response(
e,
fallback="The TTS model could not be loaded.",
)
raise HTTPException(status_code=503, detail=payload["detail"]) from e
async def _stream(task_id):
total = len(req.segments)
@@ -593,11 +662,11 @@ async def dub_generate(job_id: str, req: DubRequest):
voice_match = (req.voice_match or "per_line").lower()
_consistent_ref_memo: dict = {}
remote_audio: dict[int, str] = {}
# Strategy-transition guard: smart_fit re-mixes the *natural-rate*
# per-segment WAVs from disk. If the previous run used strict_slot,
# the on-disk WAVs are slot-squeezed ("slotted") — reusing them would
# double-compress. Force one full regen; afterwards seg_wav_kind is
# "natural" and partial regen / fit-only re-mix (regen_only=[]) work.
# Strategy-transition guard: concise, stretch_video and smart_fit all
# re-mix *natural-rate* per-segment WAVs. If the previous run used
# strict_slot, the on-disk WAVs are slot-squeezed ("slotted") — the
# missing tails cannot be recovered by a re-mix. Force one full regen;
# afterwards partial regen / fit-only re-mix (regen_only=[]) is safe.
# Jobs predating this field have unknown kind → also regen once.
# P1.3: the kind is per-track now (each language renders under its own
# strategy); the flat job["seg_wav_kind"] is only consulted for jobs
@@ -608,7 +677,7 @@ async def dub_generate(job_id: str, req: DubRequest):
_wav_kind = (
_kind_map.get(lang_code) if isinstance(_kind_map, dict) else job.get("seg_wav_kind")
)
if strategy == "smart_fit" and regen_only is not None and _wav_kind != "natural":
if strategy != "strict_slot" and regen_only is not None and _wav_kind != "natural":
regen_only = None
# Manifest: stable segment id per current index. Per-segment WAVs are
# named by stable id (dub_seg_path) so regen reuses the right audio after
@@ -759,15 +828,38 @@ async def dub_generate(job_id: str, req: DubRequest):
if os.path.exists(seg_wav_path):
try:
_t_cache_0 = time.perf_counter()
# Natural-rate caches are already the exact assembly
# input. Keep the durable path in the manifest so the
# mixer decodes it once; the old path decoded here,
# wrote an identical mix_<id> scratch WAV, then decoded
# that copy again. Header-only inspection preserves
# the resample fallback for caches made by an engine
# with a different sample rate.
if strategy != "strict_slot":
try:
cached_info = torchaudio.info(seg_wav_path)
except Exception:
cached_info = None
if (
cached_info is not None
and int(cached_info.sample_rate) == int(backend.sample_rate)
and _cached_payload_intact(seg_wav_path, cached_info)
):
all_segment_wavs.append(
(seg.start, seg.end, seg_wav_path, backend.sample_rate)
)
sync_scores.append(getattr(seg, 'sync_ratio', None) or 1.0)
_t_cache += time.perf_counter() - _t_cache_0
continue
cached_wav, cached_sr = torchaudio.load(seg_wav_path)
if cached_sr != backend.sample_rate:
import torchaudio.functional as AF
cached_wav = AF.resample(cached_wav, cached_sr, backend.sample_rate)
# Pad/trim to slot — except smart_fit, whose mix
# loop needs the natural-rate length to compute the
# audio/video split (the seg_wav_kind guard above
# guarantees these cached WAVs are natural-rate).
if strategy != "smart_fit":
# strict_slot persists slot-sized buffers. Every other
# strategy consumes natural-rate audio and lets the mix
# loop fit it to the current timeline.
if strategy == "strict_slot":
target_samples = int(seg_duration * backend.sample_rate)
current_samples = cached_wav.shape[-1]
if target_samples > current_samples:
@@ -1091,7 +1183,7 @@ async def dub_generate(job_id: str, req: DubRequest):
_num_step, req.guidance_scale, seg_speed, seg_profile, seg_effect_preset,
),
what="Dub generate",
timeout=generate_timeout_s(seg.text),
timeout=generate_timeout_s(seg.text, engine=backend),
)
_t_tts += time.perf_counter() - _t_tts_0
@@ -1164,12 +1256,15 @@ async def dub_generate(job_id: str, req: DubRequest):
if rvc_sr == backend.sample_rate:
audio_tensor = rvc_wav
target_samples = int(seg_duration * backend.sample_rate)
current_samples = audio_tensor.shape[-1]
if target_samples > current_samples:
audio_tensor = torch.nn.functional.pad(audio_tensor, (0, target_samples - current_samples))
elif current_samples > target_samples:
audio_tensor = audio_tensor[..., :target_samples]
if strategy == "strict_slot":
target_samples = int(seg_duration * backend.sample_rate)
current_samples = audio_tensor.shape[-1]
if target_samples > current_samples:
audio_tensor = torch.nn.functional.pad(
audio_tensor, (0, target_samples - current_samples)
)
elif current_samples > target_samples:
audio_tensor = audio_tensor[..., :target_samples]
except Exception as e:
yield f"data: {json.dumps({'type': 'warning', 'segment': i, 'message': f'RVC skipped: {str(e)[:120]}'})}\n\n"
@@ -1208,7 +1303,15 @@ async def dub_generate(job_id: str, req: DubRequest):
pass
_release_audio_tensors()
except Exception as e:
yield f"data: {json.dumps({'type': 'error', 'segment': i, 'error': str(e)})}\n\n"
# A task-stream error bypasses the global exception handler.
# Never publish engine exception text here: allocator errors
# carry process tables and arbitrary failures can carry paths,
# tokens, or source text. The shared helper enriches recognized
# classes using VoiceStudio-owned constants only.
from core.public_errors import stream_generation_failure
error_detail = stream_generation_failure(e)["detail"]
yield f"data: {json.dumps({'type': 'error', 'segment': i, 'error': error_detail})}\n\n"
sr = backend.sample_rate
all_segment_wavs.append(_store_mix_wav(seg.start, seg.end, torch.zeros(1, max(0, int(seg_duration * sr))), sr, f"mix_{seg_id}"))
sync_scores.append(1.0)
@@ -1356,7 +1459,21 @@ async def dub_generate(job_id: str, req: DubRequest):
seg_gain = getattr(seg_ref, "gain", None) if seg_ref is not None else None
seg_gain = seg_gain if seg_gain is not None else 1.0
seg_gain = max(0.0, min(2.0, seg_gain))
wav = _load_entry_wav((start, end, wav_path, sr), sr)
try:
wav = _load_entry_wav((start, end, wav_path, sr), sr)
except Exception as e:
# A WAV header can be readable while its payload is
# truncated. Direct cache reuse deliberately defers the
# decode to assembly, so preserve the old recovery contract
# here: warn and fill this slot with silence instead of
# aborting the entire dub.
warning = {
"type": "warning",
"segment": i,
"message": f"cached seg lost, padding silence: {str(e)[:120]}",
}
yield f"data: {json.dumps(warning)}\n\n"
wav = torch.zeros(1, max(0, int((end - start) * sr)))
adjusted = wav * seg_gain
if adjusted.ndim == 2 and adjusted.shape[0] > 1:
adjusted = adjusted.mean(dim=0, keepdim=True)
@@ -1798,7 +1915,7 @@ async def preview_segment(job_id: str, req: SegmentPreviewRequest):
from services.model_manager import generate_timeout_s
audio_tensor = await run_on_gpu_pool_guarded(
_gen, what="Dub preview generate",
timeout=generate_timeout_s(req.text),
timeout=generate_timeout_s(req.text, engine=backend),
)
sr = backend.sample_rate
+15
View File
@@ -75,6 +75,21 @@ def list_tts_backends():
return _family_payload("tts", tts_backend)
@router.get(
"/engines/{engine_id}/disk-usage",
dependencies=[Depends(require_admin_action)],
)
def engine_disk_usage(engine_id: str):
"""Measure owned engine bytes only when a catalogue row is opened."""
try:
tts_backend.get_backend_class(engine_id)
except ValueError:
raise HTTPException(status_code=404, detail="Unknown TTS engine")
from services.engine_disk_usage import disk_usage_for
return disk_usage_for(engine_id)
@router.get("/engines/asr")
def list_asr_backends():
return _family_payload("asr", asr_backend)
+433 -161
View File
@@ -8,6 +8,7 @@ import asyncio
import tempfile
import contextlib
import logging
import threading
import traceback
from typing import Optional
from fastapi import APIRouter, File, Form, UploadFile, HTTPException
@@ -32,6 +33,83 @@ router = APIRouter()
logger = logging.getLogger("omnivoice.generate")
class _TempReferenceLease:
"""Delete a request-owned reference once every abandoned reader drains."""
def __init__(self, path: str):
self.path = path
self._lock = threading.Lock()
self._active = 0
self._request_done = False
self._deleted = False
def acquire(self):
with self._lock:
if self._request_done:
raise RuntimeError("reference lease acquired after request cleanup")
self._active += 1
once_lock = threading.Lock()
released = False
def release() -> None:
nonlocal released
with once_lock:
if released:
return
released = True
self._release()
return release
def _release(self) -> None:
delete = False
with self._lock:
self._active -= 1
if self._active < 0:
raise RuntimeError("reference lease released too many times")
if self._request_done and self._active == 0 and not self._deleted:
self._deleted = True
delete = True
if delete:
with contextlib.suppress(OSError):
os.remove(self.path)
def finish_request(self) -> None:
delete = False
with self._lock:
self._request_done = True
if self._active == 0 and not self._deleted:
self._deleted = True
delete = True
if delete:
with contextlib.suppress(OSError):
os.remove(self.path)
async def _run_with_reference_lease(lease, factory):
"""Hold an ad-hoc reference through one local GPU-pool dispatch."""
if lease is None:
return await factory(None)
release = lease.acquire()
abandoned = False
try:
return await factory(release)
except GpuPoolBusyError:
# Busy means no job started; release now. The callback may already have
# done so, and the lease token is deliberately idempotent.
release()
abandoned = True
raise
except (asyncio.CancelledError, GpuJobTimeoutError):
# The guard owns release now: immediately for a queued cancellation,
# or from the worker finalizer after an in-flight job drains.
abandoned = True
raise
finally:
if not abandoned:
release()
def _profile_instruct(row):
"""Validator-safe instruct for a stored profile row.
@@ -47,6 +125,96 @@ def _profile_instruct(row):
return heal_design_instruct(row["instruct"], vd)
def _resolve_profile_conditioning(row, *, ref_text=None, instruct=None,
seed=None, language=None):
"""Resolve a ``voice_profiles`` row into generation conditioning.
Extracted verbatim from /generate's inline profile-resolution block so
other synthesis routes (POST /convert) share the exact same semantics
lock wins, ``kind`` is authoritative (0005), legacy pre-0004 rows fall
back to the is_locked/instruct inference, and #533's language fill.
Request-supplied values (``ref_text``/``instruct``/``seed``/``language``)
always win over the stored row; only gaps are filled. Returns a dict with
``ref_audio_path`` / ``ref_text`` / ``instruct`` / ``seed`` / ``language``
/ ``kind`` plus ``persist_ref_text`` True when the caller should cache
an auto-transcribed reference transcript back onto the row (#1032).
"""
out = {
"ref_audio_path": None, "ref_text": ref_text, "instruct": instruct,
"seed": seed, "language": language, "kind": None,
"persist_ref_text": False,
}
# `kind` is authoritative (0005): 'design' profiles condition on their
# deterministic rendered sample + instruct; 'clone' on the user's
# reference. Lock always wins (it pins a specific take). Rows from
# pre-0004 DBs mid-upgrade may lack the column → fall back to the legacy
# is_locked/instruct inference.
try:
profile_kind = row["kind"] or "clone"
except (KeyError, IndexError):
profile_kind = "design" if (
row["instruct"] and not row["is_locked"] and not row["ref_audio_path"]
) else "clone"
out["kind"] = profile_kind
if row["is_locked"] and row["locked_audio_path"]:
out["ref_audio_path"] = os.path.join(VOICES_DIR, row["locked_audio_path"])
if not out["ref_text"]:
out["ref_text"] = row["ref_text"]
if not out["instruct"]:
out["instruct"] = _profile_instruct(row)
if out["seed"] is None and row["seed"] is not None:
out["seed"] = row["seed"]
elif profile_kind == "design":
# Rendered sample (if present) carries the voice identity; instruct
# alone is the fallback for legacy archetype rows.
out["ref_audio_path"] = (
os.path.join(VOICES_DIR, row["ref_audio_path"]) if row["ref_audio_path"] else None
)
if out["ref_audio_path"] and not out["ref_text"] and row["ref_text"]:
out["ref_text"] = row["ref_text"]
if not out["instruct"]:
out["instruct"] = _profile_instruct(row)
if out["seed"] is None and row["seed"] is not None:
out["seed"] = row["seed"]
elif row["instruct"] and not row["is_locked"] and not row["ref_audio_path"]:
# Legacy design-shaped row (pre-0004 archetype materialization failure
# path): instruct-only conditioning.
if not out["instruct"]:
out["instruct"] = _profile_instruct(row)
if out["seed"] is None and row["seed"] is not None:
out["seed"] = row["seed"]
else:
out["ref_audio_path"] = (
os.path.join(VOICES_DIR, row["ref_audio_path"]) if row["ref_audio_path"] else None
)
if not out["ref_text"] and row["ref_text"]:
out["ref_text"] = row["ref_text"]
elif out["ref_audio_path"] and not out["ref_text"]:
# Empty stored transcript → the caller's auto-transcribe will run;
# cache its result onto the profile so it runs ONCE, not on every
# generate (#1032 perf regression).
out["persist_ref_text"] = True
if not out["instruct"] and row["instruct"]:
out["instruct"] = row["instruct"]
if out["seed"] is None and row["seed"] is not None:
out["seed"] = row["seed"]
if out["language"] == "Auto":
out["language"] = None
# #533: a profile's stored language must drive generation when the request
# didn't pin one. An EXPLICIT non-Auto request language still wins; we
# only fill the gap. `row` is a sqlite3.Row, so guard the column lookup
# for pre-language DBs mid-upgrade.
if out["language"] is None:
try:
prof_lang = row["language"]
except (KeyError, IndexError):
prof_lang = None
if prof_lang and prof_lang != "Auto":
out["language"] = prof_lang
return out
def _note_generate_progress() -> None:
"""Tell the pool guard this render just finished a unit of work (#1391).
@@ -381,6 +549,31 @@ def _is_timeout_failure(e) -> bool:
return False
def _is_media_process_launch_failure(exc: BaseException) -> bool:
"""Identify an ffmpeg/ffprobe launch ENOENT without guessing from a file name."""
if not isinstance(exc, FileNotFoundError):
return False
# A regular missing reference/model file may itself be named "ffmpeg".
# Require the innermost raise site to be Python's process launcher so that
# basename collisions keep the normal missing-file diagnosis (#1677).
traceback_cursor = exc.__traceback__
if traceback_cursor is None:
return False
while traceback_cursor.tb_next is not None:
traceback_cursor = traceback_cursor.tb_next
origin_module = traceback_cursor.tb_frame.f_globals.get("__name__", "")
if origin_module != "subprocess" and not origin_module.startswith("asyncio."):
return False
filename = getattr(exc, "filename", None)
if not filename:
return "[winerror 2]" in str(exc).lower()
return os.path.basename(str(filename)).lower() in {
"ffmpeg", "ffmpeg.exe", "ffprobe", "ffprobe.exe",
}
def _oom_friendly_reraise(e):
"""Best-effort cache flush + the user-facing OOM hint shared by both
inference paths."""
@@ -405,6 +598,21 @@ def _oom_friendly_reraise(e):
# that lost its +x bit) is NOT an OOM — don't send the user to the Flush
# button; tell them what's actually wrong.
es = str(e)
# #1677: Windows CreateProcess reports a missing executable as a bare
# ``FileNotFoundError: [WinError 2] ...`` with no filename, while POSIX
# includes the missing ffmpeg/ffprobe name. The bundled-media downloader
# now republishes PATH as soon as it finishes, but a failed/blocked
# download still needs an actionable recovery rather than the unknown-
# error dead end. Keep missing reference/model files on their own path.
for _exc in _exception_chain(e):
if _is_media_process_launch_failure(_exc):
raise RuntimeError(
"A required media program couldn't be launched. Open "
"Settings → Audio tools and use "
"Download/Repair for the media engine, then retry. If Audio "
"tools is already ready, repair the selected TTS engine and "
f"restart VoiceStudio. Underlying error: {_safe_exc_text(_exc)}"
) from e
if isinstance(e, PermissionError) or "Permission denied" in es or "Errno 13" in es:
raise RuntimeError(
f"A required engine binary couldn't be executed (permission denied). "
@@ -596,16 +804,23 @@ def _oom_friendly_reraise(e):
) from e
def _generate_timeout_s(text: str) -> float:
def _generate_timeout_s(text: str, *, execution_device=None, min_vram_gb=0.0) -> float:
"""Wall-clock budget for one generate, scaled to the request.
Thin alias for the canonical helper, which moved to
``services.model_manager.generate_timeout_s`` (#1190) so /v1/audio/speech,
batch, dub and archetype previews share it instead of each re-deriving (or,
as they did, silently keeping the flat 300s).
``min_vram_gb`` is the engine's declared VRAM floor. A GPU below it pages to
system RAM and renders slower than this machine's CPU, so it must not be
budgeted as fast hardware (#1804) — the same figure the dispatch already
hands the guard so a timeout message can name the card (#1226/#1222).
"""
from services.model_manager import generate_timeout_s
return generate_timeout_s(text)
return generate_timeout_s(
text, execution_device=execution_device, min_vram_gb=min_vram_gb,
)
def _run_inference(
@@ -695,15 +910,17 @@ def _run_backend_inference(
backend, text, language, ref_audio_path, ref_text, instruct, duration,
num_step, guidance_scale, speed, denoise, postprocess_output,
used_seed, effect_preset="broadcast",
max_chunk_chars=None, crossfade_ms=None, *, dropped_sink=None,
max_chunk_chars=None, crossfade_ms=None, *, t_shift=None,
layer_penalty_factor=None, position_temperature=None,
class_temperature=None, dropped_sink=None,
):
"""Engine-aware twin of :func:`_run_inference` (issue #312).
Runs the request through a pluggable ``TTSBackend`` adapter instead of the
VoiceStudio model directly. The adapter protocol is narrower than the
VoiceStudio-native surface engine-specific extras (``t_shift``,
``layer_penalty_factor``, ) only exist on the native path, which is why
VoiceStudio itself still goes through ``_run_inference``.
VoiceStudio model directly. A crash-isolated OmniVoice proxy advertises
``supports_native_omnivoice_controls`` and receives the same advanced
controls and per-call seed as the native path; other adapters keep the
narrower protocol unchanged.
"""
import torch
try:
@@ -718,6 +935,18 @@ def _run_backend_inference(
instruct=instruct, num_step=num_step, guidance_scale=guidance_scale,
speed=speed, denoise=denoise, postprocess_output=postprocess_output,
)
native_proxy = bool(
getattr(backend, "supports_native_omnivoice_controls", False)
)
if native_proxy:
gen_kwargs.update({
key: value for key, value in {
"t_shift": t_shift,
"layer_penalty_factor": layer_penalty_factor,
"position_temperature": position_temperature,
"class_temperature": class_temperature,
}.items() if value is not None
})
sr = backend.sample_rate
# Inline [pause Nms] markers (issue #276) work for every engine — the
@@ -727,10 +956,17 @@ def _run_backend_inference(
has_pause = len(segments) > 1 or (segments and segments[0][1] > 0)
if has_pause:
first_span = True
def _gen_span(span_text):
nonlocal first_span
# Per-span duration is left to the engine; an explicit overall
# `duration` can't be meaningfully split across spans.
return backend.generate(span_text, duration=None, **gen_kwargs)
span_kwargs = dict(gen_kwargs)
if native_proxy and first_span and used_seed is not None:
span_kwargs["seed"] = used_seed
first_span = False
return backend.generate(span_text, duration=None, **span_kwargs)
audio_out = _render_with_pauses(_gen_span, segments, sr)
else:
# Wave 1.2: sentence-boundary chunking for long text (see
@@ -747,12 +983,19 @@ def _run_backend_inference(
for i, chunk_text in enumerate(text_chunks):
if used_seed is not None:
torch.manual_seed(used_seed + i)
parts.append(backend.generate(chunk_text, duration=None, **gen_kwargs))
chunk_kwargs = dict(gen_kwargs)
if native_proxy and used_seed is not None:
chunk_kwargs["seed"] = used_seed + i
parts.append(backend.generate(
chunk_text, duration=None, **chunk_kwargs
))
_note_generate_progress()
audio_out = concatenate_audio_chunks(parts, sr, _xfade_ms,
texts=text_chunks,
sink=dropped_sink)
else:
if native_proxy and used_seed is not None:
gen_kwargs["seed"] = used_seed
audio_out = backend.generate(text, duration=duration, **gen_kwargs)
return _apply_effect_chain(
@@ -870,7 +1113,6 @@ async def _finalize_generation(
Returns ``(watermarked_tensor, meta)`` where ``meta`` carries
``id`` / ``filename`` / ``duration`` / ``gen_time``.
"""
loop = asyncio.get_running_loop()
# Invisible AudioSeal provenance watermark on the final audio. Embedding
# was previously only wired into the dub pipeline (dub_generate.py), so
# plain TTS came out unmarked despite the setting being on — and the same
@@ -882,12 +1124,9 @@ async def _finalize_generation(
# AudioSeal embedding is CPU work that holds no VRAM, so occupying a GPU
# worker with it only delays the next generate on 1-worker hosts.
if not already_marked:
from services.watermark import mark_synthetic
from services.model_manager import get_watermark_pool
audio_tensor = await loop.run_in_executor(
get_watermark_pool(),
functools.partial(mark_synthetic, audio_tensor, sample_rate,
context="generate.finalize"),
from services.watermark import mark_synthetic_async
audio_tensor = await mark_synthetic_async(
audio_tensor, sample_rate, context="generate.finalize",
)
gen_time = round(time.time() - start_time, 2)
@@ -1198,6 +1437,10 @@ async def generate_speech(
_backend = None
_engine_min_vram_gb = getattr(backend_cls, "min_vram_gb", 0.0)
_routing_notice = None
# Remote renders deliberately skip this host's capability gate. Keep the
# local fallback call's timeout device-neutral so the closure is valid
# without pretending the control plane describes the remote worker.
_routing = {"effective_device": None}
if not _remote:
# Single-active-engine memory discipline: hand back any OTHER resident
@@ -1297,6 +1540,7 @@ async def generate_speech(
ref_audio_path = None
cleanup_ref = False
ref_lease = None
used_seed = seed
resolved_profile_id = None
history_mode = None # profile.kind when a profile drives; else inferred at insert
@@ -1314,76 +1558,28 @@ async def generate_speech(
row = conn.execute("SELECT * FROM voice_profiles WHERE id=?", (profile_id,)).fetchone()
if row:
resolved_profile_id = profile_id
# `kind` is authoritative (0005): 'design' profiles condition on
# their deterministic rendered sample + instruct; 'clone' on the
# user's reference. Lock always wins (it pins a specific take).
# Rows from pre-0004 DBs mid-upgrade may lack the column → fall
# back to the legacy is_locked/instruct inference.
try:
profile_kind = row["kind"] or "clone"
except (KeyError, IndexError):
profile_kind = "design" if (row["instruct"] and not row["is_locked"] and not row["ref_audio_path"]) else "clone"
history_mode = profile_kind
if row["is_locked"] and row["locked_audio_path"]:
ref_audio_path = os.path.join(VOICES_DIR, row["locked_audio_path"])
if not ref_text:
ref_text = row["ref_text"]
if not instruct:
instruct = _profile_instruct(row)
if used_seed is None and row["seed"] is not None:
used_seed = row["seed"]
elif profile_kind == "design":
# Rendered sample (if present) carries the voice identity;
# instruct alone is the fallback for legacy archetype rows.
ref_audio_path = os.path.join(VOICES_DIR, row["ref_audio_path"]) if row["ref_audio_path"] else None
if ref_audio_path and not ref_text and row["ref_text"]:
ref_text = row["ref_text"]
if not instruct:
instruct = _profile_instruct(row)
if used_seed is None and row["seed"] is not None:
used_seed = row["seed"]
elif row["instruct"] and not row["is_locked"] and not row["ref_audio_path"]:
# Legacy design-shaped row (pre-0004 archetype materialization
# failure path): instruct-only conditioning.
if not instruct:
instruct = _profile_instruct(row)
if used_seed is None and row["seed"] is not None:
used_seed = row["seed"]
else:
ref_audio_path = os.path.join(VOICES_DIR, row["ref_audio_path"]) if row["ref_audio_path"] else None
if not ref_text and row["ref_text"]:
ref_text = row["ref_text"]
elif ref_audio_path and not ref_text:
# Empty stored transcript → the auto-transcribe below will
# run; cache its result onto the profile so it runs ONCE,
# not on every generate (#1032 perf regression).
persist_ref_text_profile_id = profile_id
if not instruct and row["instruct"]:
instruct = row["instruct"]
if used_seed is None and row["seed"] is not None:
used_seed = row["seed"]
if language == "Auto":
language = None
# #533: a profile's stored language must drive generation when the
# request didn't pin one. Without this the German (etc.) archetype
# generates with language=None and the model drifts to English —
# even though the archetype PREVIEW renders correctly (archetypes.py
# passes the language). An EXPLICIT non-Auto request language still
# wins; we only fill the gap. `row` is a sqlite3.Row, so guard the
# column lookup for pre-language DBs mid-upgrade.
if language is None:
try:
prof_lang = row["language"]
except (KeyError, IndexError):
prof_lang = None
if prof_lang and prof_lang != "Auto":
language = prof_lang
# Shared with POST /convert — see _resolve_profile_conditioning
# for the resolution rules (kind-authoritative, lock wins, #533
# language fill, #1032 transcript-cache signal).
_cond = _resolve_profile_conditioning(
row, ref_text=ref_text, instruct=instruct, seed=used_seed,
language=language,
)
history_mode = _cond["kind"]
ref_audio_path = _cond["ref_audio_path"]
ref_text = _cond["ref_text"]
instruct = _cond["instruct"]
used_seed = _cond["seed"]
language = _cond["language"]
if _cond["persist_ref_text"]:
persist_ref_text_profile_id = profile_id
elif ref_audio is not None:
try:
with tempfile.NamedTemporaryFile(delete=False, suffix=".wav") as f:
f.write(await ref_audio.read())
ref_audio_path = f.name
cleanup_ref = True
ref_lease = _TempReferenceLease(ref_audio_path)
except Exception as e:
raise HTTPException(status_code=500, detail=str(e))
@@ -1400,13 +1596,19 @@ async def generate_speech(
# built-in ASR fallback), so a timeout degrades to None rather than
# failing the whole generate.
try:
ref_text = await run_on_gpu_pool_guarded(
functools.partial(transcribe_reference, ref_audio_path),
what="Reference transcribe",
# Floor budget (#1190): a reference clip is seconds of audio,
# so the length-scaled bonus never applies — but the timeout is
# explicit here too, so no dispatch relies on a hidden default.
timeout=_generate_timeout_s(""),
ref_text = await _run_with_reference_lease(
ref_lease,
lambda release: run_on_gpu_pool_guarded(
functools.partial(transcribe_reference, ref_audio_path),
what="Reference transcribe",
# Floor budget (#1190): a reference clip is seconds of audio,
# so the length-scaled bonus never applies — but the timeout is
# explicit here too, so no dispatch relies on a hidden default.
timeout=_generate_timeout_s(
"", execution_device=_routing["effective_device"]
),
on_abandon=release,
)
)
# TimeoutError covers both the execution bound and pool saturation:
# this path is best-effort either way.
@@ -1523,7 +1725,8 @@ async def generate_speech(
local=gpu_gateway.LocalCall(
_remote_only_local_call(_target_label),
what="TTS generate",
timeout=_generate_timeout_s(text),
timeout=_generate_timeout_s(text, execution_device=_routing["effective_device"],
min_vram_gb=_engine_min_vram_gb),
min_vram_gb=_engine_min_vram_gb,
),
remote=_remote_call,
@@ -1662,19 +1865,30 @@ async def generate_speech(
"target_label": e.worker_label or _target_label,
"hint": e.hint,
})
except Exception:
except Exception as exc:
# Mid-job remote failure is NOT quietly redone here: the client
# treats a retryable error as "surface it", so the user decides
# whether to spend the same minutes again on this machine.
logger.error("Remote generation failed", exc_info=True)
from core.public_errors import stream_failure
yield _line({"type": "error", **stream_failure("generation_failed")})
# whether to spend the same minutes again on this machine. Like
# the local streaming path, this in-band frame stands in for the
# global 500 handler, so it journals the scrubbed failure and
# names a recognized cause instead of the bare generic string
# (#1607).
logger.error(
"Remote generation failed (class=%s)",
type(exc).__name__,
)
from core.public_errors import stream_generation_failure
from core import error_journal
error_journal.record(
exc, route="/generate", trace=traceback.format_exc()
)
yield _line({"type": "error", **stream_generation_failure(exc)})
finally:
if not render.done():
render.cancel()
if cleanup_ref and ref_audio_path:
with contextlib.suppress(OSError):
os.remove(ref_audio_path)
if cleanup_ref and ref_lease is not None:
ref_lease.finish_request()
return StreamingResponse(
_remote_stream_events(),
@@ -1716,6 +1930,17 @@ async def generate_speech(
instruct=instruct, num_step=num_step,
guidance_scale=guidance_scale, speed=speed,
denoise=denoise, postprocess_output=postprocess_output,
**({
key: value for key, value in {
"t_shift": t_shift,
"layer_penalty_factor": layer_penalty_factor,
"position_temperature": position_temperature,
"class_temperature": class_temperature,
"seed": used_seed + i if used_seed is not None else None,
}.items() if value is not None
} if getattr(
_backend, "supports_native_omnivoice_controls", False
) else {}),
)
sr = _backend.sample_rate
skip = getattr(_backend, "applies_own_mastering", False)
@@ -1779,33 +2004,47 @@ async def generate_speech(
if _has_pause or len(_text_chunks) <= 1:
# Single-shot pipeline, unchanged — streamed as one chunk.
if _backend is not None:
audio_tensor = await run_on_gpu_pool_guarded(
functools.partial(
_run_backend_inference,
_backend, text, language, ref_audio_path, ref_text,
instruct, duration, num_step, guidance_scale, speed,
denoise, postprocess_output, used_seed, effect_preset,
max_chunk_chars, crossfade_ms, dropped_sink=_dropped_sink,
),
what="TTS generate",
min_vram_gb=_engine_min_vram_gb,
timeout=_generate_timeout_s(text),
audio_tensor = await _run_with_reference_lease(
ref_lease,
lambda release: run_on_gpu_pool_guarded(
functools.partial(
_run_backend_inference,
_backend, text, language, ref_audio_path, ref_text,
instruct, duration, num_step, guidance_scale, speed,
denoise, postprocess_output, used_seed, effect_preset,
max_chunk_chars, crossfade_ms, t_shift=t_shift,
layer_penalty_factor=layer_penalty_factor,
position_temperature=position_temperature,
class_temperature=class_temperature,
dropped_sink=_dropped_sink,
),
what="TTS generate",
min_vram_gb=_engine_min_vram_gb,
timeout=_generate_timeout_s(text, execution_device=_routing["effective_device"],
min_vram_gb=_engine_min_vram_gb),
on_abandon=release,
)
)
sample_rate = _backend.sample_rate
else:
audio_tensor = await run_on_gpu_pool_guarded(
functools.partial(
_run_inference,
_model, text, language, ref_audio_path, ref_text,
instruct, duration, num_step, guidance_scale, speed,
t_shift, denoise, postprocess_output,
layer_penalty_factor, position_temperature,
class_temperature, used_seed, effect_preset,
max_chunk_chars, crossfade_ms, dropped_sink=_dropped_sink,
),
what="TTS generate",
min_vram_gb=_engine_min_vram_gb,
timeout=_generate_timeout_s(text),
audio_tensor = await _run_with_reference_lease(
ref_lease,
lambda release: run_on_gpu_pool_guarded(
functools.partial(
_run_inference,
_model, text, language, ref_audio_path, ref_text,
instruct, duration, num_step, guidance_scale, speed,
t_shift, denoise, postprocess_output,
layer_penalty_factor, position_temperature,
class_temperature, used_seed, effect_preset,
max_chunk_chars, crossfade_ms, dropped_sink=_dropped_sink,
),
what="TTS generate",
min_vram_gb=_engine_min_vram_gb,
timeout=_generate_timeout_s(text, execution_device=_routing["effective_device"],
min_vram_gb=_engine_min_vram_gb),
on_abandon=release,
)
)
sample_rate = _model.sampling_rate
yield _line({
@@ -1822,12 +2061,10 @@ async def generate_speech(
# (#1190): AudioSeal embedding is CPU work that owns no
# VRAM, and on a 1-worker host it used to serialize
# directly ahead of the next generate.
from services.watermark import mark_synthetic
from services.model_manager import get_watermark_pool
_preview = await asyncio.get_running_loop().run_in_executor(
get_watermark_pool(),
functools.partial(mark_synthetic, audio_tensor, sample_rate,
context="generate.stream_preview"),
from services.watermark import mark_synthetic_async
_preview = await mark_synthetic_async(
audio_tensor, sample_rate,
context="generate.stream_preview",
)
yield _line({"type": "chunk", "seq": 0, "pcm": _pcm16_b64(_preview)})
else:
@@ -1836,25 +2073,28 @@ async def generate_speech(
for i, chunk_text in enumerate(_text_chunks):
# Bounded per chunk + pool-reset on hang (#730 class);
# a timeout surfaces as an "error" event below.
raw, preview, sample_rate = await run_on_gpu_pool_guarded(
functools.partial(_render_stream_chunk, i, chunk_text),
what="TTS generate",
min_vram_gb=_engine_min_vram_gb,
# Budget scaled to THIS chunk (#1190) — the flat
# 300s here is what made long streamed renders fail
# even after the v0.3.22 scaled budget shipped.
timeout=_generate_timeout_s(chunk_text),
raw, preview, sample_rate = await _run_with_reference_lease(
ref_lease,
lambda release: run_on_gpu_pool_guarded(
functools.partial(_render_stream_chunk, i, chunk_text),
what="TTS generate",
min_vram_gb=_engine_min_vram_gb,
# Budget scaled to THIS chunk (#1190) — the flat
# 300s here is what made long streamed renders fail
# even after the v0.3.22 scaled budget shipped.
timeout=_generate_timeout_s(chunk_text, execution_device=_routing["effective_device"],
min_vram_gb=_engine_min_vram_gb),
on_abandon=release,
)
)
parts.append(raw)
# Provenance-mark the streamed copy off the GPU pool
# (#1169 mark, #1190 placement): CPU-only AudioSeal
# work must not occupy a GPU worker between chunks.
from services.watermark import mark_synthetic
from services.model_manager import get_watermark_pool
preview = await asyncio.get_running_loop().run_in_executor(
get_watermark_pool(),
functools.partial(mark_synthetic, preview, sample_rate,
context="generate.stream_preview"),
from services.watermark import mark_synthetic_async
preview = await mark_synthetic_async(
preview, sample_rate,
context="generate.stream_preview",
)
if i == 0:
# After the first render so lazy-loading engines
@@ -1869,7 +2109,7 @@ async def generate_speech(
audio_tensor = await run_on_gpu_pool_guarded(
functools.partial(_assemble_stream_chunks, parts, sample_rate),
what="TTS assemble",
timeout=_generate_timeout_s(text),
timeout=_generate_timeout_s(text, execution_device=_routing["effective_device"]),
)
_, meta = await _finalize_generation(
@@ -1898,7 +2138,7 @@ async def generate_speech(
# Client went away mid-stream — same semantics as aborting a
# classic /generate mid-render: nothing is saved.
raise
except (GpuJobTimeoutError, GpuPoolBusyError) as e:
except GpuPoolBusyError as e:
# In-band error frame carries the machine-readable retryable
# marker (#1190) — an NDJSON consumer can back off instead of
# guessing from the prose.
@@ -1907,20 +2147,44 @@ async def generate_speech(
failure = stream_failure("generation_busy")
failure["retry_after"] = getattr(e, "retry_after", 30)
yield _line({"type": "error", **failure})
except GpuJobTimeoutError:
# The worker started and spent its full execution budget. That
# is compute time, not queue pressure (#1588).
logger.error("Streaming generation exceeded its compute budget")
from core.public_errors import stream_failure
failure = stream_failure("generation_timeout")
failure["retry_after"] = 30
yield _line({"type": "error", **failure})
except ValueError:
logger.error("Streaming generation request rejected")
from core.public_errors import stream_failure
yield _line({"type": "error", **stream_failure("invalid_request")})
except Exception:
logger.error("Streaming generation failed unexpectedly")
from core.public_errors import stream_failure
yield _line({"type": "error", **stream_failure("generation_failed")})
except Exception as exc:
# A streaming request answers 200 and carries its failure as an
# in-band error frame, so it never reaches the global 500
# handler — which is where a classic /generate failure gets its
# scrubbed journal entry (Diagnostics / recent errors) AND its
# classified, actionable message. Both have to be reproduced
# here or a streaming generation failure is invisible in the
# diagnostic bundle and opaque to the user (#1607). The raw
# exception is NOT logged: it can carry a reference-clip path or
# a provider secret, and only the journal scrubs before storing.
logger.error(
"Streaming generation failed unexpectedly (class=%s)",
type(exc).__name__,
)
from core.public_errors import stream_generation_failure
from core import error_journal
error_journal.record(
exc, route="/generate", trace=traceback.format_exc()
)
yield _line({"type": "error", **stream_generation_failure(exc)})
finally:
# Ownership of the temp reference clip moves to this generator
# in stream mode (the route returns before rendering starts).
if cleanup_ref and ref_audio_path:
with contextlib.suppress(OSError):
os.remove(ref_audio_path)
if cleanup_ref and ref_lease is not None:
ref_lease.finish_request()
# Routing notice (#21): known before the stream starts, so it rides the
# same headers the classic path uses — and now also carries "your
@@ -1958,7 +2222,11 @@ async def generate_speech(
_backend, text, language, ref_audio_path, ref_text, instruct,
duration, num_step, guidance_scale, speed, denoise,
postprocess_output, used_seed, effect_preset,
max_chunk_chars, crossfade_ms, dropped_sink=_dropped_text,
max_chunk_chars, crossfade_ms, t_shift=t_shift,
layer_penalty_factor=layer_penalty_factor,
position_temperature=position_temperature,
class_temperature=class_temperature,
dropped_sink=_dropped_text,
)
else:
_local_render = functools.partial(
@@ -1969,14 +2237,19 @@ async def generate_speech(
class_temperature, used_seed, effect_preset,
max_chunk_chars, crossfade_ms, dropped_sink=_dropped_text,
)
audio_tensor = await gpu_gateway.run(
_REMOTE_OP,
local=gpu_gateway.LocalCall(
_local_render, what="TTS generate",
timeout=_generate_timeout_s(text),
min_vram_gb=_engine_min_vram_gb,
),
decision=_decision,
audio_tensor = await _run_with_reference_lease(
ref_lease,
lambda release: gpu_gateway.run(
_REMOTE_OP,
local=gpu_gateway.LocalCall(
_local_render, what="TTS generate",
timeout=_generate_timeout_s(text, execution_device=_routing["effective_device"],
min_vram_gb=_engine_min_vram_gb),
min_vram_gb=_engine_min_vram_gb,
on_abandon=release,
),
decision=_decision,
)
)
# Read after generation: engines with lazy model loading report
# their real rate only once weights are up.
@@ -2108,9 +2381,8 @@ async def generate_speech(
),
)
finally:
if cleanup_ref and ref_audio_path:
with contextlib.suppress(OSError):
os.remove(ref_audio_path)
if cleanup_ref and ref_lease is not None:
ref_lease.finish_request()
def _safe_output_path(name):
if not name:
+24 -3
View File
@@ -160,7 +160,9 @@ _OPENAI_VOICE_ALIASES = {
def _resolve_engine(model_id: str):
"""Map an OpenAI model name to a VoiceStudio backend."""
from services.tts_backend import get_backend_class, get_active_tts_backend
from services.tts_backend import (
get_backend_class, get_active_tts_backend, get_engine_instance_for,
)
# Accept OpenAI model names as pass-through to the active engine.
if model_id in ("tts-1", "tts-1-hd"):
@@ -177,8 +179,18 @@ def _resolve_engine(model_id: str):
)
from services.tts_backend import OmniVoiceBackend
if cls is OmniVoiceBackend:
# OmniVoice only ever runs as the shared active engine — the
# explicit-omnivoice request is the active-engine request.
return get_active_tts_backend()
return cls()
# Cached singleton, not a fresh cls(): SubprocessBackend engines would
# spawn a sidecar process and reload their model on EVERY request, and
# register a new atexit hook each time (get_engine_instance's contract).
# No router-local cache on top of it: the shared cache is keyed by
# CLASS precisely so id rebinds/evictions can't serve a stale instance,
# and cross-engine memory discipline is create_speech's
# evict_other_tts_engines call (the same seam /generate uses) — not a
# bespoke unload here.
return get_engine_instance_for(model_id)
except ValueError:
raise HTTPException(
status_code=400,
@@ -388,6 +400,15 @@ async def create_speech(req: SpeechRequest):
# VRAM eviction runs in get_model()'s warm-return path now, covering every
# native TTS generate (this route, WS TTS, dub, batch, audiobook).
# Single-active-engine memory discipline (MM2-01), the same call /generate
# makes before its load: hand back every OTHER resident TTS engine's model
# before this one warms up, so switching `model` ids across requests —
# explicit id → explicit id, or explicit id → the tts-1/omnivoice aliases —
# can't stack multi-GB engines/sidecars. No-op when nothing else is
# resident; opt out with OMNIVOICE_SINGLE_ENGINE_RESIDENT=0.
from services.engine_memory import evict_other_tts_engines
await evict_other_tts_engines(backend.id)
# ── #1033/#1037/#1014: warm the engine under the LOAD budget before the
# generate clock starts. The T4 verification (#1014) measured a fresh
# install's first /v1/audio/speech burning its whole 300s generate budget
@@ -454,7 +475,7 @@ async def create_speech(req: SpeechRequest):
from services.model_manager import generate_timeout_s
wav, sr = await run_on_gpu_pool_guarded(
lambda: _run_tts(backend, text, kw), what="OpenAI TTS generate",
timeout=generate_timeout_s(text))
timeout=generate_timeout_s(text, engine=backend))
except Exception as e:
# #1172/#1173: typed failures get their real status + actionable
# message (400 bad input / 503 broken engine binary) instead of a
+10 -11
View File
@@ -35,11 +35,11 @@ class _HFTokenBody(BaseModel):
token: str = Field(..., min_length=1, description="HuggingFace access token")
def _state_response() -> dict:
def _state_response(*, validate: bool = False) -> dict:
"""Return the same shape the React panel renders. Never includes raw token."""
from services import token_resolver
s = token_resolver.state()
s = token_resolver.state(validate=validate)
return {
"active": s["active"],
"sources": [asdict(row) for row in s["sources"]],
@@ -65,8 +65,7 @@ def save_hf_token(body: _HFTokenBody):
@router.delete("/hf-token")
def clear_hf_token(also_clear_hf_cli: bool = Query(False)):
"""Clear the App-source token. Optionally also call huggingface_hub.logout
to clear the canonical HF file. Returns the updated cascade state."""
"""Clear the App token and optionally recognized local Hub token files."""
from services import token_resolver
try:
token_resolver.clear_app_token(also_clear_hf_cli=also_clear_hf_cli)
@@ -82,13 +81,12 @@ def get_hf_token_state(fresh: bool = Query(False)):
``fresh=1`` drops the resolver's whoami validation cache first so the
response re-runs whoami for every source this is what the panel's
"Test now" button sends. Plain GETs (panel mounts) keep the 300s cache
so repeat Settings visits don't hammer the HF API.
"Test now" button sends. Plain GETs only inspect local token presence.
"""
from services import token_resolver
if fresh:
token_resolver.invalidate_cache()
return _state_response()
return _state_response(validate=fresh)
# ── Performance settings (INST-12) ────────────────────────────────────────
@@ -150,7 +148,7 @@ def _compute_device_state() -> dict:
caps = device_caps.detect_host_caps()
env_pin = (os.environ.get("OMNIVOICE_DEVICE") or "").strip().lower()
auto_family = next(
(f for f in ("cuda", "rocm", "xpu", "mps") if f in caps.available_families),
(f for f in device_caps.ACCELERATOR_PRIORITY if f in caps.available_families),
"cpu",
)
value = device_caps.requested_device_override()
@@ -1035,9 +1033,10 @@ def set_asr_openai_compat(body: _ASROpenAICompatBody):
from services import asr_backend, settings_store
if body.base_url is not None:
url = body.base_url.strip().rstrip("/")
if url and not url.startswith(("http://", "https://")):
raise HTTPException(status_code=400, detail="Base URL must start with http(s)://")
try:
url = asr_backend.normalize_openai_compat_asr_base_url(body.base_url)
except ValueError as exc:
raise HTTPException(status_code=400, detail=str(exc)) from exc
settings_store.set_text(asr_backend._ASR_OPENAI_COMPAT_BASE_URL_KEY, url)
if body.model is not None:
settings_store.set_text(
+44 -12
View File
@@ -76,6 +76,7 @@ _cancelled: set[str] = set()
_active_installs: set[str] = set()
_active_installs_lock = threading.Lock()
_install_tasks: set[asyncio.Task] = set()
_install_tasks_by_repo: dict[str, asyncio.Task] = {}
def _download_max_workers() -> int:
@@ -420,16 +421,11 @@ async def install_model(req: InstallModelRequest):
f"Retry in {remaining}s or check your network."
),
)
with _active_installs_lock:
if req.repo_id in _active_installs:
return {"status": "already_running", "repo_id": req.repo_id}
_active_installs.add(req.repo_id)
loop = asyncio.get_running_loop()
def _do():
token = hf_progress.current_repo_id.set(req.repo_id)
target_token = hf_progress.current_target.set("local")
_cancelled.discard(req.repo_id) # clear any stale cancel from a prior run
hf_progress.emit({
"repo_id": req.repo_id,
"filename": req.repo_id,
@@ -688,17 +684,53 @@ async def install_model(req: InstallModelRequest):
with _active_installs_lock:
_active_installs.discard(req.repo_id)
try:
task = loop.create_task(asyncio.to_thread(_do))
_install_tasks.add(task)
task.add_done_callback(_install_tasks.discard)
except Exception:
with _active_installs_lock:
with _active_installs_lock:
if req.repo_id in _active_installs:
return {"status": "already_running", "repo_id": req.repo_id}
_active_installs.add(req.repo_id)
# Admission and task publication are one atomic generation boundary:
# cancellation can never observe an admitted install without its task.
_cancelled.discard(req.repo_id)
try:
task = loop.create_task(asyncio.to_thread(_do))
_install_tasks.add(task)
_install_tasks_by_repo[req.repo_id] = task
except Exception:
_active_installs.discard(req.repo_id)
raise
raise
def install_finished(completed: asyncio.Task) -> None:
with _active_installs_lock:
_install_tasks.discard(completed)
if _install_tasks_by_repo.get(req.repo_id) is completed:
_install_tasks_by_repo.pop(req.repo_id, None)
task.add_done_callback(install_finished)
return {"status": "install_started", "repo_id": req.repo_id}
async def cancel_install_and_wait(repo_id: str) -> None:
"""Request cancellation and retain authority until its thread exits."""
from worker.async_utils import drain_task # noqa: PLC0415
with _active_installs_lock:
_cancelled.add(repo_id)
_install_cooldowns.pop(repo_id, None)
task = _install_tasks_by_repo.get(repo_id)
if task is None:
return
try:
# asyncio.to_thread cannot stop snapshot_download mid-file. Cancelling
# its wrapper would only detach the thread, so wait until the blocking
# call observes the flag or naturally returns.
await drain_task(task)
finally:
with _active_installs_lock:
current = _install_tasks_by_repo.get(repo_id)
if current is None or current is task:
_cancelled.discard(repo_id)
@router.post("/models/install/cancel")
async def cancel_install(req: InstallModelRequest):
"""Request cancellation of an in-flight install (FDL-11).
+20 -4
View File
@@ -62,6 +62,11 @@ def setup_status():
_MIN_NVIDIA_DRIVER = 555
_RAM_FAIL_GB = 8
_RAM_WARN_GB = 12
# Installed DIMMs never fully reach the OS: firmware, integrated graphics and
# kernel reservations shave off up to ~7% (an "8 GB" Windows laptop reports
# ~7.8 GB usable). Thresholds are compared with this allowance applied so the
# machines a threshold is meant to admit aren't blocked by that gap (#1618).
_RAM_RESERVED_ALLOWANCE = 0.93
def _run_cmd(args: list[str], timeout: float = 2.0) -> tuple[int, str]:
@@ -352,17 +357,28 @@ def preflight():
# ── RAM
ram = _ram_gb()
# Escape hatch (#1618): a preflight should inform, not brick setup —
# OMNIVOICE_RAM_PREFLIGHT=0 downgrades the hard block to a warning for
# users who accept the OOM risk. Same opt-out shape as
# OMNIVOICE_ASR_VRAM_PREFLIGHT.
ram_gate = os.environ.get(
"OMNIVOICE_RAM_PREFLIGHT", "1"
).strip().lower() not in ("0", "false", "no")
if ram == 0:
ram_status, ram_detail, ram_fix = (
"warn", "Could not detect system RAM.",
"Install psutil in the backend environment or ignore this warning.",
)
elif ram < _RAM_FAIL_GB:
elif ram < _RAM_FAIL_GB * _RAM_RESERVED_ALLOWANCE:
ram_status, ram_detail, ram_fix = (
"fail", f"{ram:.1f} GB total (need ≥ {_RAM_FAIL_GB} GB)",
"The app will OOM on first dub. Close other apps or upgrade RAM.",
"fail" if ram_gate else "warn",
f"{ram:.1f} GB total (need ≥ {_RAM_FAIL_GB} GB)",
"The app will OOM on first dub. Close other apps or upgrade RAM."
if ram_gate else
"RAM check disabled via OMNIVOICE_RAM_PREFLIGHT=0 — dubbing may "
"OOM on this machine.",
)
elif ram < _RAM_WARN_GB:
elif ram < _RAM_WARN_GB * _RAM_RESERVED_ALLOWANCE:
ram_status, ram_detail, ram_fix = (
"warn", f"{ram:.1f} GB total ({_RAM_WARN_GB}+ GB recommended)",
"Long videos may hit swap. Keep other apps closed during dubbing.",
+160
View File
@@ -0,0 +1,160 @@
"""Discovery contract for VoiceStudio's local speech platform.
Interfaces should discover this document instead of hard-coding whichever
dictation route the desktop happens to use. Endpoint URLs are relative so the
same response works on loopback, a tailnet GPU host, and a reverse proxy.
"""
from __future__ import annotations
import os
from typing import Literal
from fastapi import APIRouter
from pydantic import BaseModel, Field
from core.version import APP_VERSION
router = APIRouter(tags=["Speech Platform"])
SPEECH_PROTOCOL = "voicestudio.speech.v1"
STREAM_PATH = "/v1/audio/transcriptions/stream"
class EndpointCapability(BaseModel):
path: str
transport: Literal["http", "websocket", "mcp-streamable-http", "mcp-stdio"]
method: str | None = None
protocol: str | None = None
class StreamInputCapability(BaseModel):
framing: Literal["binary"] = "binary"
formats: list[str]
default_format: str
sample_rate_query: str = "sr"
end_control: dict[str, str]
class StreamOutputCapability(BaseModel):
framing: Literal["json"] = "json"
events: list[str]
final_kinds: list[str]
class SpeechFeatureCapabilities(BaseModel):
batch_transcription: bool = True
streaming_transcription: bool = True
partial_transcripts: bool = True
utterance_finals: bool = True
session_summary: bool = True
word_timestamps: bool = True
local_refinement: bool = True
acoustic_echo_cancellation: bool = True
native_dictation_control: bool = False
class SpeechAuthCapabilities(BaseModel):
loopback: Literal["none"] = "none"
remote: Literal["bearer"] = "bearer"
header: str = "Authorization: Bearer <OMNIVOICE_API_KEY>"
browser_session_endpoint: str = "/api/auth/session"
websocket_ticket_endpoint: str = "/api/auth/ws-ticket"
websocket_ticket_query_parameter: Literal["ws_ticket"] = "ws_ticket"
class SpeechCapabilities(BaseModel):
schema_: Literal["voicestudio.speech-capabilities"] = Field(
default="voicestudio.speech-capabilities",
serialization_alias="schema",
)
protocol: Literal["voicestudio.speech.v1"] = SPEECH_PROTOCOL
protocol_version: Literal["1.0"] = "1.0"
service: str = "VoiceStudio"
service_version: str = APP_VERSION
local_first: bool = True
endpoints: dict[str, EndpointCapability]
stream_input: StreamInputCapability
stream_output: StreamOutputCapability
features: SpeechFeatureCapabilities
authentication: SpeechAuthCapabilities
def speech_capabilities() -> SpeechCapabilities:
"""Return the stable, side-effect-free integration contract."""
endpoints = {
"capabilities": EndpointCapability(
path="/.well-known/voicestudio-speech",
transport="http",
method="GET",
),
"batch_transcription": EndpointCapability(
path="/v1/audio/transcriptions",
transport="http",
method="POST",
protocol="openai.audio.transcriptions",
),
"streaming_transcription": EndpointCapability(
path=STREAM_PATH,
transport="websocket",
protocol=SPEECH_PROTOCOL,
),
"mcp": EndpointCapability(
path="/mcp",
transport="mcp-streamable-http",
method="POST",
protocol="mcp",
),
"mcp_stdio": EndpointCapability(
path="python -m backend.mcp_shim",
transport="mcp-stdio",
protocol="mcp",
),
}
native_control = False
try:
control_port = int(os.environ.get("VOICESTUDIO_SPEECH_CONTROL_PORT", ""))
except (TypeError, ValueError):
control_port = 0
if 0 < control_port <= 65535:
native_control = True
endpoints["native_dictation_control"] = EndpointCapability(
path=f"http://127.0.0.1:{control_port}/v1/capabilities",
transport="http",
method="GET",
protocol=SPEECH_PROTOCOL,
)
return SpeechCapabilities(
endpoints=endpoints,
stream_input=StreamInputCapability(
formats=[
"audio/pcm;encoding=s16le;channels=1",
"audio/webm;codecs=opus",
],
default_format="audio/webm;codecs=opus",
end_control={"type": "input_audio.end"},
),
stream_output=StreamOutputCapability(
events=["session.started", "status", "partial", "final", "error"],
final_kinds=["utterance", "summary"],
),
features=SpeechFeatureCapabilities(
native_dictation_control=native_control,
),
authentication=SpeechAuthCapabilities(),
)
@router.get(
"/.well-known/voicestudio-speech",
response_model=SpeechCapabilities,
response_model_by_alias=True,
)
@router.get(
"/v1/audio/capabilities",
response_model=SpeechCapabilities,
response_model_by_alias=True,
)
async def get_speech_capabilities() -> SpeechCapabilities:
"""Advertise batch, streaming, and agent-facing speech transports."""
return speech_capabilities()
+68 -9
View File
@@ -203,8 +203,22 @@ def system_info():
"""
try:
_ffmpeg = find_ffmpeg()
from services import model_manager as _mm
from core import prefs as _prefs_mod
return {
"app_version": APP_VERSION,
"generate_timeout_s": _mm.GPU_JOB_TIMEOUT_S,
"cpu_generate_timeout_s": _mm.CPU_JOB_TIMEOUT_S,
# #1787 review fix: a saved prefs.json value for either key can be
# silently shadowed by an external env var (os.environ.setdefault
# in core.prefs.restore_env is a no-op when one is already
# present) — the Settings panel must say so rather than promise a
# restart will apply a value that never will.
"generate_timeout_shadowed": _prefs_mod.is_env_shadowed(
"OMNIVOICE_GENERATE_TIMEOUT_S"),
"cpu_generate_timeout_shadowed": _prefs_mod.is_env_shadowed(
"OMNIVOICE_CPU_GENERATE_TIMEOUT_S"),
"code_fingerprint": os.environ.get("OMNIVOICE_BUILD_FINGERPRINT", ""),
"data_dir": DATA_DIR,
"outputs_dir": OUTPUTS_DIR,
"crash_log_path": CRASH_LOG_PATH,
@@ -240,6 +254,11 @@ def system_info():
logger.exception("system_info failed — returning safe defaults")
return {
"app_version": APP_VERSION,
"generate_timeout_s": 300.0,
"cpu_generate_timeout_s": 600.0,
"generate_timeout_shadowed": False,
"cpu_generate_timeout_shadowed": False,
"code_fingerprint": os.environ.get("OMNIVOICE_BUILD_FINGERPRINT", ""),
"data_dir": DATA_DIR,
"outputs_dir": OUTPUTS_DIR,
"crash_log_path": str(CRASH_LOG_PATH),
@@ -848,6 +867,14 @@ PERSISTENT_KEYS = {
# the Rust sidecar reads OMNIVOICE_PORT at startup and the backend derives
# the LAN-share/UI ports from the others.
"OMNIVOICE_PORT", "OMNIVOICE_SHARE_PORT", "OMNIVOICE_UI_PORT",
# Per-job compute-time budgets (#1787). Both are captured at import time
# by services/model_manager.py (GPU_JOB_TIMEOUT_S / CPU_JOB_TIMEOUT_S), so
# a value saved here takes effect on the NEXT backend restart — same
# contract as OMNIVOICE_PORT above. Restored into os.environ during the
# "env_prefs" startup step (main.py), which runs before model_manager is
# first imported ("ml_imports"), so the restored value is what the module
# captures. The Settings UI must say so (RestartBadge).
"OMNIVOICE_GENERATE_TIMEOUT_S", "OMNIVOICE_CPU_GENERATE_TIMEOUT_S",
}
# Sidecar-engine install dirs (OMNIVOICE_INDEXTTS_DIR, …). The one-click
@@ -865,6 +892,16 @@ except Exception: # pragma: no cover — defensive: env panel > installer wirin
# being set so a bad value never reaches uvicorn / the share listener.
_PORT_KEYS = {"OMNIVOICE_PORT", "OMNIVOICE_SHARE_PORT", "OMNIVOICE_UI_PORT"}
# Keys whose value is a wall-clock compute-time budget in seconds (#1787).
# Validated the same way as _PORT_KEYS: reject anything that isn't a
# positive number before it reaches services/model_manager.py. Upper bound is
# generous — long enough that a legitimate multi-hour, audiobook-length CPU
# render is never blocked — but still bounded, so a fat-fingered extra digit
# (300 -> 3000000) can't turn a wedged job into one that silently occupies a
# worker for days before the guard ever fires.
_TIMEOUT_KEYS = {"OMNIVOICE_GENERATE_TIMEOUT_S", "OMNIVOICE_CPU_GENERATE_TIMEOUT_S"}
_MAX_GENERATE_TIMEOUT_S = 21600.0 # 6 hours
@router.post("/system/set-env")
async def set_env_var(body: dict):
@@ -873,7 +910,7 @@ async def set_env_var(body: dict):
Persistent keys (proxy, FFMPEG_PATH, translation provider keys, ) are
saved to ``prefs.json`` so they survive backend restarts (restored at
startup in ``main.py``). HF_TOKEN is persisted via
``huggingface_hub.login()`` (and cleared via ``logout()``). Other keys
``huggingface_hub.login()`` (and cleared with the shared token-file helper). Other keys
are set on ``os.environ`` for the running process.
The loopback-origin gate that previously lived inline here is now applied
@@ -908,6 +945,22 @@ async def set_env_var(body: dict):
status_code=400,
detail=f"Invalid port for {key}: must be between 1024 and 65535.",
)
if key in _TIMEOUT_KEYS:
try:
timeout_n = float(value)
except (TypeError, ValueError):
raise HTTPException(
status_code=400,
detail=f"Invalid timeout for {key}: '{value}' is not a number.",
)
if not (0 < timeout_n <= _MAX_GENERATE_TIMEOUT_S):
raise HTTPException(
status_code=400,
detail=(
f"Invalid timeout for {key}: must be greater than 0 "
f"and at most {_MAX_GENERATE_TIMEOUT_S:.0f} seconds."
),
)
os.environ[key] = value
logger.info("Environment variable set (length=%d)", len(value))
@@ -932,14 +985,14 @@ async def set_env_var(body: dict):
# Mirror the persistence on clear — wipe the saved token file too.
if key == "HF_TOKEN":
try:
from huggingface_hub import logout as _hf_logout
_hf_logout()
logger.info("HF token cleared from $HF_HOME/token via logout()")
except Exception as e:
logger.warning("Could not clear HF token file: %s", e)
from services.token_resolver import clear_hf_cli_tokens
clear_hf_cli_tokens()
logger.info("Local Hugging Face token files cleared")
except Exception:
raise HTTPException(status_code=500, detail="Could not clear local Hugging Face token files") from None
# HF_TOKEN persistence is handled above via huggingface_hub.login()/
# logout() — it never touches prefs.json. Everything else in
# clear_hf_cli_tokens() — it never touches prefs.json. Everything else in
# PERSISTENT_KEYS (proxy, FFMPEG_PATH, translation provider keys, …) is
# saved to prefs.json so it survives backend restarts (restored at
# startup in main.py). Non-persistent keys stay process-local.
@@ -950,7 +1003,13 @@ async def set_env_var(body: dict):
else:
prefs_delete(prefs_key)
return {"key": key, "set": bool(value)}
# #1787 review fix: tell the caller up front when the value just saved is
# being shadowed by an external env var — set at THIS process's startup,
# before our own prefs restore ran, so it predicts the next restart too.
# A response that just said {"set": True} let the Settings panel promise
# a restart would apply a value that never will.
from core.prefs import is_env_shadowed
return {"key": key, "set": bool(value), "shadowed": is_env_shadowed(key)}
@router.post("/clean-audio")
@@ -1041,7 +1100,7 @@ def asr_backends():
def hf_token_state():
"""Return the 3-source HF token cascade state for the Settings UI
(Wave 2 React panel consumes this). Never returns the raw token
only a masked preview, whoami username, and per-source validity.
only a masked preview and local presence; no outbound validation.
"""
from dataclasses import asdict
from services import token_resolver
+71 -19
View File
@@ -10,7 +10,8 @@ as they're generated. This unlocks:
Protocol:
Client sends JSON: {"text": "...", "voice": "profile_id", ...}
Server sends binary audio chunks (PCM16 @ 24kHz mono) as generated
Server sends JSON: {"type": "done", "duration_s": 4.2, "gen_time_s": 1.1}
Server sends JSON: {"type": "done", "duration_s": 4.2,
"gen_time_s": 1.1, "ttfa_ms": 180.0, "rtf": 0.262}
Server sends JSON: {"type": "error", "detail": "..."}
The chunked delivery targets <100ms time-to-first-audio (TTFA) on warm models.
@@ -33,6 +34,30 @@ logger = logging.getLogger("omnivoice.tts_stream")
# Smaller chunks = lower latency but more WebSocket overhead.
CHUNK_SAMPLES = int(os.environ.get("OMNIVOICE_STREAM_CHUNK", "4800"))
# Module seam for deterministic latency-contract tests. Keep every timing
# sample on the same monotonic clock.
_perf_counter = time.perf_counter
async def _resolve_stream_backend(engine_id: str | None):
"""Resolve the live-stream engine without bypassing host isolation."""
from services.tts_backend import (
OmniVoiceBackend,
active_backend_id,
get_active_tts_backend,
get_backend_class,
)
if engine_id:
return get_backend_class(engine_id)()
cls = get_backend_class(active_backend_id())
if cls is OmniVoiceBackend:
from services.model_manager import get_model
return get_active_tts_backend(model=await get_model())
return get_active_tts_backend()
class StreamTTSRequest(BaseModel):
"""Client request for streaming TTS."""
@@ -85,7 +110,7 @@ async def ws_tts(websocket: WebSocket):
})
continue
t0 = time.perf_counter()
t0 = _perf_counter()
text = data["text"]
# Remote GPU: this socket stays on this machine, and says so.
@@ -127,10 +152,6 @@ async def ws_tts(websocket: WebSocket):
try:
# Resolve engine
from services.tts_backend import (
get_active_tts_backend,
get_backend_class,
)
engine_id = data.get("engine")
# #1224: leave a breadcrumb when memory is already tight before
# a heavy load. /generate has done this since the 16 GB-Mac
@@ -146,13 +167,7 @@ async def ws_tts(websocket: WebSocket):
log_if_low(f"TTS stream load ({engine_id or 'active engine'})")
except Exception:
pass
if engine_id:
cls = get_backend_class(engine_id)
backend = cls()
else:
from services.model_manager import get_model
model = await get_model()
backend = get_active_tts_backend(model=model)
backend = await _resolve_stream_backend(engine_id)
# ── Routing gate (#21 — no silent CPU fallback). WebSockets have
# no response headers, so this uses frames: an error frame +
@@ -258,6 +273,11 @@ async def ws_tts(websocket: WebSocket):
from services.model_manager import run_on_gpu_pool_guarded
def _generate(sentence_text):
# Timed INSIDE the pool worker: the guarded dispatch below
# can queue behind other jobs, and queue wait is not
# synthesis (review on #1620) — under contention it would
# inflate rtf without the engine slowing at all.
_synth_t0 = _perf_counter()
from services.audio_dsp import apply_mastering, normalize_audio
from services.watermark import mark_synthetic
wav = backend.generate(sentence_text, **kw)
@@ -279,12 +299,19 @@ async def ws_tts(websocket: WebSocket):
# watermark._iter_chunks), which is inherent to marking
# ultra-short clips, not a coverage gap.
wav = mark_synthetic(wav, sr_actual, context="tts_stream.sentence")
return wav, sr_actual
return wav, sr_actual, _perf_counter() - _synth_t0
import torch
total_samples = 0
sr = backend.sample_rate
started = False
first_audio_at: float | None = None
# Synthesis time only. The wall clock below also carries socket
# delivery and the per-chunk event-loop yields, so deriving RTF
# from it reports "how slow was the client" as if it were engine
# throughput — on a slow consumer that inflates RTF without the
# engine having changed at all.
synth_time = 0.0
for sentence in sentences:
# Bounded + pool-reset on hang so a wedged generate can't
@@ -294,11 +321,12 @@ async def ws_tts(websocket: WebSocket):
# Length-scaled budget per sentence (#1190) — the flat 300s
# default is gone from every dispatch.
from services.model_manager import generate_timeout_s
wav_tensor, sr = await run_on_gpu_pool_guarded(
wav_tensor, sr, sentence_synth_s = await run_on_gpu_pool_guarded(
functools.partial(_generate, sentence),
what="TTS generate",
timeout=generate_timeout_s(sentence),
timeout=generate_timeout_s(sentence, engine=backend),
)
synth_time += sentence_synth_s
if not started:
# Send metadata after the first generation so
@@ -325,25 +353,49 @@ async def ws_tts(websocket: WebSocket):
end = min(sent_samples + CHUNK_SAMPLES, n_samples)
chunk = pcm_bytes[sent_samples * 2: end * 2]
await websocket.send_bytes(chunk)
if first_audio_at is None:
# TTFA ends when the first audio bytes have been
# handed to the socket. The previous log used the
# whole-render duration and called it TTFA.
first_audio_at = _perf_counter()
sent_samples = end
# Yield to event loop between chunks for responsiveness
await asyncio.sleep(0)
total_samples += n_samples
gen_time = round(time.perf_counter() - t0, 3)
finished_at = _perf_counter()
wall_time_raw = max(0.0, finished_at - t0)
synth_time_raw = max(0.0, synth_time)
gen_time = round(wall_time_raw, 3)
duration = round(total_samples / sr, 3)
ttfa_ms = (
round(max(0.0, first_audio_at - t0) * 1000.0, 1)
if first_audio_at is not None
else None
)
# RTF is a render metric: synthesis seconds per audio second.
rtf = (
round(synth_time_raw / (total_samples / sr), 3)
if total_samples > 0
else None
)
await websocket.send_json({
"type": "done",
"duration_s": duration,
"gen_time_s": gen_time,
"ttfa_ms": ttfa_ms,
"rtf": rtf,
"samples": total_samples,
"sample_rate": sr,
"engine": backend.id,
})
logger.info(
"TTS stream: %.1fs audio in %.1fs (TTFA=%.0fms)",
duration, gen_time, gen_time * 1000,
"TTS stream: %.1fs audio in %.1fs (TTFA=%s, RTF=%s)",
duration,
gen_time,
f"{ttfa_ms:.0f}ms" if ttfa_ms is not None else "n/a",
f"{rtf:.3f}" if rtf is not None else "n/a",
)
except Exception as e:
+380
View File
@@ -0,0 +1,380 @@
"""Speech-to-speech voice changer — Studio's Convert method (POST /convert).
The user drops (or records) a source clip, picks an existing voice profile,
and gets the same words back in that profile's voice: the active ASR backend
transcribes the clip (no word timestamps the text is all we need), the
active TTS engine re-synthesizes it conditioned on the profile's reference
audio, and by default the take is pitch-preservingly time-stretched
(ffmpeg atempo, clamped to one well-behaved 0.52.0 stage) so it lands near
the source clip's duration.
Deliberately reuses the /generate choke points instead of re-deriving them:
* profile row conditioning via ``generation._resolve_profile_conditioning``
(lock wins, ``kind`` authoritative, #533 language fill),
* engine resolution via ``services.tts_backend.resolve_generation_backend``
(never a silent OmniVoice fallback; ``require_cloning=True`` refuses
clone-less engines with the actionable switch-engine message),
* synthesis via ``generation._run_backend_inference`` on the guarded GPU
pool (#730 bound + reset; busy/timeout → retryable 503),
* provenance + persistence via ``services.watermark.mark_synthetic_async``
and ``generation._finalize_generation`` (watermark WAV in OUTPUTS_DIR
history row retention prune), marked AFTER the stretch so the take users
keep carries exactly one whole-take mark.
Local-first: no network calls; ASR-model-less installs get the same typed
409 download CTA as /transcribe; a backend mid-shutdown surfaces the global
503 ``[shutting_down]`` (ModelLoadInterruptedByShutdown main.py handler).
Reachability matches /generate: loopback bind by default, with the shared
network-share PIN / API-key middleware gating any non-loopback exposure.
"""
from __future__ import annotations
import asyncio
import functools
import logging
import os
import tempfile
import time
from fastapi import APIRouter, File, Form, HTTPException, UploadFile
router = APIRouter()
logger = logging.getLogger("omnivoice.convert")
#: ffmpeg's atempo filter is well-behaved in [0.5, 2.0] per stage. Convert
#: clamps to ONE stage by design: needing more than 2× either way means the
#: synthesized speech differs so much from the source that "matching" it
#: would produce chipmunk/slow-motion artifacts worse than the mismatch.
ATEMPO_MIN = 0.5
ATEMPO_MAX = 2.0
#: Within this relative tolerance the durations already match — stretching
#: would resample the whole take for an inaudible gain.
_MATCH_TOLERANCE = 0.02
#: Convert clips are short conversational inputs, not long-form media. Stream
#: them to disk in bounded chunks so a network-share client cannot make the
#: backend materialize an arbitrarily large multipart upload in memory.
_MAX_SOURCE_AUDIO_BYTES = 64 * 1024 * 1024
_UPLOAD_CHUNK_BYTES = 1024 * 1024
async def _copy_source_upload(audio: UploadFile, destination) -> int:
"""Stream ``audio`` into ``destination`` with the Convert upload cap."""
total = 0
while True:
chunk = await audio.read(_UPLOAD_CHUNK_BYTES)
if not chunk:
return total
total += len(chunk)
if total > _MAX_SOURCE_AUDIO_BYTES:
raise HTTPException(
status_code=413,
detail="Source audio is too large (maximum 64 MB).",
)
destination.write(chunk)
def _clamped_tempo_ratio(tts_duration_s: float, source_duration_s: float) -> "float | None":
"""The atempo ratio that fits the take into the source duration, or None.
ratio > 1 speeds the take up (it came out longer than the source),
ratio < 1 slows it down. Clamped to a single atempo stage's [0.5, 2.0];
None when either duration is unusable or they already match.
"""
if not source_duration_s or source_duration_s <= 0:
return None
if not tts_duration_s or tts_duration_s <= 0:
return None
ratio = tts_duration_s / source_duration_s
if abs(ratio - 1.0) <= _MATCH_TOLERANCE:
return None
return min(ATEMPO_MAX, max(ATEMPO_MIN, ratio))
async def _match_source_duration(audio_tensor, sample_rate: int, source_duration_s: float):
"""Best-effort pitch-preserving stretch of the take toward the source
clip's duration. Returns the input unchanged when no stretch is needed
or ffmpeg fails a duration mismatch is better than a failed convert."""
n_samples = int(audio_tensor.shape[-1])
ratio = _clamped_tempo_ratio(n_samples / sample_rate, source_duration_s)
if ratio is None:
return audio_tensor
target_samples = max(1, int(round(n_samples / ratio)))
from services.ffmpeg_utils import _pitch_preserving_stretch
try:
return await _pitch_preserving_stretch(audio_tensor, target_samples, sample_rate)
except Exception as e: # noqa: BLE001 — stretch is opt-in polish, never fatal
logger.warning("duration match skipped — atempo stretch failed: %s", e)
return audio_tensor
async def _transcribe_source(tmp_path: str, *, source_lease=None) -> dict:
"""Active-ASR transcription of the uploaded clip (no word timestamps).
Mirrors POST /transcribe: typed 409 + download CTA before any backend
is constructed (never a silent multi-GB auto-download), the guarded GPU
pool dispatch (#730), 504 on timeout, and the same 409 when the loader
degrades onto an engine with no weights on disk (#1185).
"""
from services.asr_backend import (
ASRModelMissingError,
ASRTimeoutError,
asr_model_missing_detail,
asr_model_missing_error,
run_transcribe_guarded,
)
missing = await asyncio.to_thread(asr_model_missing_error, purpose="transcribe")
if missing is not None:
raise HTTPException(
status_code=409,
detail={**missing, "message": asr_model_missing_detail(missing)},
)
def _run():
# `load_*`, not `get_*`: the loader runs ensure_loaded() and degrades
# past an engine whose deep import chain is broken (#1185).
from services.asr_backend import load_active_asr_backend
backend = load_active_asr_backend()
return backend.transcribe(tmp_path, word_timestamps=False)
from services.model_manager import _gpu_pool
release = source_lease.acquire() if source_lease is not None else None
abandoned = False
try:
return await run_transcribe_guarded(
_gpu_pool,
_run,
what="Voice convert",
on_abandon=release,
)
except asyncio.CancelledError:
# The guard now owns the lease token until the native worker drains.
abandoned = True
raise
except ASRTimeoutError as e:
abandoned = True
logger.warning("Convert transcription timed out: %s", e)
raise HTTPException(status_code=504, detail=str(e))
except ASRModelMissingError as e:
raise HTTPException(
status_code=409,
detail={**e.payload, "message": asr_model_missing_detail(e.payload)},
)
finally:
if release is not None and not abandoned:
release()
@router.post("/convert")
async def convert_speech(
audio: UploadFile = File(...),
profile_id: str = Form(...),
match_duration: bool = Form(True),
):
"""Convert a spoken clip into an existing voice profile's voice.
Multipart form: ``audio`` (the source clip), ``profile_id`` (an existing
voice profile), optional ``match_duration`` (default on atempo the take
toward the source clip's length, clamped to 0.52.0×).
Returns JSON ``{audio_url, text, duration_s, id}`` the take is saved to
OUTPUTS_DIR and served from the ``/audio`` mount like every other take.
"""
from core.db import db_conn
from api.routers.generation import _resolve_profile_conditioning, _TempReferenceLease
# ── Profile first: strict 404, unlike /generate's silent skip — Convert
# has no meaning without a target voice.
with db_conn() as conn:
row = conn.execute(
"SELECT * FROM voice_profiles WHERE id=?", (profile_id,)
).fetchone()
if not row:
raise HTTPException(
status_code=404,
detail="That voice profile doesn't exist. It may have been deleted from another tab.",
)
cond = _resolve_profile_conditioning(row)
# ── Save the upload before loading an engine. Every ASR backend (and
# ffprobe) needs a file path; the bounded streaming copy rejects oversized
# network-share requests without materializing them in process memory or
# starting heavyweight model work.
ext = os.path.splitext(audio.filename or "audio.wav")[1] or ".wav"
tmp = tempfile.NamedTemporaryFile(delete=False, suffix=ext)
source_lease = None
try:
try:
await _copy_source_upload(audio, tmp)
finally:
tmp.close()
source_lease = _TempReferenceLease(tmp.name)
# ── Engine gate before ASR/TTS work: the shared resolver refuses a
# clone-less engine with the actionable switch-engine message (→ 400),
# and a backend mid-shutdown raises ModelLoadInterruptedByShutdown out
# of the model load → the global 503 [shutting_down] handler.
from services.tts_backend import resolve_generation_backend
try:
backend = await resolve_generation_backend(
require_cloning=True, cloning_purpose="voice conversion",
)
except ValueError as e:
raise HTTPException(status_code=400, detail=str(e))
result = await _transcribe_source(tmp.name, source_lease=source_lease)
segments = result.get("segments", [])
text = result.get("text", "")
if not text and segments:
text = " ".join(s.get("text", "") for s in segments).strip()
# Same final-text hygiene as /transcribe: strip Whisper hallucination
# loops, then deterministic polish (leading capital + terminal
# punctuation) so the TTS input reads as typed text.
from services.refinement import collapse_repetitive_artifacts
from services.text_polish import polish_text
text = polish_text(collapse_repetitive_artifacts(text))
if not text or not text.strip():
raise HTTPException(
status_code=422,
detail=(
"No speech was recognized in the source clip, so there is "
"nothing to convert. Record or drop a clip with clear, "
"audible speech and try again."
),
)
# #308/#1032 parity with /generate: a clone profile saved without a
# transcript conditions better when its reference clip is transcribed,
# and that transcript is cached onto the row so it happens ONCE, not
# per convert. Best-effort exactly like /generate — a timeout/failure
# degrades to ref_text=None and the engine's own fallback. The ASR
# model is already warm here (the source transcribe above just used it).
if cond["ref_audio_path"] and not cond["ref_text"]:
from api.routers.generation import (
_generate_timeout_s,
_persist_profile_ref_text,
)
from services.asr_backend import transcribe_reference
from services.model_manager import run_on_gpu_pool_guarded
try:
cond["ref_text"] = await run_on_gpu_pool_guarded(
functools.partial(transcribe_reference, cond["ref_audio_path"]),
what="Reference transcribe",
timeout=_generate_timeout_s(""),
)
except TimeoutError as e:
logger.warning(
"reference transcribe hung (%s); using engine ASR fallback", e,
)
cond["ref_text"] = None
if cond["ref_text"] and cond["persist_ref_text"]:
_persist_profile_ref_text(profile_id, cond["ref_text"])
# Source duration for the optional match: the container's own length
# (ffprobe), falling back to the last ASR segment end. Best-effort —
# None just skips the stretch.
source_duration_s = None
if match_duration:
from services.ffmpeg_utils import probe_duration
source_duration_s = await probe_duration(
tmp.name, allowed_root=os.path.dirname(tmp.name),
)
if not source_duration_s and segments:
source_duration_s = max((s.get("end", 0) or 0) for s in segments) or None
# ── Same text choke point as /generate: engine-agnostic normalization
# (numbers→words, junk strip) on the fully resolved language.
from services.text_normalization import normalize_for_tts
language = cond["language"]
text = normalize_for_tts(text, language)
used_seed = cond["seed"]
if used_seed is None:
import random
used_seed = random.randint(0, 2**31 - 1)
from api.routers.generation import (
_finalize_generation,
_generate_timeout_s,
_run_backend_inference,
)
from services.model_manager import (
GpuJobTimeoutError,
GpuPoolBusyError,
run_on_gpu_pool_guarded,
)
start_time = time.time()
_render = functools.partial(
_run_backend_inference,
backend, text, language, cond["ref_audio_path"], cond["ref_text"],
cond["instruct"],
None, # duration — the model picks; match_duration owns pacing
16, 2.0, # num_step / guidance_scale (the /generate defaults)
1.0, # speed
True, True, # denoise / postprocess_output
used_seed,
)
try:
audio_tensor = await run_on_gpu_pool_guarded(
_render,
what="Voice convert",
timeout=_generate_timeout_s(
text, min_vram_gb=getattr(type(backend), "min_vram_gb", 0.0),
),
min_vram_gb=getattr(type(backend), "min_vram_gb", 0.0),
)
except GpuPoolBusyError as e:
raise HTTPException(
status_code=503, detail=str(e),
headers={"Retry-After": str(e.retry_after),
"X-OmniVoice-Retryable": "true"},
) from e
except GpuJobTimeoutError as e:
raise HTTPException(
status_code=503, detail=str(e),
headers={"Retry-After": "30", "X-OmniVoice-Retryable": "true"},
) from e
except ValueError as e:
raise HTTPException(status_code=400, detail=str(e)) from e
sample_rate = backend.sample_rate
if match_duration and source_duration_s:
audio_tensor = await _match_source_duration(
audio_tensor, sample_rate, source_duration_s,
)
# Provenance mark AFTER the stretch (one whole-take mark on the audio
# the user actually keeps), then the shared finalize tail — WAV in
# OUTPUTS_DIR, self-healing history row, retention prune, event emit.
from services.watermark import mark_synthetic_async
audio_tensor = await mark_synthetic_async(
audio_tensor, sample_rate, context="convert.finalize",
)
_, meta = await _finalize_generation(
audio_tensor, sample_rate, text=text, history_mode="convert",
ref_audio_path=cond["ref_audio_path"], language=language,
instruct=cond["instruct"], resolved_profile_id=profile_id,
used_seed=used_seed, start_time=start_time,
already_marked=True,
)
return {
"id": meta["id"],
"audio_url": f"/audio/{meta['filename']}",
"text": text,
"duration_s": meta["duration"],
"gen_time_s": meta["gen_time"],
}
finally:
if source_lease is not None:
source_lease.finish_request()
else:
try:
os.unlink(tmp.name)
except OSError:
pass
+266 -55
View File
@@ -23,13 +23,16 @@ appears and is replaced by the GPU gateway.
from __future__ import annotations
import asyncio
import contextlib
import logging
from fastapi import APIRouter, Depends, HTTPException, Request
from fastapi.responses import JSONResponse
from pydantic import BaseModel, Field
from api.dependencies import require_admin
from worker import registry, routing, service
from worker.async_utils import drain_task, to_thread_and_defer_cancellation
logger = logging.getLogger("omnivoice.worker")
@@ -158,6 +161,19 @@ def agent_status() -> dict:
return worker_agent.agent.status()
@router.get("/agent/readiness", include_in_schema=False, response_model=None)
def agent_readiness() -> JSONResponse:
"""Container readiness: 200 only after this process registered as a worker."""
from worker import agent as worker_agent # noqa: PLC0415
readiness = worker_agent.agent.readiness()
return JSONResponse(
status_code=200 if readiness["ready"] else 503,
content=readiness,
headers={} if readiness["ready"] else {"Retry-After": "2"},
)
def _refuse_when_env_pinned(worker_agent) -> None:
"""OMNIVOICE_WORKER_MODE wins over the setting everywhere else.
@@ -176,6 +192,63 @@ def _refuse_when_env_pinned(worker_agent) -> None:
)
async def _finish_cleanup(awaitable):
"""Run rollback to completion even if its HTTP task was cancelled."""
task = asyncio.create_task(awaitable)
await drain_task(task)
return task.result()
async def _set_worker_mode(worker_agent, enabled: bool) -> None:
_result, cancelled = await to_thread_and_defer_cancellation(
worker_agent.set_worker_mode_enabled, enabled
)
if cancelled:
raise asyncio.CancelledError
async def _restore_agent_transaction(
worker_agent, previous: dict, *, was_running: bool
) -> None:
"""Restore durable enrollment/settings and the exact prior live state."""
try:
await _finish_cleanup(worker_agent.agent.stop())
await _finish_cleanup(worker_agent.restore_enrollment(previous))
if was_running and not worker_agent.agent.running:
await _finish_cleanup(worker_agent.agent.start())
elif not was_running and worker_agent.agent.running:
await _finish_cleanup(worker_agent.agent.stop())
except worker_agent.EnrollmentRollbackError:
raise
except BaseException as exc:
message = (
"The previous worker state could not be restored safely. "
"Worker mode remains stopped; fix its enrollment/settings storage, then retry."
)
with contextlib.suppress(BaseException):
await _finish_cleanup(worker_agent.agent.stop())
worker_agent.agent.last_error = message
raise worker_agent.EnrollmentRollbackError(message) from exc
def _raise_agent_transaction_failure(
worker_agent, operation: BaseException, rollback: BaseException | None
) -> None:
if isinstance(operation, asyncio.CancelledError):
if rollback is not None:
logger.error(
"Worker rollback failed during request cancellation",
exc_info=(type(rollback), rollback, rollback.__traceback__),
)
raise operation
if rollback is not None:
raise HTTPException(status_code=409, detail=str(rollback)) from rollback
if isinstance(operation, Exception):
worker_agent.agent.last_error = str(operation)
raise HTTPException(status_code=409, detail=str(operation)) from operation
raise operation
@router.post("/agent/join")
async def join_control_plane(request: JoinRequest) -> dict:
"""Redeem a join code and start working for that control plane.
@@ -200,26 +273,37 @@ async def join_control_plane(request: JoinRequest) -> dict:
# says it joined and never lends anything (CodeRabbit).
_refuse_when_env_pinned(worker_agent)
async with worker_agent.agent.lifecycle:
# A rejoin replaces a working enrollment. Keep enough to put it back:
# pinning the new certificate overwrites the old one on disk, so a
# failed rejoin would otherwise leave the machine unable to reconnect
# to the control plane it was already serving.
previous = worker_agent.snapshot_enrollment()
await worker_agent.agent.stop()
try:
previous, cancelled = await to_thread_and_defer_cancellation(
worker_agent.snapshot_enrollment
)
except worker_agent.EnrollmentStateError as exc:
worker_agent.agent.last_error = str(exc)
raise HTTPException(status_code=409, detail=str(exc)) from exc
if cancelled:
raise asyncio.CancelledError
was_running = worker_agent.agent.running
# A rejoin stops a working agent before the replacement is accepted.
# Stop, acceptance and the durable setting are one transaction: every
# failure, including cancellation, restores both trust and live state.
try:
await worker_agent.agent.stop()
await worker_agent.agent.start(token_text=token)
# Success is the control plane ACCEPTING this worker, not the
# connection being scheduled — see wait_until_registered.
await worker_agent.agent.wait_until_registered()
except Exception as exc:
worker_agent.agent.last_error = str(exc)
await worker_agent.agent.stop()
await worker_agent.restore_enrollment(previous)
raise HTTPException(status_code=409, detail=str(exc)) from exc
await _set_worker_mode(worker_agent, True)
except BaseException as exc:
rollback_exc = None
try:
await _restore_agent_transaction(
worker_agent, previous, was_running=was_running
)
except BaseException as rollback_error:
rollback_exc = rollback_error
_raise_agent_transaction_failure(worker_agent, exc, rollback_exc)
worker_agent.agent.last_error = ""
# Persisted only after the join actually worked: a machine that failed
# to enrol must not come back up trying again forever.
worker_agent.set_worker_mode_enabled(True)
return worker_agent.agent.status()
@@ -235,19 +319,35 @@ async def set_agent_enabled(request: EnableRequest) -> dict:
_refuse_when_env_pinned(worker_agent)
async with worker_agent.agent.lifecycle:
if request.enabled:
try:
try:
previous, cancelled = await to_thread_and_defer_cancellation(
worker_agent.snapshot_enrollment
)
except worker_agent.EnrollmentStateError as exc:
worker_agent.agent.last_error = str(exc)
raise HTTPException(status_code=409, detail=str(exc)) from exc
if cancelled:
raise asyncio.CancelledError
was_running = worker_agent.agent.running
try:
if request.enabled:
await worker_agent.agent.start()
await worker_agent.agent.wait_until_registered()
except Exception as exc:
worker_agent.agent.last_error = str(exc)
await _set_worker_mode(worker_agent, True)
else:
await worker_agent.agent.stop()
raise HTTPException(status_code=409, detail=str(exc)) from exc
worker_agent.agent.last_error = ""
worker_agent.set_worker_mode_enabled(True)
else:
await worker_agent.agent.stop()
worker_agent.set_worker_mode_enabled(False)
await _set_worker_mode(worker_agent, False)
except BaseException as exc:
rollback_exc = None
try:
await _restore_agent_transaction(
worker_agent, previous, was_running=was_running
)
except BaseException as rollback_error:
rollback_exc = rollback_error
_raise_agent_transaction_failure(worker_agent, exc, rollback_exc)
worker_agent.agent.last_error = ""
return worker_agent.agent.status()
@@ -263,9 +363,14 @@ def create_enrollment(request: EnrollRequest) -> dict:
status_code=409,
detail="Remote workers are turned off. Enable them in Settings → System → Remote workers first.",
)
token = service.control_plane.create_enrollment(
endpoint=request.endpoint, label=request.label, ttl_seconds=request.ttl_seconds
)
try:
token = service.control_plane.create_enrollment(
endpoint=request.endpoint,
label=request.label,
ttl_seconds=request.ttl_seconds,
)
except service.EndpointCertificateError as exc:
raise HTTPException(status_code=409, detail=str(exc)) from exc
return {
"token": token.encode(),
"endpoint": token.endpoint,
@@ -275,23 +380,55 @@ def create_enrollment(request: EnrollRequest) -> dict:
}
def _persist_worker_update(
worker_id: str, request: WorkerUpdate
):
"""Write policy on a worker thread; live publication stays loop-owned."""
return registry.update_policy(
worker_id,
name=request.name,
enabled=request.enabled,
priority=request.priority,
)
@router.patch("/{worker_id}")
def update_worker(worker_id: str, request: WorkerUpdate) -> dict:
worker = registry.get(worker_id)
if worker is None:
async def update_worker(worker_id: str, request: WorkerUpdate) -> dict:
pool = service.control_plane.pool if service.control_plane.running else None
live = None
was_pending = False
if pool is not None:
# Quiesce dispatch before releasing authority for the SQLite write.
# The publication after the await restores the exact prior state, so a
# concurrent registration handoff remains quiesced for its own reason.
with registry.authority_guard():
live = pool.get(worker_id)
if live is not None:
was_pending = live.registration_pending
live.registration_pending = True
updated = None
cancelled = False
try:
updated, cancelled = await to_thread_and_defer_cancellation(
_persist_worker_update, worker_id, request
)
finally:
if pool is not None:
with registry.authority_guard():
if updated is not None:
# Pool state, including the cached record the scheduler
# reads, belongs to the app's event loop.
pool.refresh_record(updated)
current = pool.get(worker_id)
if current is live:
current.registration_pending = was_pending
if updated is None:
if cancelled:
raise asyncio.CancelledError
raise HTTPException(status_code=404, detail="No such worker.")
if request.name is not None:
registry.rename(worker_id, request.name)
if request.enabled is not None:
registry.set_enabled(worker_id, request.enabled)
if request.priority is not None:
registry.set_priority(worker_id, request.priority)
updated = registry.get(worker_id)
# Keep the live copy in step, so the scheduler and its logs do not go on
# using the name or priority this worker had when it connected.
if updated is not None and service.control_plane.running:
service.control_plane.pool.refresh_record(updated)
return updated.to_dict() if updated else {}
if cancelled:
raise asyncio.CancelledError
return updated.to_dict()
@router.post("/{worker_id}/consent")
@@ -305,7 +442,7 @@ def grant_consent(worker_id: str) -> dict:
@router.post("/{worker_id}/resume")
def clear_breaker(worker_id: str) -> dict:
async def clear_breaker(worker_id: str) -> dict:
"""Clear a paused worker's circuit breakers.
The user fixed the machine and knows it a breaker with no manual clear is
@@ -320,18 +457,53 @@ def clear_breaker(worker_id: str) -> dict:
@router.delete("/{worker_id}")
def revoke_worker(worker_id: str) -> dict:
async def revoke_worker(worker_id: str) -> dict:
"""Remove a worker — which means revoke its key, not hide the row.
Its in-flight work is released so it can be retried elsewhere rather than
waiting out a lease on a machine that will never answer again.
"""
if registry.get(worker_id) is None:
pool = service.control_plane.pool if service.control_plane.running else None
live = None
was_pending = False
if pool is not None:
with registry.authority_guard():
live = pool.get(worker_id)
if live is not None:
was_pending = live.registration_pending
live.registration_pending = True
try:
revoked, cancelled = await to_thread_and_defer_cancellation(
registry.revoke, worker_id
)
except BaseException:
if pool is not None:
with registry.authority_guard():
current = pool.get(worker_id)
if current is live:
current.registration_pending = was_pending
raise
if not revoked:
if pool is not None:
with registry.authority_guard():
current = pool.get(worker_id)
if current is live:
current.registration_pending = was_pending
if cancelled:
raise asyncio.CancelledError
raise HTTPException(status_code=404, detail="No such worker.")
registry.revoke(worker_id)
if service.control_plane.running:
service.control_plane.scheduler.on_disconnected(worker_id)
service.control_plane.pool.breakers.forget_worker(worker_id)
# The tombstone committed before any egress/session mutation. Everything
# below is loop-owned and published under the same scheduler authority read
# used by next_assignment(), so no task can bind in the handoff window.
with registry.authority_guard():
if service.control_plane.running:
if service.control_plane.servicer is not None:
service.control_plane.servicer.revoke_worker_sessions(worker_id)
service.control_plane.scheduler.on_disconnected(worker_id)
service.control_plane.pool.breakers.forget_worker(worker_id)
if cancelled:
raise asyncio.CancelledError
return {"ok": True, "revoked": worker_id}
@@ -375,7 +547,9 @@ async def submit_task(request: Request, body: SubmitTaskRequest) -> dict:
scheduler = service.control_plane.scheduler
try:
task = scheduler.submit(
submit = getattr(scheduler, "submit_async", None)
submit = submit if callable(submit) else scheduler.submit
submitted = submit(
operation=body.operation,
engine=body.engine,
model_id=body.model_id,
@@ -384,6 +558,7 @@ async def submit_task(request: Request, body: SubmitTaskRequest) -> dict:
deadline_seconds=body.deadline_seconds,
pinned_worker_id=routing.decide().worker_id or None,
)
task = await submitted if asyncio.iscoroutine(submitted) else submitted
except QueueFull as exc:
raise HTTPException(status_code=429, detail=str(exc)) from exc
@@ -505,8 +680,33 @@ async def set_inbound_enabled(request: InboundEnableRequest) -> dict:
"machine. Change that environment setting and restart VoiceStudio."
),
)
requested_bind = (
inbound_service.normalise_bind_host(request.bind)
if request.bind
else inbound_service.bind_host()
)
requested_port = request.port or inbound_service.bind_port()
if (
request.enabled
and inbound_service.node.running
and (
requested_bind != inbound_service.bind_host()
or requested_port != inbound_service.node.port
)
):
# start() is intentionally idempotent while a listener owns its
# socket. Persisting a new endpoint here would make the UI report a
# narrower/different bind while the original socket stayed live.
raise HTTPException(
status_code=409,
detail=(
"Turn off Accept connections before changing its bind address "
"or port."
),
)
if request.bind:
inbound_service.set_bind_host(request.bind)
inbound_service.set_bind_host(requested_bind)
if request.port:
inbound_service.set_bind_port(request.port)
inbound_service.set_enabled(request.enabled)
@@ -535,6 +735,7 @@ def issue_inbound_key(request: IssueKeyRequest) -> dict:
is stored, so it cannot be shown again, only replaced.
"""
from worker.inbound import service as inbound_service # noqa: PLC0415
from worker.inbound.keys import KeyLimitExceeded # noqa: PLC0415
if not inbound_service.node.running:
raise HTTPException(
@@ -544,7 +745,10 @@ def issue_inbound_key(request: IssueKeyRequest) -> dict:
"Settings → System → Remote workers → Accept connections first."
),
)
issued = inbound_service.node.keys.issue(request.label)
try:
issued = inbound_service.node.keys.issue(request.label)
except KeyLimitExceeded as exc:
raise HTTPException(status_code=409, detail=str(exc)) from exc
return {
"key_id": issued.key.key_id,
"label": issued.key.label,
@@ -555,12 +759,12 @@ def issue_inbound_key(request: IssueKeyRequest) -> dict:
@router.delete("/inbound/keys/{key_id}")
def revoke_inbound_key(key_id: str) -> dict:
async def revoke_inbound_key(key_id: str) -> dict:
"""Revoke one panel. Everyone else stays connected — the whole reason keys
are per panel rather than one shared node key."""
from worker.inbound import service as inbound_service # noqa: PLC0415
if not inbound_service.node.keys.revoke(key_id):
if not await inbound_service.node.revoke_key(key_id):
raise HTTPException(status_code=404, detail="No such key.")
return inbound_service.node.snapshot()
@@ -579,6 +783,7 @@ async def add_inbound_connection(request: ConnectRequest) -> dict:
"""Paste a connection string from a GPU machine and dial it."""
from worker.inbound import service as inbound_service # noqa: PLC0415
from worker.inbound.connection_string import InvalidConnectionString # noqa: PLC0415
from worker.inbound.connector import InboundConnectionError # noqa: PLC0415
if not service.control_plane.running:
raise HTTPException(
@@ -597,12 +802,18 @@ async def add_inbound_connection(request: ConnectRequest) -> dict:
# surfaces as "cannot connect", which is what a firewall, a wrong port
# and a dead node all say too.
raise HTTPException(status_code=400, detail=str(exc)) from exc
except InboundConnectionError as exc:
raise HTTPException(status_code=409, detail=str(exc)) from exc
return {"endpoint": connection.endpoint, "connections": inbound_service.outbound.snapshot()}
@router.delete("/inbound/connections/{endpoint}")
async def remove_inbound_connection(endpoint: str) -> dict:
from worker.inbound import service as inbound_service # noqa: PLC0415
from worker.inbound.connector import InboundConnectionError # noqa: PLC0415
await inbound_service.outbound.remove(endpoint)
try:
await inbound_service.outbound.remove(endpoint)
except InboundConnectionError as exc:
raise HTTPException(status_code=409, detail=str(exc)) from exc
return {"connections": inbound_service.outbound.snapshot()}
+15
View File
@@ -26,6 +26,21 @@ class SystemInfoResponse(BaseModel):
model_config = ConfigDict(extra="allow")
app_version: str = ""
# Effective compute-time budgets (seconds) for one synthesis job — the
# values services/model_manager.py's GPU_JOB_TIMEOUT_S / CPU_JOB_TIMEOUT_S
# captured at backend import time (#1787). A value just saved via
# /system/set-env is NOT reflected here until the next restart.
generate_timeout_s: float = 300.0
cpu_generate_timeout_s: float = 600.0
# True when an external env var (shell, `.env`, Docker, …) is currently
# shadowing a prefs.json save for this key — see core.prefs.is_env_shadowed.
generate_timeout_shadowed: bool = False
cpu_generate_timeout_shadowed: bool = False
# #1770: the desktop attach handshake's code fingerprint — whatever
# Tauri set OMNIVOICE_BUILD_FINGERPRINT to when it spawned this process,
# echoed back verbatim. Blank when unset (dev mode, a manually started
# backend). See frontend/src-tauri/src/backend.rs::code_fingerprint_is_current.
code_fingerprint: str = ""
data_dir: str
outputs_dir: str
crash_log_path: str
+10 -10
View File
@@ -159,17 +159,16 @@ models:
- repo_id: "csukuangfj/sherpa-onnx-nemo-parakeet-tdt-0.6b-v3-int8"
label: "Parakeet TDT v3 (sherpa-onnx — dictation, 25 EU langs)"
role: ASR
size_gb: 0.18
size_gb: 0.67
engine: sherpa-onnx
dictation_id: sherpa-parakeet-tdt-v3
tag: offline
curated_on: [all]
note: "Recommended live-dictation default. CPU, int8 ONNX. Requires sherpa-onnx."
note: "Multilingual European-language dictation. CPU, int8 ONNX. Requires sherpa-onnx."
- repo_id: "csukuangfj/sherpa-onnx-nemo-parakeet-tdt-0.6b-v2-int8"
label: "Parakeet TDT v2 (sherpa-onnx — dictation, English)"
role: ASR
size_gb: 0.17
size_gb: 0.66
engine: sherpa-onnx
dictation_id: sherpa-parakeet-tdt-v2
tag: offline
@@ -178,7 +177,7 @@ models:
- repo_id: "csukuangfj/sherpa-onnx-streaming-zipformer-bilingual-zh-en-2023-02-20"
label: "Zipformer Bilingual (sherpa-onnx — streaming, zh+en)"
role: ASR
size_gb: 0.13
size_gb: 0.2
engine: sherpa-onnx
dictation_id: sherpa-zipformer-bilingual-zh-en
tag: streaming
@@ -187,7 +186,7 @@ models:
- repo_id: "csukuangfj/sherpa-onnx-streaming-paraformer-bilingual-zh-en"
label: "Paraformer Bilingual (sherpa-onnx — streaming, zh+en)"
role: ASR
size_gb: 0.115
size_gb: 0.24
engine: sherpa-onnx
dictation_id: sherpa-paraformer-bilingual-zh-en
tag: streaming
@@ -196,7 +195,7 @@ models:
- repo_id: "csukuangfj/sherpa-onnx-streaming-zipformer-en-20M-2023-02-17"
label: "Zipformer Streaming EN 20M (sherpa-onnx — streaming, English)"
role: ASR
size_gb: 0.128
size_gb: 0.044
engine: sherpa-onnx
dictation_id: sherpa-zipformer-en-20m
tag: streaming
@@ -205,7 +204,7 @@ models:
- repo_id: "csukuangfj/sherpa-onnx-streaming-zipformer-zh-14M-2023-02-23"
label: "Zipformer Streaming ZH 14M (sherpa-onnx — streaming, Chinese)"
role: ASR
size_gb: 0.074
size_gb: 0.025
engine: sherpa-onnx
dictation_id: sherpa-zipformer-zh-14m
tag: streaming
@@ -214,11 +213,12 @@ models:
- repo_id: "csukuangfj/sherpa-onnx-whisper-tiny"
label: "Whisper Tiny (sherpa-onnx — dictation, 90+ langs)"
role: ASR
size_gb: 0.116
size_gb: 0.104
engine: sherpa-onnx
dictation_id: sherpa-whisper-tiny
tag: offline
note: "Multilingual offline dictation (auto-detect). CPU, int8 ONNX. Requires sherpa-onnx."
curated_on: [all]
note: "Recommended cross-platform dictation default (auto-detect). CPU, int8 ONNX. Requires sherpa-onnx."
# ── Diarisation ───────────────────────────────────────────────────────
+20
View File
@@ -15,6 +15,18 @@ def get_app_data_dir():
return os.path.expanduser("~/.omnivoice")
def _configured_hf_token_path():
"""Match Hub's token location without importing or refreshing credentials."""
default_cache = os.path.join(os.path.expanduser("~"), ".cache")
hf_home = os.environ.get("HF_HOME", os.path.join(os.environ.get("XDG_CACHE_HOME", default_cache), "huggingface"))
return os.path.expandvars(os.path.expanduser(os.environ.get("HF_TOKEN_PATH", os.path.join(hf_home, "token"))))
# Snapshot recognized locations before automatic model-cache redirection.
# Explicit cache/token overrides restrict clearing to their selected location.
HF_CLI_TOKEN_PATHS = (_configured_hf_token_path(),)
def _ensure_short_hf_cache_on_windows():
"""Redirect HuggingFace cache to a short path on Windows.
@@ -38,6 +50,14 @@ def _ensure_short_hf_cache_on_windows():
return
short_cache = os.path.join(local_app, "OmniVoice", "hf_cache")
os.makedirs(short_cache, exist_ok=True)
if "HF_TOKEN_PATH" not in os.environ:
global HF_CLI_TOKEN_PATHS
canonical = HF_CLI_TOKEN_PATHS[0]
legacy = os.path.join(short_cache, "token")
HF_CLI_TOKEN_PATHS = tuple(dict.fromkeys((canonical, legacy)))
# Keep existing app-written logins usable without copying credentials.
selected = canonical if os.path.exists(canonical) or not os.path.exists(legacy) else legacy
os.environ.setdefault("HF_TOKEN_PATH", selected)
os.environ["HF_HOME"] = short_cache
os.environ["HF_HUB_CACHE"] = short_cache
+716
View File
@@ -0,0 +1,716 @@
"""Nested subprocess ownership for desktop-managed backend operations.
The desktop owns the backend with an OS process group/Job. Engine and
installer operations also need an independently terminable subtree: killing
only their direct child on a timeout leaves uv/git/model workers holding pipes
and mutating files.
On POSIX a small supervisor is the unreaped leader of a nested process group.
A control-pipe EOF (including kernel EOF when the backend dies) kills that
group; the parent also drains the group before reaping its stable leader. On
Windows the backend retains a nested kill-on-close Job directly and assigns
the suspended operation before resuming it. The outer desktop Job remains the
terminal fallback.
Standalone/server launches use the same nested owner, preserving their
independently terminable subtree without relying on ``taskkill`` or discovery.
"""
from __future__ import annotations
import os
import signal
import struct
import subprocess
import sys
import threading
import time
from pathlib import Path
from typing import Any, Optional
_RESULT = struct.Struct("!i")
_DESKTOP_MARKER = "OMNIVOICE_DESKTOP_CONTAINED"
_DRAIN_FD_ENV = "OMNIVOICE_DESKTOP_DRAIN_FD"
def backend_drain_fd(*, required: bool = False) -> Optional[int]:
"""Validated Rust-owned drain writer inherited by the desktop backend."""
if os.name != "posix" or os.environ.get(_DESKTOP_MARKER) != "1":
return None
try:
fd = int(os.environ[_DRAIN_FD_ENV])
os.fstat(fd)
except (KeyError, ValueError, OSError) as exc:
if required:
raise RuntimeError(
"desktop backend is missing its live nested-operation drain descriptor"
) from exc
return None
return fd
def secure_backend_drain_fd() -> None:
"""Restore CLOEXEC after Rust's one intentional backend inheritance."""
fd = backend_drain_fd(required=True)
if fd is not None:
os.set_inheritable(fd, False)
class OwnedPopen:
"""Popen-compatible handle for a desktop-owned nested operation."""
def __init__(
self,
proc: subprocess.Popen,
control_fd: int,
result_fd: int,
) -> None:
self._proc = proc
self._control_fd: Optional[int] = control_fd
self._result_fd: Optional[int] = result_fd
self._returncode: Optional[int] = None
self._lock = threading.RLock()
# Popen callers use these directly (protocol pipes and log drains).
self.stdin = proc.stdin
self.stdout = proc.stdout
self.stderr = proc.stderr
@property
def pid(self) -> int:
return self._proc.pid
@property
def args(self) -> Any:
return self._proc.args
@property
def returncode(self) -> Optional[int]:
return self._returncode
def _close_control(self) -> None:
fd, self._control_fd = self._control_fd, None
if fd is not None:
try:
os.close(fd)
except OSError:
# Cleanup is idempotent; another teardown path already closed it.
pass
def _read_result(self, fallback: int) -> int:
fd, self._result_fd = self._result_fd, None
if fd is None:
return fallback
try:
payload = b""
while len(payload) < _RESULT.size:
chunk = os.read(fd, _RESULT.size - len(payload))
if not chunk:
break
payload += chunk
return _RESULT.unpack(payload)[0] if len(payload) == _RESULT.size else fallback
except OSError:
return fallback
finally:
try:
os.close(fd)
except OSError:
# The descriptor may have been closed by cancellation cleanup.
pass
def _posix_exited_unreaped(self) -> bool:
flags = os.WEXITED | os.WNOHANG | os.WNOWAIT
info = os.waitid(os.P_PID, self.pid, flags)
return info is not None and info.si_pid != 0
def _posix_exited_reaping(self) -> Optional[int]:
"""macOS fallback for :meth:`_posix_exited_unreaped` (#1656).
CPython on macOS does not expose ``os.waitid`` (HAVE_WAITID is not set
in its build), so the WNOWAIT probe is unavailable there. This
fallback *reaps* the wrapper with ``waitpid(WNOHANG)``: it returns
the wrapper's exit code once it has exited, None while it is still
running, and raises ``ChildProcessError`` when another owner already
reaped it (the same refusal the waitid probe gives).
Reaping earlier than the WNOWAIT dance loses the pre-reap group kill
in :meth:`poll`; that is safe because the supervisor's control-pipe
EOF already terminates the whole nested group (#1635 design).
"""
pid, status = os.waitpid(self.pid, os.WNOHANG)
if pid != self.pid:
return None
rc = os.waitstatus_to_exitcode(status)
# Publish on the underlying Popen so its own wait()/poll() no-op.
self._proc.returncode = rc
return rc
def _posix_exit_state_reaping(self) -> Optional[int]:
""":meth:`_posix_exited_reaping` plus one concession: if the leader
was already reaped through *this* Popen (``_proc.returncode`` known),
report that code rather than refusing reaping by our own handle is
not the foreign reaper the ECHILD refusal exists for."""
try:
return self._posix_exited_reaping()
except ChildProcessError:
return self._proc.returncode
def _signal_owned_group(self, sig: int) -> None:
# The numeric group is safe only while its direct-child leader remains
# ours and unreaped. ECHILD therefore refuses rather than guessing.
try:
os.waitid(os.P_PID, self.pid, os.WEXITED | os.WNOHANG | os.WNOWAIT)
except ChildProcessError:
return
except AttributeError:
# macOS CPython has no os.waitid (#1656). waitpid still proves
# that this exact numeric pid is our live child: ECHILD refuses a
# foreign-reaped/reused pid, while pid == self.pid records an exit
# without ever signalling the now-unowned process-group number.
try:
pid, status = os.waitpid(self.pid, os.WNOHANG)
except ChildProcessError:
return
if pid == self.pid:
self._proc.returncode = os.waitstatus_to_exitcode(status)
return
try:
os.killpg(self.pid, sig)
except ProcessLookupError:
# The owned group exited between the waitid probe and the signal.
pass
def poll(self) -> Optional[int]:
with self._lock:
if self._returncode is not None:
return self._returncode
if os.name == "posix":
try:
if hasattr(os, "waitid"):
if not self._posix_exited_unreaped():
return None
self._signal_owned_group(signal.SIGKILL)
wrapper_rc = self._proc.wait()
else:
# macOS CPython: no os.waitid (#1656) — the reaping
# probe already terminated/killed nothing; the group
# is torn down by the control-pipe EOF in _close_control.
wrapper_rc = self._posix_exit_state_reaping()
if wrapper_rc is None:
return None
except ChildProcessError:
# Never signal a potentially reused group after another
# owner reaped the stable leader.
return None
else:
wrapper_rc = self._proc.poll()
if wrapper_rc is None:
return None
self._close_control()
self._returncode = self._read_result(wrapper_rc)
return self._returncode
def wait(self, timeout: Optional[float] = None) -> int:
deadline = None if timeout is None else time.monotonic() + timeout
while True:
rc = self.poll()
if rc is not None:
return rc
if deadline is not None and time.monotonic() >= deadline:
raise subprocess.TimeoutExpired(self.args, timeout)
time.sleep(0.01)
def terminate(self) -> None:
with self._lock:
if self._returncode is not None:
return
self._close_control()
if os.name == "posix":
self._signal_owned_group(signal.SIGTERM)
else:
# Closing the control pipe asks the supervisor to terminate
# its nested Job. The stable wrapper handle is a fallback.
try:
self._proc.terminate()
except OSError:
# The wrapper exited after the return-code check.
pass
def kill(self) -> None:
with self._lock:
if self._returncode is not None:
return
self._close_control()
if os.name == "posix":
self._signal_owned_group(signal.SIGKILL)
else:
try:
self._proc.kill()
except OSError:
# The wrapper exited after the return-code check.
pass
def __getattr__(self, name: str) -> Any:
return getattr(self._proc, name)
def __del__(self) -> None:
self._close_control()
fd, self._result_fd = self._result_fd, None
if fd is not None:
try:
os.close(fd)
except OSError:
# Finalization may race explicit wait or cancellation cleanup.
pass
class WindowsJobPopen:
"""Popen-compatible handle whose child tree lives in a retained Job.
Windows Job handles already provide the stable ownership that POSIX needs
a supervisor process group for. Keeping the handle in the backend means an
abrupt backend exit closes it in the kernel and kills the whole operation
tree, without inserting a second Python process in the sidecar loader path
(#1734).
"""
def __init__(self, proc: subprocess.Popen, job: Any, kernel32: Any) -> None:
self._proc = proc
self._job = job
self._kernel32 = kernel32
self._lock = threading.RLock()
self.stdin = proc.stdin
self.stdout = proc.stdout
self.stderr = proc.stderr
@property
def pid(self) -> int:
return self._proc.pid
@property
def args(self) -> Any:
return self._proc.args
@property
def returncode(self) -> Optional[int]:
return self._proc.returncode
def _close_job(self, *, terminate: bool) -> None:
job, self._job = self._job, None
if job is None:
return
try:
if terminate:
self._kernel32.TerminateJobObject(job, 1)
finally:
self._kernel32.CloseHandle(job)
def poll(self) -> Optional[int]:
with self._lock:
rc = self._proc.poll()
if rc is None:
return None
# A successful direct child may leave helpers behind. Match the
# supervisor contract by draining the retained Job before return.
self._close_job(terminate=True)
return rc
def wait(self, timeout: Optional[float] = None) -> int:
try:
rc = self._proc.wait(timeout=timeout)
except subprocess.TimeoutExpired:
raise
with self._lock:
self._close_job(terminate=True)
return rc
def terminate(self) -> None:
with self._lock:
self._close_job(terminate=True)
def kill(self) -> None:
self.terminate()
def __getattr__(self, name: str) -> Any:
return getattr(self._proc, name)
def __del__(self) -> None:
try:
self._close_job(terminate=True)
except Exception:
pass # interpreter shutdown; closing the OS handle is best-effort
def _spawn_windows_owned(argv: list[str], kwargs: dict[str, Any]) -> WindowsJobPopen:
"""Start *argv* suspended, assign its tree to a Job, then resume it."""
import ctypes
job, kernel32, wintypes = _windows_job()
child: Optional[subprocess.Popen] = None
popen_kwargs = dict(kwargs)
supplied_env = popen_kwargs.get("env")
operation_env = dict(os.environ if supplied_env is None else supplied_env)
operation_env.pop(_DRAIN_FD_ENV, None)
operation_env.pop(_DESKTOP_MARKER, None)
popen_kwargs["env"] = operation_env
supplied_flags = int(popen_kwargs.pop("creationflags", 0))
popen_kwargs["creationflags"] = supplied_flags | 0x08000000 | 0x00000004
try:
child = subprocess.Popen(argv, **popen_kwargs)
assign = kernel32.AssignProcessToJobObject
assign.argtypes = (wintypes.HANDLE, wintypes.HANDLE)
assign.restype = wintypes.BOOL
if not assign(job, wintypes.HANDLE(child._handle)):
raise OSError(ctypes.get_last_error(), "AssignProcessToJobObject")
_resume_windows_process(kernel32, wintypes, child.pid)
return WindowsJobPopen(child, job, kernel32)
except BaseException:
kernel32.TerminateJobObject(job, 1)
if child is not None:
try:
child.kill()
except OSError:
pass # the suspended child may already have exited
try:
child.wait(timeout=5)
except (OSError, subprocess.TimeoutExpired):
pass # Job termination remains the authoritative cleanup
kernel32.CloseHandle(job)
raise
def spawn_owned(
argv: list[str], **kwargs: Any
) -> "subprocess.Popen | OwnedPopen | WindowsJobPopen":
"""Spawn an operation with a stable, independently terminable owner."""
if os.name == "nt":
return _spawn_windows_owned(argv, kwargs)
drain_fd = backend_drain_fd(required=True)
control_read, control_write = os.pipe()
result_read, result_write = os.pipe()
wrapper_argv = _supervisor_argv(
control_read,
result_write,
argv,
)
wrapper_kwargs = dict(kwargs)
wrapper_kwargs["start_new_session"] = True
pass_fds = [control_read, result_write]
if drain_fd is not None:
pass_fds.append(drain_fd)
if wrapper_kwargs.get("env") is not None:
wrapper_env = dict(wrapper_kwargs["env"])
wrapper_env[_DESKTOP_MARKER] = "1"
wrapper_env[_DRAIN_FD_ENV] = str(drain_fd)
wrapper_kwargs["env"] = wrapper_env
wrapper_kwargs["pass_fds"] = tuple(pass_fds)
try:
proc = subprocess.Popen(wrapper_argv, **wrapper_kwargs)
except BaseException:
# The finally block exclusively owns the child-side endpoints. Closing
# them here as well risks closing a reused descriptor in another thread.
for fd in (control_write, result_read):
try:
os.close(fd)
except OSError:
# A partial spawn may already have closed a parent-side endpoint.
pass
raise
finally:
for fd in (control_read, result_write):
try:
os.close(fd)
except OSError:
# Popen may have consumed an inherited child-side endpoint.
pass
return OwnedPopen(proc, control_write, result_read)
def _supervisor_argv(
control_token: int,
result_token: int,
argv: list[str],
) -> list[str]:
prefix = [sys.executable]
if not getattr(sys, "frozen", False):
prefix.append(str(Path(__file__).resolve().parents[1] / "main.py"))
return [
*prefix,
"--supervise",
str(control_token),
str(result_token),
"--",
*map(str, argv),
]
def _write_result(fd: int, returncode: int) -> None:
try:
os.write(fd, _RESULT.pack(int(returncode)))
except OSError:
# The caller may have cancelled and closed its result reader.
pass
finally:
try:
os.close(fd)
except OSError:
# Writing or cancellation may already have closed the descriptor.
pass
def _operation_env() -> dict[str, str]:
env = os.environ.copy()
# The operation intentionally does not own the Rust drain writer. Avoid
# exposing a stale numeric token which nested code could mistake as valid.
env.pop(_DRAIN_FD_ENV, None)
env.pop(_DESKTOP_MARKER, None)
return env
def _supervise_posix(control_fd: int, result_fd: int, argv: list[str]) -> int:
def cancel_on_eof() -> None:
try:
while os.read(control_fd, 1):
pass
except OSError:
# Closing the control descriptor is itself a cancellation signal.
pass
os.killpg(os.getpgrp(), signal.SIGKILL)
threading.Thread(target=cancel_on_eof, daemon=True).start()
try:
child = subprocess.Popen(argv, close_fds=True, env=_operation_env())
rc = child.wait()
except OSError:
rc = 127
_write_result(result_fd, rc)
# Drain children which outlived the operation before the stable group
# leader exits. SIGKILL intentionally includes this supervisor.
os.killpg(os.getpgrp(), signal.SIGKILL)
return rc # unreachable
def _windows_job() -> tuple[Any, Any, Any]:
import ctypes
import ctypes.wintypes as wintypes
kernel32 = ctypes.WinDLL("kernel32", use_last_error=True)
kernel32.CloseHandle.argtypes = (wintypes.HANDLE,)
kernel32.CloseHandle.restype = wintypes.BOOL
kernel32.TerminateJobObject.argtypes = (wintypes.HANDLE, wintypes.UINT)
kernel32.TerminateJobObject.restype = wintypes.BOOL
kernel32.ReadFile.argtypes = (
wintypes.HANDLE,
ctypes.c_void_p,
wintypes.DWORD,
ctypes.POINTER(wintypes.DWORD),
ctypes.c_void_p,
)
kernel32.ReadFile.restype = wintypes.BOOL
kernel32.WriteFile.argtypes = (
wintypes.HANDLE,
ctypes.c_void_p,
wintypes.DWORD,
ctypes.POINTER(wintypes.DWORD),
ctypes.c_void_p,
)
kernel32.WriteFile.restype = wintypes.BOOL
create = kernel32.CreateJobObjectW
create.argtypes = (ctypes.c_void_p, wintypes.LPCWSTR)
create.restype = wintypes.HANDLE
job = create(None, None)
if not job:
raise OSError(ctypes.get_last_error(), "CreateJobObjectW")
class BasicLimits(ctypes.Structure):
_fields_ = [
("PerProcessUserTimeLimit", ctypes.c_longlong),
("PerJobUserTimeLimit", ctypes.c_longlong),
("LimitFlags", wintypes.DWORD),
("MinimumWorkingSetSize", ctypes.c_size_t),
("MaximumWorkingSetSize", ctypes.c_size_t),
("ActiveProcessLimit", wintypes.DWORD),
("Affinity", ctypes.c_size_t),
("PriorityClass", wintypes.DWORD),
("SchedulingClass", wintypes.DWORD),
]
class IoCounters(ctypes.Structure):
_fields_ = [(name, ctypes.c_ulonglong) for name in (
"ReadOperationCount", "WriteOperationCount", "OtherOperationCount",
"ReadTransferCount", "WriteTransferCount", "OtherTransferCount",
)]
class ExtendedLimits(ctypes.Structure):
_fields_ = [
("BasicLimitInformation", BasicLimits),
("IoInfo", IoCounters),
("ProcessMemoryLimit", ctypes.c_size_t),
("JobMemoryLimit", ctypes.c_size_t),
("PeakProcessMemoryUsed", ctypes.c_size_t),
("PeakJobMemoryUsed", ctypes.c_size_t),
]
info = ExtendedLimits()
info.BasicLimitInformation.LimitFlags = 0x00002000 # KILL_ON_JOB_CLOSE
set_info = kernel32.SetInformationJobObject
set_info.argtypes = (wintypes.HANDLE, ctypes.c_int, ctypes.c_void_p, wintypes.DWORD)
set_info.restype = wintypes.BOOL
if not set_info(job, 9, ctypes.byref(info), ctypes.sizeof(info)):
error = ctypes.get_last_error()
kernel32.CloseHandle(job)
raise OSError(error, "SetInformationJobObject")
return job, kernel32, wintypes
def _resume_windows_process(kernel32: Any, wintypes: Any, pid: int) -> None:
import ctypes
class ThreadEntry(ctypes.Structure):
_fields_ = [
("dwSize", wintypes.DWORD),
("cntUsage", wintypes.DWORD),
("th32ThreadID", wintypes.DWORD),
("th32OwnerProcessID", wintypes.DWORD),
("tpBasePri", wintypes.LONG),
("tpDeltaPri", wintypes.LONG),
("dwFlags", wintypes.DWORD),
]
kernel32.CreateToolhelp32Snapshot.argtypes = (wintypes.DWORD, wintypes.DWORD)
kernel32.CreateToolhelp32Snapshot.restype = wintypes.HANDLE
kernel32.Thread32First.argtypes = (wintypes.HANDLE, ctypes.POINTER(ThreadEntry))
kernel32.Thread32First.restype = wintypes.BOOL
kernel32.Thread32Next.argtypes = (wintypes.HANDLE, ctypes.POINTER(ThreadEntry))
kernel32.Thread32Next.restype = wintypes.BOOL
kernel32.OpenThread.argtypes = (wintypes.DWORD, wintypes.BOOL, wintypes.DWORD)
kernel32.OpenThread.restype = wintypes.HANDLE
kernel32.ResumeThread.argtypes = (wintypes.HANDLE,)
kernel32.ResumeThread.restype = wintypes.DWORD
snapshot = kernel32.CreateToolhelp32Snapshot(0x00000004, 0)
invalid = ctypes.c_void_p(-1).value
if snapshot == invalid:
raise OSError(ctypes.get_last_error(), "CreateToolhelp32Snapshot")
try:
entry = ThreadEntry(dwSize=ctypes.sizeof(ThreadEntry))
found = kernel32.Thread32First(snapshot, ctypes.byref(entry))
while found:
if entry.th32OwnerProcessID == pid:
thread = kernel32.OpenThread(0x0002, False, entry.th32ThreadID)
if not thread:
raise OSError(ctypes.get_last_error(), "OpenThread")
try:
if kernel32.ResumeThread(thread) == 0xFFFFFFFF:
raise OSError(ctypes.get_last_error(), "ResumeThread")
return
finally:
kernel32.CloseHandle(thread)
found = kernel32.Thread32Next(snapshot, ctypes.byref(entry))
finally:
kernel32.CloseHandle(snapshot)
raise OSError("suspended operation thread was not found")
def _supervise_windows(control_fd: int, result_fd: int, argv: list[str]) -> int:
import ctypes
job, kernel32, wintypes = _windows_job()
cancelled = threading.Event()
job_lock = threading.Lock()
job_open = True
def terminate_job() -> None:
with job_lock:
if job_open:
kernel32.TerminateJobObject(job, 1)
def cancel_on_eof() -> None:
byte = ctypes.create_string_buffer(1)
count = wintypes.DWORD()
while kernel32.ReadFile(
wintypes.HANDLE(control_fd), byte, 1, ctypes.byref(count), None
) and count.value:
pass
kernel32.CloseHandle(wintypes.HANDLE(control_fd))
cancelled.set()
terminate_job()
threading.Thread(target=cancel_on_eof, daemon=True).start()
child: Optional[subprocess.Popen] = None
rc = 127
try:
child = subprocess.Popen(
argv,
close_fds=True,
env=_operation_env(),
creationflags=0x08000000 | 0x00000004, # NO_WINDOW | SUSPENDED
)
assign = kernel32.AssignProcessToJobObject
assign.argtypes = (wintypes.HANDLE, wintypes.HANDLE)
assign.restype = wintypes.BOOL
if not assign(job, wintypes.HANDLE(child._handle)):
raise OSError(ctypes.get_last_error(), "AssignProcessToJobObject")
if cancelled.is_set():
terminate_job()
else:
_resume_windows_process(kernel32, wintypes, child.pid)
rc = child.wait()
# A successful direct child may leave helpers behind; terminate the
# nested stable Job before reporting completion.
terminate_job()
except OSError:
terminate_job()
if child is not None:
try:
# Assignment itself may have failed, leaving this suspended
# process outside the nested Job. Terminate it through its
# stable process handle before waiting; never strand an
# unassigned operation or rely on the outer desktop Job.
child.kill()
except OSError:
# The suspended child may have exited during Job teardown.
pass
try:
child.wait(timeout=5)
except (OSError, subprocess.TimeoutExpired):
# The outer desktop Job remains the terminal containment fallback.
pass
finally:
payload = _RESULT.pack(int(rc))
payload_buffer = ctypes.create_string_buffer(payload)
written = wintypes.DWORD()
kernel32.WriteFile(
wintypes.HANDLE(result_fd),
payload_buffer,
len(payload),
ctypes.byref(written),
None,
)
kernel32.CloseHandle(wintypes.HANDLE(result_fd))
with job_lock:
job_open = False
kernel32.CloseHandle(job)
return rc
def supervisor_main(args: list[str]) -> int:
if len(args) < 5 or args[0] != "--supervise" or args[3] != "--":
return 2
control_fd = int(args[1])
result_fd = int(args[2])
argv = args[4:]
secure_backend_drain_fd()
if os.name == "posix":
return _supervise_posix(control_fd, result_fd, argv)
return _supervise_windows(control_fd, result_fd, argv)
def _main() -> int:
return supervisor_main(sys.argv[1:])
if __name__ == "__main__":
raise SystemExit(_main())
+2 -1
View File
@@ -274,7 +274,8 @@ _BASE_SCHEMA = """
started_at REAL,
finished_at REAL,
lease_expires_at REAL,
grace_expires_at REAL
grace_expires_at REAL,
deadlines_json TEXT
);
CREATE INDEX IF NOT EXISTS idx_remote_attempts_task ON remote_task_attempts(task_id);
CREATE INDEX IF NOT EXISTS idx_remote_attempts_worker ON remote_task_attempts(worker_id, state);
+76 -4
View File
@@ -35,7 +35,8 @@ import sys
from dataclasses import dataclass
from typing import Literal
DeviceFamily = Literal["cuda", "rocm", "mps", "xpu", "cpu"]
DeviceFamily = Literal["cuda", "rocm", "mps", "xpu", "npu", "cpu"]
ACCELERATOR_PRIORITY = ("cuda", "rocm", "xpu", "npu", "mps")
# Stable substring stamped onto notes that represent a real kernel-launch risk
# (arch/driver mismatch) — as opposed to advisory notes (multi-GPU, VRAM query
@@ -177,6 +178,23 @@ def gfx_for_hsa_override(value: str) -> str | None:
#: The ROCm kernel driver interface. Its absence, or its presence without
#: permission, are the two commonest reasons a ROCm host silently runs on CPU.
_KFD_DEVICE = "/dev/kfd"
_DXG_DEVICE = "/dev/dxg"
_DXG_RUNTIME_PATHS = (
"/usr/lib/libdxcore.so",
"/usr/lib/librocdxg.so",
"/usr/share/rocdxg/dids.conf",
)
def _rocm_requires_dxg_detection(version: object) -> bool:
"""Whether WSL's ROCDXG bridge still needs its explicit opt-in."""
try:
parts = str(version).split(".")
return (int(parts[0]), int(parts[1])) < (7, 13)
except (IndexError, TypeError, ValueError):
# Unknown versions get the conservative advice. The variable is
# harmless on newer runtimes and necessary on every older one.
return True
def why_no_gpu(torch) -> tuple[str, ...]:
@@ -230,6 +248,40 @@ def why_no_gpu(torch) -> tuple[str, ...]:
# /dev/kfd only exists on Linux; on any other platform its absence
# says nothing, so don't invent a reason.
if sys.platform.startswith("linux"):
if not os.path.exists(_KFD_DEVICE) and os.path.exists(_DXG_DEVICE):
if not os.access(_DXG_DEVICE, os.R_OK | os.W_OK):
return (
f"ROCm {hip} is installed and {_DXG_DEVICE} exists, "
"but this process cannot open it — pass "
"--device /dev/dxg to the WSL container",
)
dxg_detection = os.environ.get("HSA_ENABLE_DXG_DETECTION", "").strip()
if dxg_detection == "0":
return (
f"ROCm {hip} is installed and {_DXG_DEVICE} is reachable, "
"but HSA_ENABLE_DXG_DETECTION=0 explicitly disables the "
"WSL GPU bridge; remove it or set it to 1",
)
if _rocm_requires_dxg_detection(hip) and dxg_detection != "1":
return (
f"ROCm {hip} is installed and {_DXG_DEVICE} is "
"reachable, but this pre-7.13 runtime requires "
"HSA_ENABLE_DXG_DETECTION=1 inside WSL containers",
)
missing = [
path for path in _DXG_RUNTIME_PATHS if not os.path.exists(path)
]
if missing:
return (
f"ROCm {hip} can reach {_DXG_DEVICE}, but the WSL "
"ROCDXG runtime mounts are incomplete; missing: "
f"{', '.join(missing)}",
)
return (
f"ROCm {hip} and the WSL ROCDXG bridge are reachable, "
"but no GPU was enumerated — verify the AMD Windows "
"driver, librocdxg/ROCm compatibility, and host `rocminfo`",
)
if not os.path.exists(_KFD_DEVICE):
return (
f"ROCm {hip} is installed but {_KFD_DEVICE} is not "
@@ -482,9 +534,13 @@ def _probe() -> HostCaps:
# is the whole truth in that case (CodeRabbit, #1425).
notes.extend(why_no_gpu(torch))
# ── Intel XPU via IPEX ───────────────────────────────────────────────
# Older builds register XPU through IPEX; modern torch exposes it directly.
try:
import intel_extension_for_pytorch # noqa: F401
except Exception:
# Optional IPEX may be absent or incompatible; still probe native torch XPU.
pass
try:
if hasattr(torch, "xpu") and torch.xpu.is_available():
detected.append("xpu")
if not device_name:
@@ -495,7 +551,23 @@ def _probe() -> HostCaps:
pass
notes.append("XPU VRAM not queried (unreliable across IPEX versions)")
except Exception:
# IPEX absent or XPU probe failed — no XPU on this host.
# XPU probe failed — no usable XPU on this host.
pass
# Vendor extensions may register an NPU with torch. Probe only an already
# registered backend; never install or import an optional vendor package.
try:
if hasattr(torch, "npu") and torch.npu.is_available():
detected.append("npu")
if not device_name:
try:
device_name = torch.npu.get_device_name(0)
except Exception:
# An unavailable display name does not invalidate a usable NPU.
pass
notes.append("NPU VRAM not queried")
except Exception:
# Missing or broken vendor backends mean no usable NPU; continue probing.
pass
# ── Apple Silicon MPS ────────────────────────────────────────────────
@@ -528,7 +600,7 @@ def _probe() -> HostCaps:
# Preferred family by priority; cpu when nothing accelerated was detected.
family: DeviceFamily = "cpu"
for pref in ("cuda", "rocm", "xpu", "mps"):
for pref in ACCELERATOR_PRIORITY:
if pref in detected:
family = pref # type: ignore[assignment]
break
+64
View File
@@ -21,6 +21,7 @@ Check shape:
"""
from __future__ import annotations
import importlib
import os
import platform
import shutil
@@ -367,10 +368,44 @@ def run_diagnostics(include_network: bool = True, deep: bool = False) -> dict:
counts = {OK: 0, WARN: 0, FAIL: 0}
for c in checks:
counts[c["status"]] += 1
engine_execution = []
for family in ("tts", "asr"):
active = "unknown"
try:
module = importlib.import_module(f"services.{family}_backend")
active = module.active_backend_id()
row = next((item for item in module.list_backends() if item.get("id") == active), None)
if row is not None:
engine_execution.append({
"family": family,
"engine_id": active,
**row["execution_evidence"],
})
except Exception: # noqa: BLE001 - evidence must not break diagnostics
# Preserve the other family's successful evidence and make this
# collection failure explicit without exposing exception text.
engine_execution.append({
"family": family,
"engine_id": active,
"implementation_variant": None,
"declared_device_families": [],
"evidence_state": "collection_failed",
"actual_execution_provider": None,
"actual_execution_device": None,
"gpu_name": None,
"gpu_architecture": None,
"precision_or_quantization": None,
"cpu_fallback_reason": None,
"cpu_fallback_stage": None,
"parent_memory_observable": None,
"runtime_versions": {},
})
return {
"app_version": APP_VERSION,
"platform": scrub_text(platform.platform()),
"checks": checks,
"engine_execution": engine_execution,
"summary": {
"ok": counts[FAIL] == 0,
"passed": counts[OK],
@@ -395,6 +430,35 @@ def format_text(report: dict) -> str:
lines.append(f"{tag[c['status']]} {c['label']}: {c['detail']}")
if c.get("hint"):
lines.append(f" hint: {c['hint']}")
if report.get("engine_execution"):
lines.append("")
lines.append("Engine execution evidence:")
for item in report["engine_execution"]:
if item.get("actual_execution_provider"):
provider = item["actual_execution_provider"]
elif item.get("evidence_state") == "subprocess_loaded_provider_unreported":
provider = "loaded child; provider not reported"
else:
provider = "not loaded"
precision = item.get("precision_or_quantization") or "unknown"
device = item.get("actual_execution_device") or "unknown"
gpu = item.get("gpu_name") or "none"
architecture = item.get("gpu_architecture") or "unknown"
fallback_stage = item.get("cpu_fallback_stage") or "none"
fallback_reason = item.get("cpu_fallback_reason") or "none"
versions = ",".join(
f"{name}={version}"
for name, version in sorted(item.get("runtime_versions", {}).items())
) or "none"
visible = "yes" if item.get("parent_memory_observable") else "no"
lines.append(
f" {item['family']}:{item['engine_id']} provider={provider}; "
f"device={device}; gpu={gpu}; architecture={architecture}; "
f"precision={precision}; fallback-stage={fallback_stage}; "
f"fallback-reason={fallback_reason}; runtimes={versions}; "
f"evidence-state={item.get('evidence_state', 'unknown')}; "
f"parent-memory-visible={visible}"
)
s = report["summary"]
lines.append("")
lines.append(
+34 -8
View File
@@ -23,9 +23,17 @@ logger = logging.getLogger("omnivoice.events")
_listeners: list[asyncio.Queue] = []
_lock = asyncio.Lock()
# The loop that serves /ws/events, captured on first use. Sync FastAPI
# endpoints (rename/delete profile, revoke consent) run in threadpool workers
# where `asyncio.get_running_loop()` raises, which used to silently drop their
# events — the UI then never refetched the voice list (#1158 class).
_serving_loop: asyncio.AbstractEventLoop | None = None
async def subscribe() -> asyncio.Queue:
"""Register a new listener. Returns a Queue that receives event dicts."""
global _serving_loop
_serving_loop = asyncio.get_running_loop()
q: asyncio.Queue = asyncio.Queue(maxsize=64)
async with _lock:
_listeners.append(q)
@@ -57,11 +65,29 @@ def emit(kind: str, payload: dict[str, Any] | None = None) -> None:
}
event_str = json.dumps(event)
try:
loop = asyncio.get_running_loop()
loop.create_task(_broadcast(event_str))
caller_loop = asyncio.get_running_loop()
except RuntimeError:
# No event loop running (unlikely in FastAPI context but safe)
caller_loop = None
target_loop = _serving_loop or caller_loop
if target_loop is None:
# No serving loop yet — nobody to notify; dropping is correct.
logger.debug("No event loop — event dropped: %s", kind)
return
try:
if caller_loop is target_loop:
target_loop.create_task(_broadcast(event_str))
else:
# Sync endpoints and async producers on a foreign loop must both
# hand off: the lock and listener queues belong to serving_loop.
target_loop.call_soon_threadsafe(_schedule_broadcast, event_str)
except RuntimeError:
# The serving loop closed between capture and use (app shutdown).
logger.debug("Event loop closed — event dropped: %s", kind)
def _schedule_broadcast(event_str: str) -> None:
"""Run `_broadcast` on the serving loop; called via call_soon_threadsafe."""
asyncio.get_running_loop().create_task(_broadcast(event_str))
async def _broadcast(event_str: str) -> None:
@@ -73,11 +99,11 @@ async def _broadcast(event_str: str) -> None:
q.put_nowait(event_str)
except asyncio.QueueFull:
# Slow consumer — drop oldest, then push. Not a race (#1163):
# every queue op runs on the single event loop, and there is
# no await between the QueueFull and this get_nowait/put_nowait
# pair — no consumer can interleave, so get_nowait cannot raise
# QueueEmpty here. emit() from a foreign thread drops the event
# before ever touching a queue (see the RuntimeError branch).
# every queue op runs on the single event loop (a foreign
# thread's emit() hands off via call_soon_threadsafe first),
# and there is no await between the QueueFull and this
# get_nowait/put_nowait pair — no consumer can interleave, so
# get_nowait cannot raise QueueEmpty here.
try:
q.get_nowait()
q.put_nowait(event_str)
+39
View File
@@ -52,6 +52,7 @@ _REDACTED_VALUE = "***REDACTED***"
# One-line "what to do" per docs-taxonomy key. Keys mirror error_docs_map's
# taxonomy; the docs URL itself stays owned by error_docs_map.
_HINTS: dict[str, str] = {
"GPU_OOM": "Close other GPU-heavy apps or unload models, then retry. You can also choose CPU in Settings → Performance & Device or select a smaller TTS engine.",
"WORKER_AT_CAPACITY": "Wait for a running job on that worker to finish, or choose another available worker and retry.",
"MODEL_NOT_INSTALLED": "Install or enable this engine on the worker machine, then refresh its capabilities and retry.",
"MODEL_NOT_DOWNLOADED": "Open Models, install this model on the selected worker, then retry when the download completes.",
@@ -290,6 +291,9 @@ def append_hf_mirror_hint(text: str) -> str:
# must NOT be added: its bare "timed out" trigger would stamp a "video server"
# hint on a model-load timeout that leaks through the 500 handler.
_CONTEXT_FREE_HINT_CLASSES = frozenset({
# Device allocator signatures are specific enough to attach the shared
# recovery without exposing CUDA's process table or filesystem paths.
"GPU_OOM",
"SOCKS_PROXY_SUPPORT_MISSING",
"SSL_HANDSHAKE_FAILURE",
# Its trigger is an exact OpenSSL string, so it cannot be confused with
@@ -323,6 +327,38 @@ def append_hint(text: str) -> str:
return f"{text}{hint}" if hint else text
_GPU_OOM_SIGNATURES = (
"cuda out of memory",
"cuda error: out of memory",
"cuda_error_out_of_memory",
"mps backend out of memory",
"hip out of memory",
"out of memory on device",
)
def is_gpu_oom(error: BaseException | str) -> bool:
"""Recognize device OOMs through wrappers without importing torch."""
pending: list[BaseException] = [error] if isinstance(error, BaseException) else []
seen: set[int] = set()
while pending:
current = pending.pop()
if id(current) in seen:
continue
seen.add(id(current))
if type(current).__name__ == "OutOfMemoryError":
return True
if any(signature in str(current).lower() for signature in _GPU_OOM_SIGNATURES):
return True
if current.__cause__ is not None:
pending.append(current.__cause__)
if current.__context__ is not None:
pending.append(current.__context__)
if isinstance(error, str):
return any(signature in error.lower() for signature in _GPU_OOM_SIGNATURES)
return False
def classify(reason: str) -> str:
"""Map a failure reason to a docs-taxonomy key, or "" when unknown.
@@ -330,6 +366,8 @@ def classify(reason: str) -> str:
backend log / diagnostic names the same class the UI deeplink will use.
"""
low = (reason or "").lower()
if is_gpu_oom(low):
return "GPU_OOM"
if "pkg_resources" in low:
return "PKG_RESOURCES_MISSING"
if "quarantine" in low or "is damaged" in low or "gatekeeper" in low:
@@ -519,6 +557,7 @@ def classify(reason: str) -> str:
or "unable to download video" in low
or "remote end closed" in low
or "timed out" in low
or "the page needs to be reloaded" in low
):
return "VIDEO_DOWNLOAD_NETWORK"
# #1227: Windows Smart App Control / WDAC / AppLocker refused to load a
+34
View File
@@ -0,0 +1,34 @@
"""Terminate a desktop-contained backend when its owning shell disappears."""
from __future__ import annotations
import os
import sys
import threading
from typing import BinaryIO, Callable
def _watch_parent_pipe(reader: BinaryIO, exit_process: Callable[[int], None]) -> None:
"""Block until the desktop-owned stdin pipe closes, then exit immediately."""
try:
while reader.read(1):
pass
except (OSError, ValueError):
# A broken or already-closed parent-owned pipe is equivalent to EOF.
pass
exit_process(0)
def arm_desktop_parent_watchdog() -> bool:
"""Use stdin EOF as an unforgeable parent-liveness signal for desktop runs."""
if os.environ.get("OMNIVOICE_DESKTOP_CONTAINED") != "1":
return False
reader = getattr(sys.stdin, "buffer", None)
if reader is None:
return False
threading.Thread(
target=_watch_parent_pipe,
args=(reader, os._exit),
name="desktop-parent-watchdog",
daemon=True,
).start()
return True
+43 -16
View File
@@ -7,6 +7,7 @@ only the unguessable capability token crosses loopback HTTP.
from __future__ import annotations
import json
import logging
import os
import re
import secrets
@@ -14,6 +15,8 @@ import stat
from core.config import DATA_DIR
logger = logging.getLogger("omnivoice.path_authorization")
_TOKEN_RE = re.compile(r"[0-9a-f]{64}\Z")
_KINDS = {
"models_dir",
@@ -40,25 +43,49 @@ def consume(token: str, expected_kind: str) -> str:
if expected_kind not in _KINDS or not _TOKEN_RE.fullmatch(token or ""):
raise PathAuthorizationError("Invalid or expired desktop authorization")
root = _AUTH_DIR
# Distinguish "the store exists but this token isn't in it" (expired /
# already consumed / never issued — normal, no server-side signal) from
# "the store doesn't exist at all" (the desktop app and this backend are
# very likely pointed at different data directories, e.g. a dev backend
# started without OMNIVOICE_DATA_DIR, or a stale custom data folder — see
# #1781). The client-facing message is byte-identical either way (never
# leak local filesystem paths, or even which case occurred, over HTTP —
# CWE-200); the mismatch case additionally gets a server log line so it's
# diagnosable instead of a silent 403. That log line is deliberately
# path-free too (CWE-532: per-user filesystem paths, e.g. a home
# directory username, are sensitive and don't belong in application
# logs) — it names the failure mode, not the directory.
try:
entries = os.scandir(root)
except FileNotFoundError as exc:
logger.warning(
"path authorization store does not exist; the desktop app and "
"this backend likely resolved different data directories "
"(see #1781)"
)
raise PathAuthorizationError("Invalid or expired desktop authorization") from exc
except OSError as exc:
raise PathAuthorizationError("Invalid or expired desktop authorization") from exc
candidate = None
try:
for entry in os.scandir(root):
if not _TOKEN_RE.fullmatch(entry.name.removesuffix(".json")):
continue
if not entry.is_file(follow_symlinks=False):
continue
try:
with open(entry.path, "r", encoding="utf-8") as handle:
probe = json.load(handle)
except (OSError, UnicodeError, json.JSONDecodeError):
continue # Ignore corrupt/stale capabilities; they authorize nothing.
if isinstance(probe, dict) and secrets.compare_digest(
str(probe.get("token", "")), token
):
candidate = entry.path
break
with entries:
for entry in entries:
if not _TOKEN_RE.fullmatch(entry.name.removesuffix(".json")):
continue
if not entry.is_file(follow_symlinks=False):
continue
try:
with open(entry.path, "r", encoding="utf-8") as handle:
probe = json.load(handle)
except (OSError, UnicodeError, json.JSONDecodeError):
continue # Ignore corrupt/stale capabilities; they authorize nothing.
if isinstance(probe, dict) and secrets.compare_digest(
str(probe.get("token", "")), token
):
candidate = entry.path
break
if candidate is None:
raise OSError("capability not found")
raise PathAuthorizationError("Invalid or expired desktop authorization")
claimed = os.path.join(root, f".consuming-{os.getpid()}-{secrets.token_hex(16)}")
os.replace(candidate, claimed)
except OSError as exc:
+50
View File
@@ -91,3 +91,53 @@ def resolve(key: str, *, env: Optional[str] = None, default: Any = None) -> Any:
if v:
return v
return get(key, default)
# ── external-override detection (#1787 review fix) ──────────────────────────
# restore_env() below uses os.environ.setdefault(), so a value already present
# in the process's environment (shell profile, `.env`, Docker `-e`, systemd
# unit, …) silently wins over anything saved in prefs.json — the setdefault
# call is a no-op. That is the right behavior (env stays authoritative,
# matching resolve()'s contract above), but a Settings control that persists a
# value to prefs.json must not tell the user it "took effect after restart"
# when an external source will keep shadowing it on every future restart too.
#
# _EXTERNALLY_PROVIDED records, once per process start, every bare key that
# was ALREADY present in os.environ the moment restore_env() ran — i.e.
# before our own setdefault() calls could have put it there, and before any
# value our Settings UI ever wrote (Settings only ever writes prefs.json plus
# the CURRENT process's os.environ; it never touches a shell profile or `.env`
# file). Snapshotting unconditionally — not only for keys prefs.json already
# has an entry for — means is_env_shadowed() also answers correctly for a key
# a user is about to save for the FIRST time. Membership is stable for the
# life of the process (nothing removes an inherited env var), and since a
# plain restart re-inherits the same shell / container environment, it is
# also a reliable predictor for the NEXT start: if the external source is
# still exporting the key, the next restart will be shadowed again the same
# way.
_EXTERNALLY_PROVIDED: frozenset[str] = frozenset()
def restore_env(data: dict) -> None:
"""Restore ``env.*`` prefs into ``os.environ`` (startup only).
Called once from main.py's ``env_prefs`` step, before any user code reads
``os.environ``. Snapshots which keys were already externally provided
see :func:`is_env_shadowed` then applies every saved ``env.*`` pref via
``setdefault`` (never overriding an explicitly-set env var).
"""
global _EXTERNALLY_PROVIDED
_EXTERNALLY_PROVIDED = frozenset(os.environ.keys())
for k, v in data.items():
if not k.startswith("env.") or not v:
continue
os.environ.setdefault(k[len("env."):], str(v))
def is_env_shadowed(key: str) -> bool:
"""Whether *key* was already present in the environment from a source
other than our own prefs restore, as of the last time :func:`restore_env`
ran. If prefs.json holds (or will hold) a saved value for *key*, that
value is being silently ignored and will be again on the next restart
unless the external source is removed."""
return key in _EXTERNALLY_PROVIDED
+51
View File
@@ -28,6 +28,15 @@ def stream_failure(code: str) -> dict[str, object]:
"detail": "Generation capacity is busy. Try again shortly.",
"retryable": True,
},
"generation_timeout": {
"code": "generation_timeout",
"detail": (
"Generation exceeded the compute-time limit. The backend is "
"still running; try a shorter passage, or raise the "
"compute-time budget in Settings → Performance & Device."
),
"retryable": True,
},
"invalid_request": {
"code": "invalid_request",
"detail": "The generation request could not be processed.",
@@ -65,6 +74,48 @@ def stream_failure(code: str) -> dict[str, object]:
return dict(failures.get(code, failures["generation_failed"]))
def stream_generation_failure(error: BaseException | object) -> dict[str, object]:
"""``generation_failed`` stream metadata, enriched with the actual cause.
The bare "Generation failed. Check the selected engine and try again." is
the floor for an *unrecognized* failure. When the private exception DOES
classify to a known failure class a corrupt model cache, an unreachable
Hugging Face mirror, a missing ffmpeg/ffprobe, a Windows paging-file limit,
a SOCKS/TLS proxy problem, the stable VoiceStudio-owned remediation for
that class is appended so the user can self-diagnose instead of guessing
which engine or which failure. This is the same enrichment the classic
(non-streaming) ``/generate`` 500 already gets via
:func:`public_exception_response`; the in-band streaming error frame
replaces the global 500 handler for a streaming request and used to bypass
it entirely (#1607).
Only VoiceStudio-owned constants are copied never a substring of
``error`` (Constitution I). Never raises: a diagnosis failure must not
replace the failure being diagnosed.
"""
payload = stream_failure("generation_failed")
try:
enriched = public_exception_response(error, fallback=str(payload["detail"]))
except Exception:
return payload
hint = enriched.get("hint")
if hint:
payload["detail"] = enriched["detail"]
payload["hint"] = hint
topic = enriched.get("docs_topic")
if topic:
payload["docs_topic"] = topic
try:
from core import error_docs_map
url = error_docs_map.ERROR_DOCS.get(topic, "")
except Exception:
url = ""
if url:
payload["docs_url"] = url
return payload
def public_failure(
logger: logging.Logger,
log_message: str,
+1 -1
View File
@@ -24,7 +24,7 @@ from pathlib import Path
# tests/test_app_version.py::test_all_version_files_in_lockstep and bumped by
# release.yml's version-bump job, so it stays equal to
# pyproject/tauri.conf/Cargo/package.json.
_FALLBACK_VERSION = "0.5.0"
_FALLBACK_VERSION = "0.5.2"
def _fallback_version() -> str:
+4 -4
View File
@@ -56,15 +56,15 @@ class Confucius4Backend(SubprocessBackend):
id = "confucius4-tts"
display_name = (
"Confucius4-TTS (LLM, 14 langs, cross-lingual zero-shot clone, CUDA/CPU, Apache-2.0)"
"Confucius4-TTS (LLM, 14 langs, cross-lingual zero-shot clone, Apache-2.0)"
)
supports_voice_design = False # timbre comes from a reference clip
# Upstream vocoder rate (config target_sample_rate) — confirmed 22 050 Hz by
# a live run (2026-07-02); still re-read from the sidecar's ready/audio frames.
_DEFAULT_SAMPLE_RATE = 22050
# CUDA fast path + CPU fallback, both exercised (CPU end-to-end validated).
# No MPS claim — upstream has no Metal path.
gpu_compat = ("cuda", "cpu")
# Match device propagation into upstream .to(device). XPU/NPU routing is
# contract-tested, not a claim of physical-hardware synthesis validation.
gpu_compat = ("cuda", "rocm", "xpu", "npu", "cpu")
@classmethod
def is_available(cls) -> tuple[bool, str]:
+13 -2
View File
@@ -104,7 +104,7 @@ def _ensure_clone_on_sys_path() -> None:
def _load_model(stdout):
"""Cold-construct the Confucius4 model (CUDA, else CPU — both validated)."""
"""Cold-construct using an available torch accelerator, with CPU fallback."""
global _model
if _model is not None:
return _model
@@ -115,7 +115,18 @@ def _load_model(stdout):
import torch
from confuciustts.cli.inference import ConfuciusTTS # type: ignore[import-not-found]
device = "cuda" if torch.cuda.is_available() else "cpu"
try:
# Existing manually provisioned venvs may predate torch.accelerator.
current_accelerator = getattr(getattr(torch, "accelerator", None), "current_accelerator", None)
if current_accelerator is None:
device = torch.device("cuda") if torch.cuda.is_available() else None
else:
device = current_accelerator(check_available=True)
device = device.type if device is not None else "cpu" # 'cuda', 'npu', 'mps', 'xpu', 'cpu'
except Exception:
device = "cpu" # Broken accelerator drivers must not block CPU loading.
if device == "mps":
device = "cpu" # MPS was slower than CPU in the existing validation run
_send(stdout, {"op": "progress", "stage": "loading_model", "percent": 50})
_model = ConfuciusTTS(config_path=_config_path(), device=device)
+6 -1
View File
@@ -115,7 +115,12 @@ def _load_runtime(stdout):
from dots_tts.runtime import DotsTtsRuntime # type: ignore[import-not-found]
repo = os.environ.get("OMNIVOICE_DOTS_TTS_MODEL", _DEFAULT_REPO)
default_precision = "bfloat16" if torch.cuda.is_available() else "float32"
# Match DotsTtsRuntime's own CUDA/CPU selection. Its _check_torch_env
# rejects half precision without CUDA, even when an XPU/NPU is available.
try:
default_precision = "bfloat16" if torch.cuda.is_available() else "float32"
except Exception:
default_precision = "float32" # Probe failure must not force half precision.
precision = os.environ.get("OMNIVOICE_DOTS_TTS_PRECISION", default_precision)
optimize = os.environ.get("OMNIVOICE_DOTS_TTS_OPTIMIZE", "0") == "1"
+18
View File
@@ -28,6 +28,7 @@ packages. The parent only ever spawns it as a subprocess.
from __future__ import annotations
import logging
import math
import os
import re
from typing import TYPE_CHECKING
@@ -164,6 +165,23 @@ class IndexTTS2Backend(SubprocessBackend):
from engines.indextts.bootstrap import resolve_indextts_venv
return resolve_indextts_venv()
@property
def recv_timeout_s(self) -> float:
# IndexTTS was the only sidecar left on the 60s class default while
# pockettts and omnivoice-subprocess both raised theirs. infer() is one
# blocking upstream call, so a long passage legitimately outruns 60s and
# the parent's watchdog killed a healthy synthesis (#1611). main.py also
# heartbeats during infer(), which is what actually proves liveness —
# this deadline is the ceiling for a sidecar that has gone genuinely
# silent. OMNIVOICE_INDEXTTS_RECV_TIMEOUT_S tunes it.
try:
v = float(os.environ.get("OMNIVOICE_INDEXTTS_RECV_TIMEOUT_S", "900"))
except (ValueError, TypeError):
return 900.0
if not math.isfinite(v): # reject inf/nan so the deadline can't be disabled
return 900.0
return max(30.0, v)
@classmethod
def sidecar_script(cls):
from engines.indextts.bootstrap import INDEXTTS_SIDECAR_SCRIPT
+87 -7
View File
@@ -63,11 +63,13 @@ Restrictions:
from __future__ import annotations
import base64
import contextlib
import json
import os
import struct
import sys
import tempfile
import threading
import traceback
@@ -117,11 +119,59 @@ EMOTION_KWARGS_ALLOWLIST = frozenset({
# ── wire protocol ─────────────────────────────────────────────────────────
#: Seconds between keep-alive progress frames during a long blocking call.
_HEARTBEAT_S = 5.0
#: Serializes _send across threads (the heartbeat below + the main loop) so
#: concurrent length+body writes can't interleave and corrupt the framing.
_send_lock = threading.Lock()
def _send(stream, obj: dict) -> None:
body = json.dumps(obj, separators=(",", ":")).encode("utf-8")
stream.write(struct.pack("!I", len(body)))
stream.write(body)
stream.flush()
with _send_lock:
stream.write(struct.pack("!I", len(body)))
stream.write(body)
stream.flush()
@contextlib.contextmanager
def _heartbeat(stdout, stage: str):
"""Emit a progress frame every ~5s for the duration of the block.
IndexTTS spends the whole of a cold load and the whole of ``infer()``
inside one blocking upstream call, saying nothing on the wire. The parent
reads that silence two ways, and BOTH kill a perfectly healthy synthesis
of a long passage (#1611):
* ``SubprocessBackend.generate`` re-arms its recv watchdog on every
frame, so with no frames it hard-kills the sidecar at recv_timeout_s;
* each frame also reports activity to the GPU pool's execution clock
(#1367), so with no frames the outer generate budget expires and
blames the hardware.
Raising the deadline alone therefore does not fix long-text generation
the sidecar has to prove it is alive. Percent climbs 1..99 because the
upstream call exposes no real progress; it is a liveness signal, not a
measurement.
"""
stop = threading.Event()
def _beat() -> None:
pct = 1
while not stop.wait(_HEARTBEAT_S):
pct = min(pct + 1, 99)
try:
_send(stdout, {"op": "progress", "stage": stage, "percent": pct})
except Exception:
return # pipe gone — the main loop will surface it
hb = threading.Thread(target=_beat, name=f"indextts-{stage}-heartbeat", daemon=True)
hb.start()
try:
yield
finally:
stop.set()
hb.join(timeout=_HEARTBEAT_S + 1)
def _recv(stream):
@@ -160,14 +210,40 @@ def _torch_bf16_supported() -> bool:
return False
#: Model-config filenames to look for, most-preferred first, per version.
#: IndexTeam/IndexTTS-2.5 ships ``config.yaml``; VoiceStudio used to demand
#: ``config_v2_5.yaml``, a name that exists in no upstream revision, so the
#: install failed until the user hand-renamed the file (#1611). Both names are
#: accepted now — the hand-renamed installs must keep working untouched — and
#: the renamed one wins, because a user who created it did so deliberately.
_CFG_NAMES = {
"2.5": ("config_v2_5.yaml", "config.yaml"),
"2": ("config.yaml",),
}
def _resolve_cfg_path(model_dir: str, *, version: str) -> str:
"""First accepted config that exists in ``model_dir``.
Falls back to the last candidate when none exist, so the failure surfaces
as upstream's own "no such file" naming a real expected path rather than
a name no upstream release has ever shipped.
"""
names = _CFG_NAMES.get(version, _CFG_NAMES["2"])
for name in names:
candidate = os.path.join(model_dir, name)
if os.path.isfile(candidate):
return candidate
return os.path.join(model_dir, names[-1])
def _model_init_kwargs(
repo_dir: str, *, version: str, reduced_precision: bool,
) -> dict:
"""Build version-specific constructor arguments for IndexTTS 2.5 or 2."""
model_dir = os.path.join(repo_dir, "checkpoints")
cfg_name = "config_v2_5.yaml" if version == "2.5" else "config.yaml"
kwargs = {
"cfg_path": os.path.join(model_dir, cfg_name),
"cfg_path": _resolve_cfg_path(model_dir, version=version),
"model_dir": model_dir,
"use_cuda_kernel": False,
"use_deepspeed": False,
@@ -216,7 +292,8 @@ def _load_model(stdout) -> object:
model_kw = _model_init_kwargs(
repo_dir, version=_model_version, reduced_precision=reduced_precision,
)
_model = IndexTTS2(**model_kw)
with _heartbeat(stdout, "loading_model"):
_model = IndexTTS2(**model_kw)
_send(stdout, {"op": "progress", "stage": "loading_model", "percent": 100})
return _model
@@ -276,7 +353,10 @@ def _handle_synthesize(msg: dict, stdout) -> None:
tmp_path = tmp.name
try:
infer_kw["output_path"] = tmp_path
model.infer(**infer_kw)
# A long passage keeps infer() busy for minutes with nothing on the
# wire; without this the parent kills the sidecar mid-synthesis (#1611).
with _heartbeat(stdout, "synthesizing"):
model.infer(**infer_kw)
pcm_b64, sr, n_samples = _wav_to_pcm_b64(tmp_path)
finally:
try:
+11 -15
View File
@@ -29,15 +29,11 @@ Do NOT import ``main.py`` from the parent process — it runs under a
different venv (``transformers==5.0.0``) and importing it in-process would
re-introduce the exact conflict this isolation exists to avoid.
Hardware honesty (cross-platform rule): MOSS-TTS-v1.5's upstream documents
only CUDA and CPU. There is **no documented or tested MPS path** the
custom ``trust_remote_code`` modelling code and the separate audio
tokenizer are unverified on Apple Silicon. We therefore advertise
``gpu_compat = ("cuda", "cpu")`` and the sidecar selects ``cuda`` when
present else ``cpu`` it never silently routes to MPS where it might
crash. On Apple Silicon the engine honestly resolves to CPU (slow but
correct), and the engine is opt-in regardless, so it never becomes a
broken default on any platform.
Hardware routing follows the sidecar's runtime-available PyTorch accelerator:
CUDA/ROCm, XPU, or a registered NPU. MPS remains excluded; CPU is the fallback.
XPU/NPU routing is covered with mocked device contracts, not physical-hardware
synthesis certification; users need a compatible torch/vendor runtime in the
isolated engine venv.
"""
from __future__ import annotations
@@ -85,13 +81,13 @@ class MossTTSV15Backend(SubprocessBackend):
id = "moss-tts-v15"
display_name = (
"MOSS-TTS-v1.5 (8B, 31 langs, zero-shot clone, CUDA/CPU, Apache-2.0)"
"MOSS-TTS-v1.5 (8B, 31 langs, zero-shot clone, Apache-2.0)"
)
supports_voice_design = False # requires ref audio for timbre cloning
_DEFAULT_SAMPLE_RATE = 24000
# Honest hardware surface: upstream documents CUDA + CPU only. MPS is
# undocumented / untested, so we do NOT claim it (cross-platform rule).
gpu_compat = ("cuda", "cpu")
# Accelerator routing requires its matching runtime in the isolated venv.
# MPS remains untested and is deliberately excluded.
gpu_compat = ("cuda", "rocm", "xpu", "npu", "cpu")
# ── availability ───────────────────────────────────────────────────────
@@ -111,7 +107,7 @@ class MossTTSV15Backend(SubprocessBackend):
return False, (
"MOSS-TTS-v1.5 venv not found. Set OMNIVOICE_MOSS_TTS_V15_DIR "
"to your MOSS-TTS clone (the directory containing pyproject.toml) "
"and restart VoiceStudio. CUDA or CPU only (no MPS). See "
"and restart VoiceStudio. Install the matching PyTorch runtime. See "
"docs/engines/moss-tts-v15.md for the full install walk-through."
)
if not MOSS_TTS_V15_SIDECAR_SCRIPT.exists():
@@ -119,7 +115,7 @@ class MossTTSV15Backend(SubprocessBackend):
"MOSS-TTS-v1.5 sidecar script missing at "
f"{MOSS_TTS_V15_SIDECAR_SCRIPT} — reinstall VoiceStudio."
)
return True, "ok (CUDA when present, else CPU)"
return True, "ok (runtime-available accelerator or CPU; no MPS)"
@classmethod
def venv_python(cls):
+22 -7
View File
@@ -138,11 +138,12 @@ _state = None
def _load_model(stdout):
"""Cold-construct the MOSS-TTS-v1.5 processor + model.
Device selection is CUDA-or-CPU only MOSS's upstream documents no MPS
path and the custom ``trust_remote_code`` modelling code is untested on
Apple Silicon, so we never route to MPS where it might crash. dtype is
bf16 on CUDA, fp32 on CPU (bf16 CPU ops are spotty). Emits progress
frames so the parent can surface the multi-GB cold-load latency.
Device selection uses the torch.accelerator API to support any backend
(CUDA, NPU, XPU, etc.) automatically. MPS is excluded MOSS's upstream
``trust_remote_code`` modelling code is untested on Apple Silicon. dtype is
bf16 on GPU-class accelerators, fp32 on CPU (bf16 CPU ops are spotty).
Emits progress frames so the parent can surface the multi-GB cold-load
latency.
"""
global _state
if _state is not None:
@@ -154,8 +155,22 @@ def _load_model(stdout):
from transformers import AutoModel, AutoProcessor
repo, revision = _model_source()
device = "cuda" if torch.cuda.is_available() else "cpu"
dtype = torch.bfloat16 if device == "cuda" else torch.float32
# current_accelerator() returns None on CPU-only builds (no accelerator
# compiled in) or when no accelerator is available; fall back to "cpu".
# Existing manually provisioned venvs may predate torch.accelerator.
current_accelerator = getattr(getattr(torch, "accelerator", None), "current_accelerator", None)
try:
if current_accelerator is None:
accel = torch.device("cuda") if torch.cuda.is_available() else None
else:
accel = current_accelerator(check_available=True)
except Exception:
# Optional drivers can fail during probing; CPU loading remains usable.
accel = None
device = accel.type if accel is not None else "cpu" # 'cuda', 'npu', 'mps', 'xpu', 'cpu'
if device == "mps":
device = "cpu" # MOSS is untested on MPS; fall back to CPU for safety
dtype = torch.bfloat16 if device != "cpu" else torch.float32
# "sdpa" works on CUDA + CPU and needs no extra dep. flash_attention_2
# (Ampere+ CUDA, optional flash-attn) is opt-in via env.
attn = os.environ.get("OMNIVOICE_MOSS_TTS_V15_ATTN", "sdpa")
@@ -98,6 +98,7 @@ def _platform_slug() -> str:
darwin-x86_64
windows-x86_64
linux-x86_64
linux-aarch64
"""
system = platform.system().lower()
machine = platform.machine().lower()
@@ -107,6 +108,8 @@ def _platform_slug() -> str:
return "darwin-x86_64"
if system == "windows":
return "windows-x86_64"
if system == "linux" and machine in ("arm64", "aarch64"):
return "linux-aarch64"
# Linux + everything else falls into the linux slug.
return "linux-x86_64"
@@ -1,7 +1,9 @@
"""omnivoice-subprocess: the resident OmniVoice TTS engine in a crash-isolated
sidecar process (#730/#1190).
The default ``omnivoice`` engine runs in-process on the GPU ``ThreadPoolExecutor``.
The ``omnivoice`` engine runs in-process on CUDA, ROCm, and CPU. On MPS it is
resolved to :class:`OmniVoiceMPSSubprocessBackend` so a fatal native allocator
exit cannot take down the local API process.
When a generate or load there exceeds its execution budget the pool is "reset"
but the abandoned worker *thread* cannot be killed (Python cannot interrupt a
native torch/MPS call), so it holds the MPS device until it finishes on its
@@ -13,16 +15,11 @@ timeout the parent's watchdog calls ``proc.kill()``, reclaiming the child's
VRAM/device, and the next request transparently respawns a fresh sidecar. That
is the one thing the in-process engine structurally cannot do.
OPT-IN (Settings -> Engines, or ``OMNIVOICE_TTS_BACKEND=omnivoice-subprocess``);
the in-process ``omnivoice`` stays the default so existing users see no change.
The explicit ``omnivoice-subprocess`` id remains available on every host for
operators who want the same containment elsewhere.
Tradeoff vs the in-process engine: identical model and quality, a little extra
per-call overhead (one stdio round-trip), and it does not carry the native
advanced-parameter surface (``t_shift`` / ``layer_penalty_factor`` /
``position_temperature`` / ``class_temperature``) or parent-side seed
determinism, because the generic ``backend.generate`` path does not forward
those. Acceptable for unattended / reaction-triggered use where reliability
matters more than those controls.
Tradeoff vs the in-process engine: identical model, controls, seed behavior,
and quality, with a little extra per-call overhead (one stdio round-trip).
Unlike IndexTTS / dots.tts / Supertonic-3, this sidecar runs under the PARENT
interpreter (``venv_python() -> sys.executable``): the goal here is crash
@@ -51,10 +48,15 @@ class OmniVoiceSubprocessBackend(SubprocessBackend):
id = "omnivoice-subprocess"
display_name = "OmniVoice (subprocess-isolated, killable on timeout)"
_DEFAULT_SAMPLE_RATE = 24000
gpu_compat = ("cuda", "mps", "cpu")
gpu_compat = ("cuda", "rocm", "mps", "cpu")
# Match OmniVoiceBackend: the measured floor below which a render that
# should take seconds runs for minutes (the #1226/#1222 4 GB reports).
min_vram_gb = 6.0
# Packaged Windows hosts can spend more than the base 30 seconds starting
# the shared Python runtime before this stdlib-only sidecar emits ready.
# Keep the bound below the 300-second generation budget while avoiding the
# repeated false kill captured in #1711.
spawn_ready_timeout_s = 120.0
@classmethod
def is_available(cls) -> tuple[bool, str]:
@@ -102,4 +104,34 @@ class OmniVoiceSubprocessBackend(SubprocessBackend):
return ["multi"]
__all__ = ["OmniVoiceSubprocessBackend"]
class OmniVoiceMPSSubprocessBackend(OmniVoiceSubprocessBackend):
"""Effective ``omnivoice`` implementation on MPS.
Native torch/MPS allocator failures can terminate the process without a
catchable Python exception. Keeping the same engine id and model surface in
a child makes that failure recoverable while Settings, APIs, and saved
projects continue to refer to ``omnivoice``.
"""
id = "omnivoice"
display_name = "VoiceStudio (k2-fsa/OmniVoice, 600+ languages)"
supports_native_omnivoice_controls = True
def generate(self, text: str, **kw):
from services.model_manager import make_room_before_generate
make_room_before_generate()
try:
return super().generate(text, **kw)
except RuntimeError as exc:
if "sidecar closed pipe mid-generate" not in str(exc):
raise
raise RuntimeError(
"The isolated OmniVoice engine stopped during generation, "
"usually because macOS reclaimed it under memory pressure. "
"The VoiceStudio backend is still running. Close memory-heavy "
"apps or select a smaller TTS engine, then retry."
) from exc
__all__ = ["OmniVoiceMPSSubprocessBackend", "OmniVoiceSubprocessBackend"]
@@ -50,6 +50,8 @@ OMNIVOICE_SAMPLE_RATE = 24000
_GEN_KW_ALLOWLIST = (
"language", "instruct", "duration", "num_step", "guidance_scale",
"speed", "denoise", "postprocess_output", "preprocess_prompt",
"t_shift", "layer_penalty_factor", "position_temperature",
"class_temperature", "audio_chunk_duration", "audio_chunk_threshold",
)
_model = None
@@ -183,6 +185,12 @@ def _handle_synthesize(msg: dict, stdout) -> None:
ref_text = msg.get("ref_text") or None
gen_kw = {k: msg[k] for k in _GEN_KW_ALLOWLIST if k in msg}
seed = msg.get("seed")
if seed is not None:
import torch
torch.manual_seed(int(seed))
audios = model.generate(
text=text, ref_audio=ref_audio, ref_text=ref_text, **gen_kw
)
+35 -1
View File
@@ -151,6 +151,40 @@ def _pocket_language(raw) -> str:
)
_TRUTHY = {"1", "true", "yes", "on"}
def _has_24l_config(language: str) -> bool:
"""Whether the installed pocket-tts ships a 24-layer checkpoint for
``language`` (it/de/es/pt/fr in 2.1.0; english has none)."""
try:
from pocket_tts.models.tts_model import CONFIGS_DIR # type: ignore[import-not-found] # noqa: PLC0415
except Exception as exc: # noqa: BLE001 — absence of the package is not fatal here
# Log it, though: if a future pocket-tts moves CONFIGS_DIR, the 24L
# opt-in would otherwise go silently inert.
print(f"pockettts sidecar: 24l config probe failed: {exc!r}", file=sys.stderr)
return False
from pathlib import Path # noqa: PLC0415
return (Path(CONFIGS_DIR) / f"{language}_24l.yaml").is_file()
def _model_config_name(language: str) -> str:
"""Pocket-tts config name to load: the 6-layer default, or the 24-layer
checkpoint when OMNIVOICE_POCKETTTS_24L is set and one exists for the
language. Opt-in only defaults keep the fast model; the 24-layer variant
trades roughly 4x transformer compute for better prosody.
French is the exception: pocket-tts 2.1.0 only ships a 24-layer French
model and load_model(language="french") raises, so French always maps to
french_24l regardless of the env var."""
if language == "french":
return "french_24l"
if os.environ.get("OMNIVOICE_POCKETTTS_24L", "").strip().lower() not in _TRUTHY:
return language
return f"{language}_24l" if _has_24l_config(language) else language
def _load_model(stdout, language: str):
"""Cold-construct the PocketTTS model for ``language`` (cached per language).
Emits progress frames for the parent watchdog. Raises on failure (e.g.
@@ -178,7 +212,7 @@ def _load_model(stdout, language: str):
try:
from pocket_tts import TTSModel # type: ignore[import-not-found] # noqa: PLC0415
model = TTSModel.load_model(language=language)
model = TTSModel.load_model(language=_model_config_name(language))
_MODELS[language] = model
finally:
stop.set()
+122 -24
View File
@@ -9,6 +9,24 @@ _backend_dir = os.path.dirname(os.path.abspath(__file__))
if _backend_dir not in sys.path:
sys.path.insert(0, _backend_dir)
# PyInstaller re-executes this entry module when the frozen backend binary is
# launched. Nested operation supervisors therefore dispatch here, before math,
# logging, FastAPI, torch, or any application initialization. Source launches
# use this same entry contract so frozen/source behavior cannot drift.
if __name__ == "__main__" and len(sys.argv) > 1 and sys.argv[1] == "--supervise":
from core.contained_subprocess import supervisor_main
raise SystemExit(supervisor_main(sys.argv[1:]))
# Rust clears CLOEXEC only for the backend exec. Re-arm PEP 446 immediately:
# nested supervisors receive this descriptor solely through explicit pass_fds,
# so a third-party close_fds=False child cannot hold the desktop drain barrier.
from core.contained_subprocess import secure_backend_drain_fd # noqa: E402
secure_backend_drain_fd()
import math # noqa: E402
# Windows: run every child process (ffmpeg, engine sidecars, yt-dlp, demucs, …)
# WITHOUT popping a console window. The backend itself is spawned console-less by
# the Tauri shell, so on Windows each console subprocess it launches would
@@ -59,6 +77,7 @@ os.environ.setdefault("FOR_DISABLE_CONSOLE_CTRL_HANDLER", "1")
# (utils.hf_progress.SafeFileWrapper — same wrapper the patched hub tqdm
# already uses for its own fp.)
from utils.hf_progress import SafeFileWrapper as _SafeStdio # noqa: E402
from core.parent_liveness import arm_desktop_parent_watchdog # noqa: E402
# Force UTF-8 stdio before wrapping (#1155): on Windows the spawned backend's
# stdout defaults to cp1252, and any library that prints user text (kittentts
@@ -71,6 +90,11 @@ for _stream in (sys.stdout, sys.stderr):
except Exception: # noqa: BLE001 — pythonw/frozen builds may lack reconfigure
pass
# The desktop keeps the backend's stdin pipe open for its own lifetime. EOF is
# therefore a stable ownership signal that survives PID reuse and lets a child
# terminate even when the shell crashes before its normal process-tree teardown.
arm_desktop_parent_watchdog()
if not getattr(sys.stdout, "_is_safe_wrapper", False):
sys.stdout = _SafeStdio(sys.stdout)
if not getattr(sys.stderr, "_is_safe_wrapper", False):
@@ -369,19 +393,36 @@ def _env_flag(name: str, default: bool = False) -> bool:
_EAGER = _env_flag("OMNIVOICE_EAGER_INIT", default=("pytest" in sys.modules))
def _env_float(name: str, default: float) -> float:
"""Parse a float env override, rejecting negative and non-finite values.
Shared by the preload-delay / timeout knobs: NaN would silently never
fire, a negative would fire during startup I/O, so both fall back to the
default instead (the bug class CodeRabbit flagged on the watermark knob
in PR #1577 — latent in the older copies too, closed here for all)."""
raw = os.environ.get(name, "")
try:
value = float(raw) if raw.strip() else default
except ValueError:
return default
return value if math.isfinite(value) and value >= 0 else default
def _capture_preload_delay_s() -> float:
"""Seconds after boot before the dictation (capture ASR) model warms.
Late enough that it never competes with startup I/O or the TTS preload;
overridable via OMNIVOICE_CAPTURE_PRELOAD_DELAY (mostly for tests)."""
raw = os.environ.get("OMNIVOICE_CAPTURE_PRELOAD_DELAY", "")
try:
v = float(raw)
if v >= 0:
return v
except (TypeError, ValueError):
pass
return 30.0
return _env_float("OMNIVOICE_CAPTURE_PRELOAD_DELAY", 30.0)
def _watermark_preload_delay_s() -> float:
"""Seconds after boot before the AudioSeal generator warm-up fires.
Own knob, NOT ``_capture_preload_delay_s`` + offset: a capture-specific
env override must not retime the watermark warm too, and the two cold
imports shouldn't fire on the same tick (CodeRabbit, PR #1577). Default
35s sits ~5s past the capture-ASR warm for the same reason."""
return _env_float("OMNIVOICE_PRELOAD_WATERMARK_DELAY", 35.0)
def _capture_preload_ram_ok(min_free_bytes: int = 4 * 1024**3) -> bool:
@@ -398,14 +439,7 @@ def _capture_preload_ram_ok(min_free_bytes: int = 4 * 1024**3) -> bool:
def _mcp_start_timeout_s() -> float:
"""Seconds to wait for the MCP session manager to start before giving up
and serving without it (#632). Overridable via OMNIVOICE_MCP_START_TIMEOUT_S."""
raw = os.environ.get("OMNIVOICE_MCP_START_TIMEOUT_S", "")
try:
v = float(raw)
if v > 0:
return v
except (TypeError, ValueError):
pass
return 30.0
return max(_env_float("OMNIVOICE_MCP_START_TIMEOUT_S", 30.0), 0.001)
async def _serve_mcp(session_manager, ready: "asyncio.Event", stop: "asyncio.Event") -> None:
@@ -551,13 +585,14 @@ def _phase_a_build_inner() -> None:
pass # never block startup on the migration; it retries next launch
# Restore persisted env vars from prefs.json (Settings UI writes them
# there so they survive backend restarts) — before any user code reads
# os.environ, and never overriding an explicitly-set env var.
# os.environ, and never overriding an explicitly-set env var. Also
# snapshots which keys an external source (shell, `.env`, Docker, …)
# already provided, so a Settings control can tell the user their saved
# value is being shadowed instead of silently promising it will apply
# (core.prefs.is_env_shadowed — #1787 review fix).
try:
from core.prefs import _load as _load_all_prefs
_prefs = _load_all_prefs()
for _k, _v in _prefs.items():
if _k.startswith("env.") and _v:
os.environ.setdefault(_k[len("env."):], str(_v))
from core.prefs import _load as _load_all_prefs, restore_env
restore_env(_load_all_prefs())
except Exception:
pass # prefs.json missing or broken — fine on first run
# yt-dlp user-update overlay: must run before anything imports yt_dlp so
@@ -638,6 +673,7 @@ def _phase_a_build_inner() -> None:
events,
capture,
capture_ws,
speech_platform,
dictation,
openai_compat,
tts_stream,
@@ -650,14 +686,15 @@ def _phase_a_build_inner() -> None:
settings as settings_router, # Phase 1 AUTH-03: HF token save/clear/state
media_tools as media_tools_router, # Audio tools: ffmpeg/ffprobe/yt-dlp
auth as auth_router,
voice_convert, # Studio Convert: speech-to-speech via ASR → TTS
)
from api.routers import mcp_bindings as _mcp_bindings_router # noqa: E402
from api.routers import workers as workers_router # noqa: E402
_router_modules.extend([
system, profiles, exports, generation, dub_core, dub_generate,
system, profiles, exports, generation, voice_convert, dub_core, dub_generate,
dub_export, dub_translate, projects, glossary, engines, tools,
stories, setup, gallery, archetypes, describe_voice, community,
batch, watermark, events, capture, capture_ws, dictation,
batch, watermark, events, capture, capture_ws, speech_platform, dictation,
openai_compat, tts_stream, marketplace, personas, sonitranslate,
audiobook, longform_jobs, pronunciation, settings_router,
media_tools_router, auth_router, _mcp_bindings_router, workers_router,
@@ -852,6 +889,8 @@ async def _phase_b(app: FastAPI) -> None:
# #1174: arm model loads for THIS run — an in-process relaunch may carry a
# stale shutting-down flag from a previous lifespan.
model_loads_reset_shutdown()
from services.model_manager import begin_watermark_pool_lifecycle
begin_watermark_pool_lifecycle()
app.state.idle_task = asyncio.create_task(idle_worker())
app.state.worker_task = asyncio.create_task(task_manager.worker())
# Warm the TTS model in the background so first /generate is instant.
@@ -905,6 +944,50 @@ async def _phase_b(app: FastAPI) -> None:
else:
logger.info("Capture ASR preload disabled; dictation ASR will load on first use.")
# Watermark: warm the AudioSeal generator in the background so the first
# mark_synthetic doesn't serialize the audioseal import + model load
# inside the first synthesis (measured ~42 s inline on a cold filesystem,
# 2026-08-17 macOS report — 3 s short of the client's 90 s timeout).
# Small model on CPU; deferred a few seconds past the capture-ASR warm so
# the two cold imports don't contend for the same disk, and no RAM guard
# is needed. Runs on the watermark pool — where the model is used — not
# the shared default executor.
if _env_flag("OMNIVOICE_PRELOAD_WATERMARK", default=True):
async def _preload_watermark():
await asyncio.sleep(_watermark_preload_delay_s())
loop = asyncio.get_running_loop()
from services import watermark as _watermark
# Gate BEFORE touching get_watermark_pool(): the pool is lazy so
# hosts with watermarking disabled never spawn its thread, and
# creating it unconditionally would break that invariant. The
# race with a first embed is benign — pool creation is itself
# lock-guarded.
if not _watermark.will_mark():
logger.debug("Watermark preload skipped (disabled or audioseal absent)")
return
from services.model_manager import get_watermark_pool
# Default startup may warm an existing local checkpoint but may
# not fetch one. Only an explicit user opt-in permits a download.
raw_preload = os.environ.get("OMNIVOICE_PRELOAD_WATERMARK", "")
allow_download = raw_preload.strip().lower() in {"1", "true", "yes", "on"}
try:
await loop.run_in_executor(
get_watermark_pool(),
lambda: _watermark.prefetch_generator(
allow_download=allow_download
),
)
except Exception:
# prefetch_generator swallows its own errors; this guards the
# setup half (imports, pool construction) so a broken warm-up
# is visible now, not as an unretrieved exception at shutdown.
logger.warning("Watermark preload task failed", exc_info=True)
app.state.watermark_preload_task = asyncio.create_task(_preload_watermark())
# ── MCP session manager (Wave 2.2) ────────────────────────────────────
# Run it in its OWN task owning the full enter→exit lifecycle (anyio
# task-affinity, see _serve_mcp); only wait, with a timeout, for ready —
@@ -934,6 +1017,9 @@ async def _phase_b(app: FastAPI) -> None:
@asynccontextmanager
async def lifespan(app: FastAPI):
from api.dependencies import validate_server_admin_key
validate_server_admin_key()
# Startup watchdog (#632): a silent hang during startup (e.g. a model-load /
# MCP deadlock on some platforms) means "Application startup complete" never
# logs and the app sits forever with no error. If startup hasn't finished
@@ -1080,8 +1166,20 @@ async def lifespan(app: FastAPI):
getattr(app.state, "worker_task", None),
getattr(app.state, "preload_task", None),
getattr(app.state, "capture_preload_task", None),
getattr(app.state, "watermark_preload_task", None),
timeout=20.0,
)
# The watermark warm-up runs on its dedicated 1-worker pool. Cancellation
# detaches the asyncio future but cannot kill a thread inside AudioSeal,
# so drain it fully before lifespan teardown reports completion.
try:
from services.model_manager import shutdown_watermark_pool as _wm_drain
_wm_drain()
except Exception:
# Best-effort drain: a failure here must not abort the remaining
# shutdown steps (model unload, MCP teardown) below.
logger.warning("Watermark pool drain failed at shutdown", exc_info=True)
# Unload the model and free GPU memory
try:
import services.model_manager as mm
+297 -37
View File
@@ -7,8 +7,8 @@ Run standalone:
Tools exposed:
generate_speech text WAV audio (voice clone or design)
clone_voice base64 reference audio new voice profile
transcribe base64 audio text
clone_voice reference audio (base64, or a file path) new voice profile
transcribe audio (base64, or a file path) text
list_voices enumerate saved voice profiles
list_languages available TTS languages
list_personalities voice personality presets
@@ -17,6 +17,18 @@ Tools exposed:
Resources exposed:
voice://{profile_id} voice profile metadata
history://recent last 20 generated audio items
Output mode (OMNIVOICE_MCP_OUTPUT_MODE):
resources generate_speech returns the WAV as base64 inline (the original
contract; default)
files it returns a URL to the render (and, with a base path, a WAV
written there); nothing large ever enters the agent's context
both both of the above
File inputs (OMNIVOICE_MCP_BASE_PATH):
One directory that agents may read audio from (transcribe / clone_voice
`*_path` arguments) and receive files in (files mode). It is the security
boundary: with no base path configured, path-shaped inputs are refused.
"""
from __future__ import annotations
@@ -25,6 +37,8 @@ import base64
import json
import logging
import os
import re
import stat
import sys
logger = logging.getLogger("omnivoice.mcp")
@@ -69,6 +83,244 @@ def _sniff_audio_ext(raw: bytes) -> str:
return ".wav"
# ── Output mode + the base path boundary ─────────────────────────────────
# An LLM agent that receives a WAV as base64 pays for every byte in context:
# a 1.4 s clip already brushes per-result token caps, and a paragraph of
# narration blows them outright. The ElevenLabs MCP settled this with an
# OUTPUT_MODE (files / resources / both) and a BASE_PATH that doubles as the
# security boundary for file-shaped inputs; the same two knobs here, named in
# the OMNIVOICE_* family the rest of the server reads.
_OUTPUT_MODES = ("resources", "files", "both")
_MAX_INPUT_BYTES = 200 * 1024 * 1024
_SAFE_AUDIO_ID = re.compile(r"^[A-Za-z0-9_-]{1,64}$")
def _output_mode() -> str:
"""How generate_speech hands audio back (OMNIVOICE_MCP_OUTPUT_MODE).
'resources' is the original base64-inline contract and stays the default
so existing integrations see no change; 'files' returns a URL to the
render (plus a WAV under the base path when one is configured); 'both'
returns everything. Anything unrecognized falls back to 'resources' with
a warning rather than failing the tool."""
mode = os.environ.get("OMNIVOICE_MCP_OUTPUT_MODE", "resources").strip().lower()
if mode not in _OUTPUT_MODES:
logger.warning(
"OMNIVOICE_MCP_OUTPUT_MODE=%r is not one of %s; using 'resources'",
mode, _OUTPUT_MODES,
)
return "resources"
return mode
def _base_path() -> "str | None":
"""The one directory agents may read audio from and receive files in
(OMNIVOICE_MCP_BASE_PATH), realpath'd; None when unset."""
raw = os.environ.get("OMNIVOICE_MCP_BASE_PATH", "").strip()
if not raw:
return None
return os.path.realpath(os.path.expanduser(raw))
def _resolve_under_base(path: str) -> str:
"""Absolute realpath of ``path`` when it lies inside the base path.
Relative paths resolve against the base; absolute paths must already be
inside it. Both sides are realpath'd, so a symlink pointing outward cannot
smuggle a read in. Raises ValueError with an agent-legible reason when no
base path is configured or the path escapes it."""
base = _base_path()
if base is None:
raise ValueError(
"OMNIVOICE_MCP_BASE_PATH is not set; file paths are refused until it "
"names a directory"
)
candidate = os.path.realpath(os.path.join(base, os.path.expanduser(path)))
if not _path_is_under_base(base, candidate):
raise ValueError(f"{path!r} resolves outside OMNIVOICE_MCP_BASE_PATH")
return candidate
def _opened_file_is_confined(fd: int, resolved: str, base: str) -> bool:
"""Verify that an opened descriptor still names a file under ``base``."""
proc_fd = f"/proc/self/fd/{fd}"
if os.path.exists(proc_fd):
return _path_is_under_base(base, os.path.realpath(proc_fd))
try:
current = os.path.realpath(resolved)
return _path_is_under_base(base, current) and os.path.samestat(
os.fstat(fd), os.stat(current, follow_symlinks=False)
)
except OSError:
return False
def _path_is_under_base(base: str, candidate: str) -> bool:
try:
common = os.path.commonpath([base, candidate])
except ValueError: # different drives on Windows
return False
return os.path.normcase(common) == os.path.normcase(base)
def _open_under_base(path: str, flags: int, *, mode: int = 0o600) -> tuple[int, str]:
"""Open ``path`` without following a component replaced after validation."""
base = _base_path()
if base is None:
raise ValueError(
"OMNIVOICE_MCP_BASE_PATH is not set; file paths are refused until it "
"names a directory"
)
resolved = _resolve_under_base(path)
relative = os.path.relpath(resolved, base)
parts = [part for part in relative.split(os.sep) if part not in ("", ".")]
if not parts or parts[0] == os.pardir:
raise ValueError(f"{path!r} resolves outside OMNIVOICE_MCP_BASE_PATH")
no_follow = getattr(os, "O_NOFOLLOW", 0)
close_on_exec = getattr(os, "O_CLOEXEC", 0)
binary = getattr(os, "O_BINARY", 0)
file_flags = flags | no_follow | close_on_exec | binary
supports_dir_fd = os.open in getattr(os, "supports_dir_fd", ())
directory_flag = getattr(os, "O_DIRECTORY", 0)
if supports_dir_fd and directory_flag:
directory_flags = os.O_RDONLY | directory_flag | no_follow | close_on_exec
directory_fd = os.open(base, directory_flags)
try:
for component in parts[:-1]:
next_fd = os.open(component, directory_flags, dir_fd=directory_fd)
os.close(directory_fd)
directory_fd = next_fd
fd = os.open(parts[-1], file_flags, mode, dir_fd=directory_fd)
finally:
os.close(directory_fd)
else:
fd = os.open(resolved, file_flags, mode)
if not _opened_file_is_confined(fd, resolved, base):
os.close(fd)
raise ValueError(f"{path!r} resolves outside OMNIVOICE_MCP_BASE_PATH")
return fd, resolved
def _read_input_audio(
audio_base64: "str | None",
audio_path: "str | None",
*,
label: str = "audio_base64",
too_big: str = "audio exceeds 200 MB limit",
) -> "tuple[bytes | None, str | None]":
"""Audio bytes from exactly one of the two input lanes, or (None, error).
The base64 lane keeps its data-URI tolerance and 200 MB cap; the path lane
is honored only inside the base path (the security boundary) and applies
the same cap to the file's size before reading it."""
if bool(audio_base64) == bool(audio_path):
return None, f"pass exactly one of {label} or the matching *_path argument"
if audio_path:
try:
fd, _resolved = _open_under_base(audio_path, os.O_RDONLY)
except ValueError as e:
return None, str(e)
except FileNotFoundError:
return None, f"no such file under OMNIVOICE_MCP_BASE_PATH: {audio_path!r}"
except OSError as e:
return None, f"could not safely read {audio_path!r}: {e}"
with os.fdopen(fd, "rb") as handle:
info = os.fstat(handle.fileno())
if not stat.S_ISREG(info.st_mode):
return None, f"{audio_path!r} is not a regular file"
if info.st_size > _MAX_INPUT_BYTES:
return None, too_big
raw = handle.read(_MAX_INPUT_BYTES + 1)
if len(raw) > _MAX_INPUT_BYTES:
return None, too_big
if not raw:
return None, f"{label} is empty"
return raw, None
encoded = (
audio_base64.split(",", 1)[-1]
if audio_base64.startswith("data:")
else audio_base64
)
max_encoded_bytes = 4 * ((_MAX_INPUT_BYTES + 2) // 3)
if len(encoded) > max_encoded_bytes:
return None, too_big
raw = _decode_ref_audio(audio_base64)
if raw is None:
return None, f"{label} is not valid base64"
if not raw:
return None, f"{label} is empty"
if len(raw) > _MAX_INPUT_BYTES:
return None, too_big
return raw, None
def _write_output(audio_id: str, raw: bytes) -> str:
"""Land a render under the base path as ``<audio_id>.wav``; returns the path."""
if not _SAFE_AUDIO_ID.fullmatch(audio_id):
raise ValueError("backend returned an invalid X-Audio-Id header")
base = _base_path()
os.makedirs(base, exist_ok=True)
filename = f"{audio_id}.wav"
fd, path = _open_under_base(filename, os.O_WRONLY | os.O_CREAT | os.O_EXCL)
with os.fdopen(fd, "wb") as handle:
handle.write(raw)
return path
def _post_timeout_s() -> float:
"""Seconds the tools wait on a backend POST (OMNIVOICE_MCP_TIMEOUT_S,
default 120). A CPU host renders a paragraph in minutes and serializes
generations, so an agent behind another render used to hit the fixed
budget with an empty-message timeout; the knob follows the backend's own
OMNIVOICE_GENERATE_TIMEOUT_S when a deployment raises that."""
raw = os.environ.get("OMNIVOICE_MCP_TIMEOUT_S", "").strip()
try:
value = float(raw) if raw else 120.0
except ValueError:
logger.warning("OMNIVOICE_MCP_TIMEOUT_S=%r is not a number; using 120", raw)
return 120.0
return value if value > 0 else 120.0
def _maybe_number(value):
"""A response-header number as a number, or the raw text (e.g. '?')."""
try:
return float(value)
except (TypeError, ValueError):
return value
def _speech_result(audio_id: str, gen_time, duration, raw: bytes, api_base: str) -> dict:
"""The generate_speech reply shaped by the output mode.
The backend already keeps every render on disk and serves it at
``/audio/<audio_id>.wav``, so files mode costs nothing but a URL - plus one
write when a base path invites the WAV into the agent's own directory."""
if not _SAFE_AUDIO_ID.fullmatch(audio_id):
raise ValueError("backend returned an invalid X-Audio-Id header")
mode = _output_mode()
out = {
"audio_id": audio_id,
"generation_time_s": gen_time,
"audio_duration_s": duration,
"format": "wav",
"output_mode": mode,
}
if mode in ("files", "both"):
out["audio_url"] = f"{api_base.rstrip('/')}/audio/{audio_id}.wav"
if _base_path() is not None:
out["output_path"] = _write_output(audio_id, raw)
else:
out["note"] = "set OMNIVOICE_MCP_BASE_PATH to also receive the WAV as a file"
if mode in ("resources", "both"):
out["wav_base64"] = base64.b64encode(raw).decode("ascii")
return out
# ── Lazy imports — keeps startup fast when not using MCP ────────────────
@@ -147,7 +399,7 @@ def create_mcp_server():
async def _api_post_form(path: str, data: dict, files: dict | None = None):
import httpx
async with httpx.AsyncClient(base_url=_api_base(), timeout=120) as c:
async with httpx.AsyncClient(base_url=_api_base(), timeout=_post_timeout_s()) as c:
r = await c.post(path, data=data, files=files or {})
r.raise_for_status()
return r
@@ -190,8 +442,12 @@ def create_mcp_server():
steps: Diffusion steps (8=fast/draft, 16=balanced, 32=quality).
Returns:
JSON with audio_id, generation_time, audio_duration, and
base64-encoded WAV data.
JSON with audio_id, generation_time_s, audio_duration_s and the
audio itself shaped by OMNIVOICE_MCP_OUTPUT_MODE: base64 WAV data
('resources', the default), a URL to the render plus a WAV under
OMNIVOICE_MCP_BASE_PATH when one is set ('files'), or all of the
above ('both'). Prefer 'files' for LLM agents: nothing large
enters the context.
"""
# Per-agent voice binding (Wave 2.2): explicit arg wins; otherwise
# resolve this client's bound profile, then the global default.
@@ -218,18 +474,10 @@ def create_mcp_server():
r = await _api_post_form("/generate", data=form)
audio_id = r.headers.get("X-Audio-Id", "unknown")
gen_time = r.headers.get("X-Gen-Time", "?")
duration = r.headers.get("X-Audio-Duration", "?")
gen_time = _maybe_number(r.headers.get("X-Gen-Time", "?"))
duration = _maybe_number(r.headers.get("X-Audio-Duration", "?"))
wav_b64 = base64.b64encode(r.content).decode("ascii")
return (
f'{{"audio_id":"{audio_id}",'
f'"generation_time_s":{gen_time},'
f'"audio_duration_s":{duration},'
f'"format":"wav",'
f'"wav_base64":"{wav_b64}"}}'
)
return json.dumps(_speech_result(audio_id, gen_time, duration, r.content, _api_base()))
@mcp.tool()
async def list_voices() -> str:
@@ -266,30 +514,39 @@ def create_mcp_server():
)
@mcp.tool()
async def transcribe(audio_base64: str, language: str | None = None) -> str:
async def transcribe(
audio_base64: str | None = None,
audio_path: str | None = None,
language: str | None = None,
) -> str:
"""Transcribe spoken audio to text.
Pass exactly one of audio_base64 or audio_path.
Args:
audio_base64: Base64-encoded audio bytes (wav/mp3/webm/m4a).
audio_path: Path to an audio file under OMNIVOICE_MCP_BASE_PATH
(relative to it, or absolute inside it). The base path is the
security boundary: with none configured, paths are refused.
Prefer this lane for LLM agents - the audio never enters the
agent's context.
language: Optional language hint; omit for auto-detect.
Returns:
JSON with the recognized text, language, and duration.
"""
try:
raw = base64.b64decode(audio_base64, validate=True)
except Exception:
return '{"error":"audio_base64 is not valid base64"}'
# 200 MB cap — same spirit as voicebox's transcribe gate. Keeps a
# buggy/hostile agent from posting an unbounded blob.
if len(raw) > 200 * 1024 * 1024:
return '{"error":"audio exceeds 200 MB limit"}'
# 200 MB cap on both lanes — same spirit as voicebox's transcribe
# gate. Keeps a buggy/hostile agent from posting an unbounded blob.
raw, err = _read_input_audio(audio_base64, audio_path)
if err:
return json.dumps({"error": err})
data = {}
if language:
data["language"] = language
r = await _api_post_form(
"/transcribe", data=data,
files={"audio": ("audio.wav", raw, "application/octet-stream")},
files={"audio": (f"audio{_sniff_audio_ext(raw)}", raw,
"application/octet-stream")},
)
return str(r.json())
@@ -319,15 +576,17 @@ def create_mcp_server():
@mcp.tool()
async def clone_voice(
name: str,
ref_audio_base64: str,
ref_audio_base64: str | None = None,
ref_text: str = "",
instruct: str = "",
language: str = "Auto",
ref_audio_path: str | None = None,
) -> str:
"""Clone a new voice profile from a reference audio sample.
The new voice is immediately available for use with generate_speech
(pass the returned profile_id as the profile_id argument).
(pass the returned profile_id as the profile_id argument). Pass
exactly one of ref_audio_base64 or ref_audio_path.
Args:
name: A human-friendly name for the cloned voice.
@@ -338,19 +597,20 @@ def create_mcp_server():
quality for some engines).
instruct: Optional style instruction (e.g. 'whisper', 'excited').
language: Language of the reference audio (ISO code or 'Auto').
ref_audio_path: Path to the reference audio under
OMNIVOICE_MCP_BASE_PATH (relative to it, or absolute inside
it); refused when no base path is configured. Prefer this
lane for LLM agents - the clip never enters the context.
Returns:
JSON with the new profile's id, name, and kind.
"""
# Reject oversized inputs before decoding (base64 is always larger
# than raw, so this is a safe lower bound on the decoded size).
if len(ref_audio_base64) > 200 * 1024 * 1024:
return '{"error":"reference audio exceeds 200 MB limit"}'
raw = _decode_ref_audio(ref_audio_base64)
if raw is None:
return '{"error":"ref_audio_base64 is not valid base64"}'
if not raw:
return '{"error":"ref_audio_base64 is empty"}'
raw, err = _read_input_audio(
ref_audio_base64, ref_audio_path,
label="ref_audio_base64", too_big="reference audio exceeds 200 MB limit",
)
if err:
return json.dumps({"error": err})
import httpx
try:
r = await _api_post_form(
@@ -0,0 +1,18 @@
"""Retain the dispatch-time deadline policy across worker/control-plane loss."""
from alembic import op
import sqlalchemy as sa
revision = "0011_remote_attempt_deadlines"
down_revision = "0010_remote_worker_schema"
branch_labels = None
depends_on = None
def upgrade() -> None:
columns = op.get_bind().execute(sa.text("PRAGMA table_info(remote_task_attempts)"))
if not any(row[1] == "deadlines_json" for row in columns):
op.add_column("remote_task_attempts", sa.Column("deadlines_json", sa.Text(), nullable=True))
def downgrade() -> None:
op.drop_column("remote_task_attempts", "deadlines_json")
+1
View File
@@ -190,6 +190,7 @@ class ParseSubtitleTextRequest(BaseModel):
class DubIngestUrlRequest(BaseModel):
url: str
job_id: Optional[str] = None
source_lang: Optional[str] = None
# When true and the URL is a caption-bearing host (YouTube, Vimeo, TED…),
# ask yt-dlp to also download the original-language + any additional
# sub_langs as VTT. The UI uses this to seed a transcript without running
+6 -1
View File
@@ -33,7 +33,12 @@ WS_TICKET_PREFIX = "ovs_ws_ticket_"
_TOKEN_BYTES = 32
_ENCODED_TOKEN_LENGTH = 43
_TOKEN_BODY_RE = re.compile(rf"^[A-Za-z0-9_-]{{{_ENCODED_TOKEN_LENGTH}}}$")
_ALLOWED_WS_PATHS = frozenset({"/ws/events", "/ws/transcribe"})
# Every ticketed WebSocket route. The first-party mirror is ``ALLOWED_WS_PATHS``
# in frontend/src/api/authSession.ts — a route missing here mints a 422 and the
# UI consumer fails silently (#1769 added /ws/tts for the live dub preview).
_ALLOWED_WS_PATHS = frozenset(
{"/ws/events", "/ws/transcribe", "/ws/tts", "/v1/audio/transcriptions/stream"}
)
_ADMIN_CAPABILITIES = frozenset({"consume", "admin"})
_KEY_GENERATION_INFO = b"omnivoice-admin-key-generation-v1"
+278 -55
View File
@@ -24,12 +24,15 @@ faster-whisper because it's available on every platform we ship to).
from __future__ import annotations
import asyncio
import ipaddress
import logging
import os
import re
import contextlib
import threading
import time
import weakref
from urllib.parse import urlsplit
from utils.containment import contain_system_exit
from abc import ABC, abstractmethod
@@ -88,10 +91,9 @@ def reset_pool_after_wedge(executor, *, what: str = "ASR") -> bool:
# ── Consecutive-timeout streak → recommend the crash-isolated engine ────────
# A pool reset restores *capacity*, but the wedged CTranslate2/whisperx thread
# keeps its VRAM until the process exits. When guarded transcribes keep timing
# out back-to-back in one session, resets clearly aren't recovering the
# underlying hang — the durable fix is the crash-isolated sidecar engine
# A timed-out CTranslate2/whisperx thread keeps its worker and VRAM until the
# native call exits. When guarded transcribes keep timing out back-to-back in
# one session, the durable fix is the crash-isolated sidecar engine
# (services.subprocess_asr, #393), whose child process CAN be hard-killed to
# reclaim the hung call and its VRAM. We only *recommend* it (log + error
# message); we never switch engines automatically (owner rule: no silent
@@ -146,41 +148,89 @@ def _isolated_engine_hint(streak: int) -> str:
async def run_transcribe_guarded(executor, fn, *, what: str = "ASR",
timeout: float = ASR_TRANSCRIBE_TIMEOUT_S,
timeout_env: str = "OMNIVOICE_ASR_TRANSCRIBE_TIMEOUT_S"):
timeout_env: str = "OMNIVOICE_ASR_TRANSCRIBE_TIMEOUT_S",
reset_on_timeout: bool = False,
on_abandon=None):
"""Run a blocking transcribe ``fn`` in ``executor`` with a hard wall-clock
bound. On timeout, raise :class:`ASRTimeoutError` with guidance instead of
letting the request hang forever.
``run_in_executor`` cannot cancel the underlying thread, so a wedged
transcribe (a CTranslate2 / whisperx / VAD hang seen on some Windows + CUDA
setups, #730) keeps occupying its GPU-pool worker. With a 12 worker pool
that starves every *other* request including TTS generate and the next
thing the user does surfaces as "Can't reach the local backend" even though
the process is alive. So on timeout we also ``reset()`` the pool when it
supports it (``_ResilientGpuPool``): the wedged thread is abandoned and the
next submit gets a fresh worker, restoring capacity without an app restart.
The orphaned thread still holds its VRAM until the process exits, which is
why the message still recommends a smaller ASR model / Flush as the durable
fix. Executors without ``reset`` (a plain ThreadPoolExecutor in tests) just
get the bound + actionable error.
A future cannot cancel the underlying thread, so a timed-out
in-process CTranslate2/whisperx call still owns its model and device. The
default deliberately leaves that worker accounted for: swapping in a fresh
pool and immediately retrying the same backend overlaps two native calls,
which produced the Windows access violation in #1669. A caller backed by a
genuinely killable process may opt into ``reset_on_timeout``.
``on_abandon`` is called once after a timed-out or cancelled worker can no
longer access its inputs. Queued work cancelled before it starts calls it
immediately; running work calls it from the worker finalizer. Normal
completion leaves cleanup with the caller.
"""
loop = asyncio.get_running_loop()
# Same SystemExit containment as the TTS pool (#1133 class): an ASR
# dependency written as a CLI must not be able to shut the backend down.
fut = loop.run_in_executor(executor, contain_system_exit(fn, what))
inner = contain_system_exit(fn, what)
abandon_lock = threading.Lock()
abandon_state = {
"requested": False,
"finished": False,
"callback_called": False,
}
def _fire_abandon_callback() -> None:
if on_abandon is None:
return
with abandon_lock:
if abandon_state["callback_called"]:
return
abandon_state["callback_called"] = True
try:
on_abandon()
except Exception: # noqa: BLE001 — cleanup cannot hide the ASR result
logger.exception("%s abandon cleanup failed", what)
def _job():
try:
return inner()
finally:
with abandon_lock:
abandon_state["finished"] = True
abandoned = abandon_state["requested"]
if abandoned:
_fire_abandon_callback()
concurrent_fut = executor.submit(_job)
fut = asyncio.wrap_future(concurrent_fut, loop=loop)
def _abandon() -> None:
cancelled_before_start = concurrent_fut.cancel()
with abandon_lock:
abandon_state["requested"] = True
finished = abandon_state["finished"]
fut.cancel()
if cancelled_before_start or finished:
_fire_abandon_callback()
try:
result = await asyncio.wait_for(fut, timeout=timeout)
# Shield the wrapper so timeout does not discard our ability to tell a
# queued cancellation from a native thread that is still running.
result = await asyncio.wait_for(asyncio.shield(fut), timeout=timeout)
except asyncio.CancelledError:
_abandon()
raise
except asyncio.TimeoutError:
# Free the poisoned pool so a hung transcribe can't keep starving TTS /
# other ASR work (the "can't reach backend" symptom, #730).
reset_pool_after_wedge(executor, what=what)
_abandon()
if reset_on_timeout:
reset_pool_after_wedge(executor, what=what)
streak = _note_transcribe_timeout()
msg = (
f"{what} transcription exceeded {timeout:.0f}s and was abandoned — "
"the backend is running, but the ASR model is too heavy for the "
"available compute. Most often the GPU is VRAM-starved: the resident "
"TTS model and a large ASR model (large-v3) contend for memory. "
"Capacity was restored automatically, but for a durable fix Flush the "
"The native call cannot be killed safely, so its capacity remains "
"reserved until it exits. For a durable fix Flush the "
"TTS model to free VRAM, pick a smaller ASR model in "
f"Model Catalogue → Models, or set ASR to CPU. (Raise {timeout_env} "
"for very long transcribes.)"
@@ -310,6 +360,16 @@ class ASRBackend(ABC):
# broken GPU path, strictly worse than the honest `cpu_fallback`.)
gpu_compat: tuple[str, ...] = ("cpu",)
def execution_evidence_loaded(self) -> bool:
"""Whether this instance has live model state worth reporting."""
if getattr(self, "runs_out_of_process", False):
proc = getattr(self, "_proc", None)
return proc is not None and proc.poll() is None
return any(
getattr(self, attr, None) is not None
for attr in ("_model", "_asr", "_pipeline", "_pipe", "_transcriber", "_rec")
)
@classmethod
@abstractmethod
def is_available(cls) -> tuple[bool, str]:
@@ -956,6 +1016,8 @@ class FasterWhisperBackend(ASRBackend):
# (after the #551 compute_type / #255 OOM→CPU fallback chain).
self._device: str | None = None
self._compute_type: str | None = None
self._fallback_reason: str | None = None
self._fallback_stage: str | None = None
@classmethod
def is_available(cls) -> tuple[bool, str]:
@@ -1033,6 +1095,8 @@ class FasterWhisperBackend(ASRBackend):
except Exception: # noqa: BLE001 — cache clear is best-effort
pass
device = "cpu"
self._fallback_reason = "CUDA memory was exhausted while loading the engine"
self._fallback_stage = "model_load"
candidates = _compute_type_candidates(device)
compute_type = candidates[0]
continue
@@ -1991,6 +2055,42 @@ _ASR_OPENAI_COMPAT_MODEL_KEY = "asr.openai_compat.model"
_ASR_OPENAI_COMPAT_SECRET_NAME = "asr_openai_compat_key"
def normalize_openai_compat_asr_base_url(value: str) -> str:
"""Normalize a safe ASR endpoint, allowing plain HTTP only on loopback."""
base = (value or "").strip().rstrip("/")
if not base:
return ""
try:
parsed = urlsplit(base)
_ = parsed.port
except (TypeError, ValueError) as exc:
raise ValueError("Invalid OpenAI-compatible ASR base URL") from exc
scheme = parsed.scheme.lower()
if (
scheme not in {"http", "https"}
or not parsed.hostname
or parsed.username is not None
or parsed.password is not None
or parsed.query
or parsed.fragment
):
raise ValueError(
"OpenAI-compatible ASR base URL must be a credential-free HTTP(S) URL"
)
host = parsed.hostname.lower()
loopback = host == "localhost"
if not loopback:
try:
address = ipaddress.ip_address(host)
address = getattr(address, "ipv4_mapped", None) or address
loopback = address.is_loopback
except ValueError:
loopback = False
if scheme == "http" and not loopback:
raise ValueError("Non-loopback OpenAI-compatible ASR endpoints require HTTPS")
return base
def resolve_openai_compat_asr_base_url() -> str:
from services import settings_store
return (
@@ -2054,7 +2154,7 @@ def probe_openai_compat_server(
maps to a translated message:
not_configured no base URL anywhere
invalid_url base URL without an http(s):// scheme
invalid_url malformed URL or non-loopback HTTP endpoint
ok 2xx ``model_found`` says whether the configured
model appears in the server's list (None = unknown)
ok_no_models 404/405/501 reachable, but no /models endpoint
@@ -2069,7 +2169,7 @@ def probe_openai_compat_server(
from core.scrub import scrub_text
base = (base_url if base_url is not None else resolve_openai_compat_asr_base_url()).strip().rstrip("/")
configured_base = base_url if base_url is not None else resolve_openai_compat_asr_base_url()
mdl = (model if model is not None else resolve_openai_compat_asr_model()).strip()
if api_key is None:
key = resolve_openai_compat_asr_api_key()
@@ -2085,9 +2185,11 @@ def probe_openai_compat_server(
"model_found": None,
"detail": None,
}
if not base:
if not configured_base.strip():
return out
if not base.startswith(("http://", "https://")):
try:
base = normalize_openai_compat_asr_base_url(configured_base)
except ValueError:
out["status"] = "invalid_url"
return out
@@ -2098,7 +2200,7 @@ def probe_openai_compat_server(
try:
with httpx.Client(
timeout=httpx.Timeout(timeout_s, connect=min(5.0, timeout_s)),
follow_redirects=True,
follow_redirects=False,
) as client:
resp = client.get(f"{base}/models", headers=headers)
except httpx.TimeoutException as exc:
@@ -2161,13 +2263,20 @@ class OpenAICompatASRBackend(ASRBackend):
gpu_compat = ("cpu",) # network client only — no local compute
def __init__(self):
self._base_url = resolve_openai_compat_asr_base_url()
self._base_url = normalize_openai_compat_asr_base_url(
resolve_openai_compat_asr_base_url()
)
self._model = resolve_openai_compat_asr_model()
@classmethod
def is_available(cls) -> tuple[bool, str]:
if not resolve_openai_compat_asr_base_url():
base_url = resolve_openai_compat_asr_base_url()
if not base_url:
return False, "Configure a server endpoint in Model Catalogue → Engines"
try:
normalize_openai_compat_asr_base_url(base_url)
except ValueError as exc:
return False, str(exc)
try:
import openai # noqa: F401
except ImportError:
@@ -2175,13 +2284,18 @@ class OpenAICompatASRBackend(ASRBackend):
return True, "ready"
def _client(self):
from openai import OpenAI
from openai import DefaultHttpxClient, OpenAI
api_key = resolve_openai_compat_asr_api_key() or "not-needed"
# max_retries=0: mirrors llm_skills.resolve_skill_client — a
# rate-limited/slow server retrying inside the SDK would blow past
# whatever bounded timeout the caller (dub transcribe, dictation)
# expects from a single call.
return OpenAI(base_url=self._base_url, api_key=api_key, max_retries=0)
return OpenAI(
base_url=self._base_url,
api_key=api_key,
max_retries=0,
http_client=DefaultHttpxClient(follow_redirects=False),
)
def transcribe(self, audio_path: str, *, word_timestamps: bool = True) -> dict:
logger.info(
@@ -2360,6 +2474,8 @@ _LAST_ERRORS: dict[str, str] = {}
# failing ASR wholesale. Per-process by design: repairing the env requires a
# reinstall / ``uv sync --reinstall`` and an app restart anyway.
_DEEP_IMPORT_BROKEN: dict[str, str] = {}
_RUNTIME_EVIDENCE: dict[str, dict] = {}
_RUNTIME_INSTANCES: weakref.WeakValueDictionary[str, "ASRBackend"] = weakref.WeakValueDictionary()
def _deep_import_reason(cls: type["ASRBackend"], exc: ImportError) -> str:
@@ -2390,6 +2506,7 @@ def list_backends() -> list[dict]:
"""
from core.device_caps import detect_host_caps
from core.scrub import scrub_text
from services.engine_evidence import snapshot as execution_snapshot
from services.engine_routing import routing_fields
caps = detect_host_caps()
@@ -2414,6 +2531,24 @@ def list_backends() -> list[dict]:
_LAST_ERRORS[bid] = scrub_text(msg)
isolation = "subprocess" if getattr(cls, "_is_subprocess_isolated", False) else "in-process"
gpu_compat = getattr(cls, "gpu_compat", ("cpu",))
routing = routing_fields(gpu_compat, caps)
# Cached load-time facts are valid only while their exact backend still
# owns live model state. Recompute from that instance so unload/reaping
# cannot leave ghost GPU/provider evidence in diagnostics.
instance = (
_ISOLATED_INSTANCES.get(bid)
if isolation == "subprocess"
else _RUNTIME_INSTANCES.get(bid)
)
execution_evidence = execution_snapshot(
engine_id=bid,
engine_cls=cls,
instance=instance,
routing=routing,
caps=caps,
)
if execution_evidence["evidence_state"] == "not_loaded":
_RUNTIME_EVIDENCE.pop(bid, None)
out.append({
"id": bid,
"display_name": cls.display_name,
@@ -2425,7 +2560,14 @@ def list_backends() -> list[dict]:
"last_error": _LAST_ERRORS.get(bid),
"isolation_mode": isolation,
"gpu_compat": list(gpu_compat),
**routing_fields(gpu_compat, caps),
**routing,
"execution_evidence": execution_evidence or execution_snapshot(
engine_id=bid,
engine_cls=cls,
instance=None,
routing=routing,
caps=caps,
),
})
return out
@@ -2564,7 +2706,10 @@ def _auto_detect() -> str:
def active_backend_id() -> str:
explicit = os.environ.get("OMNIVOICE_ASR_BACKEND")
if explicit:
return explicit
# #1582's public spelling predates the registry name. Keep it as a
# compatibility alias for the PyTorch-native Whisper implementation
# that can use ROCm/HIP; every ASR consumer resolves through here.
return "pytorch-whisper" if explicit == "omnivoice" else explicit
from core import prefs
picked = prefs.get("asr_backend")
if picked:
@@ -2663,6 +2808,21 @@ def load_active_asr_backend(*, asr_pipe=None) -> ASRBackend:
raise ASRModelMissingError(missing)
try:
backend.ensure_loaded()
from core.device_caps import detect_host_caps
from services.engine_evidence import snapshot as execution_snapshot
from services.engine_routing import routing_fields
cls = type(backend)
caps = detect_host_caps()
routing = routing_fields(getattr(cls, "gpu_compat", ("cpu",)), caps)
_RUNTIME_EVIDENCE[bid] = execution_snapshot(
engine_id=bid,
engine_cls=cls,
instance=backend,
routing=routing,
caps=caps,
)
_RUNTIME_INSTANCES[bid] = backend
return backend
except ImportError as e:
# ModuleNotFoundError and its ImportError parent ("cannot import
@@ -2992,7 +3152,7 @@ def _capture_prefers_parakeet() -> bool:
return _parakeet_mlx_installed()
def get_capture_asr_backend() -> ASRBackend:
def get_capture_asr_backend(*, skip_sherpa: bool = False) -> ASRBackend:
"""Pick the fastest ASR engine for capture / dictation.
Selection order:
@@ -3017,6 +3177,9 @@ def get_capture_asr_backend() -> ASRBackend:
Returns a cached singleton so the model stays warm between calls; the
singleton is rebuilt if the selected sherpa model changes.
``skip_sherpa`` is used only to validate a token-silent Sherpa result with
the installed capture fallback before persisting model demotion.
"""
global _capture_backend, _capture_backend_key
@@ -3025,7 +3188,7 @@ def get_capture_asr_backend() -> ASRBackend:
# call get_sherpa_dictation_backend concurrently) can't both build a model.
with _capture_backend_lock:
# 0. Honor an explicit sherpa dictation model selection.
sherpa_id = dictation_model_id()
sherpa_id = None if skip_sherpa else dictation_model_id()
if sherpa_id:
ok, _ = SherpaDictationBackend.is_available()
if ok:
@@ -3192,7 +3355,10 @@ def _capture_whisper_repo() -> str | None:
return os.environ.get("OMNIVOICE_PYTORCH_ASR_MODEL", _PYTORCH_ASR_DEFAULT)
def _recommended_asr_model(purpose: str, missing_repo: str | None) -> dict | None:
def _recommended_asr_model(
purpose: str, missing_repo: str | None, *, prefer_sherpa: bool = True,
excluded_sherpa_model_id: str | None = None,
) -> dict | None:
"""The catalog entry to offer in the download CTA.
Offline: the missing repo itself when it's in the catalog (guarantees
@@ -3212,20 +3378,38 @@ def _recommended_asr_model(purpose: str, missing_repo: str | None) -> dict | Non
by_id = {m["repo_id"]: m for m in KNOWN_MODELS}
exact = by_id.get(missing_repo) if missing_repo else None
want_sherpa = False
if purpose == "dictation":
if exact is not None and exact.get("engine") == "sherpa-onnx":
def _eligible(m: dict, *, sherpa: bool) -> bool:
if (m.get("engine") == "sherpa-onnx") != sherpa:
return False
if sherpa and m.get("dictation_id") == excluded_sherpa_model_id:
return False
return _model_supported(m)
if purpose != "dictation":
if exact is not None and _model_supported(exact):
return _shape(exact)
prefer_sherpa = False
if purpose == "dictation" and prefer_sherpa:
ok, _ = SherpaDictationBackend.is_available()
want_sherpa = ok
if not want_sherpa and exact is not None and _model_supported(exact):
if ok:
if exact is not None and _eligible(exact, sherpa=True):
return _shape(exact)
for m in KNOWN_MODELS:
if (m.get("role") == "ASR" and _eligible(m, sherpa=True)
and _model_curated(m)):
return _shape(m)
# No usable Sherpa recommendation remains (runtime unavailable, explicit
# fallback probe, or the sole curated entry is the demoted model). Offer
# the exact capture fallback so download → retry cannot loop.
if exact is not None and _eligible(exact, sherpa=False):
return _shape(exact)
for m in KNOWN_MODELS:
if m.get("role") != "ASR":
continue
if (m.get("engine") == "sherpa-onnx") != want_sherpa:
continue
if _model_curated(m) and _model_supported(m):
if _eligible(m, sherpa=False) and _model_curated(m):
return _shape(m)
return None
@@ -3256,7 +3440,9 @@ def _repo_installed(repo: str) -> bool:
def asr_model_missing_error(*, purpose: str = "transcribe",
sherpa_model_id: str | None = None,
backend_id: str | None = None) -> dict | None:
backend_id: str | None = None,
skip_sherpa: bool = False,
require_installed: bool = False) -> dict | None:
"""None when the active ASR selection can transcribe without downloading
anything; otherwise the typed ``{"error": "asr_model_missing", ...}``
payload for a 409 / SSE / WS error with a download CTA.
@@ -3268,6 +3454,11 @@ def asr_model_missing_error(*, purpose: str = "transcribe",
``?model=`` override. Installed state comes from the same HF-cache helpers
the model store uses (see :func:`_repo_installed`), so the answer matches
the Model Catalogue Models install badges.
``skip_sherpa`` probes only the non-Sherpa capture fallback; silent-model
recovery uses it before deciding whether persistent demotion is warranted.
``require_installed`` makes unknown/custom selections fail closed for that
recovery path so it can never turn the normal fail-open policy into an
implicit model download.
FAIL-OPEN rule: a repo the model catalog doesn't know (a custom
``ASR_MODEL_*`` pin, pytorch-whisper's default repo, an unrecognized
@@ -3277,27 +3468,55 @@ def asr_model_missing_error(*, purpose: str = "transcribe",
a broken preflight must degrade to the old behaviour, not block ASR.
"""
try:
prefer_sherpa_recommendation = not skip_sherpa
excluded_sherpa_model_id = None
if purpose == "dictation":
sid = sherpa_model_id or dictation_model_id()
sid = None if skip_sherpa else (sherpa_model_id or dictation_model_id())
if sid:
ok, _ = SherpaDictationBackend.is_available()
if ok:
from services import sherpa_dictation as _sd
spec = _sd.get_spec(sid)
# A recognizer observed returning silence must follow the
# same capture fallback as execution, even when the
# frontend keeps sending its persisted `?model=` value.
if spec is not None:
if _sd.is_installed(spec):
return None
return {
"error": ASR_MODEL_MISSING,
"missing_repo_id": spec.repo_id,
"recommended": _recommended_asr_model(purpose, spec.repo_id),
}
if _sd.is_demoted(spec.id):
excluded_sherpa_model_id = spec.id
else:
if _sd.is_installed(spec):
return None
return {
"error": ASR_MODEL_MISSING,
"missing_repo_id": spec.repo_id,
"recommended": _recommended_asr_model(
purpose, spec.repo_id,
),
}
repo = _capture_whisper_repo()
else:
repo = _offline_asr_repo(backend_id)
if repo is None:
if require_installed:
return {
"error": ASR_MODEL_MISSING,
"missing_repo_id": "unresolved-capture-fallback",
"recommended": None,
}
return None # explicit opt-in engine — can't (and shouldn't) preflight
from api.routers.setup.models import get_model_catalog
if require_installed:
if _repo_installed(repo):
return None
return {
"error": ASR_MODEL_MISSING,
"missing_repo_id": repo,
"recommended": _recommended_asr_model(
purpose, repo,
prefer_sherpa=prefer_sherpa_recommendation,
excluded_sherpa_model_id=excluded_sherpa_model_id,
),
}
if get_model_catalog().get(repo) is None:
return None # not installable from the CTA — fail open (see docstring)
if _repo_installed(repo):
@@ -3305,7 +3524,11 @@ def asr_model_missing_error(*, purpose: str = "transcribe",
return {
"error": ASR_MODEL_MISSING,
"missing_repo_id": repo,
"recommended": _recommended_asr_model(purpose, repo),
"recommended": _recommended_asr_model(
purpose, repo,
prefer_sherpa=prefer_sherpa_recommendation,
excluded_sherpa_model_id=excluded_sherpa_model_id,
),
}
except Exception: # noqa: BLE001 — preflight is best-effort, never a blocker
logger.warning("ASR install preflight failed — proceeding without it",
+77
View File
@@ -742,6 +742,74 @@ def _ensure_browser_playable_mp4(video_path: str) -> str:
return video_path
async def _ensure_browser_playable_mp4_for_job(job_id: str, video_path: str) -> str:
"""Normalize an upload through the job's cancellable process registry."""
is_mp4 = video_path.lower().endswith(".mp4")
vcodec, acodec = await asyncio.to_thread(_probe_codecs, video_path)
if is_mp4 and vcodec in _BROWSER_VIDEO_CODECS and acodec in _BROWSER_AUDIO_CODECS:
return video_path
target = os.path.splitext(video_path)[0] + ".mp4"
if target == video_path:
target = os.path.splitext(video_path)[0] + ".browser.mp4"
run_proc = run_proc_factory(job_id)
ffmpeg_bin = find_ffmpeg()
async def attempt(cmd: list[str]) -> int:
try:
proc, _stdout, _stderr = await run_proc(cmd, timeout=1800.0)
return proc.returncode
except asyncio.CancelledError:
raise
except Exception as exc:
logger.warning(
"Browser-media normalization process failed for %s: %s",
log_safe(video_path),
log_safe(exc),
)
return 1
rc = 1
if not is_mp4:
rc = await attempt(
[
ffmpeg_bin, "-y", "-i", video_path,
"-c:v", "copy", "-c:a", "copy",
"-movflags", "+faststart", target,
]
)
if rc == 0 and os.path.exists(target):
target_vcodec, target_acodec = await asyncio.to_thread(_probe_codecs, target)
if (
target_vcodec not in _BROWSER_VIDEO_CODECS
or target_acodec not in _BROWSER_AUDIO_CODECS
):
rc = 1
else:
rc = 1
if rc != 0:
rc = await attempt(
[
ffmpeg_bin, "-y", "-i", video_path,
"-c:v", "libx264", "-preset", "veryfast", "-crf", "23",
"-pix_fmt", "yuv420p", "-c:a", "aac", "-b:a", "192k",
"-movflags", "+faststart", target,
]
)
if rc == 0 and os.path.exists(target) and target != video_path:
try:
os.remove(video_path)
except OSError:
pass # Best effort: the normalized target is already complete.
return target
logger.warning(
"Could not transcode %s to browser-playable mp4 — the in-app "
"video player may render this file as a black box.",
log_safe(video_path),
)
return video_path
# Bounded retry for transient download failures (#579/#598). yt-dlp's own
# `retries`/`fragment_retries` cover per-fragment HTTP flakes, but a broken
# pipe ([Errno 32]) raised while the write side of a pipe closes mid-stream
@@ -1257,6 +1325,13 @@ async def ingest_pipeline(
except Exception:
dur = 0.0
# URL downloads already pass through this guard in yt_download_sync.
# Uploaded videos did not, so a valid VP9/AV1/Opus upload could be
# processed successfully but remain undecodable by the in-app WebView.
# Codec probing/transcoding is blocking; keep it off the event loop.
if source.get("kind") != "url" and input_type != "audio":
video_path = await _ensure_browser_playable_mp4_for_job(job_id, video_path)
# Content-hash cache: reuse artifacts from previous matching jobs.
content_hash = await asyncio.to_thread(compute_file_hash, audio_path)
cached = find_cached_job(content_hash, job_id)
@@ -1295,6 +1370,7 @@ async def ingest_pipeline(
"scene_cuts": scene_cuts,
"youtube_subs": youtube_subs_by_lang or None,
"input_type": input_type,
"source_lang_override": source.get("source_lang"),
}
if not put_and_save_job(
job_id, full_job, filename=filename, duration=dur, content_hash=content_hash,
@@ -1323,6 +1399,7 @@ async def ingest_pipeline(
"scene_cuts": [],
"youtube_subs": youtube_subs_by_lang or None,
"input_type": input_type,
"source_lang_override": source.get("source_lang"),
}
if not put_and_save_job(
job_id, partial, filename=filename, duration=dur, content_hash=content_hash,
+230
View File
@@ -0,0 +1,230 @@
"""Structured pre-install and measured disk costs for TTS engines."""
from __future__ import annotations
import os
import threading
import time
from functools import lru_cache
from pathlib import Path
_GIB = 1024**3
_CACHE_TTL_SECONDS = 10.0
_measurement_cache: dict[str, tuple[float, dict]] = {}
_measurement_lock = threading.Lock()
# Catalogue/build estimates. ``None`` is deliberate: unknown costs must stay
# visible instead of being silently treated as zero.
_ESTIMATES: dict[str, dict] = {
"omnivoice": {
"package_download_bytes": None,
"unique_installed_bytes": None,
"potentially_shared_bytes": None,
"temporary_free_bytes": None,
"confidence": "estimated",
"destination": "hf_model_cache",
"deduplication": None,
},
"kittentts": {
"package_download_bytes": None,
"unique_installed_bytes": None,
"potentially_shared_bytes": None,
"temporary_free_bytes": None,
"confidence": "estimated",
"destination": "hf_model_cache",
"deduplication": None,
},
}
_MODEL_REPOS = {
"omnivoice": "k2-fsa/OmniVoice",
"kittentts": "KittenML/kitten-tts-mini-0.8",
}
def _volume_root(path: Path) -> str:
"""Mount point/drive containing a possibly not-yet-created destination."""
try:
current = path.expanduser().resolve()
while not current.exists() and current.parent != current:
current = current.parent
device = current.stat().st_dev
while current.parent != current and current.parent.stat().st_dev == device:
current = current.parent
return str(current)
except OSError:
return "unknown"
def _hf_cache_path() -> Path:
configured = (
os.environ.get("HF_HUB_CACHE")
or os.environ.get("HUGGINGFACE_HUB_CACHE")
or os.environ.get("HF_HOME")
)
return Path(configured) if configured else Path.home() / ".cache" / "huggingface"
@lru_cache(maxsize=None)
def _catalog_model_bytes(engine_id: str) -> int | None:
"""Resolve the weight estimate from config/models.yaml, its source of truth."""
repo_id = _MODEL_REPOS.get(engine_id)
if repo_id is None:
return None
try:
import yaml
catalog_path = Path(__file__).resolve().parents[1] / "config" / "models.yaml"
entries = yaml.safe_load(catalog_path.read_text(encoding="utf-8"))["models"]
model = next(item for item in entries if item["repo_id"] == repo_id)
return round(float(model["size_gb"]) * _GIB)
except (OSError, KeyError, StopIteration, TypeError, ValueError):
return None
def _dir_size(path: Path) -> int:
total = 0
try:
for root, _dirs, files in os.walk(path):
for filename in files:
try:
total += os.path.getsize(os.path.join(root, filename))
except OSError:
continue
except OSError:
return 0
return total
def _sidecar_estimate(engine_id: str) -> dict | None:
try:
from services.sidecar_install import get_spec, managed_root
spec = get_spec(engine_id)
except Exception:
return None
if spec is None:
return None
model_bytes = spec.weights_bytes
dependency_bytes = spec.dependency_bytes
return {
"model_download_bytes": model_bytes,
"package_download_bytes": dependency_bytes,
"unique_installed_bytes": spec.required_bytes,
"potentially_shared_bytes": spec.potentially_shared_bytes,
"temporary_free_bytes": spec.temporary_free_bytes,
"confidence": spec.disk_confidence,
"destination": "engine_data",
"destination_volume": _volume_root(managed_root(spec)),
"deduplication": "uv_same_volume",
}
def estimate_for(engine_id: str) -> dict:
estimate = _sidecar_estimate(engine_id) or _ESTIMATES.get(engine_id)
if estimate is not None:
return {
"model_download_bytes": _catalog_model_bytes(engine_id),
"destination_volume": _volume_root(_hf_cache_path()),
**estimate,
}
return {
"model_download_bytes": None,
"package_download_bytes": None,
"unique_installed_bytes": None,
"potentially_shared_bytes": None,
"temporary_free_bytes": None,
"confidence": "unknown",
"destination": "unknown",
"destination_volume": "unknown",
"deduplication": None,
}
def _measure_sidecar(engine_id: str) -> dict | None:
try:
from services.sidecar_install import get_spec, managed_checkout, managed_root
spec = get_spec(engine_id)
except Exception:
return None
if spec is None:
return None
checkout = managed_checkout(spec)
if not checkout.is_dir():
return None
model = _dir_size(checkout / spec.weights_subdir)
environment = _dir_size(checkout / ".venv")
total = _dir_size(managed_root(spec))
shared_cache = _dir_size(managed_root(spec).parent / ".uv-cache")
return {
"model_bytes": model,
"environment_bytes": environment,
"cache_bytes": shared_cache,
"total_owned_bytes": total,
"confidence": "measured",
}
def _measure_model_cache(engine_id: str) -> dict | None:
repo_id = _MODEL_REPOS.get(engine_id)
if repo_id is None:
return None
try:
from huggingface_hub import scan_cache_dir
repo = next((item for item in scan_cache_dir().repos if item.repo_id == repo_id), None)
except Exception:
return None
if repo is None or repo.size_on_disk <= 0:
return None
size = int(repo.size_on_disk)
return {
"model_bytes": size,
# The model lives in this cache; cache overhead is not separately
# attributable without double-counting the same hardlinked blobs.
"environment_bytes": None,
"cache_bytes": 0,
"total_owned_bytes": size,
"confidence": "measured",
}
def actual_for(engine_id: str) -> dict:
now = time.monotonic()
cached = _measurement_cache.get(engine_id)
if cached and now - cached[0] < _CACHE_TTL_SECONDS:
return dict(cached[1])
# A cache miss can recursively walk a sidecar and the shared uv cache.
# Coalesce concurrent requests so callers cannot multiply that work.
with _measurement_lock:
now = time.monotonic()
cached = _measurement_cache.get(engine_id)
if cached and now - cached[0] < _CACHE_TTL_SECONDS:
return dict(cached[1])
actual = _measure_sidecar(engine_id) or _measure_model_cache(engine_id) or {
"model_bytes": None,
"environment_bytes": None,
"cache_bytes": None,
"total_owned_bytes": None,
"confidence": "unknown",
}
_measurement_cache[engine_id] = (now, actual)
return dict(actual)
def disk_usage_for(engine_id: str) -> dict:
"""Stable API shape consumed by the engine catalogue."""
return {"estimate": estimate_for(engine_id), "actual": actual_for(engine_id)}
def disk_summary_for(engine_id: str) -> dict:
"""Cheap list payload; measurement is deferred until the row is opened."""
return {
"estimate": estimate_for(engine_id),
"actual": {
"model_bytes": None,
"environment_bytes": None,
"cache_bytes": None,
"total_owned_bytes": None,
"confidence": "unknown",
},
}
+119
View File
@@ -0,0 +1,119 @@
"""Sanitized, reproducible execution evidence for TTS and ASR engines."""
from __future__ import annotations
import importlib.metadata
import platform
from typing import Any
def _version(distribution: str) -> str | None:
try:
return importlib.metadata.version(distribution)
except importlib.metadata.PackageNotFoundError:
return None
def _value(instance: object, *names: str) -> str | None:
for name in names:
try:
value = getattr(instance, name, None)
if value is not None and not callable(value):
text = str(value).strip()
if text and len(text) <= 80 and "/" not in text and "\\" not in text:
return text
except Exception:
continue
return None
def runtime_versions(engine_id: str) -> dict[str, str]:
"""Relevant installed library versions, never paths or environment values."""
names = {"python": platform.python_version()}
candidates = ["torch"]
low = engine_id.lower()
if "faster" in low or "whisperx" in low:
candidates.extend(["ctranslate2", "faster-whisper"])
if "sherpa" in low or "moonshine" in low:
candidates.append("onnxruntime")
if "mlx" in low:
candidates.append("mlx")
for name in candidates:
if (version := _version(name)) is not None:
names[name] = version
return names
def snapshot(
*,
engine_id: str,
engine_cls: type,
instance: object | None,
routing: dict[str, Any],
caps: object,
) -> dict[str, Any]:
"""Return fixed-shape evidence; actual fields stay null until an instance loads."""
isolated = bool(
getattr(engine_cls, "_is_subprocess_isolated", False)
or getattr(engine_cls, "runs_out_of_process", False)
)
loaded = False
probe_failed = False
if instance is not None:
try:
contract = getattr(instance, "execution_evidence_loaded", False)
loaded = bool(contract() if callable(contract) else contract)
except Exception: # noqa: BLE001 - third-party lifecycle descriptors may raise
probe_failed = True
actual_device = None
provider = None
precision = None
if loaded:
actual_device = _value(instance, "_device", "device", "execution_device")
provider = _value(instance, "_provider", "provider", "execution_provider")
precision = _value(
instance, "_compute_type", "compute_type", "_dtype", "dtype", "quantization"
)
if provider is None and actual_device is not None:
provider = actual_device
runtime_fallback_reason = _value(instance, "_fallback_reason", "fallback_reason") if loaded else None
runtime_fallback_stage = _value(instance, "_fallback_stage", "fallback_stage") if loaded else None
status = routing.get("routing_status")
fallback = status == "cpu_fallback" or runtime_fallback_reason is not None
evidence_state = "not_loaded"
if probe_failed:
evidence_state = "probe_error"
elif loaded:
evidence_state = "loaded"
if isolated and provider is None and actual_device is None:
evidence_state = "subprocess_loaded_provider_unreported"
return {
"implementation_variant": f"{engine_cls.__module__}.{engine_cls.__name__}",
"declared_device_families": list(getattr(engine_cls, "gpu_compat", ("cpu",))),
"evidence_state": evidence_state,
"actual_execution_provider": provider,
"actual_execution_device": actual_device,
"gpu_name": getattr(caps, "device_name", "") or None,
"gpu_architecture": _gpu_architecture(getattr(caps, "family", "cpu")),
"precision_or_quantization": precision,
"cpu_fallback_reason": runtime_fallback_reason or (routing.get("routing_reason") if fallback else None),
"cpu_fallback_stage": runtime_fallback_stage or ("routing_preflight" if fallback else None),
"parent_memory_observable": not isolated,
"runtime_versions": runtime_versions(engine_id),
}
def _gpu_architecture(family: str) -> str | None:
if family not in {"cuda", "rocm"}:
return "apple-silicon" if family == "mps" else None
try:
import torch
if family == "rocm":
props = torch.cuda.get_device_properties(0)
return str(getattr(props, "gcnArchName", "") or "") or None
major, minor = torch.cuda.get_device_capability(0)
return f"sm_{major}{minor}"
except Exception:
return None
+26 -11
View File
@@ -31,6 +31,30 @@ class RoutingResult(TypedDict):
routing_reason: str | None # raw, pre-scrub
def under_provisioned_vram(caps: HostCaps, min_vram_gb: float = 0.0) -> bool:
"""Is this host's DEDICATED VRAM below the engine's declared floor?
The one definition of "under-provisioned", shared by everything that acts
on the verdict: the routing caveat below, the timeout guidance, and since
#1804 — the compute-time budget itself (``model_manager
.generate_timeout_s``). It was written out inline in each of them, which is
how the budget came to disagree with the warning printed next to it.
Dedicated-VRAM families ONLY. On MPS, ``HostCaps.vram_gb`` is a heuristic
(system RAM / 2, see device_caps) for a UNIFIED memory pool; comparing it
against a floor measured on discrete CUDA hardware would tell every 8 GB Mac
its 4 GB "VRAM" is too small for an engine that runs fine there. A VRAM
figure of 0 means the probe failed don't guess from it. A floor of 0 means
the engine declares none, and inventing one is worse than staying quiet.
"""
if not min_vram_gb or min_vram_gb <= 0:
return False
if getattr(caps, "family", None) not in ("cuda", "rocm"):
return False
vram_gb = float(getattr(caps, "vram_gb", 0.0) or 0.0)
return 0 < vram_gb < float(min_vram_gb)
def _caveat(caps: HostCaps, min_vram_gb: float = 0.0) -> str | None:
"""A caveat string for an otherwise-accelerated host, or None.
@@ -51,16 +75,7 @@ def _caveat(caps: HostCaps, min_vram_gb: float = 0.0) -> str | None:
for note in caps.notes:
if KERNEL_RISK_MARKER in note:
return f"{caps.family.upper()} selected, but: {note}"
# Dedicated-VRAM families ONLY. On MPS, HostCaps.vram_gb is a heuristic
# (system RAM / 2, see device_caps) for a UNIFIED memory pool — comparing
# it against a floor measured on discrete CUDA hardware would tell every
# 8 GB Mac its 4 GB "VRAM" is too small for an engine that runs fine there.
# Different memory model, different (unmeasured) floor; don't guess.
if (
caps.family in ("cuda", "rocm")
and min_vram_gb > 0
and 0 < caps.vram_gb < min_vram_gb
):
if under_provisioned_vram(caps, min_vram_gb):
device = caps.device_name or caps.family.upper()
return (
f"{device} has {caps.vram_gb:.1f} GB VRAM; this engine wants about "
@@ -208,5 +223,5 @@ def routing_fields(
__all__ = [
"RoutingStatus", "RoutingResult", "resolve_routing", "routing_fields",
"routing_notice", "header_safe_reason",
"routing_notice", "header_safe_reason", "under_provisioned_vram",
]
+8 -1
View File
@@ -177,6 +177,9 @@ class LocalCall:
queue_timeout: Optional[float] = None
# The engine's declared VRAM floor; only shapes the timeout message.
min_vram_gb: float = 0.0
# Called once a local worker abandoned by its waiter can no longer touch
# request-owned inputs. Normal completion does not call it (#1668).
on_abandon: Optional[Callable[[], None]] = None
# Some remote-first callers cannot construct the local callable without
# loading the very model they are trying to offload. Prepare it only when
# routing/fallback actually selects this machine.
@@ -465,6 +468,7 @@ async def _run_local(call: LocalCall, *, admit: bool = False, executor=None) ->
queue_timeout=call.queue_timeout,
min_vram_gb=call.min_vram_gb,
executor=executor,
on_abandon=call.on_abandon,
)
@@ -499,7 +503,9 @@ async def _run_remote(
deadline = _default_deadline(call.operation, params.get("text"))
try:
task = scheduler.submit(
submit = getattr(scheduler, "submit_async", None)
submit = submit if callable(submit) else scheduler.submit
submitted = submit(
operation=call.operation,
engine=call.engine,
model_id=call.model_id,
@@ -508,6 +514,7 @@ async def _run_remote(
deadline_seconds=deadline,
pinned_worker_id=decision.worker_id,
)
task = await submitted if asyncio.iscoroutine(submitted) else submitted
except QueueFull as exc:
raise _NotDispatched(str(exc)) from exc
+220
View File
@@ -0,0 +1,220 @@
"""Karaoke (word-highlight) ASS builder for dub hardsub export.
Pure text-in/text-out: no ffmpeg, no models, no filesystem. ``build_ass``
turns subtitle cues into an ASS script whose lines carry ``\\k``/``\\kf``
karaoke tags, so ffmpeg's ``ass=`` filter burns a word-by-word highlight
sweep instead of the static line the SRT path renders.
Word timing sources, in order:
1. ``cue["words"]`` per-word ``{text, start, end}`` persisted at
transcribe time (services.segmentation). Used only when the words still
spell the cue's display text: after translation the persisted ASR words
are source-language tokens, so re-using their timing would burn the
wrong language. The display text is always authoritative.
2. Even split the cue text's whitespace tokens spread uniformly across
``[start, end]``. This is the compatibility path for jobs transcribed
before word persistence and for translated tracks.
Dual-layout karaoke is intentionally unsupported (out of scope): callers
must fall back to the line (SRT) burn when the dual layout is requested.
"""
from __future__ import annotations
import re
from typing import Optional, Sequence
_WS = re.compile(r"\s+")
#: Default ASS canvas. libass scales the script to the real video size, so
#: one reference resolution keeps font/margin proportions stable everywhere.
DEFAULT_PLAY_RES = (1920, 1080)
_HEADER_TEMPLATE = """[Script Info]
; Generated by VoiceStudio karaoke burn-in
ScriptType: v4.00+
PlayResX: {res_x}
PlayResY: {res_y}
WrapStyle: 0
ScaledBorderAndShadow: yes
[V4+ Styles]
Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding
Style: Default,Arial,64,&H0000E7FF,&H00FFFFFF,&H00101010,&H7F000000,0,0,0,0,100,100,0,0,1,3,1,2,96,96,48,1
[Events]
Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text
"""
def _norm(text: object) -> str:
return _WS.sub(" ", str(text or "").strip())
def _ass_time(seconds: float) -> str:
"""``H:MM:SS.CC`` (centiseconds) — the ASS event timestamp format."""
cs = max(0, int(round(float(seconds) * 100)))
h, rem = divmod(cs, 360000)
m, rem = divmod(rem, 6000)
s, c = divmod(rem, 100)
return f"{h}:{m:02d}:{s:02d}.{c:02d}"
def _ass_escape(text: str) -> str:
"""Escape a display token for an ASS Dialogue text field.
Braces would open an override block (user text like ``{\\b1}`` must render
literally, never execute); newlines become ASS hard line breaks.
"""
return (
str(text)
.replace("{", "\\{")
.replace("}", "\\}")
.replace("\r\n", "\\N")
.replace("\n", "\\N")
.replace("\r", "\\N")
)
def _cs(seconds: float) -> int:
"""Karaoke tag duration in centiseconds; ≥1 so a tag never renders as 0."""
return max(1, int(round(float(seconds) * 100)))
def even_split_words(text: str, start: float, end: float) -> list[dict]:
"""Uniformly distribute the cue text's whitespace tokens over [start, end].
The export fallback for jobs transcribed before per-word persistence and
for translated tracks (whose persisted words are source-language tokens).
"""
tokens = [tok for tok in _WS.split(str(text or "").strip()) if tok]
if not tokens:
return []
start = float(start)
dur = max(0.0, float(end) - start) / len(tokens)
return [
{"text": tok, "start": start + i * dur, "end": start + (i + 1) * dur}
for i, tok in enumerate(tokens)
]
def scale_words(
words: Sequence[dict],
orig_start: float,
orig_end: float,
new_start: float,
new_end: float,
) -> Optional[list[dict]]:
"""Map word times linearly from [orig_start, orig_end] → [new_start, new_end].
Used when Smart Fit moves a cue onto the fitted timeline: the persisted
word times live on the original timeline and must ride along. Returns
``None`` when either span is degenerate (caller should drop the words so
export falls back to an even split over the new span).
"""
orig_span = float(orig_end) - float(orig_start)
new_span = float(new_end) - float(new_start)
if orig_span <= 0 or new_span <= 0:
return None
ratio = new_span / orig_span
out: list[dict] = []
for w in words:
try:
ws = float(w["start"])
we = float(w["end"])
except (KeyError, TypeError, ValueError):
return None
out.append({
**w,
"start": round(float(new_start) + (ws - float(orig_start)) * ratio, 3),
"end": round(float(new_start) + (we - float(orig_start)) * ratio, 3),
})
return out
def _usable_words(cue: dict, text: str) -> Optional[list[tuple[str, float, float]]]:
"""Persisted words, iff well-formed AND they spell the cue's display text."""
words = cue.get("words")
if not isinstance(words, list) or not words:
return None
clean: list[tuple[str, float, float]] = []
for w in words:
if not isinstance(w, dict):
return None
wtext = _norm(w.get("text"))
try:
ws = float(w["start"])
we = float(w["end"])
except (KeyError, TypeError, ValueError):
return None
if wtext:
clean.append((wtext, ws, we))
if not clean:
return None
if _norm(" ".join(t for t, _, _ in clean)) != text:
return None
return clean
def _karaoke_text(cue: dict, text: str, start: float, end: float) -> str:
"""One Dialogue text field: ``{\\k…}`` lead-in + per-word ``{\\kf…}`` tags.
Each word's sweep runs until the next word starts (the classic karaoke
layout inter-word gaps finish the previous word's fill), and the last
word sweeps out to the cue end.
"""
words = _usable_words(cue, text) or [
(w["text"], w["start"], w["end"]) for w in even_split_words(text, start, end)
]
# Clamp into the cue span and enforce monotonic starts so malformed
# persisted data can only mistime the sweep, never corrupt the script.
clamped: list[tuple[str, float]] = []
prev = start
for wtext, ws, _ in words:
ws = min(max(ws, prev), end)
clamped.append((wtext, ws))
prev = ws
parts: list[str] = []
lead = clamped[0][1] - start
if lead > 0.005:
parts.append(f"{{\\k{_cs(lead)}}}")
for i, (wtext, ws) in enumerate(clamped):
nxt = clamped[i + 1][1] if i + 1 < len(clamped) else end
sep = " " if i + 1 < len(clamped) else ""
parts.append(f"{{\\kf{_cs(max(nxt, ws) - ws)}}}{_ass_escape(wtext)}{sep}")
return "".join(parts)
def build_ass(
cues: Sequence[dict],
*,
dual: bool = False,
play_res: tuple[int, int] = DEFAULT_PLAY_RES,
) -> str:
"""Build a karaoke ASS script from subtitle cues ({text, start, end, words?}).
One ``Default`` style; one Dialogue event per cue. ``dual`` exists for
signature parity with the line burn but dual-layout karaoke is out of
scope callers must keep the SRT line burn for dual, so requesting it
here is a contract violation, not a rendering mode.
"""
if dual:
raise ValueError(
"dual-layout karaoke is not supported; use the line (SRT) burn for dual subtitles"
)
res_x, res_y = play_res
lines = [_HEADER_TEMPLATE.format(res_x=int(res_x), res_y=int(res_y))]
for cue in cues or []:
text = _norm(cue.get("text"))
if not text:
continue
start = float(cue["start"])
end = float(cue["end"])
if end <= start:
end = start + 0.1
lines.append(
f"Dialogue: 0,{_ass_time(start)},{_ass_time(end)},Default,,0,0,0,,"
f"{_karaoke_text(cue, text, start, end)}"
)
return "\n".join(lines) + "\n"
+20
View File
@@ -154,6 +154,17 @@ def bundled_dir() -> str:
return os.path.join(media_tools_dir(), f"ffbin-{_FFBIN_COMMIT[:12]}", _platform_key())
def _publish_bundled_on_path() -> None:
"""Make a newly validated bundle visible to bare-name subprocess calls."""
directory = os.path.abspath(bundled_dir())
current = os.environ.get("PATH", "")
entries = current.split(os.pathsep) if current else []
if os.path.normcase(directory) in {os.path.normcase(entry) for entry in entries if entry}:
return
os.environ["PATH"] = os.pathsep.join([directory, *entries])
logger.info("Published the acquired media-tool directory on PATH")
def _exe(name: str) -> str:
return f"{name}.exe" if sys.platform == "win32" else name
@@ -231,12 +242,21 @@ def acquire_bundled(wait: bool = False) -> dict:
_ops["acquire"].update(state="running", progress=0.0, error=None)
if all(bundled_tool_path(t) and _binary_runs(bundled_tool_path(t)) for t in TOOLS):
# The bundle may have arrived after startup's one-time PATH publish
# (first-run acquisition is asynchronous). Make it visible to pydub
# and other dependencies that launch ffmpeg/ffprobe by bare name now,
# without requiring a backend restart (#1677).
_publish_bundled_on_path()
_set_op("acquire", state="done", progress=1.0)
return _op_snapshot()["acquire"]
def _worker():
try:
_do_acquire()
# Startup cannot publish binaries which do not exist yet. The
# background worker must complete that second half atomically with
# installation so the very next synthesis can use the tools.
_publish_bundled_on_path()
_set_op("acquire", state="done", progress=1.0, error=None)
logger.info("media-tools: bundled ffmpeg/ffprobe installed at %s", bundled_dir())
except Exception as e:
+325 -38
View File
@@ -4,8 +4,9 @@ import sys
import time
import asyncio
import logging
import queue
import threading
from concurrent.futures import ThreadPoolExecutor, Executor
from concurrent.futures import Executor, Future, ThreadPoolExecutor
from utils.containment import contain_system_exit
@@ -397,7 +398,29 @@ def __getattr__(name: str):
# (generation.py, tts_stream.py) were the last unguarded dispatch — and the
# residual on-main reports all fail on generate:start (audio). This is the same
# guard generalised so every GPU dispatch shares one recovery path.
_GENERATE_TIMEOUT_EXPLICIT = "OMNIVOICE_GENERATE_TIMEOUT_S" in os.environ
GPU_JOB_TIMEOUT_S = float(os.environ.get("OMNIVOICE_GENERATE_TIMEOUT_S", "300.0"))
_CONFIGURED_GPU_JOB_TIMEOUT_S = GPU_JOB_TIMEOUT_S
# CPU synthesis is healthy but substantially slower than accelerated inference.
# Keep a separate, bounded floor so a short render on CPU is not abandoned at
# the GPU-oriented five-minute deadline (#1588).
#
# #1787 review fix: an explicit OMNIVOICE_CPU_GENERATE_TIMEOUT_S must ALWAYS
# govern CPU dispatches, even when OMNIVOICE_GENERATE_TIMEOUT_S is ALSO
# explicit. Before this flag existed, `universal_override` below treated any
# explicit GENERATE_TIMEOUT_S as authoritative for CPU too, so the Settings
# panel's "CPU budget" row could be saved and silently never apply whenever
# the "Accelerated" row was also set — the exact defect (a control that looks
# like it works and doesn't) issue #1787 exists to remove. Setting ONLY
# OMNIVOICE_GENERATE_TIMEOUT_S keeps its historical "universal" behavior
# unchanged (test_explicit_universal_generate_timeout_wins_on_cpu) — nobody
# who already relies on that single-var override loses it. The only case that
# changes is the previously-undocumented, previously-broken combination of
# setting BOTH: the more specific (CPU) value now wins for CPU jobs, matching
# what a user who filled in both Settings rows was told would happen.
_CPU_GENERATE_TIMEOUT_EXPLICIT = "OMNIVOICE_CPU_GENERATE_TIMEOUT_S" in os.environ
CPU_JOB_TIMEOUT_S = float(os.environ.get("OMNIVOICE_CPU_GENERATE_TIMEOUT_S", "600.0"))
_CONFIGURED_CPU_JOB_TIMEOUT_S = CPU_JOB_TIMEOUT_S
# Queue-wait budget — a SEPARATE, deliberately generous clock (#1190/#1202).
# The execution bound above must never be spent waiting in line: a job queued
@@ -500,7 +523,10 @@ class GpuPoolBusyError(TimeoutError):
self.retry_after = max(1, int(round(retry_after)))
def generate_timeout_s(text: "str | None") -> float:
def generate_timeout_s(
text: "str | None", *, engine: object = None, execution_device: "str | None" = None,
min_vram_gb: float = 0.0,
) -> float:
"""THE wall-clock execution budget for one synthesis job, scaled to input.
Single source of truth for every TTS dispatch (#1190/#1202). The
@@ -511,15 +537,67 @@ def generate_timeout_s(text: "str | None") -> float:
on long inputs. Lives here (not in a router) so every router shares it
without importing generation.py.
Policy: floor at the configured OMNIVOICE_GENERATE_TIMEOUT_S, plus 1s per
40 characters past a 1200-character free allowance generous enough for
Policy: floor at the configured OMNIVOICE_GENERATE_TIMEOUT_S (accelerated
hosts) or OMNIVOICE_CPU_GENERATE_TIMEOUT_S (CPU hosts the latter wins
for CPU whenever it is itself explicit, even if the former also is; see
the #1787 comment on the module-level constants), plus 1s per 40
characters past a 1200-character free allowance generous enough for
CPU-class hardware, still bounded (a wedged job is caught in minutes, not
hours).
#1804: "accelerated" is not one performance class. A card with less VRAM
than the engine declares it needs pages to system RAM over PCIe and renders
SLOWER than the same machine's CPU would — yet, judged by device family
alone, it was handed HALF the CPU budget. That inversion is what three 4 GB
reporters hit (#1226 GTX 1650 Ti, #1222 Quadro P2000, #1804 GTX 1650), all
on the engine that declares a 6 GB floor. Every layer already knew: routing
raises a caveat, the preflight toast warns, and the timeout message names
the card. Only the budget ignored it. So an under-provisioned accelerator
now floors at the CPU budget the class of hardware it actually performs
like. ``min_vram_gb`` is the engine's declared floor; callers that pass
``engine`` get it read off the engine automatically.
"""
return max(
GPU_JOB_TIMEOUT_S,
GPU_JOB_TIMEOUT_S + (max(0, len(text or "") - 1200) / 40.0),
)
base = GPU_JOB_TIMEOUT_S
try:
from core.device_caps import detect_host_caps
caps = detect_host_caps()
family = execution_device or caps.family
if not min_vram_gb and engine is not None:
min_vram_gb = float(getattr(engine, "min_vram_gb", 0.0) or 0.0)
if execution_device is None and engine is not None:
from services.engine_routing import resolve_routing
compat = getattr(engine, "gpu_compat", None)
if compat is None:
compat = getattr(type(engine), "gpu_compat", (family, "cpu"))
if tuple(compat) == ("cpu",):
family = "cpu"
else:
family = resolve_routing(compat, caps, min_vram_gb)["effective_device"]
universal_override = (
_GENERATE_TIMEOUT_EXPLICIT
or GPU_JOB_TIMEOUT_S != _CONFIGURED_GPU_JOB_TIMEOUT_S
)
# An explicit (env-set, or runtime-changed the same way tests do)
# CPU budget is more specific than the universal override and always
# wins for CPU dispatches — see the #1787 comment above.
cpu_explicit = (
_CPU_GENERATE_TIMEOUT_EXPLICIT
or CPU_JOB_TIMEOUT_S != _CONFIGURED_CPU_JOB_TIMEOUT_S
)
if family == "cpu" and (cpu_explicit or not universal_override):
base = CPU_JOB_TIMEOUT_S
elif not universal_override and family in ("cuda", "rocm"):
from services.engine_routing import under_provisioned_vram
if under_provisioned_vram(caps, min_vram_gb):
# `max`, never a plain assignment: an operator who raised the
# accelerated budget above the CPU one must not have it cut.
base = max(base, CPU_JOB_TIMEOUT_S)
except Exception:
# Device probing is advisory here; the configured universal bound is
# still safe when a platform probe is unavailable during startup.
pass
return base + (max(0, len(text or "") - 1200) / 40.0)
def _retry_after_estimate(stats: dict) -> float:
@@ -599,7 +677,8 @@ async def run_on_gpu_pool_guarded(fn, *, what: str = "GPU job",
timeout: "float | None" = None,
executor=None,
queue_timeout: "float | None" = None,
min_vram_gb: float = 0.0):
min_vram_gb: float = 0.0,
on_abandon=None):
"""Run blocking ``fn`` on the GPU pool, bounding **execution** — not the
wait for a free worker.
@@ -627,6 +706,12 @@ async def run_on_gpu_pool_guarded(fn, *, what: str = "GPU job",
at 0 the default, and correct for every non-TTS job on this pool
(reference transcribe, watermarking, dub steps) the under-provisioned-GPU
wording is never used, because nothing measured says it applies (#1226).
``on_abandon`` is called once, after a job whose caller stopped waiting can
no longer access its inputs. A queued job that is cancelled before it
starts calls it immediately; a running thread calls it from ``_job``'s
finalizer. Normal completion never calls it. This lets request-owned temp
files outlive abandoned workers without delaying ordinary requests (#1668).
"""
loop = asyncio.get_running_loop()
ex = executor if executor is not None else _get_gpu_pool()
@@ -641,6 +726,24 @@ async def run_on_gpu_pool_guarded(fn, *, what: str = "GPU job",
# job's model-load heartbeats (#1367). A dict, not a nonlocal: the closure
# runs on a pool thread while the waiter reads from the event loop.
_ident_box: dict = {}
_abandon_lock = threading.Lock()
_abandon_state = {
"requested": False,
"finished": False,
"callback_called": False,
}
def _fire_abandon_callback() -> None:
if on_abandon is None:
return
with _abandon_lock:
if _abandon_state["callback_called"]:
return
_abandon_state["callback_called"] = True
try:
on_abandon()
except Exception: # noqa: BLE001 — cleanup cannot hide the pool result
logger.exception("%s abandon cleanup failed", _log_safe(what))
def _job():
# First thing the worker does: tell the awaiting coroutine the
@@ -657,8 +760,26 @@ async def run_on_gpu_pool_guarded(fn, *, what: str = "GPU job",
# Idents are reused by the OS; a stale heartbeat under this ident
# must not vouch for some future job on the same thread.
_MODEL_LOAD_ACTIVITY.pop(threading.get_ident(), None)
with _abandon_lock:
_abandon_state["finished"] = True
abandoned = _abandon_state["requested"]
if abandoned:
_fire_abandon_callback()
concurrent_fut = ex.submit(_job)
fut = asyncio.wrap_future(concurrent_fut, loop=loop)
def _abandon() -> None:
# Keep the concurrent future so we can distinguish a job cancelled out
# of the queue from a thread that Python cannot stop once it has begun.
cancelled_before_start = concurrent_fut.cancel()
with _abandon_lock:
_abandon_state["requested"] = True
finished = _abandon_state["finished"]
fut.cancel()
if cancelled_before_start or finished:
_fire_abandon_callback()
fut = loop.run_in_executor(ex, _job)
waiter = asyncio.ensure_future(started.wait())
try:
# Phase 1 — queue wait. Watch the future too, so a job that fails or is
@@ -671,6 +792,7 @@ async def run_on_gpu_pool_guarded(fn, *, what: str = "GPU job",
# Caller went away (client disconnect). We stop awaiting the job, so
# make sure its eventual result/exception is consumed rather than
# logged as "Future exception was never retrieved".
_abandon()
fut.add_done_callback(_swallow_abandoned)
raise
finally:
@@ -680,7 +802,7 @@ async def run_on_gpu_pool_guarded(fn, *, what: str = "GPU job",
# Never picked up: cancel it out of the queue (a not-yet-started
# concurrent future cancels cleanly) and report saturation, NOT a
# too-heavy job.
fut.cancel()
_abandon()
fut.add_done_callback(_swallow_abandoned)
stats = gpu_pool_stats(ex)
logger.warning(
@@ -751,14 +873,14 @@ async def run_on_gpu_pool_guarded(fn, *, what: str = "GPU job",
# Caller went away mid-execution. The old wait_for cancelled the
# wrapper itself; asyncio.wait does not, so do both halves here or the
# eventual result is logged as "Future exception was never retrieved".
fut.cancel()
_abandon()
fut.add_done_callback(_swallow_abandoned)
raise
except asyncio.TimeoutError as timeout_exc:
# Parity with the old wait_for semantics: cancel the asyncio wrapper;
# the worker thread keeps going regardless. Consume whatever it
# eventually produces.
fut.cancel()
_abandon()
fut.add_done_callback(_swallow_abandoned)
# Capture the stacks BEFORE reset(): reset() replaces the executor, and
# once the wedged thread is no longer a pool worker we can no longer
@@ -974,6 +1096,7 @@ def _timeout_guidance(
"""
family = "cuda" # conservative default: GPU wording if the probe fails
device_name, vram_gb = "", 0.0
_caps = None # a failed probe stays None; under_provisioned_vram() reads it safely
try:
from core.device_caps import detect_host_caps
_caps = detect_host_caps()
@@ -1021,11 +1144,9 @@ def _timeout_guidance(
# a threshold applied without knowing whose job it is would confidently
# misdiagnose most of them. And on MPS `vram_gb` is a unified-memory
# heuristic (RAM/2), not a dedicated pool to compare against.
if (
min_vram_gb > 0
and family in ("cuda", "rocm")
and 0 < vram_gb < min_vram_gb
):
from services.engine_routing import under_provisioned_vram
if under_provisioned_vram(_caps, min_vram_gb):
return common + (
f"{device_name or 'this GPU'} has {vram_gb:.1f} GB of VRAM and "
f"this engine wants about {min_vram_gb:.0f} GB — generations here "
@@ -1057,21 +1178,166 @@ def _timeout_guidance(
# doubling the effective queue depth of a streamed multi-chunk render.
# Giving it its own tiny pool removes that head-of-line blocking with no VRAM
# risk, because the work was never on the device to begin with.
_watermark_pool_singleton: "ThreadPoolExecutor | None" = None
_watermark_pool_lock = threading.Lock()
_WATERMARK_STOP = object()
def get_watermark_pool() -> ThreadPoolExecutor:
"""Dedicated 1-worker pool for provenance marking. Built lazily so hosts
with watermarking disabled never spawn the thread."""
global _watermark_pool_singleton
if _watermark_pool_singleton is None:
with _watermark_pool_lock:
if _watermark_pool_singleton is None:
_watermark_pool_singleton = ThreadPoolExecutor(
max_workers=1, thread_name_prefix="watermark",
class _WatermarkExecutor(Executor):
"""Single daemon worker with a bounded shutdown contract.
``ThreadPoolExecutor`` uses non-daemon workers that Python joins at exit,
so ``wait=False`` still delays process exit while ``wait=True`` can hang
lifespan teardown forever. AudioSeal loading is not cooperatively
cancellable; a daemon worker plus a bounded join is the only thread-based
contract that both preserves in-process model warm-up and guarantees exit.
"""
def __init__(self) -> None:
self._items: queue.Queue = queue.Queue()
self._lock = threading.Lock()
self._shutdown = False
self._thread: threading.Thread | None = None
def submit(self, fn, /, *args, **kwargs) -> Future:
future: Future = Future()
with self._lock:
if self._shutdown:
raise RuntimeError("cannot schedule new futures after shutdown")
if self._thread is None:
self._thread = threading.Thread(
target=self._run,
name="watermark_0",
daemon=True,
)
return _watermark_pool_singleton
self._thread.start()
self._items.put((future, fn, args, kwargs))
return future
def _run(self) -> None:
while True:
item = self._items.get()
if item is _WATERMARK_STOP:
return
future, fn, args, kwargs = item
if not future.set_running_or_notify_cancel():
continue
try:
future.set_result(fn(*args, **kwargs))
except (Exception, SystemExit, KeyboardInterrupt) as exc:
future.set_exception(exc)
def is_stopped(self) -> bool:
"""Whether shutdown has completed and this executor can be replaced."""
with self._lock:
return self._shutdown and (
self._thread is None or not self._thread.is_alive()
)
def is_shutdown(self) -> bool:
with self._lock:
return self._shutdown
def shutdown(
self,
wait: bool = True,
*,
cancel_futures: bool = False,
timeout: float | None = None,
) -> bool:
with self._lock:
self._shutdown = True
thread = self._thread
if cancel_futures:
while True:
try:
item = self._items.get_nowait()
except queue.Empty:
break
if item is not _WATERMARK_STOP:
item[0].cancel()
self._items.put(_WATERMARK_STOP)
if wait and thread is not None:
thread.join(timeout=timeout)
return thread is None or not thread.is_alive()
_watermark_pool_singleton: "_WatermarkExecutor | None" = None
_watermark_pool_lock = threading.Lock()
_watermark_pool_accepting = True
def begin_watermark_pool_lifecycle() -> None:
"""Open watermark submissions for a newly-started app lifespan."""
global _watermark_pool_accepting, _watermark_pool_singleton
with _watermark_pool_lock:
if (
_watermark_pool_singleton is not None
and _watermark_pool_singleton.is_stopped()
):
_watermark_pool_singleton = None
_watermark_pool_accepting = (
_watermark_pool_singleton is None
or not _watermark_pool_singleton.is_shutdown()
)
def get_watermark_pool() -> _WatermarkExecutor:
"""Dedicated 1-worker pool for provenance marking. Built lazily so hosts
with watermarking disabled never spawn the thread.
The executor is captured and returned UNDER the lock: reading the global
again after an unlocked null-check could race shutdown_watermark_pool's
reset and hand out None (CodeRabbit, PR #1577)."""
global _watermark_pool_accepting, _watermark_pool_singleton
with _watermark_pool_lock:
if not _watermark_pool_accepting:
if (
_watermark_pool_singleton is not None
and _watermark_pool_singleton.is_stopped()
):
_watermark_pool_singleton = None
_watermark_pool_accepting = True
else:
raise RuntimeError("watermark executor is shutting down")
if (
_watermark_pool_singleton is not None
and _watermark_pool_singleton.is_stopped()
):
_watermark_pool_singleton = None
if _watermark_pool_singleton is None:
_watermark_pool_singleton = _WatermarkExecutor()
return _watermark_pool_singleton
def shutdown_watermark_pool(*, timeout: float = 20.0) -> None:
"""Drain the watermark pool at app shutdown (PR #1577).
Refuse queued work and wait for the active operation: Python cannot kill
a thread inside AudioSeal loading, so returning early would let model
initialization continue during interpreter teardown. The draining pool
remains published until its worker stops, preventing concurrent producers
from creating a replacement that escapes this shutdown. A process that
keeps running after lifespan shutdown (the test suite does exactly this)
gets a fresh pool once the old worker has actually stopped."""
global _watermark_pool_accepting, _watermark_pool_singleton
with _watermark_pool_lock:
_watermark_pool_accepting = False
pool = _watermark_pool_singleton
if pool is not None:
stopped = pool.shutdown(
wait=True,
cancel_futures=True,
timeout=max(0.0, float(timeout)),
)
if stopped:
with _watermark_pool_lock:
if _watermark_pool_singleton is pool:
_watermark_pool_singleton = None
else:
logger.warning(
"Watermark worker exceeded the %.1fs shutdown deadline; "
"abandoning its daemon thread",
timeout,
)
model = None # type: ignore
@@ -1259,14 +1525,17 @@ def get_best_device():
# ── DirectML — universal Windows GPU (probe reports this as "cpu") ─
# Reached only when no torch family was detected (family == "cpu"), which is
# exactly the DirectML case — the probe classifies DirectML hosts as cpu.
try:
import torch_directml
if torch_directml.device_count() > 0:
logger.info("Using DirectML device (GPU %d)", 0)
return str(torch_directml.device(0))
except ImportError:
pass
if family == "cpu":
try:
import torch_directml
if torch_directml.device_count() > 0:
logger.info("Using DirectML device (GPU %d)", 0)
return str(torch_directml.device(0))
except ImportError:
# DirectML is optional; an absent package leaves CPU available.
pass
# Other families need an explicitly compatible loader (e.g. NPU sidecars).
return "cpu"
_COMPILE_ERR_MODULE_PREFIXES = ("torch._dynamo", "torch._inductor", "torch.fx", "triton")
@@ -2628,6 +2897,21 @@ async def preload_model():
if model is not None:
return # already loaded
# On MPS the configured ``omnivoice`` id resolves to a crash-isolated
# sidecar. Warming the native singleton here would put the same fatal MPS
# allocator risk back into the API process before the isolated engine is
# ever asked to synthesize.
try:
from core.device_caps import detect_host_caps
if detect_host_caps().family == "mps":
logger.info(
"Native TTS preload skipped: OmniVoice uses crash isolation on this host."
)
return
except Exception: # noqa: BLE001 -- preload selection must stay best-effort
logger.debug("effective TTS preload selection failed", exc_info=True)
# A machine lending its GPU has no local user to warm the model FOR. This
# preload exists to make the first /generate feel instant for the person
# sitting in front of the app; on a headless node there is nobody sitting
@@ -2717,7 +3001,10 @@ async def preload_model():
"The TTS model could not be loaded. Settings → Logs → Backend "
"has the full error."
)
_set_loading("failed", detail, error=detail)
# `sub_stage` is a public API enum and the frontend keys failure state
# off `error`. Keep the human-readable word "failed" in the detail,
# not in the state machine (#1695).
_set_loading("error", detail, error=detail)
def get_model_status():
is_loaded = model is not None
+94 -1
View File
@@ -79,6 +79,7 @@ class Segment:
text: str
speaker_id: str = "Speaker 1"
id: str = field(default_factory=lambda: str(uuid.uuid4())[:8])
extra: dict = field(default_factory=dict)
@property
def duration(self) -> float:
@@ -90,6 +91,7 @@ class Segment:
def to_dict(self) -> dict:
return {
**self.extra,
"id": self.id,
"start": round(self.start, 2),
"end": round(self.end, 2),
@@ -98,6 +100,69 @@ class Segment:
}
def _serialize_words(words: Sequence[Word]) -> list[dict]:
"""Word objects → the ``{text, start, end}`` dicts persisted on segments.
Per-word timing is kept on each segment (``Segment.extra["words"]``, so
``to_dict`` carries it onto the job) to drive the karaoke hardsub export.
"""
return [
{"text": w.text, "start": round(w.start, 3), "end": round(w.end, 3)}
for w in words
]
def _merge_segment_extra(target: Segment, incoming: Segment, *, prepend: bool) -> None:
"""Preserve editor metadata when cleanup folds ``incoming`` into ``target``."""
# Word lists must CONCATENATE in text order (the setdefault below would
# otherwise adopt the incoming list wholesale when the target has none,
# then double it). Capture both sides before setdefault runs.
raw_target_words = target.extra.get("words")
raw_incoming_words = incoming.extra.get("words")
for key, value in incoming.extra.items():
target.extra.setdefault(key, value)
target_words = raw_target_words if isinstance(raw_target_words, list) else []
incoming_words = raw_incoming_words if isinstance(raw_incoming_words, list) else []
if target_words or incoming_words:
target.extra["words"] = (
incoming_words + target_words if prepend else target_words + incoming_words
)
def joined(left: object, right: object) -> str:
return _clean(f"{left or ''} {right or ''}")
target_original = target.extra.get("text_original")
incoming_original = incoming.extra.get("text_original")
if target_original is not None or incoming_original is not None:
target.extra["text_original"] = (
joined(incoming_original, target_original)
if prepend
else joined(target_original, incoming_original)
)
raw_target_translations = target.extra.get("translations")
raw_incoming_translations = incoming.extra.get("translations")
target_translations = raw_target_translations if isinstance(raw_target_translations, dict) else {}
incoming_translations = (
raw_incoming_translations if isinstance(raw_incoming_translations, dict) else {}
)
if target_translations or incoming_translations:
merged = {}
languages = {
*target_translations.keys(),
*incoming_translations.keys(),
}
for language in languages:
target_text = target_translations.get(language)
incoming_text = incoming_translations.get(language)
merged[language] = (
joined(incoming_text, target_text)
if prepend
else joined(target_text, incoming_text)
)
target.extra["translations"] = merged
def _clean(text: str) -> str:
return _WS.sub(" ", (text or "").strip())
@@ -191,7 +256,10 @@ def _build_segments_from_words(words: Sequence[Word]) -> List[Segment]:
if not text:
buf = []
return
segments.append(Segment(start=buf_start, end=buf[-1].end, text=text))
segments.append(Segment(
start=buf_start, end=buf[-1].end, text=text,
extra={"words": _serialize_words(buf)},
))
buf = []
if not force:
buf_start = 0.0
@@ -249,6 +317,7 @@ def _build_segments_from_words(words: Sequence[Word]) -> List[Segment]:
start=buf_start,
end=left_buf[-1].end,
text=_clean(" ".join(x.text for x in left_buf)),
extra={"words": _serialize_words(left_buf)},
))
buf = list(right_buf)
buf_start = right_buf[0].start
@@ -317,12 +386,14 @@ def _merge_short(segments: List[Segment]) -> List[Segment]:
i += 1
continue
if target is prev:
_merge_segment_extra(prev, s, prepend=False)
prev.text = _clean(prev.text + " " + s.text)
prev.end = max(prev.end, s.end)
segments.pop(i)
did_merge = True
continue
if target is nxt:
_merge_segment_extra(nxt, s, prepend=True)
nxt.text = _clean(s.text + " " + nxt.text)
nxt.start = min(nxt.start, s.start)
segments.pop(i)
@@ -360,6 +431,7 @@ def _stitch_adjacent_shorts(segments: List[Segment]) -> List[Segment]:
and b.duration <= STITCH_DUR
and combined_dur <= MAX_DUR
):
_merge_segment_extra(a, b, prepend=False)
a.text = _clean(a.text + " " + b.text)
a.end = b.end
segments.pop(i + 1)
@@ -386,6 +458,11 @@ def clean_up_segments(segments: List[dict]) -> List[dict]:
text=_clean(str(s.get("text", ""))),
speaker_id=str(s.get("speaker_id") or "Speaker 1"),
id=str(s.get("id") or uuid.uuid4().hex[:8]),
extra={
key: value
for key, value in s.items()
if key not in {"id", "start", "end", "text", "speaker_id"}
},
))
except (TypeError, ValueError):
continue
@@ -429,11 +506,23 @@ def _apply_scene_cuts(segments: List[Segment], scene_cuts: Iterable[float]) -> L
or (remaining.end - cut) < MIN_DUR
):
continue
# Segment text is the joined word texts, so a whitespace-boundary
# text split maps exactly onto a word-count split of the list.
words = remaining.extra.get("words")
left_extra: dict = {}
right_extra: dict = {}
if isinstance(words, list) and words:
n_left = len(left_text.split())
if n_left and len(words) > n_left:
left_extra = {"words": words[:n_left]}
right_extra = {"words": words[n_left:]}
out.append(Segment(
start=remaining.start, end=cut, text=left_text, speaker_id=remaining.speaker_id,
extra=left_extra,
))
remaining = Segment(
start=cut, end=remaining.end, text=right_text, speaker_id=remaining.speaker_id,
extra=right_extra,
)
out.append(remaining)
return out
@@ -665,6 +754,10 @@ def _resplit_core(
piece["text"] = text
piece["start"] = s0 if k == 0 else ws[0].start
piece["end"] = s1 if k == n_runs - 1 else ws[-1].end
# dict(seg) copied the WHOLE segment's word list into every piece;
# each piece keeps only its own run's words (karaoke burn-in).
if "words" in piece:
piece["words"] = _serialize_words(ws)
if label:
piece["speaker_id"] = label
if piece_no > 0:
+38
View File
@@ -254,6 +254,31 @@ def get_text(key: str, default: Optional[str] = None) -> Optional[str]:
return default
def get_text_state(key: str) -> tuple[bool, str]:
"""Return ``(is_present, value)`` without hiding storage failures.
Rollback snapshots must distinguish a missing row from an unreadable
database. ``get_text`` deliberately collapses those cases for ordinary
preference reads, so transactional callers use this strict variant.
"""
if key == _TOKEN_KEY or key.startswith(_SECRET_PREFIX):
raise ValueError(
"get_text_state refuses to read an encrypted secret row; "
"use get_hf_token()/get_secret() for secrets"
)
from core.db import db_conn
with db_conn() as conn:
row = conn.execute(
"SELECT value FROM settings WHERE key = ?", (key,)
).fetchone()
if row is None:
return False, ""
if row[0] is None:
return True, ""
return True, str(row[0])
def set_text(key: str, value: str) -> None:
"""Persist a non-encrypted text value into the settings table.
@@ -274,6 +299,19 @@ def set_text(key: str, value: str) -> None:
)
def clear_text(key: str) -> None:
"""Remove a non-encrypted text setting, preserving a missing-row default."""
if key == _TOKEN_KEY or key.startswith(_SECRET_PREFIX):
raise ValueError(
"clear_text refuses to delete an encrypted secret row; "
"use clear_hf_token()/clear_secret() for secrets"
)
from core.db import db_conn
with db_conn() as conn:
conn.execute("DELETE FROM settings WHERE key = ?", (key,))
# ── Phase 4 Plan 04-01 (GGUF-04): per-engine quant override ────────────────
#
# Settings row "gguf_quant_override" holds either:
+117 -30
View File
@@ -110,7 +110,7 @@ class SherpaModelSpec:
# the same HF tree API on 2026-08-07 — not estimated. Every one of the seven
# was wrong before, and in both directions, which is worse than uniformly
# optimistic: the two Parakeets under-reported by ~3.8x (0.18 -> 0.67 GB),
# so the recommended default quietly downloaded four times what the picker
# so installing v3 quietly downloaded four times what the picker
# promised on a metered or small-disk machine; but the two low-RAM
# zipformers OVER-reported by ~3x (0.128 -> 0.044), making the fallback
# models look bulkier than the heavyweights they exist to rescue users
@@ -129,7 +129,6 @@ _MODELS: dict[str, SherpaModelSpec] = {
kind="offline-transducer",
size_gb=0.67,
languages="25 European languages",
recommended=True,
heavy=True,
model_type="nemo_transducer",
files={
@@ -223,6 +222,7 @@ _MODELS: dict[str, SherpaModelSpec] = {
kind="offline-whisper",
size_gb=0.104,
languages="90+ languages (auto-detect)",
recommended=True,
files={
"encoder": "tiny-encoder.int8.onnx",
"decoder": "tiny-decoder.int8.onnx",
@@ -231,7 +231,7 @@ _MODELS: dict[str, SherpaModelSpec] = {
),
}
DEFAULT_MODEL_ID = "sherpa-parakeet-tdt-v3"
DEFAULT_MODEL_ID = "sherpa-whisper-tiny"
# repo_id → model id, so the model-store list (keyed by repo_id) can be
# enriched with the dictation metadata, and so capture can map either key.
@@ -261,6 +261,23 @@ def sherpa_available() -> tuple[bool, str]:
return True, "ready"
except ImportError as e:
return False, f"sherpa-onnx not installed: {e}. Install with: uv add sherpa-onnx"
except Exception as e: # noqa: BLE001 — an availability probe must fail closed
# Native wheel failures surface as OSError/RuntimeError rather than
# ImportError (missing DLL/dylib/so, loader or runtime init failure) —
# but the set is open-ended: an extension module is free to raise
# anything at init. This is an availability question, so ANY failure to
# import means "not available", never an exception escaping to the
# caller. SherpaDictationBackend.is_available() calls this directly and
# capture_ws.ws_transcribe calls that without a guard, so an unexpected
# type here took the WebSocket down instead of falling back (#1610).
return False, f"sherpa-onnx unavailable ({type(e).__name__}): {e}"
def _usable_model_file(path: str) -> bool:
try:
return os.path.isfile(path) and os.path.getsize(path) > 0
except OSError:
return False
def _resolve_model_dir(spec: SherpaModelSpec, *, download: bool = True) -> str:
@@ -271,41 +288,114 @@ def _resolve_model_dir(spec: SherpaModelSpec, *, download: bool = True) -> str:
Restricts the fetch to the exact int8 assets we pin via ``allow_patterns``
so we never pull the bundled fp32 weights or test wavs.
"""
from huggingface_hub import constants as hf_constants
from huggingface_hub import snapshot_download
from services.hf_revisions import installed_revision, revision_for
from services.hf_revisions import revision_for
wanted = list(spec.files.values())
cache_dir = _live_hub_cache_dir()
# Probe the revision an existing installation actually resolved. Older
# releases followed ``main`` and may therefore have a different snapshot;
# retaining it preserves offline upgrades. Any network fetch still uses
# the reviewed immutable pin.
installed = installed_revision(spec.repo_id, hf_constants.HF_HUB_CACHE)
try:
return snapshot_download(
repo_id=spec.repo_id,
revision=installed,
local_files_only=True,
allow_patterns=wanted,
)
except Exception:
if not download:
raise
installed = _installed_snapshot(spec)
if installed:
return installed
if not download:
raise FileNotFoundError(f"No complete cached snapshot for {spec.repo_id}")
# A Windows cache can retain a snapshot entry whose target blob vanished,
# or a zero-byte ONNX placeholder left by an interrupted download. Hub may
# then treat that entry as already materialized and return the same broken
# snapshot. Repair those entries before asking for another download so the
# recognizer never receives a path to a file that does not resolve (#1733).
from services.hf_cache_repair import (
find_dangling_entries,
repair_repo_cache,
repo_cache_dir,
)
if find_dangling_entries(repo_cache_dir(spec.repo_id, cache_dir)):
repair = repair_repo_cache(spec.repo_id, cache_dir)
installed = _installed_snapshot(spec)
if installed:
return installed
if not repair.get("ok"):
logger.warning(
"sherpa dictation: cache repair for %s failed: %s",
spec.repo_id,
repair.get("error") or repair.get("outcome") or "unknown error",
)
logger.info("sherpa dictation: downloading %s on first use", spec.repo_id)
return snapshot_download(
snapshot = snapshot_download(
repo_id=spec.repo_id,
revision=revision_for(spec.repo_id),
allow_patterns=wanted,
cache_dir=cache_dir,
)
missing = [
name for name in wanted
if not _usable_model_file(os.path.join(snapshot, name))
]
if not missing:
return snapshot
# Verify after the Hub reports success. This catches hosts where a broken
# snapshot entry short-circuits snapshot_download. The generic repair
# removes only broken entries, preserves blobs, and retries the immutable
# installed revision.
repair = repair_repo_cache(spec.repo_id, cache_dir)
installed = _installed_snapshot(spec)
if installed:
return installed
detail = repair.get("error") or repair.get("outcome") or "repair did not restore them"
raise FileNotFoundError(
f"Sherpa model cache is incomplete for {spec.repo_id}; missing "
f"{', '.join(missing)}. Cache repair failed: {detail}. Reinstall this "
"model from Model Catalogue."
)
def _live_hub_cache_dir() -> str:
"""The effective hub root, evaluated after Settings restores the env."""
direct = os.environ.get("HF_HUB_CACHE") or os.environ.get("HUGGINGFACE_HUB_CACHE")
if direct:
return os.path.expanduser(direct)
home = os.environ.get("HF_HOME") or os.path.expanduser("~/.cache/huggingface")
return os.path.join(os.path.expanduser(home), "hub")
def _installed_snapshot(spec: SherpaModelSpec) -> str | None:
"""Complete snapshot for the recorded revision in the live cache."""
from services.hf_revisions import installed_revision
cache_dir = _live_hub_cache_dir()
revision = installed_revision(spec.repo_id, cache_dir)
snapshot = os.path.join(
cache_dir,
"models--" + spec.repo_id.replace("/", "--"),
"snapshots",
revision,
)
if all(
_usable_model_file(os.path.join(snapshot, filename))
for filename in spec.files.values()
):
return snapshot
return None
def is_installed(spec: SherpaModelSpec) -> bool:
"""True if every pinned asset is already present in the HF cache."""
try:
d = _resolve_model_dir(spec, download=False)
except Exception:
return False
return all(os.path.isfile(os.path.join(d, f)) for f in spec.files.values())
"""True if the recorded cached snapshot contains every pinned asset.
Do not use ``snapshot_download(local_files_only=True)`` for this probe.
``huggingface_hub.constants.HF_HUB_CACHE`` is fixed when that module is
first imported, while VoiceStudio can restore its cache directory later
from the durable user settings. Resolve the live root and the recorded
revision ourselves so readiness and loading cannot disagree after a cache
move, desktop relaunch, or stale snapshot (#1707).
"""
return _installed_snapshot(spec) is not None
# ── Recognizers ──────────────────────────────────────────────────────────────
@@ -397,13 +487,10 @@ def build_online_recognizer(spec: SherpaModelSpec, *, download: bool = True):
# transcribe the same bytes. It is a defect inside sherpa-onnx that the app
# cannot fix by configuration.
#
# The curated default therefore cannot be trusted to WORK just because it is
# installed — and which platforms are affected is not knowable up front, so
# hard-coding a different default per OS would only be a guess. Instead the app
# learns from what it observes: when a session hears real speech and the model
# returns nothing, that model is demoted on THIS machine and stops being
# selected. Self-correcting wherever the breakage actually is, and a no-op
# everywhere it isn't.
# Installation alone therefore cannot prove that a recognizer works. When a
# session hears real speech and the model returns nothing, that model is
# demoted on this machine and stops being selected. This self-corrects wherever
# the decoder defect appears and is a no-op everywhere it does not.
#: prefs key holding the list of model ids demoted on this machine.
PREF_SILENT_MODELS = "dictation.silent_models"
+71 -35
View File
@@ -60,6 +60,7 @@ from pathlib import Path
from typing import Callable, Optional
from core.config import DATA_DIR
from core.contained_subprocess import OwnedPopen, WindowsJobPopen, spawn_owned
logger = logging.getLogger("omnivoice.sidecar_install")
@@ -108,9 +109,18 @@ class SidecarSpec:
weights_repo_id: Optional[str] = None # HF repo downloaded into <checkout>/<weights_subdir>
weights_revision: Optional[str] = None # reviewed HF commit
weights_subdir: str = "checkpoints"
weights_config_name: str = "config.yaml" # required model config inside weights_subdir
# Model-config filenames accepted inside weights_subdir. A tuple, not a
# single name: IndexTTS 2.5's weights repo ships config.yaml, but installs
# predating #1611 were only usable after hand-renaming it to
# config_v2_5.yaml, and those must keep working without a reinstall.
weights_config_names: tuple[str, ...] = ("config.yaml",)
docs_path: str = "docs/engines" # where the manual-install fallback lives
required_bytes: int = 12 * _GIB # conservative source+venv+weights estimate for preflight
weights_bytes: Optional[int] = None
dependency_bytes: Optional[int] = None
potentially_shared_bytes: Optional[int] = None
temporary_free_bytes: Optional[int] = None
disk_confidence: str = "unknown"
# Called after a successful install/uninstall so the engine's memoised
# venv resolution re-probes (import inside the lambda — never at module load).
invalidate: Callable[[], None] = field(default=lambda: None)
@@ -146,12 +156,17 @@ SPECS: dict[str, SidecarSpec] = {
weights_repo_id="IndexTeam/IndexTTS-2.5",
weights_revision="d0aa86e75bb6f3437f3831e95056fa72842d89ef",
weights_subdir="checkpoints",
weights_config_name="config_v2_5.yaml",
weights_config_names=("config.yaml", "config_v2_5.yaml"),
docs_path="docs/engines/indextts.md",
# ~0.1 GB source + up to ~6 GB venv (torch + transformers<5) +
# ~6 GB weights. Deliberately conservative; the preflight subtracts
# whatever a partial install already put on disk.
required_bytes=12 * _GIB,
weights_bytes=6 * _GIB,
dependency_bytes=6 * _GIB,
potentially_shared_bytes=None,
temporary_free_bytes=12 * _GIB,
disk_confidence="estimated",
invalidate=_indextts_invalidate,
installed_probe=_indextts_installed,
),
@@ -307,6 +322,25 @@ def _dir_size_bytes(path: Path) -> int:
return total
def _preserved_install_bytes(spec: SidecarSpec, checkout: Path) -> tuple[int, int]:
"""Return bytes preserved for the final install and dependency peak.
Resumable weights reduce the final download requirement, but they do not
reduce uv's separate environment-build peak. Only source and a usable
existing venv count against that peak.
"""
if not _source_present(spec, checkout):
return 0, 0
weights_dir = checkout / spec.weights_subdir
weights = _dir_size_bytes(weights_dir) if spec.weights_repo_id else 0
venv_dir = checkout / ".venv"
venv = _dir_size_bytes(venv_dir)
source = max(0, _dir_size_bytes(checkout) - weights - venv)
usable_venv = venv if _venv_python(venv_dir).is_file() else 0
return source + usable_venv + weights, source + usable_venv
def disk_free_bytes(path: Path) -> int:
"""Free bytes on the volume backing *path* (nearest existing ancestor).
Never raises; 0 when the volume can't be probed."""
@@ -329,8 +363,17 @@ def disk_space_error(spec: SidecarSpec) -> Optional[str]:
root = managed_root(spec)
# A preserved predecessor is not a partial copy of the new install: the
# upgrade needs its full space until the new sidecar is verified.
already = _dir_size_bytes(managed_checkout(spec))
remaining = max(0, spec.required_bytes - already)
checkout = managed_checkout(spec)
# Credit only bytes the later steps preserve. An invalid layout or revision
# marker makes _step_fetch_source delete the whole checkout.
preserved, dependency_peak_credit = _preserved_install_bytes(spec, checkout)
remaining = max(0, spec.required_bytes - preserved)
if spec.temporary_free_bytes is not None:
# Resumable model weights are unrelated to uv's dependency-build peak.
remaining = max(
remaining,
max(0, spec.temporary_free_bytes - dependency_peak_credit),
)
free = disk_free_bytes(root)
if free <= 0:
return None # can't probe → never block on missing information
@@ -879,11 +922,11 @@ def _weights_present(spec: SidecarSpec) -> bool:
actual = marker[:2] if len(marker) >= 2 else marker + [""]
if actual != expected:
return False
return _weights_floor_ok(wdir, config_name=spec.weights_config_name)
return _weights_floor_ok(wdir, config_names=spec.weights_config_names)
def _weights_floor_ok(wdir: Path, *, config_name: str = "config.yaml") -> bool:
if not (wdir / config_name).is_file():
def _weights_floor_ok(wdir: Path, *, config_names: tuple[str, ...] = ("config.yaml",)) -> bool:
if not any((wdir / name).is_file() for name in config_names):
return False
floor = 5 * 1024 * 1024
try:
@@ -969,7 +1012,7 @@ def _step_fetch_weights(spec: SidecarSpec, job: dict) -> None:
hf_progress.unregister_listener(listener_id)
hf_progress.current_repo_id.reset(repo_token)
if not _weights_floor_ok(wdir, config_name=spec.weights_config_name):
if not _weights_floor_ok(wdir, config_names=spec.weights_config_names):
raise _StepError(
"Weight download finished but no plausible weight files were found — "
"the download was likely interrupted.",
@@ -994,6 +1037,11 @@ def _step_persist(spec: SidecarSpec, job: dict) -> None:
# ── Subprocess runner with live log capture ────────────────────────────────
def _install_containment_kwargs() -> dict:
"""Nested process-group/Job ownership is supplied by ``spawn_owned``."""
return {}
def _run_logged(job: dict, argv: list[str], *, timeout: float,
env: "dict[str, str] | None" = None) -> int:
"""Run *argv*, streaming combined stdout+stderr lines into the job log.
@@ -1008,13 +1056,12 @@ def _run_logged(job: dict, argv: list[str], *, timeout: float,
killed child a blocking ``for line in proc.stdout`` on this thread
would hang past the timeout waiting for pipe EOF.
"""
popen_kwargs: dict = {}
if os.name == "posix":
# New session → we can kill the whole process group on timeout
# instead of only the direct child.
popen_kwargs["start_new_session"] = True
# ``spawn_owned`` creates the local timeout group/Job before the operation
# starts. POSIX links it to backend death through a control pipe; Windows
# retains a kill-on-close Job handle in this backend process.
popen_kwargs = _install_containment_kwargs()
try:
proc = subprocess.Popen(
proc = spawn_owned(
argv,
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
@@ -1049,29 +1096,18 @@ def _run_logged(job: dict, argv: list[str], *, timeout: float,
def _kill_tree(proc: "subprocess.Popen") -> None:
"""Kill the child and its whole process tree, on every platform.
POSIX: the child was started in its own session, so SIGKILL the group.
Windows: ``proc.kill()`` only terminates the direct child a git/uv
helper it spawned would keep running (and writing into the checkout)
past our timeout so use ``taskkill /T`` to fell the tree.
"""
if os.name == "posix":
import signal
"""Kill an operation through its stable nested group/Job owner."""
if isinstance(proc, (OwnedPopen, WindowsJobPopen)):
# The retained supervisor/process-group or nested Job is the stable
# per-operation owner. Do not fall back to a direct PID kill.
proc.kill()
try:
os.killpg(proc.pid, signal.SIGKILL)
proc.wait(timeout=5)
except subprocess.TimeoutExpired:
return
except (ProcessLookupError, PermissionError, OSError):
pass # group already gone / not ours — fall through to plain kill
else: # Windows
try:
subprocess.run(
["taskkill", "/F", "/T", "/PID", str(proc.pid)],
capture_output=True, timeout=15,
)
return
except (OSError, subprocess.SubprocessError):
pass # taskkill unavailable/failed — fall through to plain kill
return
# A test double or a legacy caller without the nested owner can only be
# stopped through its stable direct-process handle.
try:
proc.kill()
except OSError:
+50 -24
View File
@@ -32,9 +32,9 @@ Threat-model summary (see Plan 02-01 frontmatter):
AUTH-05 installed (``HFTokenRedactor``) on the root logger.
T-02-04 compromised sidecar emitting unexpected ops: parent allowlist
``PARENT_INBOUND_OPS`` rejects everything else.
T-02-05 Tauri group-kill scope: ``start_new_session=True`` on Unix
and ``CREATE_NEW_PROCESS_GROUP`` on Windows isolate the
sidecar's process group.
T-02-05 nested containment: a retained POSIX supervisor process group or
Windows Job owns each engine operation, while still permitting
independent timeout teardown and cleanup on backend death.
"""
from __future__ import annotations
@@ -56,6 +56,7 @@ from typing import Optional
import numpy as np
import torch
from core.contained_subprocess import spawn_owned
from services.tts_backend import TTSBackend
logger = logging.getLogger("omnivoice.subprocess_backend")
@@ -362,6 +363,7 @@ class SubprocessBackend(TTSBackend):
# be a different class object from the one the subclass closed over.
# A duck-typed marker survives that.
_is_subprocess_isolated: bool = True
spawn_ready_timeout_s: float = SPAWN_READY_TIMEOUT_S
# Generation happens in the sidecar: parent-side accelerator counters
# can't see its allocations (see TTSBackend.runs_out_of_process).
@@ -383,6 +385,10 @@ class SubprocessBackend(TTSBackend):
def __init__(self) -> None:
self._proc: Optional[subprocess.Popen] = None
# A failed bounded reap must retain ownership and forbid reuse. This
# lock is separate from _lock: the receive owner joins its watchdog.
self._timeout_quarantine: list[subprocess.Popen] = []
self._timeout_quarantine_lock = threading.Lock()
# Single lock serialises spawn + every send/recv pair so two threads
# can't interleave half-frames on the same pipe.
self._lock = threading.Lock()
@@ -449,6 +455,10 @@ class SubprocessBackend(TTSBackend):
def _spawn(self) -> None:
"""Launch the sidecar if not already running. Blocks on the ready
handshake. Caller must hold self._lock."""
if not self._retry_timeout_cleanup():
raise RuntimeError(
f"{self.id} sidecar is still stopping after a timeout; retry once it exits"
)
if self._proc is not None and self._proc.poll() is None:
return # already up
@@ -470,13 +480,6 @@ class SubprocessBackend(TTSBackend):
"env": env,
"bufsize": 0, # unbuffered binary pipes
}
# Process-group isolation so the Tauri lib.rs group-kill in shutdown
# doesn't escape into other children. See T-02-05.
if sys.platform == "win32":
kwargs["creationflags"] = subprocess.CREATE_NEW_PROCESS_GROUP
else:
kwargs["start_new_session"] = True
# `venv_python()` resolves the engine's interpreter, and on a cold
# first run that is not cheap: it spawns each candidate to import the
# engine (bounded, but tens of seconds on a slow disk), and if none is
@@ -509,7 +512,7 @@ class SubprocessBackend(TTSBackend):
self.id, Path(python_path).name, Path(script_path).name,
)
try:
self._proc = subprocess.Popen([python_path, script_path], **kwargs)
self._proc = spawn_owned([python_path, script_path], **kwargs)
except OSError as exc:
raise InvalidBinaryError(
python_path,
@@ -530,7 +533,7 @@ class SubprocessBackend(TTSBackend):
# Block on the ready handshake. A sidecar that fails to emit ready
# within SPAWN_READY_TIMEOUT_S is killed and the failure is raised.
try:
frame = self._recv_with_timeout(SPAWN_READY_TIMEOUT_S)
frame = self._recv_with_timeout(self.spawn_ready_timeout_s)
except Exception:
self._force_kill()
raise
@@ -547,6 +550,7 @@ class SubprocessBackend(TTSBackend):
"""Idempotent. Sends {op:shutdown}; falls back to terminate/kill."""
proc = self._proc
if proc is None:
self._retry_timeout_cleanup()
return
try:
try:
@@ -582,6 +586,7 @@ class SubprocessBackend(TTSBackend):
pass
finally:
self._proc = None
self._retry_timeout_cleanup()
def _force_kill(self) -> None:
"""Internal: kill a sidecar that never reached the ready state."""
@@ -782,38 +787,59 @@ class SubprocessBackend(TTSBackend):
return msg
def _recv_with_timeout(self, timeout_s: float) -> Optional[dict]:
"""Recv that aborts if the sidecar goes silent.
"""Read one frame, finishing timeout cleanup before the caller can retry.
Implemented by polling the proc for liveness with a deadline. We
don't block on a `select` of the pipe because Windows can't select
on subprocess pipes keeping the implementation cross-platform
means a simpler polling loop here.
A watchdog closes the pipe on timeout; Windows cannot select on pipes.
EOF alone does not prove the owned process/supervisor has exited.
"""
# On Unix we could use selectors; on Windows the pipe is not
# selectable. Use a watchdog thread that kills the sidecar on
# timeout — that triggers EOF on stdout, so _recv returns None
# and the caller raises.
watchdog = threading.Timer(timeout_s, self._timeout_kill)
proc = self._proc
watchdog = threading.Timer(timeout_s, self._timeout_kill, args=(proc,))
watchdog.daemon = True
watchdog.start()
try:
return self._recv()
finally:
watchdog.cancel()
# cancel() cannot stop an already-running callback. Finish its
# bounded reap before another receive or generation starts.
watchdog.join()
self._touch() # any reply (or attempt) counts as recent activity
def _timeout_kill(self) -> None:
proc = self._proc
def _timeout_kill(self, proc: Optional[subprocess.Popen]) -> None:
"""Kill only the child this receive captured, then reap its owner."""
if proc is None:
return
logger.error("[%s] sidecar exceeded recv timeout; killing", self.id)
try:
logger.error(
"[%s] sidecar exceeded recv timeout; killing",
self.id,
)
proc.kill()
except Exception:
# A raced exit can make kill fail, but its owner still needs reaping.
pass
try:
proc.wait(timeout=2)
except Exception:
# Do not discard a possibly live owner, or replace a newer _proc.
with self._timeout_quarantine_lock:
if not any(item is proc for item in self._timeout_quarantine):
self._timeout_quarantine.append(proc)
else:
with self._timeout_quarantine_lock:
self._timeout_quarantine = [
item for item in self._timeout_quarantine if item is not proc
]
def _retry_timeout_cleanup(self) -> bool:
"""Retry bounded cleanup, retaining every owner that could still be live."""
with self._timeout_quarantine_lock:
pending = tuple(self._timeout_quarantine)
for proc in pending:
self._timeout_kill(proc)
with self._timeout_quarantine_lock:
return not self._timeout_quarantine
# ── stderr drain ───────────────────────────────────────────────────────
+88
View File
@@ -112,6 +112,13 @@ _FULL_NAME_TO_CODE = {
"vietnamese": "vi",
"kazakh": "kz",
"standard arabic": "ar",
# Below: inert for num2words (absent from _NUM2WORDS_LANGS, which reads
# digits natively for these scripts), present so _plain_lang_code can
# resolve them for the digit-range rule.
"korean": "ko",
"japanese": "ja",
"chinese": "zh",
"mandarin chinese": "zh",
}
# ISO codes whose num2words locale name differs.
@@ -178,6 +185,82 @@ def _num2words_lang(language: Optional[str]) -> Optional[str]:
return None
def _plain_lang_code(language: Optional[str]) -> Optional[str]:
"""Resolve a request language to a bare ISO code, with no num2words gate.
:func:`_num2words_lang` answers "may I call num2words for this?" and so
returns ``None`` for ko/ja/zh/th/vi. Rules that are not num2words-backed
need the code itself, which is what this returns.
"""
if not language:
return None
s = str(language).strip().lower()
if not s or s == "auto":
return None
code = _FULL_NAME_TO_CODE.get(s)
if code:
return code
m = _ISO_CODE_RE.match(s)
if m:
return _ISO_ALIASES.get(m.group(1), m.group(1))
return None
# ── Digit ranges ─────────────────────────────────────────────────────────────
# "20~30" loses its separator at the engine and reads as ONE number: OmniVoice
# says "이십삼" (23) for "20~30초". Speak the separator instead. Verified by
# rendering each form and transcribing it back (ko, OmniVoice):
# "20~30초" → heard "23초" ✗
# "20에서 30초" → heard "20에서 30초" ✓
# Only the tilde family is rewritten — those are unambiguously range marks
# between digits. An ASCII hyphen is left alone on purpose: it also spells
# dates, phone numbers and product codes, where "to" would be wrong.
#: Spacing is part of the form, not decoration: a Korean postposition binds to
#: the numeral ("20에서 30"), Japanese and Chinese set no spaces at all, and
#: English needs them on both sides.
_RANGE_FORM = {
"ko": "{a}에서 {b}",
"ja": "{a}から{b}",
"zh": "{a}{b}",
"en": "{a} to {b}",
}
#: ASCII tilde, wave dash, fullwidth tilde — Japanese and Korean IMEs emit the
#: latter two, so all three have to match.
#:
#: Match complete signed/decimal endpoints; reject partial numbers and product
#: codes while allowing adjacent CJK units. Guard all tilde forms so malformed
#: chains cannot be partially rewritten, including when their separators have
#: whitespace around them.
_RANGE_MARKS = "~\u301c\uff5e"
_RANGE_ENDPOINT = r"[+-]?(?:\d{1,6}(?:\.\d{1,6})?|\.\d{1,6})"
_NUM_RANGE_RE = re.compile(
rf"(?<![\d.,A-Za-z+{_RANGE_MARKS}-])({_RANGE_ENDPOINT})"
rf"\s*[{_RANGE_MARKS}]\s*({_RANGE_ENDPOINT})"
rf"(?![\d.,A-Za-z+{_RANGE_MARKS}-])"
)
def _speak_number_ranges(text: str, lang: str) -> str:
"""Speak complete tilde ranges only for languages with a verified form."""
form = _RANGE_FORM.get(lang)
if not form:
return text
def replace(match: re.Match) -> str:
before, after = match.start() - 1, match.end()
while before >= 0 and text[before].isspace():
before -= 1
while after < len(text) and text[after].isspace():
after += 1
if ((before >= 0 and text[before] in _RANGE_MARKS)
or (after < len(text) and text[after] in _RANGE_MARKS)):
return match.group(0)
return form.format(a=match.group(1), b=match.group(2))
return _NUM_RANGE_RE.sub(replace, text)
# ── Universal safety filters (all languages) ─────────────────────────────────
# Zero-width & bidi controls, C0/C1 controls (except \t \n \r), BOM, U+FFFD.
@@ -487,6 +570,11 @@ def normalize_text(text: str, language: Optional[str] = None) -> str:
if not text:
return text or ""
out = _safety_filters(text)
# Runs outside the num2words gate below: ko/ja/zh keep their digits (that
# gate returns None for them) but still need the range mark spoken.
plain = _plain_lang_code(language)
if plain:
out = _outside_brackets(out, lambda t: _speak_number_ranges(t, plain))
lang = _num2words_lang(language)
if lang:
if lang in _ABBREV_COMPILED:
+41 -18
View File
@@ -4,7 +4,7 @@ Resolution priority (highest → lowest):
1. app `settings_store.get_hf_token()` (encrypted in SQLite)
2. env `HF_TOKEN` or the legacy `HUGGING_FACE_HUB_TOKEN` env var
3. hf-cli `huggingface_hub.get_token()` (canonical ~/.cache/huggingface/token)
3. hf-cli the selected local Hub token file (`HF_TOKEN_PATH`)
For each candidate, the resolver calls `huggingface_hub.whoami(token=...)`
to verify the token is live; any HTTP error (401, 403, network) skips to
@@ -20,6 +20,7 @@ from __future__ import annotations
import hashlib
import logging
import os
from pathlib import Path
import threading
import time
from dataclasses import dataclass
@@ -46,7 +47,7 @@ class SourceState:
set: bool
masked: Optional[str]
whoami_user: Optional[str]
whoami_ok: bool
whoami_ok: Optional[bool]
# ── module-level cache ────────────────────────────────────────────────────
@@ -79,19 +80,26 @@ def _read_app() -> Optional[str]:
return None
def _clean_token(value: Optional[str]) -> Optional[str]:
if not value:
return None
return value.replace("\r", "").replace("\n", "").strip() or None
def _read_env() -> Optional[str]:
# HF docs explicitly accept either name; user may have either exported.
val = os.environ.get("HF_TOKEN") or os.environ.get("HUGGING_FACE_HUB_TOKEN")
return val or None
return _clean_token(val)
def _read_hf_cli() -> Optional[str]:
try:
import huggingface_hub
tok = huggingface_hub.get_token()
return tok or None
from huggingface_hub import constants
return _clean_token(Path(constants.HF_TOKEN_PATH).read_text(encoding="utf-8"))
except FileNotFoundError:
return None
except Exception:
logger.exception("huggingface_hub.get_token failed")
logger.warning("Could not read the local Hugging Face token file")
return None
@@ -184,17 +192,17 @@ def on_401(active_source: Source) -> Optional[ResolvedToken]:
return resolve(skip=frozenset({active_source}))
def state() -> dict:
def state(*, validate: bool = False) -> dict:
"""Return one SourceState per priority position so the Settings UI can
render the cascade table. Includes a masked token + whoami result;
never includes the raw token."""
never includes the raw token. Reads are local unless validation is explicitly requested."""
rows: list[SourceState] = []
active: Optional[Source] = None
for source in _PRIORITY:
token = _READERS[source]()
if token:
username = _validate(source, token)
ok = username is not None
username = _validate(source, token) if validate else None
ok = (username is not None) if validate else None
rows.append(SourceState(
source=source,
set=True,
@@ -249,15 +257,30 @@ def save_app_token(token: str) -> None:
invalidate_cache()
def clear_hf_cli_tokens() -> None:
"""Remove recognized Hub token files without refreshing or revoking tokens."""
from core.config import HF_CLI_TOKEN_PATHS
from huggingface_hub import constants
# Hub's active path remains authoritative if imported before app config.
paths = set(HF_CLI_TOKEN_PATHS) | {constants.HF_TOKEN_PATH}
failed = False
for token_path in paths:
path = Path(token_path)
for target in (path, path.parent / "stored_tokens"):
try:
target.unlink(missing_ok=True)
except OSError:
failed = True
invalidate_cache()
if failed:
raise OSError("Could not clear all local Hugging Face token files")
def clear_app_token(also_clear_hf_cli: bool = False) -> None:
"""Remove from the encrypted settings store; optionally also call
`huggingface_hub.logout()` to clear the canonical HF file."""
"""Clear the encrypted app token, optionally recognized local Hub files."""
from services import settings_store
settings_store.clear_hf_token()
if also_clear_hf_cli:
try:
import huggingface_hub
huggingface_hub.logout()
except Exception:
logger.exception("huggingface_hub.logout failed (non-fatal)")
clear_hf_cli_tokens()
invalidate_cache()
+197 -11
View File
@@ -341,6 +341,44 @@ class TTSBackend(ABC):
Engines that don't support this will ignore the parameter.
"""
def generate_batch(
self,
texts: list[str],
*,
ref_audio=None,
ref_text=None,
instruct=None,
language=None,
duration=None,
speed=1.0,
**extras,
) -> list[torch.Tensor]:
"""Synthesize several utterances, preserving the single-item contract.
Engines with a native batch forward pass override this method. The
default keeps every existing adapter correct while giving callers one
stable seam and per-item keyword handling.
"""
if not texts:
return []
def _item(value, index):
return value[index] if isinstance(value, list) else value
return [
self.generate(
text,
ref_audio=_item(ref_audio, index),
ref_text=_item(ref_text, index),
instruct=_item(instruct, index),
language=_item(language, index),
duration=_item(duration, index),
speed=_item(speed, index),
**extras,
)
for index, text in enumerate(texts)
]
# ── Lifecycle (Phase 2 will enforce per-engine overrides) ──────────────
#
# Today every backend lazily loads its weights on first `generate()` and
@@ -369,6 +407,13 @@ class TTSBackend(ABC):
# entirely (it drives the shared model_manager singleton).
_MODEL_ATTRS: tuple[str, ...] = ("_model", "_tts")
def execution_evidence_loaded(self) -> bool:
"""Whether this instance has live model state worth reporting."""
if self.runs_out_of_process:
proc = getattr(self, "_proc", None)
return proc is not None and proc.poll() is None
return any(getattr(self, attr, None) is not None for attr in self._MODEL_ATTRS)
def unload(self) -> None:
"""Release the heavy model this backend holds, and free device caches.
@@ -521,7 +566,12 @@ def _get_clone_prompt(
):
"""Return a cached/precomputed ``VoiceClonePrompt`` for
(ref_audio, ref_text, preprocess_prompt), or ``None`` to fall back to the
inline ref path. Never raises.
inline ref path.
Raises only on a device OOM that survives a cache-drop retry (#1790): the
inline path is the same allocation on the same device, so falling back to
it after an OOM cannot succeed and has been observed taking the whole
process down instead. Every other failure still falls back silently.
``store=False`` still *reads* the cache (a hit is free) but never inserts:
it exists for single-use references a dub's per-segment ref clips are each
@@ -551,10 +601,43 @@ def _get_clone_prompt(
ref_audio, ref_text=ref_text, preprocess_prompt=preprocess_prompt
)
except Exception as e: # noqa: BLE001 — fall back, never break synthesis
logger.warning(
"voice-clone prompt precompute failed; using inline ref: %s", e
)
return None
# #1790/#1777: a GPU OOM is the one failure this fallback cannot
# absorb. `generate()`'s inline ref path runs the SAME encode on the
# SAME device — the docstring above says so, because producing
# identical output is the point — so returning None after an OOM
# guarantees a second OOM moments later, on a device with even less
# headroom than the first attempt found. Both reporters' backends
# then died with a Windows access violation (exit code
# -1073741819) seconds after this exact log line, mid-generation on
# a GPU that had just refused an 86 MiB allocation.
#
# An OOM here is also the most recoverable kind: the allocator is
# typically holding reserved-but-unallocated blocks (#1790's own
# log reports 90 MiB reserved against an 86 MiB request). Drop them
# and try once more. If it still will not fit, raise — the failure
# layer turns a device OOM into the actionable GPU_OOM message
# ("close other GPU-heavy apps or unload models…"), which is a far
# better answer than walking into a native fault.
from core.failure import is_gpu_oom
if is_gpu_oom(e):
logger.warning(
"voice-clone prompt precompute hit a device OOM (%s) — "
"releasing allocator caches and retrying once", e,
)
try:
from services.model_manager import free_vram
free_vram()
except Exception: # noqa: BLE001 — reclaim is best-effort
logger.debug("VRAM reclaim before OOM retry failed", exc_info=True)
prompt = model.create_voice_clone_prompt(
ref_audio, ref_text=ref_text, preprocess_prompt=preprocess_prompt
)
else:
logger.warning(
"voice-clone prompt precompute failed; using inline ref: %s", e
)
return None
if store:
_prompt_disk_save(key, prompt)
if not store:
@@ -631,7 +714,7 @@ class OmniVoiceBackend(TTSBackend):
id = "omnivoice"
display_name = "VoiceStudio (k2-fsa/OmniVoice, 600+ languages)"
gpu_compat = ("cuda", "mps", "cpu")
gpu_compat = ("cuda", "rocm", "mps", "cpu")
# Derived from the pool's own per-job budget (_GPU_VRAM_PER_JOB_GB = 5.0 in
# model_manager, itself measured from the ~1.6 GB forward + autoregressive
# decode and the co-loaded WhisperX on the clone path), plus room for the
@@ -717,6 +800,73 @@ class OmniVoiceBackend(TTSBackend):
)
return audios[0]
def generate_batch(self, texts: list[str], **kw) -> list[torch.Tensor]:
"""Use OmniVoice's native variable-length batch generation.
Batch callers pass per-item language, duration, speed and reference
lists. Reusable clone prompts are prepared once and handed to the
model together; an incomplete prompt batch falls back to the proven
single-item path instead of changing synthesis semantics.
"""
self._ensure_loaded()
if not texts:
return []
def _items(value):
if isinstance(value, list):
return value
return [value] * len(texts)
def _item_kwargs(index):
return {
key: value[index] if isinstance(value, list) else value
for key, value in kw.items()
}
ref_audios = _items(kw.get("ref_audio"))
ref_texts = _items(kw.get("ref_text"))
cache_ref = bool(kw.get("cache_ref", True))
preprocess_prompt = bool(kw.get("preprocess_prompt", True))
prompts = []
if any(ref_audios):
for ref_audio, ref_text in zip(ref_audios, ref_texts):
if not ref_audio:
prompts = []
break
prompt = _get_clone_prompt(
self._model,
ref_audio,
ref_text,
preprocess_prompt,
store=cache_ref,
)
if prompt is None:
prompts = []
break
prompts.append(prompt)
if any(ref_audios) and len(prompts) != len(texts):
return [self.generate(text, **_item_kwargs(i))
for i, text in enumerate(texts)]
gen_kw = dict(
language=kw.get("language"),
instruct=kw.get("instruct"),
duration=kw.get("duration"),
speed=kw.get("speed", 1.0),
denoise=kw.get("denoise", True),
postprocess_output=kw.get("postprocess_output", True),
num_step=kw.get("num_step", 16),
guidance_scale=kw.get("guidance_scale", 2.0),
preprocess_prompt=preprocess_prompt,
)
if prompts:
gen_kw["voice_clone_prompt"] = prompts
else:
gen_kw["ref_audio"] = None
gen_kw["ref_text"] = None
return self._model.generate(text=texts, **gen_kw)
def unload(self) -> None:
"""Release the OmniVoice model (MM2-02). OmniVoice shares the singleton
owned by ``model_manager``, so dropping our local ref isn't enough — we
@@ -2186,9 +2336,9 @@ _INSTALL_HINTS: dict[str, str] = {
"omnivoice-gguf":"Bundled — runs the C++ omnivoice-tts binary in bin/. Quants download lazily from Serveurperso/OmniVoice-GGUF on first generate.",
"supertonic3": "uv sync --extra supertonic (CPU-only ONNX, 31 langs, ~400 MB model on first use; OpenRAIL-M model license)",
"pockettts": "uv sync --extra pockettts (Kyutai, CPU-only, ~100 MB model on first use; MIT code + CC-BY-4.0 weights; HF-gated, review terms and set HF_TOKEN)",
"moss-tts-v15": "git clone OpenMOSS/MOSS-TTS + set OMNIVOICE_MOSS_TTS_V15_DIR (own venv, transformers==5.0; 8B, ~16 GB weights; CUDA/CPU, no MPS; Apache-2.0)",
"moss-tts-v15": "git clone OpenMOSS/MOSS-TTS + set OMNIVOICE_MOSS_TTS_V15_DIR (own venv, transformers==5.0; 8B, ~16 GB weights; CUDA/ROCm/XPU/NPU/CPU, no MPS; Apache-2.0)",
"dots-tts": "git clone rednote-hilab/dots.tts + set OMNIVOICE_DOTS_TTS_DIR (own venv, transformers==4.57; 2B, ~9 GB weights; CUDA/CPU, Linux/macOS only — no Windows; Apache-2.0)",
"confucius4-tts":"git clone netease-youdao/Confucius4-TTS + set OMNIVOICE_CONFUCIUS4_TTS_DIR (own Python 3.10 venv; 14-lang cross-lingual zero-shot clone; ~5 GB weights auto-download; CUDA/CPU, no MPS; Apache-2.0)",
"confucius4-tts":"git clone netease-youdao/Confucius4-TTS + set OMNIVOICE_CONFUCIUS4_TTS_DIR (own Python 3.10 venv; 14-lang cross-lingual zero-shot clone; ~5 GB weights auto-download; CUDA/ROCm/XPU/NPU/CPU, no MPS; Apache-2.0)",
}
@@ -2259,7 +2409,7 @@ def list_backends() -> list[dict]:
"one_click_install": bool, # services.sidecar_install can provision it in-app
"last_error": Optional[str], # cached most-recent failure
"isolation_mode": "in-process" | "subprocess",
"gpu_compat": list[str], # subset of {cuda, rocm, mps, xpu, cpu}
"gpu_compat": list[str], # subset of {cuda, rocm, mps, xpu, npu, cpu}
"supports_cloning": Optional[bool], # True/False from the class attr; None when
# model-dependent (property, e.g. mlx-audio)
"effective_device": str, # device this engine uses on THIS host
@@ -2287,12 +2437,15 @@ def list_backends() -> list[dict]:
# Routing is host-aware but the host caps are constant per process, so probe
# ONCE here and resolve each engine's effective device against the same caps.
from core.device_caps import detect_host_caps
from services.engine_disk_usage import disk_summary_for
from services.engine_evidence import snapshot as execution_snapshot
from services.engine_routing import routing_fields
caps = detect_host_caps()
installable = _sidecar_installable_ids()
out: list[dict] = []
for bid, cls in _REGISTRY.items():
cls = _effective_backend_class(bid, cls, caps.family)
try:
ok, msg = cls.is_available()
except Exception:
@@ -2319,6 +2472,12 @@ def list_backends() -> list[dict]:
# descriptor, not a bool, so report None (= model-dependent) there
# instead of an always-truthy false positive.
_clone = getattr(cls, "supports_cloning", True)
routing = routing_fields(gpu_compat, caps, getattr(cls, "min_vram_gb", 0.0))
loaded_instance = None
if _active_instance_id == bid:
loaded_instance = _active_instance
if loaded_instance is None:
loaded_instance = _ENGINE_INSTANCES.get(cls)
out.append({
"id": bid,
"display_name": cls.display_name,
@@ -2338,6 +2497,7 @@ def list_backends() -> list[dict]:
# in-app (Settings renders an Install button instead of leading
# with the manual setup snippet).
"one_click_install": bid in installable,
"disk_usage": disk_summary_for(bid),
"last_error": _LAST_ERRORS.get(bid),
"isolation_mode": isolation,
"gpu_compat": list(gpu_compat),
@@ -2345,7 +2505,14 @@ def list_backends() -> list[dict]:
"min_vram_gb": getattr(cls, "min_vram_gb", 0.0) or None,
# effective_device / routing_status / routing_reason (scrubbed);
# the reason now also carries the under-provisioned-GPU caveat.
**routing_fields(gpu_compat, caps, getattr(cls, "min_vram_gb", 0.0)),
**routing,
"execution_evidence": execution_snapshot(
engine_id=bid,
engine_cls=cls,
instance=loaded_instance,
routing=routing,
caps=caps,
),
})
# #981: mlx-audio multiplexes 7+ curated models behind one backend id
# — surface the roster + the currently-active pick so Settings can
@@ -2366,10 +2533,29 @@ def list_backends() -> list[dict]:
return out
def _effective_backend_class(
backend_id: str,
backend_cls: type[TTSBackend],
host_family: str | None = None,
) -> type[TTSBackend]:
"""Resolve host-specific containment without changing the configured id."""
if backend_id != "omnivoice":
return backend_cls
if host_family is None:
from core.device_caps import detect_host_caps
host_family = detect_host_caps().family
if host_family != "mps":
return backend_cls
from engines.omnivoice_subprocess import OmniVoiceMPSSubprocessBackend
return OmniVoiceMPSSubprocessBackend
def get_backend_class(backend_id: str) -> type[TTSBackend]:
if backend_id not in _REGISTRY:
raise ValueError(f"Unknown TTS backend: {backend_id!r}. Known: {list(_REGISTRY)}")
return _REGISTRY[backend_id]
return _effective_backend_class(backend_id, _REGISTRY[backend_id])
def cloning_capable_engine_ids() -> list[str]:
+266 -41
View File
@@ -20,12 +20,17 @@ Usage:
from __future__ import annotations
import contextlib
import logging
import math
import os
import threading
import time
import torch
from pathlib import Path
from typing import Optional
import torch
from core.prefs import resolve
logger = logging.getLogger("omnivoice.watermark")
@@ -37,6 +42,20 @@ _detector = None
_audioseal_available: Optional[bool] = None
# Monotonic stamp of the last embed/detect, for the idle release below.
_last_used = 0.0
# Per-model locks for the lazy builds below: the startup prefetch thread
# races the first embed, and both must share ONE build (a double load doubles
# the cold-start cost the prefetch exists to hide). One lock PER MODEL — a
# single shared lock made the ~42s generator prefetch block unrelated detector
# loads and the idle reaper behind it. release_idle_models acquires both, in
# this fixed order (nothing else nests them, so no cycle is possible).
_generator_lock = threading.Lock()
_detector_lock = threading.Lock()
# True when the generator exists ONLY because the startup prefetch built it
# and no embed/detect has used it since. The idle reaper grants one extra
# idle window before dropping such a model, so a first synthesis at minute
# 20 still finds it warm (code-review finding 2 on the prefetch PR).
_prefetched_unused = False
# 16-bit message: "OM" in ASCII = 0x4F 0x4D = 0100_1111 0100_1101
# This is our signature — every VoiceStudio-generated audio carries it.
@@ -50,6 +69,86 @@ OMNI_MESSAGE = [0, 1, 0, 0, 1, 1, 1, 1, 0, 1, 0, 0, 1, 1, 0, 1]
_CHUNK_SECONDS = 30
# AudioSeal vendors moshi's ``@torch_compile_lazy`` on SEANetEncoder.forward,
# so the first EMBED — not the model load, which prefetch already warms —
# calls torch.compile and drops into Inductor's C++ codegen. On hosts whose
# C++ toolchain can't serve Inductor that compile raises CppCompileError, the
# embed fail-opens, and audio ships unmarked: a macOS arm64 deployment lost
# provenance marking on 10/10 takes while paying 30-40 s for the first failed
# compile and 5-8 s for each later one (#1615).
#
# The compile is pure cost even where it succeeds. Measured on an M3 (5 s of
# 24 kHz audio, three consecutive embeds): compiled 9.70 / 0.26 / 0.23 s vs
# eager 0.30 / 0.28 / 0.27 s — a ~10 s first-embed tax to save ~0.03 s per
# later embed, on CPU work that is already bounded by the 30 s chunk loop.
# So watermarking runs eager on every platform.
def _moshi_compile_module():
"""AudioSeal's vendored moshi compile switch module, or None.
Resolved per call rather than at import: ``_check_available()`` is what
guarantees audioseal is importable, and it runs later than this module.
"""
try:
from audioseal.libs.moshi.utils import compile as moshi_compile
except Exception: # noqa: BLE001 — any import shape change degrades, not crashes
return None
return moshi_compile
_eager_lock = threading.Lock()
#: Depth of nested/concurrent eager scopes, and the switch value to put back
#: when the last one exits. One dict rather than two module scalars: the
#: fields are only meaningful together, and only under _eager_lock.
_eager_state: dict = {"depth": 0, "saved": None}
_eager_guard_warned = False
def _warn_missing_eager_guard() -> None:
global _eager_guard_warned
_eager_guard_warned = True
logger.info(
"audioseal's no_compile switch is unavailable — watermarking may run "
"through torch.compile and pay (or fail) an Inductor C++ compile (#1615)."
)
@contextlib.contextmanager
def _eager_audioseal():
"""Run the AudioSeal model eagerly, restoring the switch on the way out.
Upstream's own ``no_compile()`` saves and restores ``_compile_disabled``
per call, which is not safe when two watermark calls overlap: the first to
exit restores False while the second is still mid-embed, handing it back
the compile this whole fix exists to avoid. So the flag is reference
counted here it goes True on the outermost entry and only comes back on
the outermost exit rather than serializing embeds behind a lock, which
would cost real throughput on concurrent generations.
Degrades to a plain call if a future audioseal drops the helper
(``tests/test_watermark_no_torch_compile_1615.py`` fails loudly on that
upgrade rather than letting the compile creep back in).
"""
moshi = _moshi_compile_module()
if moshi is None:
if not _eager_guard_warned:
_warn_missing_eager_guard()
yield
return
with _eager_lock:
if _eager_state["depth"] == 0:
_eager_state["saved"] = moshi._compile_disabled
_eager_state["depth"] += 1
moshi._compile_disabled = True
try:
yield
finally:
with _eager_lock:
_eager_state["depth"] -= 1
if _eager_state["depth"] == 0:
moshi._compile_disabled = _eager_state["saved"]
_eager_state["saved"] = None
def _iter_chunks(audio: torch.Tensor, sample_rate: int):
"""Yield ≤ ~_CHUNK_SECONDS slices of (batch, channels, samples) audio
along the time axis. A sub-second tail is folded into the previous chunk
@@ -77,28 +176,82 @@ def _check_available() -> bool:
return _audioseal_available
def _get_generator():
"""Lazy-load the AudioSeal generator model."""
global _generator, _last_used
_last_used = time.monotonic()
if _generator is None:
from audioseal import AudioSeal
_generator = AudioSeal.load_generator("audioseal_wm_16bits")
_generator.eval()
logger.info("AudioSeal generator loaded (16-bit message mode)")
return _generator
def _get_generator(mark_prefetched: bool = False):
"""Lazy-load the AudioSeal generator model.
Owns the idle-reaper grace in ONE critical section: the startup prefetch
claims it (``mark_prefetched=True``) only when THIS call builds the model,
and every other call (a real embed) consumes it no call-site blocks, no
window between two lock scopes where the claim could land on an
already-used model.
"""
global _generator, _last_used, _prefetched_unused
with _generator_lock:
_last_used = time.monotonic()
if _generator is None:
from audioseal import AudioSeal
_generator = AudioSeal.load_generator("audioseal_wm_16bits")
_generator.eval()
logger.info("AudioSeal generator loaded (16-bit message mode)")
_prefetched_unused = mark_prefetched
elif not mark_prefetched:
_prefetched_unused = False
return _generator
def _get_detector():
"""Lazy-load the AudioSeal detector model."""
global _detector, _last_used
_last_used = time.monotonic()
if _detector is None:
from audioseal import AudioSeal
_detector = AudioSeal.load_detector("audioseal_detector_16bits")
_detector.eval()
logger.info("AudioSeal detector loaded (16-bit message mode)")
return _detector
with _detector_lock:
_last_used = time.monotonic()
if _detector is None:
from audioseal import AudioSeal
_detector = AudioSeal.load_detector("audioseal_detector_16bits")
_detector.eval()
logger.info("AudioSeal detector loaded (16-bit message mode)")
return _detector
def _generator_checkpoint_cached() -> bool:
"""Return whether AudioSeal can warm without contacting Hugging Face.
AudioSeal 0.2 stores the checkpoint in ``<cache>/audioseal`` even though
it uses huggingface_hub to fetch it. Keep startup local-first: an ordinary
boot may consume that file, but must never turn prefetch into a download.
"""
cache_root = os.environ.get("AUDIOSEAL_CACHE_DIR") or os.environ.get(
"XDG_CACHE_HOME"
)
root = Path(cache_root).expanduser() if cache_root else Path.home() / ".cache"
return (root / "audioseal" / "generator_base.pth").is_file()
def prefetch_generator(*, allow_download: bool = False) -> None:
"""Warm the AudioSeal generator eagerly (startup background thread).
The first ``mark_synthetic`` otherwise pays the audioseal import plus the
generator load inline measured at ~42 s on a cold filesystem (2026-08-17
macOS deployment), serialized inside the first synthesis and 3 s short of
a 90 s client timeout. Warming here overlaps that span with the TTS model
load. No-op when watermarking is off or audioseal is absent; a failure
logs and leaves the lazy path to retry on first embed. Default startup is
also cache-only; a download is allowed only when the user explicitly set
``OMNIVOICE_PRELOAD_WATERMARK=1``.
"""
try:
if not will_mark():
logger.debug("Watermark prefetch skipped (disabled or audioseal absent)")
return
if not allow_download and not _generator_checkpoint_cached():
logger.info("Watermark prefetch skipped: AudioSeal checkpoint is not cached")
return
_get_generator(mark_prefetched=True)
logger.info("AudioSeal generator prefetched in the background")
except Exception:
logger.warning(
"Watermark prefetch failed; the first embed will retry inline",
exc_info=True,
)
def release_idle_models(idle_seconds: float, *, now: Optional[float] = None) -> bool:
@@ -114,14 +267,28 @@ def release_idle_models(idle_seconds: float, *, now: Optional[float] = None) ->
Returns True if anything was released. Never raises: this runs from the
idle reaper, which must survive it.
"""
global _generator, _detector
if _generator is None and _detector is None:
return False
stamp = time.monotonic() if now is None else float(now)
if stamp - _last_used < idle_seconds:
return False
_generator = None
_detector = None
global _generator, _detector, _prefetched_unused
with _generator_lock, _detector_lock:
if _generator is None and _detector is None:
return False
stamp = time.monotonic() if now is None else float(now)
if stamp - _last_used < idle_seconds:
return False
if _prefetched_unused:
# The startup prefetch built the generator and nothing has used
# it yet. Drop the grace (one extra idle window only) instead of
# the model, so a first synthesis shortly after boot still finds
# it warm — the exact scenario the prefetch exists for.
_prefetched_unused = False
logger.info(
"Idle watermark models are prefetch-warmed but unused; "
"granting one more idle window before releasing."
)
return False
# Under the locks so a release racing the prefetch or a first embed
# can't wipe a model the lazy path just built.
_generator = None
_detector = None
logger.info("Idle timeout reached. Released the AudioSeal watermark models.")
return True
@@ -200,6 +367,62 @@ def mark_synthetic(
return marked
async def mark_synthetic_async(
waveform: torch.Tensor,
sample_rate: int,
*,
context: str,
force: bool = False,
timeout: float | None = None,
) -> torch.Tensor:
"""Dispatch marking without letting a draining pool lose finished audio."""
import asyncio
import functools
from services.model_manager import (
GpuJobTimeoutError,
GpuPoolBusyError,
get_watermark_pool,
run_on_gpu_pool_guarded,
)
try:
pool = get_watermark_pool()
except RuntimeError:
logger.warning("Watermark skipped while the prior worker is shutting down")
return waveform
job = functools.partial(
mark_synthetic, waveform, sample_rate, context=context, force=force
)
try:
if timeout is not None:
return await run_on_gpu_pool_guarded(
job, what="Audio watermark", timeout=timeout, executor=pool
)
return await asyncio.get_running_loop().run_in_executor(pool, job)
except (GpuJobTimeoutError, GpuPoolBusyError):
# Watermarking is provenance best-effort: a typed execution overrun or
# queue saturation must not discard synthesis that already completed.
logger.warning("Watermark skipped after its bounded dispatch expired")
return waveform
except asyncio.CancelledError:
# A queued future is cancelled during pool teardown. Caller-driven
# cancellation while the pool is live must retain normal semantics.
if not pool.is_shutdown():
raise
logger.warning("Watermark skipped while the pool is shutting down")
return waveform
except RuntimeError:
# Shutdown may begin after admission but before Executor.submit().
# Preserve unrelated worker failures; only lifecycle rejection is
# fail-open because finished synthesis must not be lost to teardown.
if not pool.is_shutdown():
raise
logger.warning("Watermark skipped while the pool is shutting down")
return waveform
@torch.no_grad()
def embed_watermark(
waveform: torch.Tensor,
@@ -243,13 +466,14 @@ def embed_watermark(
# AudioSeal operates at 16kHz internally; it handles resampling, but
# we need to inform it of the source rate for correct embedding.
watermarked = torch.cat(
[
generator(seg, sample_rate=sample_rate, message=msg)
for seg in _iter_chunks(audio, sample_rate)
],
dim=-1,
)
with _eager_audioseal():
watermarked = torch.cat(
[
generator(seg, sample_rate=sample_rate, message=msg)
for seg in _iter_chunks(audio, sample_rate)
],
dim=-1,
)
# Restore original shape
if len(original_shape) == 2:
@@ -260,7 +484,7 @@ def embed_watermark(
return watermarked
except Exception as e:
logger.warning("Watermark embedding failed (passing through original): %s", e)
logger.warning("Watermark embedding failed (passing through original): %s", e, exc_info=True)
return waveform
@@ -307,12 +531,13 @@ def detect_watermark(
# embedding does, and a splice where only part of the file is
# VoiceStudio audio still registers (a whole-file average would dilute it).
best_conf, decoded_msg = -1.0, None
for seg in _iter_chunks(audio, sample_rate):
result = detector.detect_watermark(seg, sample_rate=sample_rate, message_threshold=0.5)
seg_conf = float(result[0]) if isinstance(result, tuple) else 0.0
if seg_conf > best_conf:
best_conf = seg_conf
decoded_msg = result[1] if isinstance(result, tuple) and len(result) > 1 else None
with _eager_audioseal():
for seg in _iter_chunks(audio, sample_rate):
result = detector.detect_watermark(seg, sample_rate=sample_rate, message_threshold=0.5)
seg_conf = float(result[0]) if isinstance(result, tuple) else 0.0
if seg_conf > best_conf:
best_conf = seg_conf
decoded_msg = result[1] if isinstance(result, tuple) and len(result) > 1 else None
confidence = max(best_conf, 0.0)
# Decode message bits
@@ -337,7 +562,7 @@ def detect_watermark(
}
except Exception as e:
logger.warning("Watermark detection failed: %s", e)
logger.warning("Watermark detection failed: %s", e, exc_info=True)
return {
"is_watermarked": False,
"confidence": 0.0,
+1
View File
@@ -0,0 +1 @@
"""Dependency-free client for VoiceStudio's local speech platform."""
+278
View File
@@ -0,0 +1,278 @@
"""CLI/module bridge for terminals, editor extensions, and agent hooks.
The desktop app must be running for native dictation control. Batch
transcription can also target a standalone or remote VoiceStudio backend.
"""
from __future__ import annotations
import argparse
import ipaddress
import json
import mimetypes
import os
from pathlib import Path
import secrets
import sys
from typing import Any
from urllib import error, request
from urllib.parse import urlsplit
DEFAULT_CONTROL_URL = "http://127.0.0.1:3902"
DEFAULT_ENGINE_URL = "http://127.0.0.1:3900"
class SpeechClientError(RuntimeError):
pass
class _RejectCredentialRedirect(request.HTTPRedirectHandler):
def redirect_request(self, req, fp, code, msg, headers, newurl): # noqa: ARG002
raise SpeechClientError("VoiceStudio refused a credentialed redirect")
def _join_url(base_url: str, path: str) -> str:
return f"{base_url.rstrip('/')}/{path.lstrip('/')}"
def _decode_error(exc: error.HTTPError) -> str:
try:
body = exc.read().decode("utf-8", errors="replace")
except Exception:
body = ""
try:
detail = json.loads(body)
except (TypeError, json.JSONDecodeError):
detail = body.strip()
return f"HTTP {exc.code}: {detail or exc.reason}"
def _is_loopback_host(host: str | None) -> bool:
if not host:
return False
if host.lower() == "localhost":
return True
try:
return ipaddress.ip_address(host).is_loopback
except ValueError:
return False
def _open(req: request.Request, timeout: float = 300.0) -> tuple[bytes, str]:
target = urlsplit(req.full_url)
scheme = target.scheme.lower()
if scheme not in {"http", "https"}:
raise SpeechClientError("VoiceStudio URLs must use http:// or https://")
credentialed = bool(req.get_header("Authorization"))
if credentialed and scheme != "https" and not _is_loopback_host(target.hostname):
raise SpeechClientError("Remote VoiceStudio credentials require https://")
try:
opener = (
request.build_opener(_RejectCredentialRedirect())
if credentialed
else request.build_opener()
)
with opener.open(req, timeout=timeout) as response: # noqa: S310
return response.read(), response.headers.get("Content-Type", "")
except error.HTTPError as exc:
raise SpeechClientError(_decode_error(exc)) from exc
except error.URLError as exc:
raise SpeechClientError(f"VoiceStudio is unavailable: {exc.reason}") from exc
def _json_request(method: str, url: str, payload: Any | None = None) -> Any:
data = None if payload is None else json.dumps(payload).encode("utf-8")
headers = {"Accept": "application/json"}
if data is not None:
headers["Content-Type"] = "application/json"
body, _ = _open(request.Request(url, data=data, headers=headers, method=method), timeout=10.0)
try:
return json.loads(body)
except json.JSONDecodeError as exc:
raise SpeechClientError("VoiceStudio returned invalid JSON") from exc
def _encode_multipart(
*,
filename: str,
audio: bytes,
fields: dict[str, str],
boundary: str | None = None,
) -> tuple[bytes, str]:
boundary = boundary or f"voicestudio-{secrets.token_hex(16)}"
marker = boundary.encode("ascii")
parts: list[bytes] = []
for name, value in fields.items():
parts.extend(
[
b"--" + marker + b"\r\n",
f'Content-Disposition: form-data; name="{name}"\r\n\r\n'.encode(),
value.encode("utf-8"),
b"\r\n",
]
)
safe_filename = Path(filename).name.replace('"', "") or "audio.wav"
content_type = mimetypes.guess_type(safe_filename)[0] or "application/octet-stream"
if Path(safe_filename).suffix.lower() in {".wav", ".wave"}:
content_type = "audio/wav"
parts.extend(
[
b"--" + marker + b"\r\n",
(
'Content-Disposition: form-data; name="file"; '
f'filename="{safe_filename}"\r\n'
).encode(),
f"Content-Type: {content_type}\r\n\r\n".encode(),
audio,
b"\r\n--" + marker + b"--\r\n",
]
)
return b"".join(parts), f"multipart/form-data; boundary={boundary}"
def _control(args: argparse.Namespace, action: str) -> int:
method = "GET" if action in {"status", "capabilities"} else "POST"
path = {
"status": "/v1/status",
"capabilities": "/v1/capabilities",
"start": "/v1/dictation/start",
"stop": "/v1/dictation/stop",
"toggle": "/v1/dictation/toggle",
}[action]
result = _json_request(method, _join_url(args.control_url, path))
print(json.dumps(result, ensure_ascii=False, indent=2))
return 0
def _read_audio(path: str, stdin_filename: str) -> tuple[bytes, str]:
if path == "-":
return sys.stdin.buffer.read(), stdin_filename
audio_path = Path(path)
try:
return audio_path.read_bytes(), audio_path.name
except OSError as exc:
display_name = path.replace("\\", "/").rsplit("/", 1)[-1] or "audio input"
reason = exc.strerror or type(exc).__name__
raise SpeechClientError(f"could not read '{display_name}': {reason}") from exc
def _response_text(body: bytes, content_type: str) -> str:
decoded = body.decode("utf-8", errors="replace")
if "json" not in content_type.lower():
return decoded
try:
payload = json.loads(decoded)
except json.JSONDecodeError:
return decoded
if isinstance(payload, dict) and isinstance(payload.get("text"), str):
return payload["text"]
return decoded
def _transcribe(args: argparse.Namespace) -> int:
audio, filename = _read_audio(args.audio, args.stdin_filename)
fields = {
"model": args.model,
"response_format": args.response_format,
}
if args.language:
fields["language"] = args.language
body, content_type = _encode_multipart(filename=filename, audio=audio, fields=fields)
headers = {"Content-Type": content_type, "Accept": "application/json, text/plain"}
api_key = os.environ.get("OMNIVOICE_API_KEY", "").strip()
if api_key:
headers["Authorization"] = f"Bearer {api_key}"
output_session_id = None
if args.insert:
session = _json_request(
"POST", _join_url(args.control_url, "/v1/output/sessions")
)
output_session_id = session["session_id"]
session_needs_cleanup = output_session_id is not None
try:
response_body, response_type = _open(
request.Request(
_join_url(args.engine_url, "/v1/audio/transcriptions"),
data=body,
headers=headers,
method="POST",
)
)
if output_session_id is not None:
_json_request(
"POST",
_join_url(
args.control_url,
f"/v1/output/sessions/{output_session_id}/insert",
),
{"text": _response_text(response_body, response_type)},
)
session_needs_cleanup = False
finally:
if session_needs_cleanup:
try:
_json_request(
"DELETE",
_join_url(args.control_url, f"/v1/output/sessions/{output_session_id}"),
)
except Exception: # noqa: BLE001
# Best-effort cleanup must not replace the original failure or
# KeyboardInterrupt that brought control into this finally.
pass
sys.stdout.buffer.write(response_body)
if response_body and not response_body.endswith(b"\n"):
sys.stdout.buffer.write(b"\n")
return 0
def _parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(
prog="voicestudio-speech",
description="Control and consume VoiceStudio's local speech platform.",
)
parser.add_argument(
"--control-url",
default=os.environ.get("VOICESTUDIO_SPEECH_URL", DEFAULT_CONTROL_URL),
)
parser.add_argument(
"--engine-url",
default=os.environ.get("VOICESTUDIO_URL", DEFAULT_ENGINE_URL),
)
subparsers = parser.add_subparsers(dest="command", required=True)
for command in ("status", "capabilities", "start", "stop", "toggle"):
subparsers.add_parser(command)
transcribe = subparsers.add_parser("transcribe")
transcribe.add_argument("audio", help="audio file, or - for stdin")
transcribe.add_argument("--stdin-filename", default="audio.wav")
transcribe.add_argument("--model", default="whisper-1")
transcribe.add_argument("--language")
transcribe.add_argument(
"--format",
dest="response_format",
choices=("json", "text", "verbose_json", "srt", "vtt"),
default="text",
)
transcribe.add_argument(
"--insert",
action="store_true",
help="insert the result into the app focused when this command starts",
)
return parser
def main(argv: list[str] | None = None) -> int:
args = _parser().parse_args(argv)
try:
if args.command == "transcribe":
return _transcribe(args)
return _control(args, args.command)
except (SpeechClientError, KeyError) as exc:
print(f"voicestudio-speech: {exc}", file=sys.stderr)
return 2
if __name__ == "__main__":
raise SystemExit(main())
@@ -7,13 +7,15 @@ old silence-only guard missed it (the buzz is loud, not silent) so the garbage
was cached and served.
These tests cover the fix *without the 5 GB model / a GPU*: they drive the pure
``_spectral_flatness`` / ``_is_unusable_audio`` helpers with synthetic signals,
and assert the render constants didn't regress. The real end-to-end render is
verified manually (spectral flatness back in the speech range + Whisper ASR).
``_spectral_flatness`` / ``_is_unusable_audio`` helpers with synthetic tones and
tracked speech demo renders, and assert the render constants did not regress.
"""
from __future__ import annotations
import math
from pathlib import Path
import soundfile as sf
import pytest
@@ -39,6 +41,13 @@ def _white_noise() -> "torch.Tensor":
return 0.5 * (torch.rand(N, generator=g) * 2 - 1)
def _two_tone_buzz() -> "torch.Tensor":
"""Two inharmonic partials — the other shape a collapsed render takes."""
t = torch.arange(N, dtype=torch.float32) / SR
s = torch.sin(2 * math.pi * 180.0 * t) + 0.6 * torch.sin(2 * math.pi * 361.0 * t)
return 0.8 * s / s.abs().max()
def _speech_like() -> "torch.Tensor":
"""Broadband + harmonic + amplitude-modulated — a coarse stand-in for voiced
speech: several harmonics (formant-ish), additive noise (consonants), and a
@@ -85,6 +94,71 @@ def test_speech_like_is_usable():
assert arch._is_unusable_audio(_speech_like()) is False
# ── Threshold stays between the two things it has to separate ───────────────
# Synthetic broadband speech has much higher flatness than real voiced audio.
# Measure both sides of the threshold against actual inputs, including the
# existing demo renders that the old thresholds rejected.
def test_tonal_ceiling_is_measured_not_assumed():
"""Derive the tonal side of the margin instead of trusting a literal.
A bare constant would keep passing if `_spectral_flatness` stopped scoring
tones near zero, so measure the degenerate signals here and require the
threshold to clear the worst of them tenfold.
"""
tones = [
arch._spectral_flatness(_pure_tone(80.0)),
arch._spectral_flatness(_pure_tone(220.0)),
arch._spectral_flatness(_two_tone_buzz()),
]
assert all(t is not None for t in tones)
assert max(tones) * 10 < arch._DEGENERATE_FLATNESS
_SAMPLES = Path(__file__).resolve().parents[1] / "assets" / "samples"
_SPEECH_FIXTURES = [
"demo_voice.wav",
"demo_clone_output.wav",
*[f"voice_design/demo_voice_design_{name}.wav" for name in (
"audiobook_uk_narrator", "aussie_podcaster", "bedtime_storyteller",
"gravelly_villain", "indian_support_agent", "mandarin_sichuan", "us_news_anchor",
)],
*[f"dictation/{name}.wav" for name in (
"en_conversational", "en_technical", "fr_reservation",
)],
*[f"demo/dubbing/{name}.src.wav" for name in (
"source", "dubbed_es", "dubbed_fr", "dubbed_ja", "dubbed_zh",
)],
]
@pytest.mark.parametrize("fixture", _SPEECH_FIXTURES)
def test_real_shipped_speech_clears_quality_floor(fixture):
# These are existing tracked demo renders, not synthesized stand-ins or
# asserted measurements. Loading PCM needs neither a model nor a network.
audio, _sample_rate = sf.read(_SAMPLES / fixture, dtype="float32", always_2d=True)
speech = torch.from_numpy(audio.T)
flatness = arch._spectral_flatness(speech)
assert flatness is not None
assert flatness > arch._DEGENERATE_FLATNESS * 10
assert arch._is_unusable_audio(speech) is False
def test_flatness_is_not_clip_length_dependent():
"""Repeating a signal must not change what it measures.
The whole-clip FFT this replaced failed exactly here: its frequency
resolution grew with duration, so the same audio measured 0.0229 at 3 s and
~0 at 12 s (100% drift). Framed, the drift is under 0.1%.
"""
short = _speech_like()
long = torch.cat([short] * 4)
a, b = arch._spectral_flatness(short), arch._spectral_flatness(long)
assert a is not None and b is not None
assert abs(a - b) / a < 0.02
# ── Constants didn't regress ────────────────────────────────────────────────
def test_preview_render_constants():
# 16 steps under-converged on the social script; the fix bumped it.
+62 -7
View File
@@ -11,6 +11,7 @@ that the error message tells the user what to do.
import asyncio
import os
import sys
import threading
import time
import pytest
@@ -75,17 +76,71 @@ def test_fast_transcribe_passes_through():
pool.shutdown(wait=True)
def test_timeout_defers_abandon_cleanup_until_running_worker_finishes():
"""A timed-out native worker may still be reading request-owned inputs."""
pool = ThreadPoolExecutor(max_workers=1)
started = threading.Event()
finish = threading.Event()
cleaned = threading.Event()
def _slow():
started.set()
finish.wait(timeout=5)
assert not cleaned.is_set()
return "done"
async def _go():
with pytest.raises(ASRTimeoutError):
await run_transcribe_guarded(
pool,
_slow,
what="Convert",
timeout=0.05,
on_abandon=cleaned.set,
)
assert started.is_set()
assert not cleaned.is_set()
finish.set()
await asyncio.to_thread(cleaned.wait, 2)
assert cleaned.is_set()
try:
asyncio.run(_go())
finally:
finish.set()
pool.shutdown(wait=True)
def test_normal_completion_keeps_abandon_cleanup_with_caller():
pool = ThreadPoolExecutor(max_workers=1)
cleaned = threading.Event()
async def _go():
result = await run_transcribe_guarded(
pool,
lambda: "done",
what="Convert",
timeout=5,
on_abandon=cleaned.set,
)
assert result == "done"
assert not cleaned.is_set()
try:
asyncio.run(_go())
finally:
pool.shutdown(wait=True)
def test_timeout_error_is_a_timeouterror_subclass():
# Routers that catch broad TimeoutError (openai_compat) must also catch ours.
assert issubclass(ASRTimeoutError, TimeoutError)
def test_timeout_resets_a_resilient_pool_to_restore_capacity():
# #730: a wedged transcribe holds its GPU-pool worker forever; with a 1-2
# worker pool that starves TTS generate and surfaces as "can't reach
# backend". On timeout, run_transcribe_guarded must reset() a pool that
# supports it (the real _ResilientGpuPool) so the next submit gets a fresh
# worker — capacity restored without an app restart.
def test_timeout_does_not_overlap_an_in_process_native_worker():
# #1669: reset() cannot kill the old native thread. A fresh pool let the
# retry enter the same whisperx/CTranslate2 model concurrently and the
# process died with 0xC0000005. Keep the old worker accounted for instead.
class _FakePool(ThreadPoolExecutor):
def __init__(self):
super().__init__(max_workers=1)
@@ -105,7 +160,7 @@ def test_timeout_resets_a_resilient_pool_to_restore_capacity():
await run_transcribe_guarded(pool, _hang, what="Dub", timeout=0.2)
asyncio.run(_go())
assert pool.reset_calls == 1
assert pool.reset_calls == 0
pool.shutdown(wait=False)
+36
View File
@@ -112,6 +112,42 @@ class TestEnqueue:
job = client.get(f"/batch/jobs/{job_id}").json()
assert job["filename"] == "test.mp4"
@pytest.mark.asyncio
async def test_upload_is_persisted_in_bounded_chunks(self, batch, tmp_path):
class RecordingUpload:
def __init__(self):
self.read_sizes = []
self.remaining = b"video"
async def read(self, size):
self.read_sizes.append(size)
chunk, self.remaining = self.remaining[:size], self.remaining[size:]
return chunk
upload = RecordingUpload()
destination = tmp_path / "video.mp4"
await batch._save_upload(upload, str(destination))
assert destination.read_bytes() == b"video"
assert upload.read_sizes == [batch._UPLOAD_CHUNK_BYTES, batch._UPLOAD_CHUNK_BYTES]
@pytest.mark.asyncio
async def test_failed_upload_removes_partial_file(self, batch, tmp_path):
class FailingUpload:
calls = 0
async def read(self, _size):
self.calls += 1
if self.calls == 1:
return b"partial"
raise OSError("upload interrupted")
destination = tmp_path / "video.mp4"
with pytest.raises(OSError, match="upload interrupted"):
await batch._save_upload(FailingUpload(), str(destination))
assert not destination.exists()
class TestListJobs:
def test_empty(self, client):
+346
View File
@@ -0,0 +1,346 @@
"""Stable nested operation ownership (model-free, cross-platform seams)."""
import ctypes
import builtins
import os
import runpy
import subprocess
import sys
import threading
import time
import types
from ctypes import wintypes
from pathlib import Path
import pytest
from core import contained_subprocess as owned
class _Call:
def __init__(self, fn):
self.fn = fn
def __call__(self, *args):
return self.fn(*args)
def test_supervisor_argv_uses_entry_module_for_source_and_frozen_binary(monkeypatch):
monkeypatch.delattr(owned.sys, "frozen", raising=False)
source = owned._supervisor_argv(3, 4, ["operation"])
assert source[:2] == [sys.executable, str(Path(owned.__file__).parents[1] / "main.py")]
assert source[2:] == ["--supervise", "3", "4", "--", "operation"]
monkeypatch.setattr(owned.sys, "frozen", True, raising=False)
frozen = owned._supervisor_argv(3, 4, ["operation"])
assert frozen == [sys.executable, "--supervise", "3", "4", "--", "operation"]
def test_source_main_dispatches_supervisor_before_heavy_imports(monkeypatch):
calls = []
fake = types.ModuleType("core.contained_subprocess")
fake.supervisor_main = lambda args: calls.append(args) or 23
monkeypatch.setitem(sys.modules, "core.contained_subprocess", fake)
main_path = Path(owned.__file__).parents[1] / "main.py"
monkeypatch.setattr(
sys,
"argv",
[str(main_path), "--supervise", "3", "4", "--", "operation"],
)
original_import = builtins.__import__
def guard_heavy_import(name, *args, **kwargs):
if name == "math":
raise AssertionError("supervisor dispatch reached application imports")
return original_import(name, *args, **kwargs)
monkeypatch.setattr(builtins, "__import__", guard_heavy_import)
with pytest.raises(SystemExit, match="23"):
runpy.run_path(str(main_path), run_name="__main__")
assert calls == [["--supervise", "3", "4", "--", "operation"]]
@pytest.mark.skipif(os.name != "posix", reason="Unix drain pipe contract")
def test_drain_fd_is_explicitly_inherited_by_wrapper_but_not_operation(monkeypatch):
drain_read, drain_write = os.pipe()
monkeypatch.setenv("OMNIVOICE_DESKTOP_CONTAINED", "1")
monkeypatch.setenv("OMNIVOICE_DESKTOP_DRAIN_FD", str(drain_write))
owned.secure_backend_drain_fd()
assert not os.get_inheritable(drain_write)
implicit_probe = subprocess.check_output(
[
sys.executable,
"-c",
"import os; "
"fd=int(os.environ['OMNIVOICE_DESKTOP_DRAIN_FD']); "
"\ntry: os.fstat(fd); print('leaked')"
"\nexcept OSError: print('closed')",
],
close_fds=False,
text=True,
)
assert implicit_probe.strip() == "closed"
script = (
"import os,time; token=os.environ.get('OMNIVOICE_DESKTOP_DRAIN_FD'); "
"marker=os.environ.get('OMNIVOICE_DESKTOP_CONTAINED'); "
"\nif token is None and marker is None: state='stripped'"
"\nelse:"
"\n try: os.fstat(int(token)); state='leaked'"
"\n except OSError: state='closed'"
"\nprint(state, flush=True); time.sleep(60)"
)
proc = owned.spawn_owned(
[sys.executable, "-c", script],
stdout=subprocess.PIPE,
text=True,
)
try:
assert proc.stdout.readline().strip() == "stripped"
os.close(drain_write)
drain_write = -1
os.set_blocking(drain_read, False)
with pytest.raises(BlockingIOError):
os.read(drain_read, 1) # wrapper still holds the only writer
proc.kill()
proc.wait(timeout=5)
deadline = time.monotonic() + 2
while time.monotonic() < deadline:
try:
if os.read(drain_read, 1) == b"":
break
except BlockingIOError:
time.sleep(0.01)
else:
pytest.fail("wrapper exit did not close the desktop drain writer")
finally:
if drain_write >= 0:
os.close(drain_write)
os.close(drain_read)
if proc.poll() is None:
proc.kill()
proc.wait(timeout=5)
def test_invalid_or_missing_desktop_drain_fd_fails_safe(monkeypatch):
monkeypatch.setenv("OMNIVOICE_DESKTOP_CONTAINED", "1")
monkeypatch.setenv("OMNIVOICE_DESKTOP_DRAIN_FD", "not-an-fd")
with pytest.raises(RuntimeError, match="missing its live.*drain descriptor"):
owned.spawn_owned([sys.executable, "-c", "print('unsafe')"])
monkeypatch.delenv("OMNIVOICE_DESKTOP_DRAIN_FD")
with pytest.raises(RuntimeError, match="missing its live.*drain descriptor"):
owned.secure_backend_drain_fd()
monkeypatch.delenv("OMNIVOICE_DESKTOP_CONTAINED")
assert owned.backend_drain_fd(required=True) is None
proc = owned.spawn_owned(
[sys.executable, "-c", "print('standalone')"],
stdout=subprocess.PIPE,
text=True,
)
assert proc.stdout.readline().strip() == "standalone"
assert proc.wait(timeout=5) == 0
def test_windows_operation_is_in_kill_on_close_job_before_resume(monkeypatch):
"""The child gets no instruction before stable nested Job assignment."""
events = []
job_closed = threading.Event()
job = 99
def close_handle(handle):
value = getattr(handle, "value", handle)
events.append(("close", value))
if value == job:
job_closed.set()
return True
kernel = type("Kernel", (), {})()
kernel.AssignProcessToJobObject = _Call(
lambda assigned_job, process: events.append(("assign", assigned_job, process)) or True
)
kernel.TerminateJobObject = _Call(
lambda assigned_job, code: events.append(("terminate", assigned_job, code)) or True
)
kernel.WriteFile = _Call(
lambda handle, payload, size, written, overlap: events.append(("write", size)) or True
)
kernel.CloseHandle = _Call(close_handle)
def read_control(*_args):
job_closed.wait(2)
return False
kernel.ReadFile = _Call(read_control)
monkeypatch.setattr(owned, "_windows_job", lambda: (job, kernel, wintypes))
monkeypatch.setattr(
owned,
"_resume_windows_process",
lambda _kernel, _types, pid: events.append(("resume", pid)),
)
class Child:
_handle = 77
pid = 123
def wait(self, timeout=None):
events.append(("wait", timeout))
return 0
monkeypatch.setattr(
owned.subprocess,
"Popen",
lambda *args, **kwargs: events.append(("spawn", kwargs["creationflags"])) or Child(),
)
assert owned._supervise_windows(11, 12, ["operation.exe"]) == 0
assert job_closed.wait(1)
names = [event[0] for event in events]
assert names.index("assign") < names.index("resume") < names.index("wait")
assert names.index("wait") < names.index("terminate") < names.index("write")
def test_windows_assignment_failure_kills_suspended_unowned_child(monkeypatch):
"""A child outside the nested Job must be killed through its stable handle."""
events = []
job_closed = threading.Event()
job = 99
def close_handle(handle):
value = getattr(handle, "value", handle)
events.append(("close", value))
if value == job:
job_closed.set()
return True
kernel = type("Kernel", (), {})()
kernel.AssignProcessToJobObject = _Call(
lambda assigned_job, process: events.append(("assign", assigned_job, process))
or False
)
kernel.TerminateJobObject = _Call(
lambda assigned_job, code: events.append(("terminate", assigned_job, code)) or True
)
kernel.WriteFile = _Call(
lambda handle, payload, size, written, overlap: events.append(("write", size)) or True
)
kernel.CloseHandle = _Call(close_handle)
def read_control(*_args):
job_closed.wait(2)
return False
kernel.ReadFile = _Call(read_control)
monkeypatch.setattr(owned, "_windows_job", lambda: (job, kernel, wintypes))
monkeypatch.setattr(ctypes, "get_last_error", lambda: 5, raising=False)
class Child:
_handle = 77
pid = 123
def kill(self):
events.append(("kill",))
def wait(self, timeout=None):
events.append(("wait", timeout))
return 1
monkeypatch.setattr(
owned.subprocess,
"Popen",
lambda *args, **kwargs: events.append(("spawn", kwargs["creationflags"])) or Child(),
)
assert owned._supervise_windows(11, 12, ["operation.exe"]) == 127
assert job_closed.wait(1)
names = [event[0] for event in events]
assert names.index("assign") < names.index("terminate") < names.index("kill")
assert names.index("kill") < names.index("wait") < names.index("write")
def test_windows_direct_job_owner_assigns_before_resume(monkeypatch):
"""Windows skips the extra Python wrapper but retains pre-start Job ownership."""
events = []
job = 99
kernel = type("Kernel", (), {})()
kernel.AssignProcessToJobObject = _Call(
lambda assigned_job, process: events.append(("assign", assigned_job, process)) or True
)
kernel.TerminateJobObject = _Call(
lambda assigned_job, code: events.append(("terminate", assigned_job, code)) or True
)
kernel.CloseHandle = _Call(
lambda handle: events.append(("close", getattr(handle, "value", handle))) or True
)
monkeypatch.setattr(owned, "_windows_job", lambda: (job, kernel, wintypes))
monkeypatch.setattr(
owned,
"_resume_windows_process",
lambda _kernel, _types, pid: events.append(("resume", pid)),
)
class Child:
_handle = 77
pid = 123
args = ["operation.exe"]
stdin = None
stdout = object()
stderr = object()
returncode = None
def poll(self):
return self.returncode
def wait(self, timeout=None):
events.append(("wait", timeout))
return self.returncode
def kill(self):
events.append(("kill",))
child = Child()
def fake_popen(argv, **kwargs):
events.append(("spawn", argv, kwargs))
return child
monkeypatch.setattr(owned.subprocess, "Popen", fake_popen)
proc = owned._spawn_windows_owned(
["operation.exe"],
{
"env": {
"KEEP": "yes",
"OMNIVOICE_DESKTOP_CONTAINED": "1",
"OMNIVOICE_DESKTOP_DRAIN_FD": "42",
},
"creationflags": 0x00000200,
},
)
names = [event[0] for event in events]
assert names[:3] == ["spawn", "assign", "resume"]
spawn_argv, spawn_kwargs = events[0][1:]
assert spawn_argv == ["operation.exe"]
assert spawn_kwargs["creationflags"] == 0x08000204
assert spawn_kwargs["env"] == {"KEEP": "yes"}
assert proc.stdout is child.stdout
child.returncode = 0
assert proc.poll() == 0
assert [event[0] for event in events][-2:] == ["terminate", "close"]
def test_spawn_owned_selects_direct_windows_job_path(monkeypatch):
sentinel = object()
calls = []
monkeypatch.setattr(owned.os, "name", "nt")
monkeypatch.setattr(
owned,
"_spawn_windows_owned",
lambda argv, kwargs: calls.append((argv, kwargs)) or sentinel,
)
assert owned.spawn_owned(["sidecar.exe"], text=True) is sentinel
assert calls == [(["sidecar.exe"], {"text": True})]
@@ -0,0 +1,126 @@
"""macOS fallback for the os.waitid probe (#1656).
CPython on macOS does not expose os.waitid, so OwnedPopen's WNOWAIT dance
crashed with AttributeError on every poll after the first spawn. These tests
simulate that platform (monkeypatch os.waitid away) and pin the fallback:
poll/wait/kill must work, exit codes must be real, and an already-reaped
leader must be refused (ChildProcessError path), never signalled blind.
"""
import os
import subprocess
import sys
import time
import pytest
from core import contained_subprocess as owned
def _make_owned(argv):
cr, cw = os.pipe()
rr, rw = os.pipe()
proc = subprocess.Popen(argv, start_new_session=True)
os.close(cw)
os.close(rw) # result writer gone: _read_result falls back to wrapper rc
return owned.OwnedPopen(proc, cr, rr), proc
@pytest.fixture()
def no_waitid(monkeypatch):
monkeypatch.delattr(os, "waitid", raising=False)
def test_poll_running_then_exited_without_waitid(no_waitid):
h, _ = _make_owned([sys.executable, "-c", "import time; time.sleep(1.5)"])
try:
assert h.poll() is None, "running child must poll None"
h._proc.wait()
deadline = time.monotonic() + 5
rc = None
while rc is None and time.monotonic() < deadline:
rc = h.poll()
time.sleep(0.05)
assert rc == 0
assert h.poll() == 0
finally:
h._close_control()
if h._result_fd is not None:
os.close(h._result_fd)
def test_poll_reports_real_exit_code_without_waitid(no_waitid):
h, _ = _make_owned([sys.executable, "-c", "raise SystemExit(3)"])
try:
deadline = time.monotonic() + 5
while h.poll() is None and time.monotonic() < deadline:
time.sleep(0.05)
assert h.poll() == 3
finally:
h._close_control()
if h._result_fd is not None:
os.close(h._result_fd)
def test_wait_returns_after_kill_without_waitid(no_waitid):
h, _ = _make_owned([sys.executable, "-c", "import time; time.sleep(30)"])
try:
h.kill()
rc = h.wait(timeout=5)
assert rc != 0
finally:
h._close_control()
if h._result_fd is not None:
os.close(h._result_fd)
def test_reaped_by_own_popen_reports_code_without_waitid(no_waitid):
h, proc = _make_owned([sys.executable, "-c", "pass"])
try:
proc.wait() # reaped through OUR handle: known code, not a refusal
assert h.poll() == 0
finally:
h._close_control()
if h._result_fd is not None:
os.close(h._result_fd)
def test_foreign_reaped_leader_is_refused_without_waitid(no_waitid):
h, proc = _make_owned([sys.executable, "-c", "pass"])
try:
# Reap OUTSIDE this handle: Popen never learns the code, so poll must
# refuse (None) rather than guess or signal a maybe-reused group.
while True:
pid, _ = os.waitpid(proc.pid, os.WNOHANG)
if pid == proc.pid:
break
time.sleep(0.05)
assert h.poll() is None
finally:
h._close_control()
if h._result_fd is not None:
os.close(h._result_fd)
def test_kill_after_pid_reuse_does_not_signal_without_waitid(no_waitid, monkeypatch):
"""A foreign-reaped leader's reused numeric pid must not authorize killpg."""
import signal as _signal
h, proc = _make_owned([sys.executable, "-c", "pass"])
try:
while True:
pid, _ = os.waitpid(proc.pid, os.WNOHANG)
if pid == proc.pid:
break
time.sleep(0.05)
# Model the numeric pid being reused: kill(pid, 0) would succeed even
# though waitpid still reports that the original child is no longer
# ours. The old guard therefore reached killpg and fails this test.
monkeypatch.setattr(os, "kill", lambda _pid, _sig: None)
signalled = []
monkeypatch.setattr(os, "killpg", lambda pid, sig: signalled.append((pid, sig)))
h._signal_owned_group(_signal.SIGKILL)
assert signalled == []
finally:
h._close_control()
if h._result_fd is not None:
os.close(h._result_fd)
@@ -1,16 +1,15 @@
"""A dictation model that decodes nothing gets demoted, not re-selected forever.
`sherpa-parakeet-tdt-v3` is the curated default, and on Windows it installs
cleanly, loads without error, and returns an empty token list for clear speech
On Windows, `sherpa-parakeet-tdt-v3` installs cleanly, loads without error,
and returns an empty token list for clear speech
(both quantisations, both decoding methods, sherpa-onnx 1.13.3 and 1.13.4)
while whisper and zipformer transcribe the same bytes. The defect is inside
sherpa-onnx's NeMo-TDT decoder — unfixable from here by configuration.
Hard-coding a different default per OS would be a guess: we have evidence for
one platform only. So the app observes instead. When a session hears real
speech and the model returns nothing, that model is demoted ON THIS MACHINE and
stops being auto-selected, which self-corrects wherever the breakage actually
is and is a no-op everywhere it isn't.
Whisper Tiny is now the cross-platform default, while Parakeet remains
selectable. Runtime demotion still protects users who select a recognizer that
loads successfully but decodes nothing: it is demoted on this machine and the
next session follows the capture fallback.
These tests pin the demotion round trip and, critically, that the user can
always take back control by re-picking the model.
+2 -2
View File
@@ -1,13 +1,13 @@
"""A dictation model that decodes NOTHING must fall back, not fail silently.
Found on Windows with the curated default `sherpa-parakeet-tdt-v3`: the model
Found on Windows with `sherpa-parakeet-tdt-v3`: the model
downloads, loads with zero errors, and is correctly detected as a TDT model
(`num_durations: 5`) then returns an empty token list for clear speech.
Measured against the same 18.9s WAV, on the same machine, same sherpa-onnx:
sherpa-whisper-tiny -> "Alright, here we are. I hope that's all..."
sherpa-zipformer-en-20m -> "ANTS BOTH IN WHAT DISGUISED THIS THAT..."
parakeet-tdt-v3 (int8) -> '' <-- the curated default
parakeet-tdt-v3 (int8) -> ''
parakeet-tdt-v3 (fp32) -> ''
parakeet-tdt-v2 (int8) -> ''
+37
View File
@@ -0,0 +1,37 @@
from __future__ import annotations
import asyncio
import io
import threading
import pytest
from fastapi import UploadFile
@pytest.mark.asyncio
async def test_preview_ffmpeg_does_not_block_event_loop(monkeypatch, tmp_path):
from api.routers import dub_core
loop = asyncio.get_running_loop()
started = asyncio.Event()
release = threading.Event()
def slow_ffmpeg(*_args, **_kwargs):
loop.call_soon_threadsafe(started.set)
assert release.wait(timeout=2)
monkeypatch.setattr(dub_core, "PREVIEW_DIR", str(tmp_path))
monkeypatch.setattr(dub_core, "find_ffmpeg", lambda: "ffmpeg")
monkeypatch.setattr(dub_core.subprocess, "run", slow_ffmpeg)
upload = UploadFile(filename="preview.mp4", file=io.BytesIO(b"video"))
before = loop.time()
task = asyncio.create_task(dub_core.preview_upload(upload))
try:
await asyncio.wait_for(started.wait(), timeout=1)
assert loop.time() - before < 0.5
finally:
release.set()
result = await task
assert result["audioUrl"].endswith(".wav")
@@ -0,0 +1,71 @@
from __future__ import annotations
import asyncio
import threading
from concurrent.futures import ThreadPoolExecutor
import pytest
@pytest.mark.asyncio
async def test_abandoned_reader_keeps_adhoc_reference_until_worker_finishes(tmp_path):
from api.routers.generation import (
_TempReferenceLease,
_run_with_reference_lease,
)
from services.model_manager import run_on_gpu_pool_guarded
reference = tmp_path / "reference.wav"
reference.write_bytes(b"voice")
lease = _TempReferenceLease(str(reference))
started = threading.Event()
release_worker = threading.Event()
worker_read = threading.Event()
def read_reference():
started.set()
assert release_worker.wait(timeout=2)
assert reference.read_bytes() == b"voice"
worker_read.set()
with ThreadPoolExecutor(max_workers=1) as executor:
task = asyncio.create_task(
_run_with_reference_lease(
lease,
lambda on_abandon: run_on_gpu_pool_guarded(
read_reference,
executor=executor,
timeout=1,
on_abandon=on_abandon,
),
)
)
assert await asyncio.to_thread(started.wait, 1)
task.cancel()
cancelled = await asyncio.gather(task, return_exceptions=True)
assert isinstance(cancelled[0], asyncio.CancelledError)
lease.finish_request()
assert reference.exists()
release_worker.set()
assert await asyncio.to_thread(worker_read.wait, 1)
for _ in range(100):
if not reference.exists():
break
await asyncio.sleep(0.01)
assert not reference.exists()
def test_normal_request_deletes_adhoc_reference_immediately(tmp_path):
from api.routers.generation import _TempReferenceLease
reference = tmp_path / "reference.wav"
reference.write_bytes(b"voice")
lease = _TempReferenceLease(str(reference))
release = lease.acquire()
release()
lease.finish_request()
assert not reference.exists()
@@ -0,0 +1,70 @@
from __future__ import annotations
import asyncio
import threading
from concurrent.futures import ThreadPoolExecutor
import pytest
@pytest.mark.asyncio
async def test_abandon_callback_waits_for_running_worker_to_finish():
from services.model_manager import run_on_gpu_pool_guarded
started = threading.Event()
release = threading.Event()
cleaned = threading.Event()
def job():
started.set()
assert release.wait(timeout=2)
with ThreadPoolExecutor(max_workers=1) as executor:
task = asyncio.create_task(
run_on_gpu_pool_guarded(
job,
executor=executor,
timeout=1,
on_abandon=cleaned.set,
)
)
assert await asyncio.to_thread(started.wait, 1)
task.cancel()
cancelled = await asyncio.gather(task, return_exceptions=True)
assert isinstance(cancelled[0], asyncio.CancelledError)
assert not cleaned.is_set()
release.set()
assert await asyncio.to_thread(cleaned.wait, 1)
@pytest.mark.asyncio
async def test_queued_cancellation_releases_without_running_job():
from services.model_manager import GpuPoolBusyError, run_on_gpu_pool_guarded
hog_started = threading.Event()
release_hog = threading.Event()
cleaned = threading.Event()
queued_job_ran = threading.Event()
def hog():
hog_started.set()
assert release_hog.wait(timeout=2)
with ThreadPoolExecutor(max_workers=1) as executor:
hog_future = executor.submit(hog)
assert hog_started.wait(timeout=1)
try:
with pytest.raises(GpuPoolBusyError):
await run_on_gpu_pool_guarded(
queued_job_ran.set,
executor=executor,
timeout=1,
queue_timeout=0.05,
on_abandon=cleaned.set,
)
assert cleaned.is_set()
assert not queued_job_ran.is_set()
finally:
release_hog.set()
hog_future.result(timeout=1)
+465 -4
View File
@@ -17,18 +17,31 @@ import json
import math
import array
import base64
import io
import os
import subprocess
import sys
import time
import asyncio
from pathlib import Path
import pytest
from services.subprocess_backend import SubprocessBackend, RECV_TIMEOUT_S
from services.tts_backend import get_backend_class
from engines.omnivoice_subprocess import OmniVoiceSubprocessBackend
from services.subprocess_backend import (
RECV_TIMEOUT_S,
SubprocessBackend,
)
from services.tts_backend import OmniVoiceBackend, get_backend_class, list_backends
from engines.omnivoice_subprocess import (
OmniVoiceMPSSubprocessBackend,
OmniVoiceSubprocessBackend,
)
# ── stub sidecar (model-free) ──────────────────────────────────────────────
STUB_SIDECAR = r'''
import sys, json, struct, time, math, array, base64
import sys, os, json, struct, time, math, array, base64, subprocess
def _send(o):
b = json.dumps(o, separators=(",", ":")).encode()
@@ -60,9 +73,20 @@ while True:
sys.exit(0)
elif op == "synthesize":
t = m.get("text", "")
if t == "CRASH":
os._exit(137)
if t == "HANG":
while True: # wedge forever; the parent must hard-kill us
time.sleep(1)
if t == "HANG_CHILD":
subprocess.Popen([
sys.executable,
"-c",
"import os,time; time.sleep(1); "
"open(os.environ['OMNIVOICE_TIMEOUT_MARKER'], 'w').write('bad')",
])
while True:
time.sleep(1)
# Emit progress frames before the audio when asked, to exercise the
# parent's progress-consuming recv loop (the cold-load fix).
if t.startswith("PROG:"):
@@ -98,6 +122,80 @@ def test_registry_resolves_to_subprocess_backend():
assert get_backend_class("omnivoice-subprocess") is OmniVoiceSubprocessBackend
@pytest.mark.parametrize(
("family", "expected_name"),
[("mps", "OmniVoiceMPSSubprocessBackend"), ("cuda", "OmniVoiceBackend"),
("cpu", "OmniVoiceBackend")],
)
def test_omnivoice_is_crash_isolated_only_on_mps(monkeypatch, family, expected_name):
from core.device_caps import HostCaps
available = (family, "cpu") if family != "cpu" else ("cpu",)
monkeypatch.setattr(
"core.device_caps.detect_host_caps",
lambda: HostCaps(family=family, available_families=available),
)
resolved = get_backend_class("omnivoice")
assert resolved.__name__ == expected_name
if family != "mps":
assert resolved is OmniVoiceBackend
def test_engine_catalogue_reports_effective_mps_isolation(monkeypatch):
from core.device_caps import HostCaps
from services import tts_backend
monkeypatch.setattr(tts_backend, "_REGISTRY", {"omnivoice": OmniVoiceBackend})
monkeypatch.setattr(
"core.device_caps.detect_host_caps",
lambda: HostCaps(family="mps", available_families=("mps", "cpu")),
)
monkeypatch.setattr(
"engines.omnivoice_subprocess.OmniVoiceSubprocessBackend.is_available",
classmethod(lambda cls: (True, "ready")),
)
row = next(item for item in list_backends() if item["id"] == "omnivoice")
assert row["isolation_mode"] == "subprocess"
def test_mps_startup_does_not_preload_native_model(monkeypatch):
from core.device_caps import HostCaps
from services import model_manager
monkeypatch.setattr(
"core.device_caps.detect_host_caps",
lambda: HostCaps(family="mps", available_families=("mps", "cpu")),
)
monkeypatch.setenv("OMNIVOICE_TTS_BACKEND", "omnivoice")
monkeypatch.setattr(model_manager, "model", None)
async def fail_load():
raise AssertionError("native OmniVoice must not load in the API process on MPS")
monkeypatch.setattr(model_manager, "_load_model_with_timeout", fail_load)
asyncio.run(model_manager.preload_model())
def test_streaming_mps_path_does_not_load_native_model(monkeypatch):
from api.routers.tts_stream import _resolve_stream_backend
from services import model_manager, tts_backend
sentinel = object()
monkeypatch.setattr(tts_backend, "active_backend_id", lambda: "omnivoice")
monkeypatch.setattr(
tts_backend, "get_backend_class", lambda _id: OmniVoiceMPSSubprocessBackend,
)
monkeypatch.setattr(tts_backend, "get_active_tts_backend", lambda: sentinel)
async def fail_load():
raise AssertionError("streaming must not load native OmniVoice on MPS")
monkeypatch.setattr(model_manager, "get_model", fail_load)
assert asyncio.run(_resolve_stream_backend(None)) is sentinel
def test_is_marked_subprocess_isolated():
# list_backends() detects isolation via this duck-typed marker, not issubclass.
assert getattr(OmniVoiceSubprocessBackend, "_is_subprocess_isolated", False) is True
@@ -136,11 +234,81 @@ def test_base_default_recv_timeout_is_60s():
assert _PlainBackend().recv_timeout_s == 60.0
def test_sidecar_spawn_delegates_all_containment_to_nested_owner(monkeypatch, tmp_path):
from services import subprocess_backend as backend_module
captured = {}
class StubProcess:
stderr = io.BytesIO()
@staticmethod
def poll():
return None
def fake_spawn(argv, **kwargs):
captured.update(kwargs)
return StubProcess()
monkeypatch.setattr(_PlainBackend, "venv_python", classmethod(lambda cls: Path(sys.executable)))
monkeypatch.setattr(
_PlainBackend,
"sidecar_script",
classmethod(lambda cls: tmp_path / "stub.py"),
)
monkeypatch.setattr(backend_module, "spawn_owned", fake_spawn)
monkeypatch.setattr(backend_module, "_ensure_reaper_running", lambda: None)
backend = _PlainBackend()
monkeypatch.setattr(backend, "_recv_with_timeout", lambda _timeout: {"op": "ready"})
try:
backend._spawn()
assert not ({"start_new_session", "creationflags", "preexec_fn"} & captured.keys())
finally:
backend._proc = None
def test_omnivoice_subprocess_recv_timeout_overrides_default():
b = OmniVoiceSubprocessBackend()
assert b.recv_timeout_s == 300.0 # aligns with the generate budget
def test_omnivoice_subprocess_has_longer_spawn_budget_than_other_sidecars():
assert _PlainBackend.spawn_ready_timeout_s == 30.0
assert OmniVoiceSubprocessBackend.spawn_ready_timeout_s == 120.0
def test_spawn_uses_backend_specific_ready_timeout(monkeypatch, tmp_path):
_use_stub(monkeypatch, tmp_path / "unused.py")
backend = OmniVoiceSubprocessBackend()
observed = []
class StubProcess:
stderr = io.BytesIO()
@staticmethod
def poll():
return None
monkeypatch.setattr(
"services.subprocess_backend.spawn_owned",
lambda *_args, **_kwargs: StubProcess(),
)
monkeypatch.setattr(
backend,
"_recv_with_timeout",
lambda timeout: observed.append(timeout) or {"op": "ready"},
)
monkeypatch.setattr("services.subprocess_backend._ensure_reaper_running", lambda: None)
try:
backend._spawn()
finally:
backend._proc = None
assert observed == [120.0]
def test_omnivoice_subprocess_recv_timeout_env_override(monkeypatch):
monkeypatch.setenv("OMNIVOICE_SIDECAR_RECV_TIMEOUT_S", "120")
assert OmniVoiceSubprocessBackend().recv_timeout_s == 120.0
@@ -205,6 +373,48 @@ def test_wedged_sidecar_is_hard_killed_and_recovers(stub_sidecar, monkeypatch):
b.shutdown()
def test_mps_proxy_survives_fatal_child_exit_and_recovers(stub_sidecar, monkeypatch):
_use_stub(monkeypatch, stub_sidecar)
monkeypatch.setattr(
"services.model_manager.make_room_before_generate", lambda: None,
)
b = OmniVoiceMPSSubprocessBackend()
try:
with pytest.raises(RuntimeError, match="backend is still running"):
b.generate("CRASH")
assert b._proc is not None and b._proc.poll() is not None
assert b.generate("ok").shape[1] == 24000
finally:
b.shutdown()
def test_desktop_timeout_kills_engine_subtree_before_late_mutation(
stub_sidecar, monkeypatch, tmp_path
):
marker = tmp_path / "late-engine-mutation"
monkeypatch.setenv("OMNIVOICE_DESKTOP_CONTAINED", "1")
drain_read, drain_write = os.pipe()
monkeypatch.setenv("OMNIVOICE_DESKTOP_DRAIN_FD", str(drain_write))
monkeypatch.setenv("OMNIVOICE_TIMEOUT_MARKER", str(marker))
_use_stub(monkeypatch, stub_sidecar)
monkeypatch.setattr(
OmniVoiceSubprocessBackend,
"recv_timeout_s",
property(lambda self: 0.3),
)
b = OmniVoiceSubprocessBackend()
try:
with pytest.raises(RuntimeError):
b.generate("HANG_CHILD")
time.sleep(1.2)
assert not marker.exists()
assert b.generate("ok").shape[1] == 24000
finally:
b.shutdown()
os.close(drain_write)
os.close(drain_read)
def test_generate_does_not_deadlock_when_called_on_gpu_pool_worker(stub_sidecar, monkeypatch):
# Regression: /v1/audio/speech and /generate dispatch backend.generate() via
# run_on_gpu_pool_guarded, i.e. ON a gpu-pool worker. generate() must NOT
@@ -222,3 +432,254 @@ def test_generate_does_not_deadlock_when_called_on_gpu_pool_worker(stub_sidecar,
assert tensor.shape[1] == 24000
finally:
b.shutdown()
def test_sidecar_forwards_native_controls_and_applies_seed(monkeypatch):
import torch
from engines.omnivoice_subprocess import main as sidecar
calls = []
seeds = []
frames = []
class FakeModel:
sampling_rate = 24000
def generate(self, **kwargs):
calls.append(kwargs)
return [torch.zeros(1, 16)]
monkeypatch.setattr(sidecar, "_load_model", lambda _stdout: FakeModel())
monkeypatch.setattr(sidecar, "_send", lambda _stdout, frame: frames.append(frame))
real_manual_seed = torch.manual_seed
monkeypatch.setattr(
torch, "manual_seed", lambda seed: (seeds.append(seed), real_manual_seed(seed))[1],
)
sidecar._handle_synthesize({
"text": "hello",
"seed": 123,
"t_shift": 0.4,
"layer_penalty_factor": 0.2,
"position_temperature": 0.7,
"class_temperature": 0.8,
"audio_chunk_duration": 10,
"audio_chunk_threshold": 0.6,
}, object())
assert seeds == [123]
assert calls == [{
"text": "hello",
"ref_audio": None,
"ref_text": None,
"t_shift": 0.4,
"layer_penalty_factor": 0.2,
"position_temperature": 0.7,
"class_temperature": 0.8,
"audio_chunk_duration": 10,
"audio_chunk_threshold": 0.6,
}]
assert frames[-1]["op"] == "audio"
def test_generation_proxy_forwards_native_controls_and_seed():
import torch
from api.routers.generation import _run_backend_inference
calls = []
class Proxy:
id = "omnivoice"
display_name = "OmniVoice"
sample_rate = 24000
applies_own_mastering = True
supports_native_omnivoice_controls = True
def generate(self, text, **kwargs):
calls.append((text, kwargs))
return torch.zeros(1, 240)
_run_backend_inference(
Proxy(), "hello", "en", None, None, None, None,
16, 2.0, 1.0, False, False, 321,
t_shift=0.4, layer_penalty_factor=0.2,
position_temperature=0.7, class_temperature=0.8,
)
assert calls == [("hello", {
"duration": None,
"language": "en",
"ref_audio": None,
"ref_text": None,
"instruct": None,
"num_step": 16,
"guidance_scale": 2.0,
"speed": 1.0,
"denoise": False,
"postprocess_output": False,
"t_shift": 0.4,
"layer_penalty_factor": 0.2,
"position_temperature": 0.7,
"class_temperature": 0.8,
"seed": 321,
})]
def test_timeout_reaps_captured_process_before_recv_returns(monkeypatch):
import threading
class Process:
def __init__(self):
self.killed = threading.Event()
self.reaped = False
self.wait_entered = threading.Event()
self.release_wait = threading.Event()
def kill(self):
self.killed.set() # EOF may arrive before the process is reaped.
def wait(self, timeout):
assert timeout is not None
self.wait_entered.set()
assert self.release_wait.wait(2)
self.reaped = True
return -9
proc = Process()
backend = OmniVoiceSubprocessBackend()
backend._proc = proc
def recv():
assert proc.killed.wait(2)
return None
monkeypatch.setattr(backend, '_recv', recv)
returned = threading.Event()
results = []
def receive():
results.append(backend._recv_with_timeout(0.01))
returned.set()
reader = threading.Thread(target=receive)
reader.start()
try:
assert proc.wait_entered.wait(2)
assert not returned.wait(0.05), "EOF must not release the caller before process cleanup"
finally:
proc.release_wait.set()
reader.join(2)
backend._proc = None
assert not reader.is_alive()
assert returned.is_set()
assert results == [None]
assert proc.reaped
def test_timeout_never_kills_a_replacement_process(monkeypatch):
from unittest.mock import Mock
import services.subprocess_backend as module
class ManualTimer:
def __init__(self, _timeout, callback, args=()):
self.callback = lambda: callback(*args)
self.daemon = False
def start(self):
pass
def cancel(self):
pass
def join(self):
pass
timers = []
def timer(*args, **kwargs):
result = ManualTimer(*args, **kwargs)
timers.append(result)
return result
monkeypatch.setattr(module.threading, 'Timer', timer)
backend = OmniVoiceSubprocessBackend()
original, replacement = Mock(), Mock()
backend._proc = original
def recv():
backend._proc = replacement
timers[0].callback()
return None
monkeypatch.setattr(backend, '_recv', recv)
try:
backend._recv_with_timeout(1)
original.kill.assert_called_once()
replacement.kill.assert_not_called()
finally:
backend._proc = None
@pytest.mark.parametrize("failure", ["wait", "kill"])
def test_timeout_quarantine_blocks_reuse_and_retains_cleanup_handle(failure):
class StuckProcess:
stdin = None
def __init__(self):
self.exited = False
self.kill_calls = 0
def poll(self):
return 0 if self.exited else None
def kill(self):
self.kill_calls += 1
if failure == "kill" and not self.exited:
raise PermissionError("kill failed")
def terminate(self):
pass
def wait(self, timeout):
if not self.exited:
raise subprocess.TimeoutExpired("stuck-sidecar", timeout)
return 0
backend = OmniVoiceSubprocessBackend()
proc = StuckProcess()
backend._proc = proc
try:
backend._timeout_kill(proc)
with pytest.raises(RuntimeError, match="still stopping"):
backend._spawn()
backend.shutdown()
# Even after shutdown clears the current slot, ownership survives;
# retry must not silently start a second process next to this one.
before = proc.kill_calls
with pytest.raises(RuntimeError, match="still stopping"):
backend._spawn()
assert proc.kill_calls > before
finally:
proc.exited = True
backend.shutdown()
def test_timeout_quarantine_does_not_clear_or_kill_replacement():
from unittest.mock import Mock
backend = OmniVoiceSubprocessBackend()
original = Mock()
original.wait.side_effect = subprocess.TimeoutExpired("old-sidecar", 2)
replacement = Mock()
replacement.poll.return_value = None
backend._proc = replacement
try:
backend._timeout_kill(original)
with pytest.raises(RuntimeError, match="still stopping"):
backend._spawn()
assert backend._proc is replacement
replacement.kill.assert_not_called()
# Once the captured owner is reaped, reuse of the healthy replacement
# is allowed without starting or terminating another process.
original.wait.side_effect = None
original.wait.return_value = 0
backend._spawn()
assert backend._proc is replacement
replacement.kill.assert_not_called()
finally:
original.wait.side_effect = None
backend._proc = None
backend.shutdown()
+76
View File
@@ -0,0 +1,76 @@
"""#1618 — RAM preflight must not hard-block the machines it means to admit.
An "8 GB" machine reports ~7.8 GB usable (firmware/iGPU/kernel reservations),
so comparing reported RAM against the marketing-size threshold blocked exactly
the boundary hardware the 8 GB rule intends to allow. The check now applies
``_RAM_RESERVED_ALLOWANCE`` to both thresholds, and
``OMNIVOICE_RAM_PREFLIGHT=0`` downgrades a genuine fail to a warning.
"""
import pytest
from api.routers.setup import wizard
def _ram_check(monkeypatch, ram_gb: float, env: str | None = None) -> dict:
# Keep the preflight hermetic: stub the probes that hit the network or
# auto-acquire media tools, so each RAM assertion stays fast and offline.
monkeypatch.setattr(wizard, "_network_check", lambda: {
"id": "network", "label": "Network", "status": "pass",
"detail": "stubbed", "fix": None, "mirror_reachable": True,
})
import services.media_tools as media_tools
monkeypatch.setattr(media_tools, "summary", lambda auto_acquire=True: None)
monkeypatch.setattr(wizard, "_ram_gb", lambda: ram_gb)
if env is None:
monkeypatch.delenv("OMNIVOICE_RAM_PREFLIGHT", raising=False)
else:
monkeypatch.setenv("OMNIVOICE_RAM_PREFLIGHT", env)
resp = wizard.preflight()
checks = resp["checks"] if isinstance(resp, dict) else resp.checks
for c in checks:
c = c if isinstance(c, dict) else c.model_dump()
if c["id"] == "ram":
return c
raise AssertionError("no ram check in preflight response")
def test_8gb_installed_reporting_7_84_usable_is_not_blocked(monkeypatch):
"""The #1618 report: 7.84 GB usable on an 8 GB laptop was a hard fail."""
check = _ram_check(monkeypatch, 7.84)
assert check["status"] != "fail"
def test_boundary_at_allowance_passes_the_fail_gate(monkeypatch):
check = _ram_check(
monkeypatch, wizard._RAM_FAIL_GB * wizard._RAM_RESERVED_ALLOWANCE
)
assert check["status"] != "fail"
def test_genuinely_low_ram_still_fails(monkeypatch):
check = _ram_check(monkeypatch, 6.0)
assert check["status"] == "fail"
@pytest.mark.parametrize("env", ["0", "false", "no"])
def test_escape_hatch_downgrades_fail_to_warn(monkeypatch, env):
check = _ram_check(monkeypatch, 6.0, env=env)
assert check["status"] == "warn"
assert "OMNIVOICE_RAM_PREFLIGHT" in (check["fix"] or "")
def test_escape_hatch_not_triggered_by_other_values(monkeypatch):
check = _ram_check(monkeypatch, 6.0, env="1")
assert check["status"] == "fail"
def test_12gb_installed_reporting_11_8_usable_passes_clean(monkeypatch):
"""Same reservation gap at the warn threshold: 12 GB installed ≈ 11.8."""
check = _ram_check(monkeypatch, 11.8)
assert check["status"] == "pass"
def test_warn_band_between_thresholds(monkeypatch):
check = _ram_check(monkeypatch, 9.0)
assert check["status"] == "warn"
+30 -1
View File
@@ -76,6 +76,35 @@ class TestUnloadOnABC:
)
def test_omnivoice_native_batch_preserves_per_item_controls():
"""The adapter forwards variable-length batch controls to OmniVoice."""
import torch
tts = _load_tts_backend_module()
calls = []
class _Model:
sampling_rate = 24000
def generate(self, **kwargs):
calls.append(kwargs)
return [torch.zeros(1, 12000), torch.zeros(1, 24000)]
backend = tts.OmniVoiceBackend(model=_Model())
outputs = backend.generate_batch(
["short", "long"],
language=["en", "es"],
duration=[0.5, 1.0],
speed=[1.0, 0.8],
)
assert [output.shape[-1] for output in outputs] == [12000, 24000]
assert calls[0]["text"] == ["short", "long"]
assert calls[0]["language"] == ["en", "es"]
assert calls[0]["duration"] == [0.5, 1.0]
assert calls[0]["speed"] == [1.0, 0.8]
class TestUnloadDefaultBehavior:
"""The default no-op must actually be safe to call."""
@@ -154,4 +183,4 @@ class TestExistingSubclassesInherit:
assert callable(getattr(cls, "unload", None)), (
f"{cls.__name__} has no callable unload() — even via the "
"ABC inheritance. Did someone shadow it?"
)
)
+857 -86
View File
File diff suppressed because it is too large Load Diff
+61
View File
@@ -0,0 +1,61 @@
"""Cancellation helpers for work that cannot be stopped mid-call."""
from __future__ import annotations
import asyncio
from collections.abc import Callable
from typing import Any, TypeVar
_Result = TypeVar("_Result")
async def drain_task(task: asyncio.Task[Any]) -> None:
"""Wait for ``task`` even if the waiter is cancelled again."""
while not task.done():
try:
await asyncio.shield(task)
except asyncio.CancelledError:
continue
except BaseException:
break
if task.done():
try:
task.result()
except BaseException:
pass
async def to_thread_and_drain_on_cancel(
function: Callable[..., _Result], /, *args: Any
) -> _Result:
"""Run a blocking call without detaching it when its waiter is cancelled."""
thread_task = asyncio.create_task(asyncio.to_thread(function, *args))
try:
return await asyncio.shield(thread_task)
except asyncio.CancelledError:
await drain_task(thread_task)
raise
async def to_thread_and_defer_cancellation(
function: Callable[..., _Result], /, *args: Any
) -> tuple[_Result, bool]:
"""Finish a durable call and report cancellation after its result is known.
Authority writes need their event-loop publication even when the HTTP
caller disappears while SQLite is committing. Returning the cancellation
flag lets the caller publish that result first, then propagate cancellation.
"""
thread_task = asyncio.create_task(asyncio.to_thread(function, *args))
try:
return await asyncio.shield(thread_task), False
except asyncio.CancelledError:
await drain_task(thread_task)
return thread_task.result(), True
__all__ = [
"drain_task",
"to_thread_and_defer_cancellation",
"to_thread_and_drain_on_cancel",
]
+44 -8
View File
@@ -59,10 +59,18 @@ _VRAM_PER_JOB_BYTES = 5 * 1024**3
# cpu — oversubscription just thrashes
_ALWAYS_SERIAL = frozenset({"mps", "mlx", "cpu", ""})
# Absolute ceiling regardless of how much memory a card reports. Beyond this
# the bottleneck stops being VRAM and starts being scheduler overhead and
# host-side I/O contention.
_MAX_DERIVED = 4
# Absolute protocol ceiling regardless of how much memory a peer reports.
# Beyond this the bottleneck stops being VRAM and starts being scheduler
# overhead and host-side I/O contention. It is public because every wire
# boundary must clamp to the same number; a UINT32_MAX heartbeat must not grow
# a scheduler queue that local derivation would never create.
MAX_CONCURRENT_TASKS = 4
def clamp_concurrency(value: int, *, allow_zero: bool = False) -> int:
"""Bound an advertised concurrency value to the server's safe range."""
minimum = 0 if allow_zero else 1
return max(minimum, min(MAX_CONCURRENT_TASKS, int(value)))
# Bounds on how long a parked slot is held. The caller passes the timed-out
# job's own execution budget — the longest its thread can still legitimately be
@@ -104,7 +112,7 @@ def derive_concurrency(
budget = max(min_model_bytes, _VRAM_PER_JOB_BYTES)
if budget <= 0:
return 1
return max(1, min(_MAX_DERIVED, int(free_memory_bytes // budget)))
return clamp_concurrency(int(free_memory_bytes // budget))
@dataclass
@@ -149,6 +157,9 @@ class WorkerCapacity:
resident_models: set[str] = field(default_factory=set)
slots: dict[str, ModelSlot] = field(default_factory=dict)
def __post_init__(self) -> None:
self.max_concurrent_tasks = clamp_concurrency(self.max_concurrent_tasks)
@staticmethod
def slot_key(engine: str, model_id: str) -> str:
return f"{engine}:{model_id}"
@@ -198,6 +209,21 @@ class WorkerCapacity:
)
slot.active += 1
def reserve_unknown(self) -> None:
"""Consume worker-wide capacity for claimed work we cannot classify.
Reconciliation will tell the peer to cancel a terminal or unknown
attempt, but until that cancellation lands it is still using the GPU.
"""
self.active_tasks += 1
def release_unknown(self) -> bool:
"""Release one exact reconciled claim with no model-slot identity."""
if self.active_tasks <= 0:
return False
self.active_tasks -= 1
return True
def release(
self,
engine: str,
@@ -267,8 +293,12 @@ class WorkerCapacity:
) -> None:
"""Adopt a heartbeat snapshot. The worker is the source of truth for
what it is actually running."""
self.active_tasks = max(0, active_tasks)
reported_ceiling = self.active_tasks + max(0, available_slots)
self.active_tasks = clamp_concurrency(active_tasks, allow_zero=True)
bounded_available = clamp_concurrency(available_slots, allow_zero=True)
bounded_available = min(
bounded_available, MAX_CONCURRENT_TASKS - self.active_tasks
)
reported_ceiling = self.active_tasks + bounded_available
if reported_ceiling > 0:
# Adopted, not merely grown. The worker computes this as its own
# ``max_concurrent_tasks``, so a ceiling we refuse to lower is one
@@ -305,4 +335,10 @@ class WorkerCapacity:
}
__all__ = ["ModelSlot", "WorkerCapacity", "derive_concurrency"]
__all__ = [
"MAX_CONCURRENT_TASKS",
"ModelSlot",
"WorkerCapacity",
"clamp_concurrency",
"derive_concurrency",
]

Some files were not shown because too many files have changed in this diff Show More