Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8c6cb9a9ac | ||
|
|
da05973f78 | ||
|
|
860f62a9a0 | ||
|
|
e63e0a6634 | ||
|
|
c38b7d0bbb | ||
|
|
1746868fc7 | ||
|
|
3b9ca6401d | ||
|
|
07e9b8dc84 | ||
|
|
478590b689 | ||
|
|
da3667a369 | ||
|
|
77802c5f87 | ||
|
|
6ae92dbc47 | ||
|
|
9dc41252c1 | ||
|
|
2aa5f38012 | ||
|
|
082d7260d5 | ||
|
|
86c7f6bb35 | ||
|
|
eed4632933 | ||
|
|
1fd321d145 | ||
|
|
2b6fd847ce | ||
|
|
50a71fd176 | ||
|
|
750e81bcf7 | ||
|
|
8a7ab51784 | ||
|
|
7ed58cb5b7 | ||
|
|
20f7329f59 | ||
|
|
c52441188e | ||
|
|
49f9fe3b06 | ||
|
|
f6d1e0c2df | ||
|
|
d217553e2e | ||
|
|
d6efb14849 | ||
|
|
a222974568 | ||
|
|
ed9e40253c | ||
|
|
86c37c8d04 | ||
|
|
a4ffd5229b | ||
|
|
4b3e998de3 | ||
|
|
8b2ec7ba51 | ||
|
|
af8810be53 | ||
|
|
1ef42a6b21 | ||
|
|
31baa43650 | ||
|
|
834b305c05 | ||
|
|
418b1b271d | ||
|
|
a6113762a4 | ||
|
|
ad6dae49cd | ||
|
|
fbede643c1 | ||
|
|
743f4918cb | ||
|
|
543f0f6f18 | ||
|
|
ec3e12cd6e | ||
|
|
37b498c891 | ||
|
|
e6e8f16f55 | ||
|
|
b1f900c996 | ||
|
|
a8581f899e | ||
|
|
eda99113ec | ||
|
|
d2e2d6073a | ||
|
|
80a1f22175 | ||
|
|
0f13d49363 | ||
|
|
a5ebdebf5d | ||
|
|
fa745366ab | ||
|
|
c1bc91fe5b | ||
|
|
5b01b65d1e | ||
|
|
fecde7fca5 | ||
|
|
c7883c107c | ||
|
|
17c0894a85 | ||
|
|
39cdd4b4db | ||
|
|
8d903eed8e | ||
|
|
e477251b35 | ||
|
|
57501ef8ef | ||
|
|
8b1b1e9b51 | ||
|
|
863d304cd5 | ||
|
|
11b3b87b02 | ||
|
|
80e7de7e96 | ||
|
|
7a7a1d379d | ||
|
|
65044b6cfd | ||
|
|
06c675fbc8 | ||
|
|
737f04f42d | ||
|
|
7182c8d312 | ||
|
|
9ee1f9c1ed | ||
|
|
fa16980938 | ||
|
|
6450fd931f | ||
|
|
a0154778a4 | ||
|
|
1c5c16e21e | ||
|
|
49892a6066 | ||
|
|
05a8b58b53 | ||
|
|
1f17560cde | ||
|
|
7a7cccbea2 | ||
|
|
e3d3913f52 | ||
|
|
ad52d32760 | ||
|
|
1ddf79a6de | ||
|
|
308dda0cac | ||
|
|
3775a2acd9 | ||
|
|
fb80d88f34 | ||
|
|
d874239c73 | ||
|
|
85f8328b7c | ||
|
|
6a15030e31 | ||
|
|
cef375e700 | ||
|
|
84290db784 | ||
|
|
810e5e350b | ||
|
|
c9fefe39f3 | ||
|
|
c3137da680 | ||
|
|
281287d3e7 | ||
|
|
5c6d0f7f8f | ||
|
|
819fd6e4a3 | ||
|
|
ebc03bb117 | ||
|
|
e040d19513 | ||
|
|
c3b4cc8de6 | ||
|
|
68c48e384b | ||
|
|
9110b01c75 | ||
|
|
9579d1c82e | ||
|
|
de1d9d66eb | ||
|
|
38b6d11815 | ||
|
|
a17b6a8172 | ||
|
|
a9195b8e40 | ||
|
|
0be2859f75 | ||
|
|
40e52a1fa7 | ||
|
|
4c4c23d69b | ||
|
|
6bf4316554 | ||
|
|
a8c90343b7 | ||
|
|
06389e1ae2 | ||
|
|
8f06b4968f | ||
|
|
020febe327 | ||
|
|
6a7bb1a508 | ||
|
|
5f519cbfd0 | ||
|
|
e272217561 | ||
|
|
4002801c00 | ||
|
|
645aa3d451 | ||
|
|
e1e329e7f0 | ||
|
|
8ba4407d8a | ||
|
|
47d441311e | ||
|
|
9ed477c0ca | ||
|
|
dadb642e19 | ||
|
|
b12fa3a13c | ||
|
|
119816d2b1 | ||
|
|
ba77bd9443 | ||
|
|
8e97b2bd01 | ||
|
|
13d3f3fb13 | ||
|
|
9482788898 | ||
|
|
a390da1590 | ||
|
|
f53be49e86 | ||
|
|
724e573746 | ||
|
|
6c5a88260b | ||
|
|
75a7b21193 | ||
|
|
df3b6f402a | ||
|
|
f26d5cc041 | ||
|
|
b900130f29 | ||
|
|
75efda84b1 | ||
|
|
8538702ecb | ||
|
|
11584c6f4c | ||
|
|
a9d1464acb | ||
|
|
3c4c8ed679 | ||
|
|
33a6fc1d03 | ||
|
|
a33db40613 | ||
|
|
8c47e8b92f | ||
|
|
d6f034c754 | ||
|
|
f840303549 | ||
|
|
fc98d07b39 | ||
|
|
637e24d8a1 | ||
|
|
08e5d0d5cd | ||
|
|
c52fdd263f | ||
|
|
2a3257be78 | ||
|
|
7d77f1532b | ||
|
|
519cc9c1c9 | ||
|
|
4273f8149b | ||
|
|
65a535a638 | ||
|
|
b5e9e1bf58 | ||
|
|
dad6bb1cef | ||
|
|
904d90d6a7 | ||
|
|
583189a478 | ||
|
|
58eba1e0cd | ||
|
|
52962e47c6 | ||
|
|
6772ad363f | ||
|
|
63a8bf4165 | ||
|
|
f3da7b0ef3 | ||
|
|
552711b1d8 | ||
|
|
eca0972511 | ||
|
|
1685f6aba9 | ||
|
|
c780744b11 | ||
|
|
aeb0c4ea6a | ||
|
|
e35f066a9b | ||
|
|
67c61fe357 | ||
|
|
50c6bfbd01 | ||
|
|
18c543484b | ||
|
|
54c89453a4 | ||
|
|
8c69dff00f | ||
|
|
968d05faea | ||
|
|
07f2976756 | ||
|
|
a65a0a95af | ||
|
|
be1be89a69 | ||
|
|
15ef5da87d | ||
|
|
e59bd03150 | ||
|
|
34b5eaabf1 | ||
|
|
b6c4640b7d | ||
|
|
d6b1326aae | ||
|
|
052eb1050b | ||
|
|
22e52c64a4 | ||
|
|
0c71d506ec | ||
|
|
50cba2881c | ||
|
|
750aa2eb57 | ||
|
|
ddb8e4dd86 | ||
|
|
ba041b0723 | ||
|
|
c5f3355951 | ||
|
|
68653ca408 | ||
|
|
df18ac177e | ||
|
|
6cc67c31f6 | ||
|
|
1bf772466c | ||
|
|
42acd7d12f | ||
|
|
7fe80a5ab8 | ||
|
|
23df0e80ad | ||
|
|
6dafebeebe | ||
|
|
05ee8c90b2 | ||
|
|
5bad0e765a | ||
|
|
fc2d811471 | ||
|
|
d83c4c39b6 | ||
|
|
dc28c4d52d | ||
|
|
b33d9452ee | ||
|
|
596a0861c2 | ||
|
|
651b621e4a | ||
|
|
151d58f024 | ||
|
|
61f99debb4 | ||
|
|
2ce74581c2 | ||
|
|
3cd67811e5 | ||
|
|
d9f51da3c6 | ||
|
|
ad4b3547d4 | ||
|
|
41c50ed9c8 | ||
|
|
9d30de27b4 | ||
|
|
c8c833062b | ||
|
|
a628ba161e | ||
|
|
63e647bd91 |
@@ -63,10 +63,9 @@ jobs:
|
||||
experimental: false
|
||||
- os: macos-14
|
||||
platform: darwin-arm64
|
||||
# Metal build path is unpublished upstream (Pitfall 1 in
|
||||
# 04-RESEARCH.md); experimental so a failed Metal build doesn't
|
||||
# block — the SPIKE-01 ADR records the in-process fallback.
|
||||
experimental: true
|
||||
# Apple Silicon Metal build compiles cleanly with -DGGML_METAL=ON
|
||||
# at the pinned SHA (#2105); non-experimental to catch regressions.
|
||||
experimental: false
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
|
||||
@@ -6,6 +6,10 @@ on:
|
||||
tags: ['v*']
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
release_tag:
|
||||
description: "Existing version tag to package using this workflow from main (optional)"
|
||||
type: string
|
||||
default: ''
|
||||
publish:
|
||||
description: "Publish the tagged Electron release after all platforms pass"
|
||||
type: boolean
|
||||
@@ -19,9 +23,12 @@ permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: electron-release-${{ github.ref }}
|
||||
group: electron-release-${{ inputs.release_tag || github.ref_name }}
|
||||
cancel-in-progress: false
|
||||
|
||||
env:
|
||||
RELEASE_REF: ${{ inputs.release_tag && format('refs/tags/{0}', inputs.release_tag) || github.ref }}
|
||||
|
||||
jobs:
|
||||
validate:
|
||||
# The transition tag is assembled after the manual Tauri draft succeeds.
|
||||
@@ -29,9 +36,13 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
ref: ${{ env.RELEASE_REF }}
|
||||
- name: Require an exact version tag
|
||||
env:
|
||||
REF: ${{ github.ref }}
|
||||
REF: ${{ env.RELEASE_REF }}
|
||||
WORKFLOW_REF: ${{ github.ref }}
|
||||
RELEASE_TAG_OVERRIDE: ${{ inputs.release_tag }}
|
||||
ALLOW_UNSIGNED: ${{ inputs.allow_unsigned }}
|
||||
DISPATCH_ACTOR: ${{ github.actor }}
|
||||
RERUN_ACTOR: ${{ github.triggering_actor }}
|
||||
@@ -42,8 +53,12 @@ jobs:
|
||||
echo "Only the repository owner may accept unsigned installers"; exit 1;
|
||||
}
|
||||
fi
|
||||
if [ -n "$RELEASE_TAG_OVERRIDE" ]; then
|
||||
test "$WORKFLOW_REF" = refs/heads/main || { echo "Tag overrides require the workflow from main"; exit 1; }
|
||||
fi
|
||||
VERSION=$(node -p "require('./frontend/package.json').version")
|
||||
test "$REF" = "refs/tags/v$VERSION" || { echo "Dispatch on the exact version tag"; exit 1; }
|
||||
test "$REF" = "refs/tags/v$VERSION" || { echo "Select the exact version tag"; exit 1; }
|
||||
test "$(git rev-parse HEAD)" = "$(git rev-parse "$REF^{commit}")" || { echo "Checkout does not match the release tag"; exit 1; }
|
||||
package:
|
||||
needs: validate
|
||||
runs-on: ${{ matrix.runner }}
|
||||
@@ -81,6 +96,8 @@ jobs:
|
||||
CSC_IDENTITY_AUTO_DISCOVERY: 'false'
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
ref: ${{ env.RELEASE_REF }}
|
||||
- uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: '22'
|
||||
@@ -129,6 +146,11 @@ jobs:
|
||||
CSC_KEY_PASSWORD: ${{ secrets.ELECTRON_CSC_KEY_PASSWORD }}
|
||||
working-directory: electron
|
||||
run: |
|
||||
# An empty CSC_LINK is interpreted as the working directory by the
|
||||
# signer. Omit absent credentials rather than passing empty strings.
|
||||
if [ -z "${CSC_LINK:-}" ]; then
|
||||
unset CSC_LINK CSC_KEY_PASSWORD
|
||||
fi
|
||||
bun x electron-builder --config electron-builder.config.mjs ${{ matrix.flags }} --publish never
|
||||
node tests/packaging-contract.mjs --artifact
|
||||
node tests/update-package-contract.mjs --platform ${{ matrix.platform }} --arch ${{ matrix.arch }}
|
||||
@@ -174,11 +196,13 @@ jobs:
|
||||
contents: write
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
TAG: ${{ github.ref_name }}
|
||||
TAG: ${{ inputs.release_tag || github.ref_name }}
|
||||
SUNSET_TAG: ${{ vars.TAURI_SUNSET_TAG }}
|
||||
PUBLISH: ${{ inputs.publish }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
ref: ${{ env.RELEASE_REF }}
|
||||
- uses: actions/download-artifact@v4
|
||||
with:
|
||||
pattern: electron-release-*
|
||||
|
||||
@@ -29,6 +29,7 @@ Binding for every AI agent (Claude, Codex, Cursor, review bots, …). CLAUDE.md
|
||||
- Local-first: no new required network calls; any HF download gated on installed-ness or explicit user action; all synthetic audio through the `mark_synthetic` chokepoint.
|
||||
- Every user-facing string via i18n, present in ALL 21 `frontend/src/i18n/locales/*.json` with real translations.
|
||||
- Docs-sync in the same PR. CHANGELOG Unreleased: quiet one-liners ending `(#N)` + `— thanks @user!` for community work, under a short `**Highlights**` list.
|
||||
- Tagged release announcements lead with the biggest user-visible change; redesigns need real UI screenshots and migrations need installer links and steps. Verify all contributor credits from the tag comparison and included PRs; list authors and bug reporters separately (see `docs/RELEASING.md`).
|
||||
- Versioning: `frontend/package.json` is the single source of truth; never bump without the owner asking.
|
||||
- `frontend/package.json` dep changes require regenerating root `bun.lock` (Docker runs `--frozen-lockfile`).
|
||||
- Issues: absorb or decline — never defer to a future version. Check the open-PR queue before implementing community-reported fixes.
|
||||
|
||||
@@ -8,6 +8,84 @@ the frozen-backend fallback mirror it for their toolchains.
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
## [0.5.4] — 2026-09-17
|
||||
|
||||
**More reliable setup, generation, and dubbing.** This patch detects incomplete desktop runtimes before startup, repairs them without overwriting an existing Tauri installation, and makes engine failures easier to recover from. It also fixes gated model downloads, subtitle imports, and saved-voice language errors.
|
||||
|
||||
**Download**
|
||||
|
||||
| Platform | Installer |
|
||||
| --- | --- |
|
||||
| Windows x64 | [Installer](https://github.com/debpalash/VoiceStudio/releases/download/v0.5.4/VoiceStudio-Electron-0.5.4-win-x64.exe) |
|
||||
| macOS Apple Silicon | [DMG](https://github.com/debpalash/VoiceStudio/releases/download/v0.5.4/VoiceStudio-Electron-0.5.4-mac-arm64.dmg) |
|
||||
| macOS Intel | [DMG](https://github.com/debpalash/VoiceStudio/releases/download/v0.5.4/VoiceStudio-Electron-0.5.4-mac-x64.dmg) |
|
||||
| Linux x64 | [AppImage](https://github.com/debpalash/VoiceStudio/releases/download/v0.5.4/VoiceStudio-Electron-0.5.4-linux-x64.AppImage) · [deb](https://github.com/debpalash/VoiceStudio/releases/download/v0.5.4/VoiceStudio-Electron-0.5.4-linux-x64.deb) |
|
||||
|
||||
Already using Electron? Install over your existing app and keep your data. If prompted, choose **Install local runtime** to refresh its dependencies. Moving from Tauri? Back up your data directory with the app closed, install Electron, then verify your voices and projects before removing Tauri. Follow the [migration guide](https://github.com/debpalash/VoiceStudio/blob/v0.5.4/docs/electron-migration.md). Tauri v0.5.3 remains the final Tauri release; its updater cannot install Electron.
|
||||
|
||||
**Highlights**
|
||||
|
||||
- Repair incomplete Electron runtimes and recover gated model downloads (#2179, #2173)
|
||||
- More reliable transcription, engine deadlines, and dubbing streams (#2165, #2109, #2111, #2138)
|
||||
- Preserve subtitle text and legacy manuscript encodings across desktop and web (#2077, #2151, #2073)
|
||||
- Clear language guidance and capability checks before voice generation (#2172, #2174, #2175)
|
||||
|
||||
### Fixed
|
||||
|
||||
- Forward saved Hugging Face tokens when downloading gated model weights and dependencies (#2173) — thanks @shivsin25!
|
||||
- Explain unsupported saved-profile languages consistently in Electron, web, and streaming generation (#2175) — thanks @shivsin25!
|
||||
- List every installed Kokoro language and accept its displayed name, including British English (#2174) — thanks @drakeo338!
|
||||
|
||||
- Validate Python dependencies before reusing a desktop runtime and offer setup for incomplete environments (#2176)
|
||||
- Check active model cloning support before starting voice conversion (#2147)
|
||||
- Accept both valid SIGKILL diagnostics in the desktop lifecycle regression check (#2170)
|
||||
|
||||
- Repair CTranslate2 loading safely across ASR and translation, and retain the loaded Whisper model during CPU fallback (#2165) — thanks @guruthechosen!
|
||||
- Avoid pedalboard wheels that crash on unsupported CPU instructions (#2080) — thanks @D3nii!
|
||||
- Include cuDNN 8 compatibility libraries for CTranslate2 in CUDA containers (#2072) — thanks @basil-k-aji-dev!
|
||||
- Preserve audio reads, writes, and reference amplitude without TorchCodec (#2083) — thanks @Moep90!
|
||||
- Give isolated engines request-sized deadlines, validate timeout overrides, and distinguish hangs from crashes (#2109) (#2111) — thanks @SurefireStudios and @LMGXENON!
|
||||
- Keep dubbing streams alive during quiet steps and delay model cleanup until native refinement ends (#2138) — thanks @denemon!
|
||||
- Locate ffprobe beside ffmpeg without changing parent directory names (#2107) — thanks @kapelame!
|
||||
- Resample MLX output chunks to the declared rate before joining them (#2106) — thanks @kapelame!
|
||||
- Read database migration configuration on Chinese, Japanese, and Korean Windows (#2075) — thanks @kevin9327!
|
||||
- Preserve milliseconds and carry rounded subtitle timestamps across second boundaries (#2074) — thanks @kevin9327!
|
||||
- Decode UTF-16 and Windows-1252 subtitle and manuscript imports in Electron, web, and backend routes (#2073) — thanks @kevin9327!
|
||||
- Preserve numeric subtitle dialogue while recognizing mixed indexed and unindexed cues (#2151) — thanks @shivsin25!
|
||||
- Parse pasted WebVTT cues while separating metadata, identifiers, empty cues, and complete timing lines (#2077) — thanks @kevin9327!
|
||||
- Normalize Argos language aliases without silently changing Traditional Chinese to Simplified (#2143, #2152) — thanks @gyanu2507 and @rollroyces!
|
||||
- Clarify Blackwell import-crash diagnostics without blaming missing kernels (#2084) — thanks @Moep90!
|
||||
- Distinguish architecture preflight rejection from independent compile-stack failures (#2085) — thanks @Moep90!
|
||||
- Require the pinned Apple Silicon GGUF build to pass and document runtime preflight conditions (#2115) — thanks @LMGXENON and @martinezpl!
|
||||
- Correct the Windows Rustup installation command in tooling and documentation (#2066) — thanks @Rukhaam!
|
||||
- Show local setup guidance when remote native engine installation is unavailable (#2166)
|
||||
- Show scrubbed native error tails and exit codes for failed dubbing extraction (#2167)
|
||||
|
||||
### CI
|
||||
|
||||
- Handle missing Electron signing credentials and retry packaging fixes without moving release tags (#2157)
|
||||
|
||||
### Contributors
|
||||
|
||||
- @D3nii — compatibility with CPUs unsupported by newer pedalboard wheels.
|
||||
- @LMGXENON — engine deadlines, timeout diagnostics, and Apple Silicon build documentation.
|
||||
- @Moep90 — audio I/O fallbacks and GPU compatibility guidance.
|
||||
- @Rukhaam — Windows toolchain setup guidance.
|
||||
- @Shivendra-Coherent and @shivsin25 — numeric subtitle dialogue, gated downloads, and profile-language errors.
|
||||
- @SurefireStudios — request-sized sidecar generation deadlines.
|
||||
- @basil-k-aji-dev — cuDNN compatibility libraries in CUDA containers.
|
||||
- @denemon — dubbing stream keepalives and cancellation cleanup.
|
||||
- @guruthechosen — CTranslate2 repair and Whisper CPU recovery.
|
||||
- @gyanu2507 and @rollroyces — Argos language normalization and regression coverage.
|
||||
- @kapelame — ffprobe discovery and MLX audio resampling.
|
||||
- @kevin9327 — subtitle timing, text encodings, WebVTT, and Windows database migrations.
|
||||
- @drakeo338 — Kokoro supported-language reporting.
|
||||
- @debpalash — integration, Electron runtime recovery, localization, regression coverage, and release maintenance.
|
||||
|
||||
### Bug reports
|
||||
|
||||
- Thanks to @YChhunsann, @martinezpl, @denemon, @adeelahmadsiddique, @TehSmoo, @kmsitcomputer, @raya-mansouri, @infinitete, and @kor1998 for the reports behind the fixes above.
|
||||
|
||||
## [0.5.3] — 2026-09-17
|
||||
|
||||
**Highlights**
|
||||
@@ -67,6 +145,7 @@ the frozen-backend fallback mirror it for their toolchains.
|
||||
- Transcribing an M4A file with PyTorch Whisper works, instead of failing with "Format not recognised" (#2042, #2039)
|
||||
- PyTorch Whisper runs on 6 GB NVIDIA cards instead of falling back to CPU, because its memory check now fits the model it loads (#2044, #2041)
|
||||
- MCP tools wait as long as the backend does, so a long transcription no longer fails at 120 s with an empty error (#2043, #2040)
|
||||
- Installing a gated model now sends your Hugging Face token on the fast download path too, so pyannote diarisation and gated engine weights stop failing with "401 Unauthorized" when the token is saved in Settings (#2173, #2163)
|
||||
- Generating on an older NVIDIA GPU (Tesla T4, and other pre-Ampere cards) no longer kills the backend on the first request — CUDA graphs are not captured below sm_80 (#2135)
|
||||
- "Disable torch.compile" in Settings → Performance now works on macOS and Linux, not only Windows; it was greyed out on the platforms that needed it (#2135)
|
||||
- Setting `TORCH_COMPILE_DISABLE=1` in the environment now actually disables torch.compile, for the in-process engine and engine subprocesses alike (#2135)
|
||||
|
||||
@@ -44,7 +44,9 @@ For anything new: prefer what's already pinned in `pyproject.toml` / `frontend/p
|
||||
|
||||
**Docs-sync (hard rule, owner-set 2026-06-11):** any change that alters something these docs describe — README.md, `.github/CONTRIBUTING.md`, `.github/SECURITY.md`, `.github/SUPPORT.md`, LICENSE, or `docs/**` (install flows, Docker tag semantics, platform support, versioning/release behavior, review process, supported versions) — must update those docs **in the same PR** as the change. If a doc impact is discovered after merge, the docs fix is the immediate next commit, not backlog. Stale docs are treated as bugs.
|
||||
|
||||
**Release notes / changelog (hard rule, owner-set 2026-06-16):** every tagged release gets a **high-quality, user-facing `## [X.Y.Z] — DATE` section in `CHANGELOG.md`** before (or in the same hour as) the tag — never the "Auto-generated release for vX.Y.Z…" fallback. `release.yml` extracts that section verbatim as the GitHub Release body (the `Extract CHANGELOG section for tag` step), so a missing/empty section ships a bare release. Quality bar (owner-restyled 2026-07-17, replaces the old bold-lead paragraphs): **quiet and scannable** — a short `**Highlights**` bullet list first (plain words, one line each), then `### Changed` / `### Added` / `### Docs` / `### Fixed` / `### License` / `### CI` subsections where each entry is a **single one-liner** with the `(#NNN)` issue/PR ref and contributor credit (`— thanks @user!`) where applicable. Written for users, grouped by theme, no multi-line paragraphs, **not** raw commit dumps. This applies to **preview builds too**: preview release notes summarize what's new on `main` since the last stable, in the same style. Workflow: as features merge, keep `## [Unreleased]` current; at release time rename it to the version + date. If a release was already cut with the fallback body, the next action is to backfill `CHANGELOG.md` **and** `gh release edit <tag>` the live body — not backlog.
|
||||
**Release notes / changelog (hard rule, owner-set 2026-06-16):** every tagged release gets a **high-quality, user-facing `## [X.Y.Z] — DATE` section in `CHANGELOG.md`** before (or in the same hour as) the tag — never the "Auto-generated release for vX.Y.Z…" fallback. The desktop release workflow extracts that section verbatim as the GitHub Release body, so a missing/empty section ships a bare release. Quality bar (owner-restyled 2026-07-17, replaces the old bold-lead paragraphs): **quiet and scannable change entries** — after an optional tagged-release introduction (see the presentation rule below), a short `**Highlights**` bullet list (plain words, one line each), then `### Changed` / `### Added` / `### Docs` / `### Fixed` / `### License` / `### CI` subsections where each entry is a **single one-liner** with the `(#NNN)` issue/PR ref and contributor credit (`— thanks @user!`) where applicable. Change entries are written for users, grouped by theme, with no multi-line entry paragraphs or raw commit dumps. This applies to **preview builds too**: preview release notes summarize what's new on `main` since the last stable, in the same style. Workflow: as features merge, keep `## [Unreleased]` current; at release time rename it to the version + date. If a release was already cut with the fallback body, the next action is to backfill `CHANGELOG.md` **and** `gh release edit <tag>` the live body — not backlog.
|
||||
|
||||
**Release presentation and credits (owner-set 2026-09-17):** Tagged release announcements lead with the biggest user-visible change; redesigns need real UI screenshots and migrations need installer links and steps. Verify all contributor credits from the tag comparison and included PRs; list authors and bug reporters separately (see `docs/RELEASING.md`). Keep Highlights to 3–5 bullets; the release introduction can include prose, images, and a download table before the concise change entries.
|
||||
|
||||
**Localization (hard rule):** No hardcoded non-English (CJK) **user-facing text** anywhere in the codebase except the translation layer (`frontend/src/i18n/`). All UI strings go through i18n (`t('...')` keys in `locales/*.json`); native language names live in `i18n/index.ts` (`LANGUAGES`). Functional CJK is allowed and tracked via the allowlist in `tests/test_no_hardcoded_cjk.py` — text-processing regexes, model/engine vocabulary & identifiers (e.g. CosyVoice speaker IDs), localized error matching, demo/eval data, and test fixtures. CI fails on any hardcoded CJK outside the allowlist; to add legitimate functional CJK, extend `_ALLOWED_FILES` there with a justification.
|
||||
|
||||
|
||||
+10
-6
@@ -1,25 +1,29 @@
|
||||
# Alembic configuration for VoiceStudio.
|
||||
# Run from anywhere: alembic -c <repo>/alembic.ini <command>
|
||||
# Default commands:
|
||||
# alembic upgrade head — apply all pending migrations
|
||||
# alembic revision -m "…" — create a new migration
|
||||
# alembic current — show current schema version
|
||||
# alembic upgrade head - apply all pending migrations
|
||||
# alembic revision -m "..." - create a new migration
|
||||
# alembic current - show current schema version
|
||||
#
|
||||
# Keep this file ASCII: alembic reads it in the locale code page, which on a
|
||||
# Chinese, Japanese or Korean Windows cannot decode UTF-8 punctuation
|
||||
# (tests/test_alembic_ini_locale.py).
|
||||
#
|
||||
# DB URL is resolved dynamically from core.config (env aware).
|
||||
# See backend/migrations/env.py.
|
||||
|
||||
[alembic]
|
||||
# %(here)s = this file's directory. Alembic resolves bare relative paths
|
||||
# against the process CWD, not the ini — and the app doesn't always start
|
||||
# against the process CWD, not the ini - and the app doesn't always start
|
||||
# from the repo root (`tauri dev` runs the backend with
|
||||
# cwd=frontend/src-tauri), which made startup migrations die with
|
||||
# "Path doesn't exist: backend/migrations" the first time one was pending.
|
||||
script_location = %(here)s/backend/migrations
|
||||
prepend_sys_path = %(here)s/backend
|
||||
# Split multi-path options on os.pathsep, not the legacy space/comma/colon
|
||||
# set — a colon-split would shred "C:\..." absolute paths on Windows.
|
||||
# set - a colon-split would shred "C:\..." absolute paths on Windows.
|
||||
path_separator = os
|
||||
# sqlalchemy.url is set programmatically in env.py — do NOT set it here.
|
||||
# sqlalchemy.url is set programmatically in env.py - do NOT set it here.
|
||||
sqlalchemy.url =
|
||||
|
||||
[loggers]
|
||||
|
||||
@@ -186,7 +186,8 @@ async def audiobook_import(file: UploadFile = File(...)) -> dict:
|
||||
except ValueError as e:
|
||||
raise HTTPException(status_code=400, detail=f"couldn't parse PDF: {e}")
|
||||
else:
|
||||
script = chapterize_plaintext(data.decode("utf-8", "ignore"))
|
||||
from services.text_upload import decode_text_upload
|
||||
script = chapterize_plaintext(decode_text_upload(data))
|
||||
if not script.strip():
|
||||
raise HTTPException(status_code=400, detail="no text found in the file")
|
||||
plan = parse_audiobook_script(script)
|
||||
|
||||
@@ -1030,11 +1030,14 @@ async def retry_batch_job(job_id: str):
|
||||
if not translation_engines.is_ready(provider):
|
||||
raise HTTPException(409, "Configure the selected translation provider before retrying")
|
||||
if provider == "argos" and job.get("source_lang"):
|
||||
status = await asyncio.to_thread(
|
||||
translation_engines.argos_pack_status,
|
||||
job["source_lang"],
|
||||
job["langs"],
|
||||
)
|
||||
try:
|
||||
status = await asyncio.to_thread(
|
||||
translation_engines.argos_pack_status,
|
||||
job["source_lang"],
|
||||
job["langs"],
|
||||
)
|
||||
except ValueError as exc:
|
||||
raise HTTPException(status_code=422, detail=str(exc)) from exc
|
||||
if any(not pair["installed"] for pair in status["pairs"]):
|
||||
raise HTTPException(409, "Install the required Argos language packs before retrying")
|
||||
|
||||
|
||||
+104
-34
@@ -272,12 +272,10 @@ async def dub_import_srt(job_id: str, file: UploadFile = File(...)):
|
||||
raise HTTPException(status_code=400, detail=f"Could not read uploaded file: {e}") from e
|
||||
if not raw_bytes:
|
||||
raise HTTPException(status_code=400, detail="Uploaded SRT file is empty.")
|
||||
# Most SRT files are UTF-8 (with or without BOM); fall back to latin-1
|
||||
# so legacy Windows-encoded subs don't blow up the import.
|
||||
try:
|
||||
text = raw_bytes.decode("utf-8-sig")
|
||||
except UnicodeDecodeError:
|
||||
text = raw_bytes.decode("latin-1", errors="replace")
|
||||
# Most SRT files are UTF-8, but Windows subtitle tools also save UTF-16
|
||||
# (with a BOM) and Windows-1252; decode those instead of finding no cues.
|
||||
from services.text_upload import decode_text_upload
|
||||
text = decode_text_upload(raw_bytes)
|
||||
|
||||
from services.srt_parser import parse_srt
|
||||
result = parse_srt(text)
|
||||
@@ -869,9 +867,73 @@ _CHUNK_TRANSCRIBE_ATTEMPTS = max(1, int(os.environ.get("OMNIVOICE_TRANSCRIBE_CHU
|
||||
#: by Chrome's ~5 min no-response cap and by reverse-proxy idle timeouts,
|
||||
#: which the UI can only report as the generic "stream dropped" guess.
|
||||
ASR_LOAD_KEEPALIVE_S = float(os.environ.get("OMNIVOICE_ASR_LOAD_KEEPALIVE_S", "15.0"))
|
||||
#: Seconds between `ping` events while a post-transcript step runs on an
|
||||
#: executor (diarization, clone extraction, reference-text refinement, the
|
||||
#: ASR unload / TTS restore). #2108: the per-segment refinement re-runs ASR
|
||||
#: once per segment — 113 passes, ~19 min on an M1 Pro CPU — and was awaited
|
||||
#: bare, so the stream went byte-silent for the whole stretch, the webview
|
||||
#: severed it, and the UI could only say "stream ended early".
|
||||
POST_ASR_PING_S = 5.0
|
||||
|
||||
|
||||
_sse_event = dub_pipeline.sse_event
|
||||
|
||||
|
||||
class _ASRWorkLifetime:
|
||||
"""Keep model cleanup behind native work even after its waiter is cancelled."""
|
||||
|
||||
def __init__(self):
|
||||
import threading
|
||||
self._lock = threading.Lock()
|
||||
self._closed = threading.Event()
|
||||
self._cleaned = False
|
||||
|
||||
def run(self, fn):
|
||||
with self._lock:
|
||||
if self._closed.is_set():
|
||||
raise RuntimeError("Transcription stream has ended")
|
||||
return fn()
|
||||
|
||||
def stop(self):
|
||||
# Reject queued work before it can touch an unloaded model.
|
||||
self._closed.set()
|
||||
|
||||
def cleanup(self, fn):
|
||||
self.stop()
|
||||
with self._lock:
|
||||
if self._cleaned:
|
||||
return
|
||||
# Claim cleanup before invoking even a non-idempotent unload.
|
||||
self._cleaned = True
|
||||
return fn()
|
||||
|
||||
|
||||
async def _ping_while(fut):
|
||||
"""Yield `ping` events every POST_ASR_PING_S until ``fut`` settles.
|
||||
|
||||
Every await in the transcribe stream body that can outlast a few seconds
|
||||
goes through here so the connection never goes byte-silent. The result
|
||||
(or exception) stays on ``fut`` for the caller to read.
|
||||
|
||||
Leaving early — the client disconnected, or the body raised at a `yield` —
|
||||
cancels ``fut`` exactly as a bare ``await fut`` would have, so a wrapped
|
||||
run_transcribe_guarded still runs its abandon path instead of refining on
|
||||
after disconnection. Native threads are not cancelled by Future.cancel();
|
||||
_ASRWorkLifetime orders model cleanup behind their actual completion. Nothing is awaited in the finally: it also runs under GeneratorExit.
|
||||
A failure that lands after we left is still marked retrieved, so it cannot
|
||||
surface later as "exception was never retrieved" (CodeRabbit, #2138) —
|
||||
the same done-callback the TTS-load keepalive uses.
|
||||
"""
|
||||
fut.add_done_callback(lambda f: f.cancelled() or f.exception())
|
||||
try:
|
||||
while True:
|
||||
done, _ = await asyncio.wait({fut}, timeout=POST_ASR_PING_S)
|
||||
if done:
|
||||
return
|
||||
yield _sse_event("ping", {})
|
||||
finally:
|
||||
if not fut.done():
|
||||
fut.cancel()
|
||||
_prep_event_helper = dub_pipeline.prep_event # alias; we keep the module-local _prep_event below for the inline one-liner shape
|
||||
|
||||
#: User-facing warning emitted when auto voice cloning is skipped because the
|
||||
@@ -1045,6 +1107,7 @@ async def dub_transcribe_stream(
|
||||
# _gen_body parks the loaded backend here; the normal unload clears it;
|
||||
# gen()'s `finally` unloads whatever is still parked, on EVERY exit.
|
||||
_loaded_asr: dict = {"backend": None}
|
||||
_asr_work = _ASRWorkLifetime()
|
||||
# Same shape, same reason, for the TTS offload (#1191): offload_tts_for_asr()
|
||||
# moves the TTS model to CPU, and only _gen_body's success path moved it
|
||||
# back — so an abort/error/disconnect stranded it there, silently making
|
||||
@@ -1447,7 +1510,7 @@ async def dub_transcribe_stream(
|
||||
for _attempt in range(1, _CHUNK_TRANSCRIBE_ATTEMPTS + 1):
|
||||
# Run as a task and poll so pings keep the EventSource alive.
|
||||
task = asyncio.ensure_future(run_transcribe_guarded(
|
||||
_gpu_pool, _transcribe_chunk,
|
||||
_gpu_pool, lambda: _asr_work.run(_transcribe_chunk),
|
||||
what=f"Dub chunk {i + 1}/{chunks_n}",
|
||||
timeout=transcribe_timeout_s,
|
||||
timeout_env=transcribe_timeout_env,
|
||||
@@ -1855,16 +1918,10 @@ async def dub_transcribe_stream(
|
||||
"heuristic",
|
||||
)
|
||||
|
||||
fut_diar = loop.run_in_executor(_gpu_pool, _diarize)
|
||||
final_segs = None
|
||||
diar_warning = None
|
||||
labels_source = "heuristic"
|
||||
while True:
|
||||
done, pending = await asyncio.wait([fut_diar], timeout=5.0)
|
||||
if done:
|
||||
final_segs, diar_warning, labels_source = done.pop().result()
|
||||
break
|
||||
yield _sse_event("ping", {})
|
||||
fut_diar = loop.run_in_executor(_gpu_pool, lambda: _asr_work.run(_diarize))
|
||||
async for _ping in _ping_while(fut_diar):
|
||||
yield _ping
|
||||
final_segs, diar_warning, labels_source = fut_diar.result()
|
||||
if job.get("aborted") or task_manager.is_cancelled(job_id):
|
||||
yield _sse_event("aborted", {})
|
||||
return
|
||||
@@ -1931,12 +1988,9 @@ async def dub_transcribe_stream(
|
||||
labels_source=labels_source,
|
||||
),
|
||||
)
|
||||
while True:
|
||||
done, pending = await asyncio.wait([fut_clones], timeout=5.0)
|
||||
if done:
|
||||
clones = done.pop().result()
|
||||
break
|
||||
yield _sse_event("ping", {})
|
||||
async for _ping in _ping_while(fut_clones):
|
||||
yield _ping
|
||||
clones = fut_clones.result()
|
||||
if clones:
|
||||
from services.speaker_clone import refine_ref_texts
|
||||
# Bound the re-transcribe like every other ASR dispatch in
|
||||
@@ -1946,11 +2000,14 @@ async def dub_transcribe_stream(
|
||||
# and raises — keep the original (unrefined) clones, matching
|
||||
# refine_ref_text's own "failure is a strict no-op" fallback.
|
||||
try:
|
||||
clones = await run_transcribe_guarded(
|
||||
fut_refine = asyncio.ensure_future(run_transcribe_guarded(
|
||||
_gpu_pool,
|
||||
lambda: refine_ref_texts(clones, _asr_backend),
|
||||
lambda: _asr_work.run(lambda: refine_ref_texts(clones, _asr_backend)),
|
||||
what="Dub clone ref-text refine",
|
||||
)
|
||||
))
|
||||
async for _ping in _ping_while(fut_refine):
|
||||
yield _ping
|
||||
clones = fut_refine.result()
|
||||
except ASRTimeoutError as e:
|
||||
logger.warning(
|
||||
"clone ref-text refine timed out; keeping original ref_text: %s", e
|
||||
@@ -1973,23 +2030,29 @@ async def dub_transcribe_stream(
|
||||
# reviewers, on the first version of this fix).
|
||||
_seg_clone_dir = _safe_job_dir(job_id) or os.path.dirname(vocals_for_clone)
|
||||
os.makedirs(_seg_clone_dir, exist_ok=True)
|
||||
seg_clones = await loop.run_in_executor(
|
||||
fut_seg_refs = loop.run_in_executor(
|
||||
_cpu_pool, lambda: extract_segment_refs(
|
||||
vocals_for_clone, final_segs,
|
||||
_seg_clone_dir,
|
||||
seg_ids=seg_ids_for_clone,
|
||||
),
|
||||
)
|
||||
async for _ping in _ping_while(fut_seg_refs):
|
||||
yield _ping
|
||||
seg_clones = fut_seg_refs.result()
|
||||
if seg_clones:
|
||||
from services.speaker_clone import refine_ref_texts
|
||||
# Same guard as the per-speaker refine above (#730):
|
||||
# keep the original seg_clones on a wedge/timeout.
|
||||
try:
|
||||
seg_clones = await run_transcribe_guarded(
|
||||
fut_refine = asyncio.ensure_future(run_transcribe_guarded(
|
||||
_gpu_pool,
|
||||
lambda: refine_ref_texts(seg_clones, _asr_backend),
|
||||
lambda: _asr_work.run(lambda: refine_ref_texts(seg_clones, _asr_backend)),
|
||||
what="Dub segment ref-text refine",
|
||||
)
|
||||
))
|
||||
async for _ping in _ping_while(fut_refine):
|
||||
yield _ping
|
||||
seg_clones = fut_refine.result()
|
||||
except ASRTimeoutError as e:
|
||||
logger.warning(
|
||||
"segment ref-text refine timed out; keeping original ref_text: %s", e
|
||||
@@ -2044,13 +2107,19 @@ async def dub_transcribe_stream(
|
||||
# (CodeRabbit review, #1198 — normal-completion half).
|
||||
if _asr_backend:
|
||||
try:
|
||||
await loop.run_in_executor(_gpu_pool, _asr_backend.unload)
|
||||
fut_unload = loop.run_in_executor(_gpu_pool, lambda: _asr_work.cleanup(_asr_backend.unload))
|
||||
async for _ping in _ping_while(fut_unload):
|
||||
yield _ping
|
||||
fut_unload.result()
|
||||
except Exception as e:
|
||||
logger.warning("Failed to unload ASR backend: %s", e)
|
||||
# Unload attempted once — don't retry from gen()'s finally.
|
||||
_loaded_asr["backend"] = None
|
||||
|
||||
await loop.run_in_executor(_cpu_pool, restore_tts_after_asr)
|
||||
fut_restore = loop.run_in_executor(_cpu_pool, restore_tts_after_asr)
|
||||
async for _ping in _ping_while(fut_restore):
|
||||
yield _ping
|
||||
fut_restore.result()
|
||||
# Debt paid — don't make gen()'s finally repeat it.
|
||||
_tts_offloaded["v"] = False
|
||||
|
||||
@@ -2092,6 +2161,7 @@ async def dub_transcribe_stream(
|
||||
yield _sse_event("error", stream_failure("transcription_failed"))
|
||||
yield _sse_event("done", {})
|
||||
finally:
|
||||
_asr_work.stop()
|
||||
# Last-resort VRAM release (see _loaded_asr above): covers crashes,
|
||||
# early terminal-error returns, and client disconnects
|
||||
# (GeneratorExit bypasses the except, never this finally).
|
||||
@@ -2117,7 +2187,7 @@ async def dub_transcribe_stream(
|
||||
# (CodeRabbit review, #1198).
|
||||
try:
|
||||
_fut = asyncio.get_running_loop().run_in_executor(
|
||||
_gpu_pool, _b.unload
|
||||
_gpu_pool, lambda: _asr_work.cleanup(_b.unload)
|
||||
)
|
||||
# Restore the TTS model only AFTER the ASR weights are
|
||||
# freed — the same ordering the success path enforces, so
|
||||
@@ -2126,7 +2196,7 @@ async def dub_transcribe_stream(
|
||||
except RuntimeError:
|
||||
# No running loop (interpreter teardown) — best effort.
|
||||
try:
|
||||
_b.unload()
|
||||
_asr_work.cleanup(_b.unload)
|
||||
except Exception as e:
|
||||
logger.warning("Failed to unload ASR backend: %s", e)
|
||||
_submit_tts_restore()
|
||||
|
||||
@@ -70,6 +70,13 @@ def _unique_stamp() -> str:
|
||||
|
||||
_SAFE_LANG = re.compile(r"^[A-Za-z0-9_-]{1,32}$")
|
||||
|
||||
#: Seconds of silence on a `/tasks/stream` before a keepalive comment goes out.
|
||||
#: A task that is busy but quiet — ffmpeg on a long video, a slow TTS segment,
|
||||
#: a job queued behind another — leaves the stream byte-silent, and byte-silent
|
||||
#: SSE gets severed by the desktop webview, Chrome's ~5 min cap or a proxy's
|
||||
#: idle timeout (#1196, #2108). Comments are invisible to every consumer.
|
||||
TASK_STREAM_KEEPALIVE_S = 15.0
|
||||
|
||||
|
||||
def _job_dir_or_400(job_id: str) -> str:
|
||||
if not re.fullmatch(r"[A-Za-z0-9_-]{1,64}", job_id or ""):
|
||||
@@ -273,14 +280,21 @@ async def stream_task(task_id: str, after_seq: int = 0):
|
||||
await task_manager.add_listener(task_id, q)
|
||||
try:
|
||||
while True:
|
||||
evt = await q.get()
|
||||
try:
|
||||
evt = await asyncio.wait_for(q.get(), timeout=TASK_STREAM_KEEPALIVE_S)
|
||||
except asyncio.TimeoutError:
|
||||
yield ": keepalive\n\n"
|
||||
continue
|
||||
if evt is None:
|
||||
break
|
||||
yield evt
|
||||
finally:
|
||||
await task_manager.remove_listener(task_id, q)
|
||||
|
||||
return StreamingResponse(_reader(), media_type="text/event-stream")
|
||||
return StreamingResponse(
|
||||
_reader(), media_type="text/event-stream",
|
||||
headers={"Cache-Control": "no-cache, no-transform", "X-Accel-Buffering": "no"},
|
||||
)
|
||||
|
||||
|
||||
@router.get("/jobs")
|
||||
@@ -1722,11 +1736,8 @@ async def dub_download_audio(
|
||||
|
||||
|
||||
def _format_srt_time(seconds):
|
||||
h = int(seconds // 3600)
|
||||
m = int((seconds % 3600) // 60)
|
||||
s = int(seconds % 60)
|
||||
ms = int((seconds % 1) * 1000)
|
||||
return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}"
|
||||
from services.srt_parser import format_cue_timestamp
|
||||
return format_cue_timestamp(seconds, ",")
|
||||
|
||||
def _pick_subtitle_text(seg: dict, dual: bool) -> str:
|
||||
"""One line per subtitle cue, unless dual=true and an original exists.
|
||||
@@ -1810,11 +1821,8 @@ async def dub_export_srt(
|
||||
)
|
||||
|
||||
def _format_vtt_time(seconds):
|
||||
h = int(seconds // 3600)
|
||||
m = int((seconds % 3600) // 60)
|
||||
s = int(seconds % 60)
|
||||
ms = int((seconds % 1) * 1000)
|
||||
return f"{h:02d}:{m:02d}:{s:02d}.{ms:03d}"
|
||||
from services.srt_parser import format_cue_timestamp
|
||||
return format_cue_timestamp(seconds, ".")
|
||||
|
||||
@router.get("/dub/vtt/{job_id}")
|
||||
@router.get("/dub/vtt/{job_id}/{filename}")
|
||||
|
||||
@@ -790,6 +790,28 @@ async def dub_translate(req: TranslateRequest):
|
||||
f"switch the Engine dropdown to another provider."
|
||||
)
|
||||
return JSONResponse(status_code=400, content={"error": friendly})
|
||||
# The package imports without its native dep; the *translator*
|
||||
# needs CTranslate2, whose library is rejected outright by kernels
|
||||
# that refuse an executable stack (#692). Repair it (a one-bit ELF
|
||||
# patch), and if that is impossible say so in one actionable 400
|
||||
# instead of the opaque 500 every segment used to produce.
|
||||
try:
|
||||
from core.execstack import ensure_ctranslate2_loadable
|
||||
|
||||
ensure_ctranslate2_loadable()
|
||||
except Exception as e: # noqa: BLE001 — repair must not block translation
|
||||
logger.debug("exec-stack repair unavailable (%s) — continuing", e)
|
||||
try:
|
||||
import argostranslate.translate # noqa: F401
|
||||
except Exception as e: # noqa: BLE001 — OSError here, not ImportError
|
||||
friendly = (
|
||||
f"The '{provider}' engine's CTranslate2 runtime could not be "
|
||||
"loaded in this backend."
|
||||
+ " Switch the Engine dropdown to NLLB (local) or an online "
|
||||
"provider, or reinstall the backend, then retry."
|
||||
)
|
||||
return JSONResponse(status_code=400, content={"error": friendly, "detail": {"code": "argos_runtime_unavailable", "message": friendly}})
|
||||
|
||||
target_codes = list(dict.fromkeys(
|
||||
seg.target_lang if seg.target_lang else req.target_lang
|
||||
for seg in req.segments
|
||||
|
||||
@@ -21,12 +21,12 @@ import os
|
||||
import threading
|
||||
from time import perf_counter
|
||||
|
||||
from fastapi import APIRouter, Depends, HTTPException
|
||||
from fastapi import APIRouter, Depends, HTTPException, Request
|
||||
from huggingface_hub import utils as hf_utils
|
||||
from huggingface_hub.errors import HFValidationError
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from api.dependencies import require_admin, require_admin_action, require_desktop
|
||||
from api.dependencies import require_admin, require_admin_action, require_desktop, is_loopback
|
||||
from core import prefs
|
||||
from core.engine_licenses import LICENSE_GATED_ENGINES
|
||||
from services import tts_backend, asr_backend, llm_backend, translation_engines
|
||||
@@ -85,6 +85,9 @@ def _family_payload(family: str, module):
|
||||
|
||||
for backend in backends:
|
||||
engine_id = backend.get("id")
|
||||
if engine_id == active == "mlx-audio":
|
||||
# Constructor resolves model preferences only; never loads weights.
|
||||
backend["supports_cloning"] = tts_backend.MLXAudioBackend().supports_cloning
|
||||
if engine_id in LICENSE_GATED_ENGINES:
|
||||
backend["license_required"] = True
|
||||
try:
|
||||
@@ -116,18 +119,29 @@ def _is_hf_repo_id(value: str) -> bool:
|
||||
return True
|
||||
|
||||
|
||||
def _request_install_capability(payload, request):
|
||||
allowed = bool(request and request.client and is_loopback(request.client.host))
|
||||
result = dict(payload)
|
||||
result["backends"] = [dict(entry) for entry in payload["backends"]]
|
||||
for entry in result["backends"]:
|
||||
if entry.get("one_click_install") and not allowed:
|
||||
entry["one_click_install"] = False
|
||||
entry["local_install_required"] = True
|
||||
return result
|
||||
|
||||
|
||||
@router.get("/engines")
|
||||
def list_all_engines():
|
||||
def list_all_engines(request: Request):
|
||||
return {
|
||||
"tts": _family_payload("tts", tts_backend),
|
||||
"tts": _request_install_capability(_family_payload("tts", tts_backend), request),
|
||||
"asr": _family_payload("asr", asr_backend),
|
||||
"llm": _family_payload("llm", llm_backend),
|
||||
}
|
||||
|
||||
|
||||
@router.get("/engines/tts")
|
||||
def list_tts_backends():
|
||||
return _family_payload("tts", tts_backend)
|
||||
def list_tts_backends(request: Request):
|
||||
return _request_install_capability(_family_payload("tts", tts_backend), request)
|
||||
|
||||
|
||||
@router.get(
|
||||
@@ -409,10 +423,10 @@ async def uninstall_translation_engine(engine_id: str):
|
||||
"/engines/audiocpp/runtime/install/status",
|
||||
dependencies=[Depends(require_admin)],
|
||||
)
|
||||
def audiocpp_runtime_install_status():
|
||||
def audiocpp_runtime_install_status(request: Request):
|
||||
from services import audiocpp_runtime_install
|
||||
|
||||
return audiocpp_runtime_install.status()
|
||||
return {**audiocpp_runtime_install.status(), "install_allowed": bool(request.client and is_loopback(request.client.host))}
|
||||
|
||||
|
||||
@router.post(
|
||||
@@ -485,7 +499,7 @@ def install_sidecar_engine(engine_id: str):
|
||||
"/engines/sidecar/{engine_id}/install/status",
|
||||
dependencies=[Depends(require_admin)],
|
||||
)
|
||||
def sidecar_install_status(engine_id: str):
|
||||
def sidecar_install_status(engine_id: str, request: Request = None):
|
||||
"""Step-by-step status of the sidecar install job (poll while running).
|
||||
|
||||
Shape: ``{engine_id, installed, managed, install_dir, job}`` where job is
|
||||
@@ -494,7 +508,7 @@ def sidecar_install_status(engine_id: str):
|
||||
"""
|
||||
from services import sidecar_install
|
||||
try:
|
||||
return sidecar_install.get_status(engine_id)
|
||||
return {**sidecar_install.get_status(engine_id), "install_allowed": bool(request and request.client and is_loopback(request.client.host))}
|
||||
except KeyError:
|
||||
raise HTTPException(
|
||||
status_code=404,
|
||||
|
||||
@@ -143,7 +143,7 @@ def _resolve_profile_conditioning(row, *, ref_text=None, instruct=None,
|
||||
out = {
|
||||
"ref_audio_path": None, "ref_text": ref_text, "instruct": instruct,
|
||||
"seed": seed, "language": language, "kind": None,
|
||||
"persist_ref_text": False,
|
||||
"persist_ref_text": False, "language_from_profile": False,
|
||||
}
|
||||
# `kind` is authoritative (0005): 'design' profiles condition on their
|
||||
# deterministic rendered sample + instruct; 'clone' on the user's
|
||||
@@ -212,6 +212,12 @@ def _resolve_profile_conditioning(row, *, ref_text=None, instruct=None,
|
||||
prof_lang = None
|
||||
if prof_lang and prof_lang != "Auto":
|
||||
out["language"] = prof_lang
|
||||
# #2156: record that the caller never asked for this language. The
|
||||
# UI omits `language` entirely while its picker reads "Auto", so a
|
||||
# profile-filled language must not be reported back as if the user
|
||||
# had picked it — an engine that can't speak it would otherwise
|
||||
# tell them to "leave language as Auto", which is what they did.
|
||||
out["language_from_profile"] = True
|
||||
return out
|
||||
|
||||
|
||||
@@ -807,6 +813,7 @@ def _oom_friendly_reraise(e):
|
||||
def _generate_timeout_s(
|
||||
text: str,
|
||||
*,
|
||||
engine: object = None,
|
||||
execution_device=None,
|
||||
min_vram_gb=0.0,
|
||||
hardware_family=None,
|
||||
@@ -827,6 +834,7 @@ def _generate_timeout_s(
|
||||
from services.model_manager import generate_timeout_s
|
||||
return generate_timeout_s(
|
||||
text,
|
||||
engine=engine,
|
||||
execution_device=execution_device,
|
||||
min_vram_gb=min_vram_gb,
|
||||
hardware_family=hardware_family,
|
||||
@@ -1041,6 +1049,18 @@ _LANGUAGE_REJECTION_SIGNATURES = (
|
||||
"unsupported language code",
|
||||
)
|
||||
|
||||
# Engine-specific rejections that ALREADY name the engine and what it supports,
|
||||
# so #1257's generic rewrite deliberately leaves them alone — re-wrapping them
|
||||
# only nests "Engine's own message:" twice. They still have to be recognised as
|
||||
# language rejections for #2156's provenance check, which cares about the
|
||||
# *cause* of the language, not the quality of the wording.
|
||||
_SELF_DESCRIBING_LANGUAGE_REJECTIONS = (
|
||||
# services/tts_backend.py: "…doesn't support language='Persian'. Kokoro
|
||||
# supports: …" — mlx-audio's Kokoro, the engine reported in #2156.
|
||||
"doesn't support language",
|
||||
"does not support language",
|
||||
)
|
||||
|
||||
#: `unsupported language: xx` / `unsupported language 'xx'` — but not
|
||||
#: `unsupported language model ...`.
|
||||
_LANGUAGE_REJECTION_RE = re.compile(
|
||||
@@ -1050,15 +1070,87 @@ _LANGUAGE_REJECTION_RE = re.compile(
|
||||
)
|
||||
|
||||
|
||||
def _is_language_rejection(text: str) -> bool:
|
||||
"""True when an engine failure is about the LANGUAGE it was handed.
|
||||
|
||||
Matched on the message, not the type: the engines multiplex third-party
|
||||
libraries that each raise their own class. Covers the self-describing
|
||||
wordings too — #1257's rewrite skips those, but #2156 still needs to know a
|
||||
language was refused so it can say where that language came from.
|
||||
"""
|
||||
low = text.lower()
|
||||
return (
|
||||
any(sig in low for sig in _LANGUAGE_REJECTION_SIGNATURES)
|
||||
or any(sig in low for sig in _SELF_DESCRIBING_LANGUAGE_REJECTIONS)
|
||||
or bool(_LANGUAGE_REJECTION_RE.search(text))
|
||||
)
|
||||
|
||||
|
||||
def _profile_language_rejection_detail(exc: BaseException, language) -> str:
|
||||
"""The 400 body for a language the *voice profile* supplied, not the user.
|
||||
|
||||
#2156: the UI omits `language` while its picker reads "Auto", and #533
|
||||
fills that gap from the selected profile. When the active engine can't
|
||||
speak the profile's language the engine's own message tells the user to
|
||||
"leave language as 'Auto'" — which is exactly what they did, so the advice
|
||||
cannot be acted on. Name the real source and the remedies that exist.
|
||||
"""
|
||||
return (
|
||||
f"This voice profile is saved with the language '{language}', and the "
|
||||
f"active engine can't speak it. The language picker being on \"Auto\" "
|
||||
f"does not override that — Auto fills the language in from the "
|
||||
f"profile. Set this voice profile's language to one the engine "
|
||||
f"supports, pick a supported language explicitly for this render, or "
|
||||
f"switch engine in Model Catalogue (the VoiceStudio engine has the "
|
||||
f"widest coverage). Engine's own message: {exc}"
|
||||
)
|
||||
|
||||
|
||||
def _language_rejection_payload(exc, language, *, from_profile):
|
||||
"""Stable, non-retryable error metadata for both response transports."""
|
||||
from core.public_errors import stream_failure
|
||||
failure = stream_failure("invalid_request")
|
||||
failure["terminal"] = True
|
||||
if from_profile:
|
||||
failure.update(
|
||||
code="profile_language_rejected", language=language,
|
||||
detail=_profile_language_rejection_detail(_root_language_error(exc), language),
|
||||
)
|
||||
return failure
|
||||
|
||||
|
||||
def _language_rejection_http_error(exc: BaseException, language, *, from_profile):
|
||||
"""The 400 a refused language deserves, wherever the refusal was raised.
|
||||
|
||||
A language an engine cannot speak is never retryable — not by waiting, and
|
||||
not by re-running the same request on another machine. Built here so the
|
||||
local (`ValueError`) and remote (`RemoteJobFailed`) handlers cannot drift:
|
||||
#2156 shipped the profile-aware branch on the local path only, and a remote
|
||||
render kept answering with a retryable 503 that offered "run it on this
|
||||
machine instead", which cannot help.
|
||||
"""
|
||||
root = _root_language_error(exc)
|
||||
detail = (
|
||||
_profile_language_rejection_detail(root, language)
|
||||
if from_profile else str(exc)
|
||||
)
|
||||
if from_profile:
|
||||
detail = {"code": "profile_language_rejected", "language": language, "message": detail}
|
||||
return HTTPException(status_code=400, detail=detail)
|
||||
|
||||
|
||||
def _language_rejection_or(e: BaseException, backend, language):
|
||||
"""``e`` rewritten with engine context when it's a language rejection.
|
||||
|
||||
Returns ``e`` unchanged otherwise, so this is safe to wrap any failure in.
|
||||
Matched on the message, not the type: the engines multiplex third-party
|
||||
libraries that each raise their own class.
|
||||
Deliberately narrower than :func:`_is_language_rejection`: a message that
|
||||
already names its engine and the languages it supports is left alone rather
|
||||
than nested inside a second "Engine's own message:".
|
||||
"""
|
||||
text = str(e)
|
||||
low = text.lower()
|
||||
if any(sig in low for sig in _SELF_DESCRIBING_LANGUAGE_REJECTIONS):
|
||||
return e
|
||||
if not any(sig in low for sig in _LANGUAGE_REJECTION_SIGNATURES) and not (
|
||||
_LANGUAGE_REJECTION_RE.search(text)
|
||||
):
|
||||
@@ -1067,13 +1159,23 @@ def _language_rejection_or(e: BaseException, backend, language):
|
||||
type(backend), "id", type(backend).__name__
|
||||
)
|
||||
requested = f" '{language}'" if language else ""
|
||||
return ValueError(
|
||||
rewritten = ValueError(
|
||||
f"The {engine} engine can't speak{requested}. VoiceStudio offers every "
|
||||
f"language its default engine supports, but each engine covers a "
|
||||
f"different set — pick one this engine supports, or switch engine in "
|
||||
f"Model Catalogue (the VoiceStudio engine has the widest coverage) "
|
||||
f"and generate again. Engine's own message: {e}"
|
||||
)
|
||||
# Keep the engine's own text reachable. #2156's profile message quotes the
|
||||
# engine once; without this it would quote THIS wrapper, repeating both the
|
||||
# engine-switch remedy and "Engine's own message:" twice.
|
||||
rewritten.engine_language_error = e
|
||||
return rewritten
|
||||
|
||||
|
||||
def _root_language_error(exc: BaseException) -> BaseException:
|
||||
"""The engine's own rejection, unwrapping :func:`_language_rejection_or`."""
|
||||
return getattr(exc, "engine_language_error", exc)
|
||||
|
||||
|
||||
def _persist_profile_ref_text(profile_id: str, ref_text: str) -> None:
|
||||
@@ -1568,6 +1670,9 @@ async def generate_speech(
|
||||
ref_lease = None
|
||||
used_seed = seed
|
||||
resolved_profile_id = None
|
||||
# #2156: True once a profile's stored language fills a language the caller
|
||||
# never sent, so a rejection can name the profile instead of the picker.
|
||||
language_from_profile = False
|
||||
history_mode = None # profile.kind when a profile drives; else inferred at insert
|
||||
# #1032: profile id to persist an auto-transcribed reference transcript to.
|
||||
# Set only for a plain (unlocked) clone profile whose stored ref_text is
|
||||
@@ -1596,6 +1701,7 @@ async def generate_speech(
|
||||
instruct = _cond["instruct"]
|
||||
used_seed = _cond["seed"]
|
||||
language = _cond["language"]
|
||||
language_from_profile = _cond["language_from_profile"]
|
||||
if _cond["persist_ref_text"]:
|
||||
persist_ref_text_profile_id = profile_id
|
||||
elif ref_audio is not None:
|
||||
@@ -1752,6 +1858,7 @@ async def generate_speech(
|
||||
what="TTS generate",
|
||||
timeout=_generate_timeout_s(
|
||||
text,
|
||||
engine=_backend,
|
||||
execution_device=_routing["effective_device"],
|
||||
min_vram_gb=_engine_min_vram_gb,
|
||||
hardware_family=_routing_hardware_family,
|
||||
@@ -1869,10 +1976,14 @@ async def generate_speech(
|
||||
# holding what is often its only slot until the lease lapses.
|
||||
render.cancel()
|
||||
raise
|
||||
except ValueError:
|
||||
except ValueError as e:
|
||||
logger.error("Remote generation request rejected")
|
||||
from core.public_errors import stream_failure
|
||||
yield _line({"type": "error", **stream_failure("invalid_request")})
|
||||
failure = (
|
||||
_language_rejection_payload(e, language, from_profile=language_from_profile)
|
||||
if _is_language_rejection(str(e)) else stream_failure("invalid_request")
|
||||
)
|
||||
yield _line({"type": "error", **failure})
|
||||
except gpu_gateway.ModelNotDownloaded as e:
|
||||
logger.warning("Remote model missing on %s", _target_label)
|
||||
from core.public_errors import stream_failure
|
||||
@@ -1888,13 +1999,16 @@ async def generate_speech(
|
||||
except gpu_gateway.RemoteJobFailed as e:
|
||||
logger.error("Remote generate failed on %s", _target_label)
|
||||
from core.public_errors import stream_failure
|
||||
yield _line({
|
||||
"type": "error",
|
||||
**stream_failure("generation_failed"),
|
||||
"retryable": True,
|
||||
"target_label": e.worker_label or _target_label,
|
||||
"hint": e.hint,
|
||||
})
|
||||
if _is_language_rejection(str(e)):
|
||||
yield _line({"type": "error", **_language_rejection_payload(
|
||||
e, language, from_profile=language_from_profile,
|
||||
)})
|
||||
else:
|
||||
yield _line({
|
||||
"type": "error", **stream_failure("generation_failed"),
|
||||
"retryable": True, "target_label": e.worker_label or _target_label,
|
||||
"hint": e.hint,
|
||||
})
|
||||
except Exception as exc:
|
||||
# Mid-job remote failure is NOT quietly redone here: the client
|
||||
# treats a retryable error as "surface it", so the user decides
|
||||
@@ -2052,6 +2166,7 @@ async def generate_speech(
|
||||
min_vram_gb=_engine_min_vram_gb,
|
||||
timeout=_generate_timeout_s(
|
||||
text,
|
||||
engine=_backend,
|
||||
execution_device=_routing["effective_device"],
|
||||
min_vram_gb=_engine_min_vram_gb,
|
||||
hardware_family=_routing_hardware_family,
|
||||
@@ -2078,6 +2193,7 @@ async def generate_speech(
|
||||
min_vram_gb=_engine_min_vram_gb,
|
||||
timeout=_generate_timeout_s(
|
||||
text,
|
||||
engine=_backend,
|
||||
execution_device=_routing["effective_device"],
|
||||
min_vram_gb=_engine_min_vram_gb,
|
||||
hardware_family=_routing_hardware_family,
|
||||
@@ -2124,6 +2240,7 @@ async def generate_speech(
|
||||
# even after the v0.3.22 scaled budget shipped.
|
||||
timeout=_generate_timeout_s(
|
||||
chunk_text,
|
||||
engine=_backend,
|
||||
execution_device=_routing["effective_device"],
|
||||
min_vram_gb=_engine_min_vram_gb,
|
||||
hardware_family=_routing_hardware_family,
|
||||
@@ -2200,10 +2317,14 @@ async def generate_speech(
|
||||
failure = stream_failure("generation_timeout")
|
||||
failure["retry_after"] = 30
|
||||
yield _line({"type": "error", **failure})
|
||||
except ValueError:
|
||||
except ValueError as e:
|
||||
logger.error("Streaming generation request rejected")
|
||||
from core.public_errors import stream_failure
|
||||
yield _line({"type": "error", **stream_failure("invalid_request")})
|
||||
failure = (
|
||||
_language_rejection_payload(e, language, from_profile=language_from_profile)
|
||||
if _is_language_rejection(str(e)) else stream_failure("invalid_request")
|
||||
)
|
||||
yield _line({"type": "error", **failure})
|
||||
except Exception as exc:
|
||||
# A streaming request answers 200 and carries its failure as an
|
||||
# in-band error frame, so it never reaches the global 500
|
||||
@@ -2290,6 +2411,7 @@ async def generate_speech(
|
||||
_local_render, what="TTS generate",
|
||||
timeout=_generate_timeout_s(
|
||||
text,
|
||||
engine=_backend,
|
||||
execution_device=_routing["effective_device"],
|
||||
min_vram_gb=_engine_min_vram_gb,
|
||||
hardware_family=_routing_hardware_family,
|
||||
@@ -2385,6 +2507,15 @@ async def generate_speech(
|
||||
# the client can offer "run it on this machine instead" — a resubmit
|
||||
# the user chose, with a wait they were told about.
|
||||
logger.error("Remote generate failed on %s: %s", _target_label, e)
|
||||
# #2156: a language the engine can't speak is a request problem, not a
|
||||
# worker problem. It travels home as RemoteJobFailed — caught here,
|
||||
# ahead of the ValueError branch below — so without this the user is
|
||||
# told to retry on this machine, where the same engine refuses the same
|
||||
# language. Answer it as the 400 it is, on either path.
|
||||
if _is_language_rejection(str(e)):
|
||||
raise _language_rejection_http_error(
|
||||
e, language, from_profile=language_from_profile
|
||||
) from e
|
||||
raise HTTPException(
|
||||
status_code=503,
|
||||
detail=f"{e} {e.hint or 'Run it on this machine instead, or pick another GPU.'}",
|
||||
@@ -2419,6 +2550,15 @@ async def generate_speech(
|
||||
raise HTTPException(status_code=503, detail=str(e)) from e
|
||||
except ValueError as e:
|
||||
logger.error("Validation failed: %s", e)
|
||||
# #2156: the language the engine refused was never chosen by the user —
|
||||
# it came from the selected voice profile because the picker was on
|
||||
# "Auto". The engine's own remedy ("leave language as 'Auto'") is then
|
||||
# unfollowable, so say where the language actually came from. Only this
|
||||
# scope knows that; the engine adapters never see the provenance.
|
||||
if language_from_profile and _is_language_rejection(str(e)):
|
||||
raise _language_rejection_http_error(
|
||||
e, language, from_profile=True
|
||||
) from e
|
||||
# Most ValueErrors here are VoiceStudio's own validation messages and
|
||||
# are exactly what the user should read. A few are raw library text
|
||||
# naming parameters and files the user cannot act on — those get the
|
||||
|
||||
@@ -721,17 +721,11 @@ def list_voices():
|
||||
|
||||
def _format_ts_srt(seconds: float) -> str:
|
||||
"""Format seconds as SRT timestamp: HH:MM:SS,mmm"""
|
||||
h = int(seconds // 3600)
|
||||
m = int((seconds % 3600) // 60)
|
||||
s = int(seconds % 60)
|
||||
ms = int((seconds % 1) * 1000)
|
||||
return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}"
|
||||
from services.srt_parser import format_cue_timestamp
|
||||
return format_cue_timestamp(seconds, ",")
|
||||
|
||||
|
||||
def _format_ts_vtt(seconds: float) -> str:
|
||||
"""Format seconds as VTT timestamp: HH:MM:SS.mmm"""
|
||||
h = int(seconds // 3600)
|
||||
m = int((seconds % 3600) // 60)
|
||||
s = int(seconds % 60)
|
||||
ms = int((seconds % 1) * 1000)
|
||||
return f"{h:02d}:{m:02d}:{s:02d}.{ms:03d}"
|
||||
from services.srt_parser import format_cue_timestamp
|
||||
return format_cue_timestamp(seconds, ".")
|
||||
|
||||
@@ -230,7 +230,16 @@ def _segmented_snapshot(repo_id: str, *, endpoint: "str | None", revision: str)
|
||||
from services.segmented_download import segmented_download
|
||||
from services.token_resolver import resolve as _resolve_token
|
||||
|
||||
token = _resolve_token()
|
||||
# `resolve()` returns a ResolvedToken record, not the bearer string, and
|
||||
# every consumer below is typed `token: str | None`. Handing over the
|
||||
# record fails silently rather than loudly (#2163): huggingface_hub's
|
||||
# build_hf_headers ignores a non-str token and falls back to its own
|
||||
# ambient discovery, so a token held only in VoiceStudio's settings sends
|
||||
# NO Authorization header at all and every gated file 401s; our own
|
||||
# segmented_download interpolates it into `f"Bearer {token}"` and sends a
|
||||
# malformed header carrying the raw secret. Unwrap once, here.
|
||||
_resolved = _resolve_token()
|
||||
token = _resolved.token if _resolved else None
|
||||
api = HfApi(endpoint=endpoint, token=token)
|
||||
info = api.repo_info(repo_id, repo_type="model", revision=revision)
|
||||
commit = info.sha
|
||||
|
||||
@@ -337,6 +337,7 @@ async def convert_speech(
|
||||
what="Voice convert",
|
||||
timeout=_generate_timeout_s(
|
||||
text,
|
||||
engine=backend,
|
||||
execution_device=compute_profile["effective_device"],
|
||||
min_vram_gb=compute_profile["min_vram_gb"],
|
||||
hardware_family=compute_profile.get("runtime_hardware_family"),
|
||||
|
||||
@@ -0,0 +1,245 @@
|
||||
"""Repair wheel-shipped shared libraries that request an executable stack.
|
||||
|
||||
Why this exists
|
||||
---------------
|
||||
CTranslate2 wheels up to and including 4.4.0 ship
|
||||
``ctranslate2.libs/libctranslate2-*.so`` with ``PT_GNU_STACK`` marked
|
||||
``RWE`` — a request for an executable stack. Linux kernels that refuse to
|
||||
grant it (hardened kernels, and mainline from 6.x onwards) fail the
|
||||
``dlopen`` outright::
|
||||
|
||||
ImportError: libctranslate2-d3638643.so.4.4.0: cannot enable executable
|
||||
stack as shared object requires: Invalid argument
|
||||
|
||||
Everything that links CTranslate2 dies with it: the **whisperx** and
|
||||
**faster-whisper** ASR engines (#692) *and* Argos translation, which is the
|
||||
default dub translation engine (``argostranslate.translate`` imports
|
||||
``ctranslate2``). #692 taught the ASR selector to fall through to another
|
||||
engine; it never fixed the library, so Linux users on Python 3.11 lost both
|
||||
engines. The pin is upstream and not ours to lift: whisperx 3.4.5 — the last
|
||||
release that supports Python 3.11, which is what ``.python-version``, CI and
|
||||
the installers use — requires ``ctranslate2<4.5.0``, and 4.5.0 is the first
|
||||
release whose ``.so`` drops the exec-stack request.
|
||||
|
||||
The flag is a single bit in the ELF program header, so we clear it in place
|
||||
rather than shipping a patched wheel or asking users for ``patchelf`` (which
|
||||
is not installed on a typical desktop). Inspection is a ~100-byte read with
|
||||
no imports, so :func:`ensure_ctranslate2_loadable` is cheap enough to call
|
||||
from an availability probe: it only rewrites a file when that file would
|
||||
otherwise refuse to load.
|
||||
|
||||
Everything here is a no-op off Linux (macOS/Windows have no such rejection)
|
||||
and handles malformed ELF data — a repair that cannot happen returns a reason, and the
|
||||
caller degrades exactly as it did before.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import glob
|
||||
import logging
|
||||
import os
|
||||
import struct
|
||||
import sys
|
||||
import threading
|
||||
|
||||
logger = logging.getLogger("omnivoice.execstack")
|
||||
|
||||
#: ELF segment type for the stack-permission marker, and its executable bit.
|
||||
_PT_GNU_STACK = 0x6474E551
|
||||
_PF_X = 0x1
|
||||
|
||||
#: Serialize in-process writes; flock also coordinates sidecar processes.
|
||||
_REPAIR_LOCK = threading.RLock()
|
||||
|
||||
|
||||
def _elf_header(fh) -> tuple[str, int, int, int, bool] | None:
|
||||
"""Return ``(endian_prefix, e_phoff, e_phentsize, e_phnum, is_64)`` or None.
|
||||
|
||||
None means "not an ELF file we understand" — which is a normal answer
|
||||
(a ``.so`` stub, a text file, a Mach-O), never an error.
|
||||
"""
|
||||
fh.seek(0)
|
||||
ident = fh.read(16)
|
||||
if len(ident) < 16 or ident[:4] != b"\x7fELF":
|
||||
return None
|
||||
if ident[4] not in (1, 2) or ident[5] not in (1, 2):
|
||||
return None
|
||||
is_64 = ident[4] == 2
|
||||
endian = "<" if ident[5] == 1 else ">"
|
||||
fh.seek(0, os.SEEK_END)
|
||||
size = fh.tell()
|
||||
header_size = 64 if is_64 else 52
|
||||
if size < header_size:
|
||||
return None
|
||||
fh.seek(0)
|
||||
header = fh.read(header_size)
|
||||
if len(header) != header_size:
|
||||
return None
|
||||
e_phoff = struct.unpack_from(endian + ("Q" if is_64 else "I"), header, 0x20 if is_64 else 0x1C)[0]
|
||||
e_phentsize, e_phnum = struct.unpack_from(endian + "HH", header, 0x36 if is_64 else 0x2A)
|
||||
if (not e_phnum or e_phoff < header_size or
|
||||
e_phentsize < (56 if is_64 else 32) or
|
||||
e_phoff + e_phentsize * e_phnum > size):
|
||||
return None
|
||||
# p_flags sits at a different offset per class (ELF64 puts it right after
|
||||
# p_type; ELF32 puts it last), so the caller needs the class too.
|
||||
return endian, e_phoff, e_phentsize, e_phnum, is_64
|
||||
|
||||
|
||||
def _gnu_stack_flags_offset(fh) -> tuple[int, int, str] | None:
|
||||
"""Locate the ``PT_GNU_STACK`` ``p_flags`` field.
|
||||
|
||||
Returns ``(file_offset, flags_value, endian_prefix)``, or None when the
|
||||
file is not an ELF or carries no such segment.
|
||||
"""
|
||||
parsed = _elf_header(fh)
|
||||
if parsed is None:
|
||||
return None
|
||||
endian, e_phoff, e_phentsize, e_phnum, is_64 = parsed
|
||||
flags_rel = 4 if is_64 else 24 # p_flags offset inside the program header
|
||||
for i in range(e_phnum):
|
||||
base = e_phoff + i * e_phentsize
|
||||
fh.seek(base)
|
||||
raw = fh.read(e_phentsize)
|
||||
if len(raw) < flags_rel + 4:
|
||||
continue
|
||||
(p_type,) = struct.unpack_from(endian + "I", raw, 0)
|
||||
if p_type != _PT_GNU_STACK:
|
||||
continue
|
||||
(p_flags,) = struct.unpack_from(endian + "I", raw, flags_rel)
|
||||
return base + flags_rel, p_flags, endian
|
||||
return None
|
||||
|
||||
|
||||
def has_execstack(path: str) -> bool | None:
|
||||
"""True when ``path`` requests an executable stack.
|
||||
|
||||
None when the question does not apply: unreadable, not an ELF, or no
|
||||
``PT_GNU_STACK`` segment.
|
||||
"""
|
||||
try:
|
||||
with open(path, "rb") as fh:
|
||||
found = _gnu_stack_flags_offset(fh)
|
||||
except OSError:
|
||||
return None
|
||||
if found is None:
|
||||
return None
|
||||
_offset, flags, _endian = found
|
||||
return bool(flags & _PF_X)
|
||||
|
||||
|
||||
def clear_execstack(path: str) -> tuple[bool, str]:
|
||||
"""Clear the executable-stack request on ``path``.
|
||||
|
||||
Returns ``(changed, detail)``. ``changed`` is False both when there was
|
||||
nothing to do and when the write was refused (a read-only bundle, for
|
||||
instance) — ``detail`` says which.
|
||||
"""
|
||||
try:
|
||||
# Lock and inspect the same descriptor we write: another process may
|
||||
# already have repaired it, or the wheel may have been replaced.
|
||||
with _REPAIR_LOCK, open(path, "r+b") as fh:
|
||||
# Use host capability, not an emulated target platform.
|
||||
if os.name == "posix":
|
||||
import fcntl
|
||||
fcntl.flock(fh, fcntl.LOCK_EX)
|
||||
found = _gnu_stack_flags_offset(fh)
|
||||
if found is None:
|
||||
return False, "no PT_GNU_STACK segment"
|
||||
offset, flags, endian = found
|
||||
if not flags & _PF_X:
|
||||
return False, "already non-executable"
|
||||
fh.seek(offset)
|
||||
fh.write(struct.pack(endian + "I", flags & ~_PF_X))
|
||||
fh.flush()
|
||||
os.fsync(fh.fileno())
|
||||
except OSError as e:
|
||||
return False, f"unreadable or not writable ({e.__class__.__name__})"
|
||||
return True, "cleared PT_GNU_STACK executable bit"
|
||||
|
||||
|
||||
def ctranslate2_library_paths() -> list[str]:
|
||||
"""Native libraries shipped with the installed ``ctranslate2`` wheel.
|
||||
|
||||
Found without importing ``ctranslate2`` — importing it is the very thing
|
||||
that fails when the exec-stack bit is set.
|
||||
"""
|
||||
import importlib.util
|
||||
|
||||
roots: list[str] = []
|
||||
try:
|
||||
spec = importlib.util.find_spec("ctranslate2")
|
||||
except (ImportError, ValueError): # pragma: no cover — defensive
|
||||
spec = None
|
||||
locations = list(getattr(spec, "submodule_search_locations", None) or []) if spec else []
|
||||
for pkg_dir in locations:
|
||||
roots.append(pkg_dir)
|
||||
roots.append(os.path.join(os.path.dirname(pkg_dir), "ctranslate2.libs"))
|
||||
# Frozen builds flatten the wheel into the bundle directory.
|
||||
meipass = getattr(sys, "_MEIPASS", None)
|
||||
if meipass:
|
||||
roots.append(meipass)
|
||||
roots.append(os.path.join(meipass, "ctranslate2.libs"))
|
||||
out: list[str] = []
|
||||
for root in roots:
|
||||
if not os.path.isdir(root):
|
||||
continue
|
||||
for pattern in ("libctranslate2*.so*", "libctranslate2*.dylib"):
|
||||
out.extend(sorted(glob.glob(os.path.join(root, pattern))))
|
||||
# Dedupe, preserving order.
|
||||
return list(dict.fromkeys(out))
|
||||
|
||||
|
||||
def ensure_ctranslate2_loadable() -> tuple[bool, str]:
|
||||
"""Make ``import ctranslate2`` possible on kernels that refuse exec stacks.
|
||||
|
||||
Returns ``(ok, detail)`` where ``ok`` is False only when a library needs
|
||||
the repair and could not get it — the caller should then report its
|
||||
engine unavailable with ``detail`` as the reason. The repair is idempotent. Re-probe on each call so installation or
|
||||
external repair takes effect without restarting.
|
||||
"""
|
||||
|
||||
result: tuple[bool, str]
|
||||
if sys.platform != "linux":
|
||||
# Only Linux rejects an exec-stack request at dlopen time.
|
||||
result = (True, "not applicable off Linux")
|
||||
else:
|
||||
libs = ctranslate2_library_paths()
|
||||
if not libs:
|
||||
result = (True, "no ctranslate2 library found")
|
||||
else:
|
||||
repaired: list[str] = []
|
||||
blocked: list[str] = []
|
||||
for lib in libs:
|
||||
if has_execstack(lib) is not True:
|
||||
continue
|
||||
changed, detail = clear_execstack(lib)
|
||||
if changed:
|
||||
repaired.append(os.path.basename(lib))
|
||||
logger.warning(
|
||||
"Repaired %s: %s — its executable-stack request is "
|
||||
"rejected by this kernel, which broke whisperx, "
|
||||
"faster-whisper and Argos translation (#692)",
|
||||
os.path.basename(lib), detail,
|
||||
)
|
||||
elif has_execstack(lib) is not False:
|
||||
blocked.append(f"{os.path.basename(lib)} ({detail})")
|
||||
if blocked:
|
||||
result = (
|
||||
False,
|
||||
"ctranslate2's native library requests an executable stack, "
|
||||
"which this kernel refuses, and it could not be patched: "
|
||||
+ "; ".join(blocked)
|
||||
+ ". Reinstall the backend on Python 3.12+ (which resolves "
|
||||
"ctranslate2 4.8+, without the exec-stack request), or run "
|
||||
"`patchelf --clear-execstack <library>` once.",
|
||||
)
|
||||
elif repaired:
|
||||
result = (True, "repaired " + ", ".join(repaired))
|
||||
else:
|
||||
result = (True, "no exec-stack request")
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def reset_ctranslate2_cache() -> None:
|
||||
"""Compatibility hook; recoverable probe results are no longer cached."""
|
||||
@@ -24,7 +24,7 @@ from pathlib import Path
|
||||
# tests/test_app_version.py::test_all_version_files_in_lockstep and bumped by
|
||||
# release.yml's version-bump job, so it stays equal to
|
||||
# pyproject/tauri.conf/Cargo/package.json.
|
||||
_FALLBACK_VERSION = "0.5.3"
|
||||
_FALLBACK_VERSION = "0.5.4"
|
||||
|
||||
|
||||
def _fallback_version() -> str:
|
||||
|
||||
@@ -84,6 +84,10 @@ def _is_ct_error(msg):
|
||||
def _get_model():
|
||||
global _model
|
||||
if _model is None:
|
||||
from core.execstack import ensure_ctranslate2_loadable
|
||||
ok, detail = ensure_ctranslate2_loadable()
|
||||
if not ok:
|
||||
raise ImportError(f"faster-whisper cannot load CTranslate2: {detail}")
|
||||
from faster_whisper import WhisperModel
|
||||
# Same weights as in-process faster-whisper: ASR_MODEL_FASTER selects
|
||||
# for BOTH variants, ASR_MODEL_FW stays as a sidecar-only override.
|
||||
|
||||
@@ -25,6 +25,8 @@ runs under the Confucius4 venv — never imported by the parent), and
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import math
|
||||
import os
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from services.subprocess_backend import SubprocessBackend
|
||||
@@ -99,6 +101,19 @@ class Confucius4Backend(SubprocessBackend):
|
||||
from engines.confucius4.bootstrap import CONFUCIUS4_SIDECAR_SCRIPT
|
||||
return CONFUCIUS4_SIDECAR_SCRIPT
|
||||
|
||||
@property
|
||||
def recv_timeout_s(self) -> float:
|
||||
"""Receive timeout in seconds for the Confucius4 sidecar process (#2103)."""
|
||||
# Confucius4 is an LLM-based TTS (~17x realtime on CPU); synthesis legitimately
|
||||
# outruns the 60s class default. OMNIVOICE_CONFUCIUS4_RECV_TIMEOUT_S tunes it (#2103).
|
||||
try:
|
||||
v = float(os.environ.get("OMNIVOICE_CONFUCIUS4_RECV_TIMEOUT_S", "900"))
|
||||
except (ValueError, TypeError):
|
||||
return 900.0
|
||||
if not math.isfinite(v):
|
||||
return 900.0
|
||||
return max(30.0, v)
|
||||
|
||||
@property
|
||||
def sample_rate(self) -> int:
|
||||
return self._DEFAULT_SAMPLE_RATE
|
||||
|
||||
@@ -31,6 +31,8 @@ by the parent), and ``bootstrap.py`` (venv probe + lazy bootstrap).
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import math
|
||||
import os
|
||||
import sys
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
@@ -121,6 +123,19 @@ class DotsTTSBackend(SubprocessBackend):
|
||||
from engines.dots_tts.bootstrap import DOTS_TTS_SIDECAR_SCRIPT
|
||||
return DOTS_TTS_SIDECAR_SCRIPT
|
||||
|
||||
@property
|
||||
def recv_timeout_s(self) -> float:
|
||||
"""Receive timeout in seconds for the dots.tts sidecar process (#2103)."""
|
||||
# dots.tts is a 2B autoregressive model; synthesis on CPU legitimately
|
||||
# outruns the 60s class default. OMNIVOICE_DOTS_TTS_RECV_TIMEOUT_S tunes it (#2103).
|
||||
try:
|
||||
v = float(os.environ.get("OMNIVOICE_DOTS_TTS_RECV_TIMEOUT_S", "900"))
|
||||
except (ValueError, TypeError):
|
||||
return 900.0
|
||||
if not math.isfinite(v):
|
||||
return 900.0
|
||||
return max(30.0, v)
|
||||
|
||||
# ── TTSBackend protocol ────────────────────────────────────────────────
|
||||
|
||||
@property
|
||||
|
||||
@@ -38,6 +38,8 @@ isolated engine venv.
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import math
|
||||
import os
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from services.subprocess_backend import SubprocessBackend
|
||||
@@ -127,6 +129,19 @@ class MossTTSV15Backend(SubprocessBackend):
|
||||
from engines.moss_tts_v15.bootstrap import MOSS_TTS_V15_SIDECAR_SCRIPT
|
||||
return MOSS_TTS_V15_SIDECAR_SCRIPT
|
||||
|
||||
@property
|
||||
def recv_timeout_s(self) -> float:
|
||||
"""Receive timeout in seconds for the MOSS-TTS-v1.5 sidecar process (#2103)."""
|
||||
# MOSS-TTS-v1.5 is an 8B model; synthesis legitimately outruns the
|
||||
# 60s class default. OMNIVOICE_MOSS_TTS_V15_RECV_TIMEOUT_S tunes it (#2103).
|
||||
try:
|
||||
v = float(os.environ.get("OMNIVOICE_MOSS_TTS_V15_RECV_TIMEOUT_S", "900"))
|
||||
except (ValueError, TypeError):
|
||||
return 900.0
|
||||
if not math.isfinite(v):
|
||||
return 900.0
|
||||
return max(30.0, v)
|
||||
|
||||
# ── TTSBackend protocol ────────────────────────────────────────────────
|
||||
|
||||
@property
|
||||
|
||||
@@ -113,11 +113,11 @@ This clears the quarantine xattr recursively — including on the bundled
|
||||
returns a clear error message pointing at this command rather than
|
||||
silently hanging on a Gatekeeper-killed spawn.
|
||||
|
||||
If the macOS Apple Silicon Metal build fails to materialize in Wave 1
|
||||
(no published `buildmetal.sh` in `omnivoice.cpp` per Pitfall 1), the
|
||||
GGUF engine is unavailable on Apple Silicon and the existing in-process
|
||||
`VoiceStudioBackend` remains the cloning default on that platform — no
|
||||
hard block, no error toast on launch.
|
||||
The macOS Apple Silicon Metal build compiles cleanly with `-DGGML_METAL=ON`
|
||||
at the pinned `omnivoice.cpp` SHA (#2105), enabling GPU-accelerated GGUF
|
||||
voice cloning when the packaged binary passes preflight and is permitted by
|
||||
macOS. Missing binaries, placeholders, or Gatekeeper rejection leave
|
||||
`VoiceStudioBackend` available as the in-process fallback.
|
||||
|
||||
## Smoke test
|
||||
|
||||
|
||||
@@ -35,6 +35,7 @@ Threat model (per Plan 03-01 frontmatter):
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import math
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
@@ -99,6 +100,19 @@ class Supertonic3Backend(SubprocessBackend):
|
||||
def sidecar_script(cls) -> Path:
|
||||
return SUPERTONIC3_SIDECAR_SCRIPT
|
||||
|
||||
@property
|
||||
def recv_timeout_s(self) -> float:
|
||||
"""Receive timeout in seconds for the Supertonic-3 sidecar process (#2103)."""
|
||||
# Supertonic-3 runs ONNX on CPU; cold load downloads ~400MB and long
|
||||
# synthesis benefits from more headroom than 60s. OMNIVOICE_SUPERTONIC3_RECV_TIMEOUT_S (#2103).
|
||||
try:
|
||||
v = float(os.environ.get("OMNIVOICE_SUPERTONIC3_RECV_TIMEOUT_S", "300"))
|
||||
except (ValueError, TypeError):
|
||||
return 300.0
|
||||
if not math.isfinite(v):
|
||||
return 300.0
|
||||
return max(30.0, v)
|
||||
|
||||
# ── availability ───────────────────────────────────────────────────
|
||||
|
||||
@classmethod
|
||||
|
||||
+10
-5
@@ -636,11 +636,16 @@ def _phase_a_build_inner() -> None:
|
||||
# failure in that phase takes the whole backend down — the desktop app sits
|
||||
# on "starting backend" forever and /health stays 503.
|
||||
#
|
||||
# That is not a hypothetical version: RTX 50-series (Blackwell, sm_120)
|
||||
# users have no choice but to move off the pinned torch 2.8.0, which has no
|
||||
# sm_120 kernels, and the torch 2.9.x they land on brings torchaudio 2.9
|
||||
# with it. So the one group forced to upgrade hit a hard startup crash for
|
||||
# a line that does nothing (#1931).
|
||||
# That is not a hypothetical version: #1931 came from an sm_120 (Blackwell)
|
||||
# user whose torch import crashed on Windows and who fixed it by moving to
|
||||
# torch 2.9.1, which brings torchaudio 2.9 with it. Someone already working
|
||||
# around one problem then hit a hard startup crash on a line that does
|
||||
# nothing (#1931).
|
||||
#
|
||||
# The pin is not missing sm_120 kernels: torch 2.8.0 from the cu128 index
|
||||
# lists sm_120 in get_arch_list(). CU128_ARCHS in
|
||||
# tests/test_cuda_arch_compat.py records the same list, captured verbatim
|
||||
# from a real cu128 build in #1285.
|
||||
if hasattr(torchaudio, "set_audio_backend"):
|
||||
torchaudio.set_audio_backend("soundfile")
|
||||
from utils import hf_progress
|
||||
|
||||
@@ -167,6 +167,8 @@ async def run_transcribe_guarded(executor, fn, *, what: str = "ASR",
|
||||
immediately; running work calls it from the worker finalizer. Normal
|
||||
completion leaves cleanup with the caller.
|
||||
"""
|
||||
from services.inference_cancellation import InferenceCancellation
|
||||
cancellation = InferenceCancellation()
|
||||
loop = asyncio.get_running_loop()
|
||||
# Same SystemExit containment as the TTS pool (#1133 class): an ASR
|
||||
# dependency written as a CLI must not be able to shut the backend down.
|
||||
@@ -192,7 +194,8 @@ async def run_transcribe_guarded(executor, fn, *, what: str = "ASR",
|
||||
|
||||
def _job():
|
||||
try:
|
||||
return inner()
|
||||
with cancellation.activate():
|
||||
return inner()
|
||||
finally:
|
||||
with abandon_lock:
|
||||
abandon_state["finished"] = True
|
||||
@@ -204,6 +207,7 @@ async def run_transcribe_guarded(executor, fn, *, what: str = "ASR",
|
||||
fut = asyncio.wrap_future(concurrent_fut, loop=loop)
|
||||
|
||||
def _abandon() -> None:
|
||||
cancellation.cancel()
|
||||
cancelled_before_start = concurrent_fut.cancel()
|
||||
with abandon_lock:
|
||||
abandon_state["requested"] = True
|
||||
@@ -286,6 +290,26 @@ def _ctranslate2_cudnn_ok() -> tuple[bool, str]:
|
||||
return True, "ready"
|
||||
|
||||
|
||||
def _ctranslate2_execstack_ok() -> tuple[bool, str]:
|
||||
"""Make CTranslate2 importable on kernels that refuse an executable stack.
|
||||
|
||||
ctranslate2 ≤4.4.0 — the version whisperx 3.4.5 pins, and 3.4.5 is the
|
||||
newest release that supports the Python 3.11 we ship — marks its native
|
||||
library's stack ``RWE``. Kernels that refuse the request fail the dlopen
|
||||
with "cannot enable executable stack", killing whisperx, faster-whisper
|
||||
*and* Argos translation (#692). :mod:`core.execstack` clears that one bit
|
||||
in place, so call this BEFORE importing either engine; it cheaply rechecks the library and
|
||||
only writes when the library would otherwise refuse to load.
|
||||
"""
|
||||
try:
|
||||
from core.execstack import ensure_ctranslate2_loadable
|
||||
|
||||
return ensure_ctranslate2_loadable()
|
||||
except Exception as e: # noqa: BLE001 — a broken repair must not block ASR
|
||||
logger.debug("exec-stack repair unavailable (%s) — continuing", e)
|
||||
return True, "repair probe unavailable"
|
||||
|
||||
|
||||
def _decode_audio_16k_mono(audio_path: str):
|
||||
"""Decode `audio_path` to a 16 kHz mono float32 waveform using VoiceStudio's
|
||||
*validated* ffmpeg, instead of whisperx.load_audio's bare ``"ffmpeg"`` PATH
|
||||
@@ -657,6 +681,9 @@ class WhisperXBackend(ASRBackend):
|
||||
|
||||
@classmethod
|
||||
def is_available(cls) -> tuple[bool, str]:
|
||||
ct2_ok, ct2_detail = _ctranslate2_execstack_ok()
|
||||
if not ct2_ok:
|
||||
return False, f"whisperx cannot load CTranslate2: {ct2_detail}"
|
||||
try:
|
||||
import whisperx # noqa: F401
|
||||
except ImportError as e:
|
||||
@@ -684,6 +711,14 @@ class WhisperXBackend(ASRBackend):
|
||||
# → speechbrain, or a stray k2_fsa redirect import aborts ASR on Windows
|
||||
# (#630/#611/#647). No-op on macOS/Linux and when speechbrain is absent.
|
||||
_harden_speechbrain_lazy_imports()
|
||||
# #692: repair CTranslate2's exec-stack request before the import that
|
||||
# would be rejected by it. Memoized, so this is free after the probe.
|
||||
ct2_ok, ct2_detail = _ctranslate2_execstack_ok()
|
||||
if not ct2_ok:
|
||||
# ImportError (not RuntimeError): this IS a native-import failure,
|
||||
# and the sentinel lets load_active_asr_backend degrade to the next
|
||||
# engine instead of failing ASR wholesale (#1185).
|
||||
raise ImportError(f"whisperx cannot load CTranslate2: {ct2_detail}")
|
||||
import whisperx
|
||||
# #723: re-check the CUDA pick against *currently free* VRAM — the TTS
|
||||
# model may have claimed the card since __init__. A too-big load dies
|
||||
@@ -1020,6 +1055,9 @@ class FasterWhisperBackend(ASRBackend):
|
||||
|
||||
@classmethod
|
||||
def is_available(cls) -> tuple[bool, str]:
|
||||
ct2_ok, ct2_detail = _ctranslate2_execstack_ok()
|
||||
if not ct2_ok:
|
||||
return False, f"faster-whisper cannot load CTranslate2: {ct2_detail}"
|
||||
try:
|
||||
import faster_whisper # noqa: F401
|
||||
except ImportError as e:
|
||||
@@ -1034,6 +1072,9 @@ class FasterWhisperBackend(ASRBackend):
|
||||
def _ensure_model(self):
|
||||
if self._model is not None:
|
||||
return
|
||||
ct2_ok, ct2_detail = _ctranslate2_execstack_ok() # #692, see WhisperX
|
||||
if not ct2_ok:
|
||||
raise ImportError(f"faster-whisper cannot load CTranslate2: {ct2_detail}")
|
||||
from faster_whisper import WhisperModel
|
||||
# Device / compute-type auto-pick:
|
||||
# - CUDA present → GPU fp16
|
||||
@@ -1445,9 +1486,42 @@ class PyTorchWhisperBackend(ASRBackend):
|
||||
f"Underlying: {e}"
|
||||
) from e
|
||||
|
||||
#: Batch sizes to try on CUDA, largest first. The VRAM preflight only sizes
|
||||
#: the *weights*; generation adds an encoder/decoder workspace that scales
|
||||
#: with the batch, and `return_timestamps="word"` keeps every layer's
|
||||
#: cross-attention for the whole batch — gigabytes at batch 16. A card with
|
||||
#: room for the model can therefore still OOM at the first transcribe, which
|
||||
#: used to lose that chunk entirely (the dub retried the same batch size and
|
||||
#: gave up, leaving a hole in the transcript). Step down, then use CPU.
|
||||
_CUDA_BATCH_LADDER = (16, 4, 1)
|
||||
_CUDA_BATCH_LADDER_WORD_TS = (8, 2, 1)
|
||||
|
||||
@staticmethod
|
||||
def _is_oom(exc: BaseException) -> bool:
|
||||
try:
|
||||
import torch
|
||||
|
||||
if isinstance(exc, torch.cuda.OutOfMemoryError):
|
||||
return True
|
||||
except Exception: # noqa: BLE001 — classification must not raise
|
||||
pass
|
||||
return "out of memory" in str(exc).lower()
|
||||
|
||||
def _rebuild_on_cpu(self) -> None:
|
||||
"""Move the existing pipeline to CPU without resolving any model files."""
|
||||
import torch
|
||||
|
||||
# Keep the loaded checkpoint, tokenizer and feature extractor. Looking
|
||||
# up the default model here could download a different model offline.
|
||||
self._pipe.model.to(device="cpu", dtype=torch.float32)
|
||||
self._pipe.device = torch.device("cpu")
|
||||
try:
|
||||
torch.cuda.empty_cache()
|
||||
except Exception:
|
||||
pass # Some builds have no CUDA cache to release.
|
||||
|
||||
def transcribe(self, audio_path: str, *, word_timestamps: bool = True) -> dict:
|
||||
import soundfile as sf
|
||||
import torch
|
||||
self._ensure_pipe()
|
||||
# #2039: libsndfile cannot open MP4/M4A (AAC), which /transcribe and
|
||||
# the MCP tool both accept. Those decode through the validated ffmpeg
|
||||
@@ -1460,15 +1534,67 @@ class PyTorchWhisperBackend(ASRBackend):
|
||||
audio_np, sr = _decode_audio_16k_mono(audio_path), 16000
|
||||
if audio_np.ndim > 1:
|
||||
audio_np = audio_np.mean(axis=1)
|
||||
bs = 16 if torch.cuda.is_available() else 2
|
||||
result = self._pipe(
|
||||
{"array": audio_np, "sampling_rate": sr},
|
||||
return_timestamps="word" if word_timestamps else True,
|
||||
chunk_length_s=15,
|
||||
batch_size=bs,
|
||||
)
|
||||
|
||||
def _run(batch_size: int):
|
||||
return self._pipe(
|
||||
{"array": audio_np, "sampling_rate": sr},
|
||||
return_timestamps="word" if word_timestamps else True,
|
||||
chunk_length_s=15,
|
||||
batch_size=batch_size,
|
||||
)
|
||||
|
||||
if self._on_cuda():
|
||||
ladder = (
|
||||
self._CUDA_BATCH_LADDER_WORD_TS if word_timestamps
|
||||
else self._CUDA_BATCH_LADDER
|
||||
)
|
||||
for i, bs in enumerate(ladder):
|
||||
try:
|
||||
result = _run(bs)
|
||||
break
|
||||
except Exception as e: # noqa: BLE001 — only OOM is retryable
|
||||
if not self._is_oom(e):
|
||||
raise
|
||||
try:
|
||||
import torch
|
||||
|
||||
torch.cuda.empty_cache()
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
if i + 1 < len(ladder):
|
||||
logger.warning(
|
||||
"PyTorch Whisper CUDA OOM at batch_size=%d — "
|
||||
"retrying at %d. Free VRAM (Flush models, close "
|
||||
"other GPU apps) for full-speed ASR.",
|
||||
bs, ladder[i + 1],
|
||||
)
|
||||
continue
|
||||
# Smallest batch still OOMs: finish on CPU rather than
|
||||
# return an empty chunk the caller cannot distinguish
|
||||
# from silence.
|
||||
logger.warning(
|
||||
"PyTorch Whisper CUDA OOM even at batch_size=1 — "
|
||||
"transcribing on CPU (slower, same model). Detail: %s", e,
|
||||
)
|
||||
self._rebuild_on_cpu()
|
||||
result = _run(2)
|
||||
else:
|
||||
result = _run(2)
|
||||
return result if isinstance(result, dict) else {"chunks": [], "raw": result}
|
||||
|
||||
def _on_cuda(self) -> bool:
|
||||
"""Whether the built pipeline actually sits on a CUDA device.
|
||||
|
||||
`torch.cuda.is_available()` is the wrong question: `_pick_device()` may
|
||||
have chosen CPU on a CUDA host (low free VRAM), and a CPU pipeline must
|
||||
not be handed a CUDA-sized batch.
|
||||
"""
|
||||
try:
|
||||
device = getattr(self._pipe, "device", None)
|
||||
return "cuda" in str(device).lower()
|
||||
except Exception: # noqa: BLE001
|
||||
return False
|
||||
|
||||
|
||||
# ── NeMo Parakeet TDT (NVIDIA — Open ASR Leaderboard SOTA, 25 langs) ────────
|
||||
|
||||
|
||||
@@ -204,6 +204,42 @@ def _safe_torchaudio_save(
|
||||
fmt, e,
|
||||
)
|
||||
torchaudio.save(path_or_buf, tensor, sample_rate, format=fmt)
|
||||
except (ImportError, RuntimeError) as e:
|
||||
if isinstance(e, RuntimeError) and "could not load libtorchcodec" not in str(e).lower():
|
||||
raise _describe_write_failure(e, path_or_buf) from e
|
||||
# torchaudio >= 2.9 routes save() through TorchCodec, which needs
|
||||
# FFmpeg *shared libraries* on the system. Where those are absent the
|
||||
# write raises ImportError and every generation fails. #1931 guarded
|
||||
# set_audio_backend() against that torchaudio but left save() itself
|
||||
# unprotected; arm64 CUDA hosts reach it unavoidably, since torch
|
||||
# 2.8.0 publishes no aarch64 wheel. soundfile is already a locked
|
||||
# dependency and the tensor is normalized by this point, so hand it to
|
||||
# the audited sibling helper rather than failing the request.
|
||||
logger.warning(
|
||||
"torchaudio.save needs TorchCodec (%s); writing via soundfile", e
|
||||
)
|
||||
if hasattr(path_or_buf, "seek") and hasattr(path_or_buf, "truncate"):
|
||||
try:
|
||||
path_or_buf.seek(0)
|
||||
path_or_buf.truncate(0)
|
||||
except (OSError, io.UnsupportedOperation):
|
||||
pass # Non-seekable streams cannot be rewound; preserve fallback behavior.
|
||||
_subtype = {
|
||||
"wav": "FLOAT" if bits_per_sample == 32 else "PCM_16",
|
||||
"flac": "PCM_16",
|
||||
"ogg": "VORBIS",
|
||||
"mp3": "MPEG_LAYER_III",
|
||||
}.get(fmt, "PCM_16")
|
||||
try:
|
||||
_safe_soundfile_write(
|
||||
path_or_buf,
|
||||
tensor.transpose(0, 1).contiguous().numpy(),
|
||||
sample_rate,
|
||||
subtype=_subtype,
|
||||
format=fmt.upper(),
|
||||
)
|
||||
except Exception as e2:
|
||||
raise _describe_write_failure(e2, path_or_buf) from e2
|
||||
except Exception as e:
|
||||
# #1221: libsndfile reports OS-level write failures as a bare
|
||||
# "LibsndfileError: System error." — no path, no errno, nothing the
|
||||
@@ -267,6 +303,7 @@ def _safe_soundfile_write(
|
||||
sample_rate: int,
|
||||
*,
|
||||
subtype: str = "PCM_16",
|
||||
format: str | None = None,
|
||||
) -> None:
|
||||
"""Sibling helper for the one in-tree ``sf.write`` site.
|
||||
|
||||
@@ -284,6 +321,10 @@ def _safe_soundfile_write(
|
||||
subtype: Soundfile subtype string. ``"PCM_16"`` (default) for
|
||||
standard 16-bit PCM WAV; ``"PCM_24"``, ``"FLOAT"`` etc.
|
||||
also work.
|
||||
format: Container format (``"WAV"``, ``"FLAC"``, ``"OGG"``,
|
||||
``"MP3"``). ``None`` lets soundfile infer it from the path's
|
||||
extension — which it cannot do for a file-like object, so
|
||||
callers passing a buffer must name it.
|
||||
|
||||
Raises:
|
||||
ValueError: if the array is empty.
|
||||
@@ -321,7 +362,7 @@ def _safe_soundfile_write(
|
||||
samples = np.ascontiguousarray(samples)
|
||||
|
||||
_ensure_audio_parent(path)
|
||||
sf.write(path, samples, sample_rate, subtype=subtype)
|
||||
sf.write(path, samples, sample_rate, subtype=subtype, format=format)
|
||||
|
||||
|
||||
def atomic_save_wav(
|
||||
|
||||
@@ -64,6 +64,23 @@ from core.logging_utils import log_safe
|
||||
|
||||
logger = logging.getLogger("omnivoice.dub_pipeline")
|
||||
|
||||
|
||||
def _media_process_error(tool: str, returncode: int, stderr: bytes, *, paths=()) -> str:
|
||||
"""Keep the actionable end of native diagnostics without leaking paths."""
|
||||
from core.scrub import scrub_text
|
||||
detail = stderr.decode(errors="replace")
|
||||
# Input/output may live outside home directories (mounted media, Windows
|
||||
# drive roots). Remove the exact command paths before generic scrubbing.
|
||||
for path in sorted((str(p) for p in paths if p), key=len, reverse=True):
|
||||
for variant in {path, path.replace("\\", "/"), path.replace("/", "\\")}:
|
||||
detail = detail.replace(variant, "[redacted path]")
|
||||
detail = scrub_text(detail).strip()
|
||||
tail = detail[-2000:]
|
||||
if len(detail) > 2000:
|
||||
tail = "…" + tail
|
||||
return f"{tool} exited with code {returncode}" + (f": {tail}" if tail else ". No diagnostic output.")
|
||||
|
||||
|
||||
# ── Module-level state ──────────────────────────────────────────────────────
|
||||
# These used to live in dub_core.py. The router now re-exports them for
|
||||
# backward compat during the transition.
|
||||
@@ -1326,7 +1343,7 @@ async def ingest_pipeline(
|
||||
"-ar", "16000", "-ac", "1", audio_path, "-y",
|
||||
])
|
||||
if p.returncode != 0:
|
||||
msg = (stderr.decode(errors="replace") or f"ffmpeg returned exit code {p.returncode}").strip()[:500]
|
||||
msg = _media_process_error("FFmpeg", p.returncode, stderr, paths=(video_path, audio_path, job_dir))
|
||||
raise Exception(msg)
|
||||
# Second, FULL-QUALITY extraction for source separation. audio.wav
|
||||
# is deliberately 16 kHz mono — that's what ASR wants — but Demucs
|
||||
@@ -1476,7 +1493,7 @@ async def ingest_pipeline(
|
||||
elif evt[0] == "done":
|
||||
rc, stderr_full = evt[1], evt[2]
|
||||
if rc != 0:
|
||||
raise Exception(stderr_full.decode(errors="replace")[:500])
|
||||
raise Exception(_media_process_error("Demucs", rc, stderr_full, paths=(audio_hq_path, audio_path, job_dir)))
|
||||
# Stems land under the INPUT's basename ("audio_hq" when the
|
||||
# full-quality extraction succeeded, "audio" on its fallback).
|
||||
demucs_out = os.path.join(
|
||||
|
||||
@@ -199,10 +199,12 @@ def mark_flashinfer_runtime_failure(reason: str) -> None:
|
||||
def _cuda_arch_supported_for_compile() -> "tuple[bool, str]":
|
||||
"""Check the GPU's architecture against this torch build's arch list.
|
||||
|
||||
New GPU architectures (e.g. Blackwell sm_120, issue #278) routinely break
|
||||
torch.compile/Triton before upstream support lands: the eager model runs
|
||||
via PTX forward-compat, but Inductor/Triton kernel compilation targets the
|
||||
new arch directly and fails mid-generation. If the device's arch tag is
|
||||
A new GPU architecture routinely breaks torch.compile/Triton before
|
||||
upstream support lands (issue #278): the eager model runs via PTX
|
||||
forward-compat, but Inductor/Triton kernel compilation targets the new arch
|
||||
directly and fails mid-generation. Blackwell sm_120 was that case; it no
|
||||
longer is on the pinned torch 2.8.0+cu128, where this probe can return
|
||||
supported; independent compiler/runtime failures still need eager fallback. If the device's arch tag is
|
||||
absent from this build's arch list we treat compile as unsupported and use
|
||||
eager. The comparison is delegated to ``core.device_caps.arch_unsupported``
|
||||
so it stays CUDA/ROCm-aware — a ROCm build lists ``gfx…`` names, and the
|
||||
|
||||
@@ -174,12 +174,12 @@ def _binary_runs(path: str) -> bool:
|
||||
if cached is not None:
|
||||
return cached
|
||||
try:
|
||||
subprocess.run(
|
||||
result = subprocess.run(
|
||||
[path, "-version"],
|
||||
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
|
||||
timeout=10, check=False,
|
||||
)
|
||||
ok = True
|
||||
ok = result.returncode == 0
|
||||
except (OSError, subprocess.TimeoutExpired, subprocess.SubprocessError) as e:
|
||||
logger.warning(
|
||||
"Rejecting non-runnable ffmpeg/ffprobe candidate %s: %s",
|
||||
@@ -309,8 +309,11 @@ def find_ffprobe():
|
||||
try:
|
||||
ffmpeg_path = find_ffmpeg()
|
||||
if ffmpeg_path:
|
||||
candidate = ffmpeg_path.replace("ffmpeg", "ffprobe")
|
||||
if os.path.isfile(candidate):
|
||||
candidate = os.path.join(
|
||||
os.path.dirname(ffmpeg_path),
|
||||
os.path.basename(ffmpeg_path).replace("ffmpeg", "ffprobe"),
|
||||
)
|
||||
if os.path.isfile(candidate) and _binary_runs(candidate):
|
||||
return candidate
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
"""Request cancellation visible to killable sidecars on executor threads.
|
||||
|
||||
In-process native calls remain accounted for until they return; only sidecars
|
||||
can safely terminate early. A scope belongs to one job, never to a pool thread.
|
||||
"""
|
||||
from contextlib import contextmanager
|
||||
import threading
|
||||
|
||||
_local = threading.local()
|
||||
|
||||
|
||||
class InferenceCancellation:
|
||||
def __init__(self):
|
||||
self.cancelled = threading.Event()
|
||||
|
||||
def cancel(self):
|
||||
self.cancelled.set()
|
||||
|
||||
@contextmanager
|
||||
def activate(self):
|
||||
previous = getattr(_local, "scope", None)
|
||||
_local.scope = self
|
||||
try:
|
||||
if self.cancelled.is_set():
|
||||
raise RuntimeError("Inference request was cancelled before execution")
|
||||
yield
|
||||
finally:
|
||||
_local.scope = previous
|
||||
|
||||
|
||||
def current_cancellation():
|
||||
return getattr(_local, "scope", None)
|
||||
@@ -1,11 +1,12 @@
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
import asyncio
|
||||
import logging
|
||||
import math
|
||||
import os
|
||||
import queue
|
||||
import re
|
||||
import sys
|
||||
import threading
|
||||
import time
|
||||
from concurrent.futures import Executor, Future, ThreadPoolExecutor
|
||||
|
||||
from utils.containment import contain_system_exit
|
||||
@@ -527,6 +528,7 @@ def generate_timeout_s(
|
||||
text: "str | None", *, engine: object = None, execution_device: "str | None" = None,
|
||||
min_vram_gb: float = 0.0, hardware_family: "str | None" = None,
|
||||
vram_gb: "float | None" = None,
|
||||
_include_sidecar_grace: bool = True,
|
||||
) -> float:
|
||||
"""THE wall-clock execution budget for one synthesis job, scaled to input.
|
||||
|
||||
@@ -562,6 +564,7 @@ def generate_timeout_s(
|
||||
claiming the card is under-provisioned in user-facing diagnostics.
|
||||
"""
|
||||
base = GPU_JOB_TIMEOUT_S
|
||||
explicit_budget = _GENERATE_TIMEOUT_EXPLICIT or GPU_JOB_TIMEOUT_S != _CONFIGURED_GPU_JOB_TIMEOUT_S
|
||||
try:
|
||||
from core.device_caps import detect_host_caps
|
||||
caps = detect_host_caps()
|
||||
@@ -588,6 +591,7 @@ def generate_timeout_s(
|
||||
)
|
||||
if family == "cpu" and (cpu_explicit or not universal_override):
|
||||
base = CPU_JOB_TIMEOUT_S
|
||||
explicit_budget = cpu_explicit
|
||||
elif not universal_override and family in (
|
||||
"cuda", "rocm", "vulkan", "xpu",
|
||||
):
|
||||
@@ -610,7 +614,23 @@ def generate_timeout_s(
|
||||
# Device probing is advisory here; the configured universal bound is
|
||||
# still safe when a platform probe is unavailable during startup.
|
||||
pass
|
||||
return base + (max(0, len(text or "") - 1200) / 40.0)
|
||||
|
||||
# If the engine specifies its own sidecar receive timeout (e.g. SubprocessBackend
|
||||
# engines like Confucius, Dots, Moss, Supertonic), the outer execution budget
|
||||
# must not cut the sidecar off early (#2103). A bounded 5s grace period ensures
|
||||
# the sidecar's watchdog timer fires and surfaces its actionable timeout error
|
||||
# before the outer pool cancellation cuts it off.
|
||||
sidecar_grace = 0.0
|
||||
if not explicit_budget and engine is not None and hasattr(engine, "recv_timeout_s"):
|
||||
try:
|
||||
sidecar_timeout = float(engine.recv_timeout_s)
|
||||
if math.isfinite(sidecar_timeout) and sidecar_timeout > 0:
|
||||
base = max(base, sidecar_timeout)
|
||||
sidecar_grace = 5.0 if _include_sidecar_grace else 0.0
|
||||
except (TypeError, ValueError):
|
||||
pass # Invalid optional engine metadata cannot disable the outer guard.
|
||||
|
||||
return base + (max(0, len(text or "") - 1200) / 40.0) + sidecar_grace
|
||||
|
||||
|
||||
def _retry_after_estimate(stats: dict) -> float:
|
||||
@@ -726,6 +746,8 @@ async def run_on_gpu_pool_guarded(fn, *, what: str = "GPU job",
|
||||
finalizer. Normal completion never calls it. This lets request-owned temp
|
||||
files outlive abandoned workers without delaying ordinary requests (#1668).
|
||||
"""
|
||||
from services.inference_cancellation import InferenceCancellation
|
||||
cancellation = InferenceCancellation()
|
||||
loop = asyncio.get_running_loop()
|
||||
ex = executor if executor is not None else _get_gpu_pool()
|
||||
# Resolved at CALL time, not def time, so monkeypatching/reloading the
|
||||
@@ -768,7 +790,8 @@ async def run_on_gpu_pool_guarded(fn, *, what: str = "GPU job",
|
||||
except RuntimeError:
|
||||
pass # loop already closed (caller vanished) — still run the job
|
||||
try:
|
||||
return _inner()
|
||||
with cancellation.activate():
|
||||
return _inner()
|
||||
finally:
|
||||
# Idents are reused by the OS; a stale heartbeat under this ident
|
||||
# must not vouch for some future job on the same thread.
|
||||
@@ -783,6 +806,7 @@ async def run_on_gpu_pool_guarded(fn, *, what: str = "GPU job",
|
||||
fut = asyncio.wrap_future(concurrent_fut, loop=loop)
|
||||
|
||||
def _abandon() -> None:
|
||||
cancellation.cancel()
|
||||
# Keep the concurrent future so we can distinguish a job cancelled out
|
||||
# of the queue from a thread that Python cannot stop once it has begun.
|
||||
cancelled_before_start = concurrent_fut.cancel()
|
||||
@@ -1565,10 +1589,13 @@ def _is_compile_runtime_failure(exc: BaseException) -> bool:
|
||||
"""True when an exception originates in the torch.compile stack (Dynamo /
|
||||
Inductor / Triton / FX / CUDA-graph trees) rather than in the model itself.
|
||||
|
||||
#278: on GPU architectures Triton doesn't support yet (e.g. Blackwell
|
||||
sm_120), the compiled model dies mid-generation with errors like
|
||||
"Detected that you are using FX to symbolically trace a dynamo-optimized
|
||||
function" or an AssertionError out of torch/_inductor/cudagraph_trees.py.
|
||||
#278: an independent compile-stack failure can surface during generation
|
||||
as an AssertionError out of torch/_inductor/cudagraph_trees.py. An
|
||||
architecture missing from the running torch build's arch list is rejected
|
||||
earlier by should_torch_compile(), before this runtime fallback applies.
|
||||
#278 also quotes "Detected that you are using FX to symbolically trace a
|
||||
dynamo-optimized function"; Dynamo raises that on any device, CPU included,
|
||||
so it is a compile-stack error to catch here but never an arch signal.
|
||||
Walks the exception chain and checks (a) the exception type's module,
|
||||
(b) the message, (c) the traceback file paths — the cudagraph case is a
|
||||
bare AssertionError, so the traceback check is load-bearing.
|
||||
|
||||
@@ -1543,10 +1543,15 @@ def _step_fetch_weights(spec: SidecarSpec, job: dict) -> None:
|
||||
# other model download in the app — see setup/download.py): the
|
||||
# source checkout is unpinned upstream `main` anyway, and hf_hub
|
||||
# checksum-verifies each artifact. Hence the B615 waiver below.
|
||||
# Unwrap to the bearer string: snapshot_download takes `token: str |
|
||||
# None`, and a ResolvedToken record is ignored in favour of ambient
|
||||
# discovery, so gated engine weights 401 for a user whose token lives
|
||||
# in VoiceStudio's settings rather than HF's own cache (#2163).
|
||||
_resolved = resolve_token()
|
||||
kwargs: dict = {
|
||||
"repo_id": spec.weights_repo_id,
|
||||
"local_dir": str(wdir),
|
||||
"token": resolve_token(),
|
||||
"token": _resolved.token if _resolved else None,
|
||||
}
|
||||
if spec.weights_revision:
|
||||
kwargs["revision"] = spec.weights_revision
|
||||
|
||||
@@ -24,7 +24,7 @@ from dataclasses import dataclass
|
||||
|
||||
|
||||
# Captures: HH MM SS sep(`,` or `.`) ms (1-3 digits)
|
||||
_TS = r"(\d{1,2}):([0-5]?\d):([0-5]?\d)[,.](\d{1,3})"
|
||||
_TS = r"(?:(\d{1,2}):)?([0-5]?\d):([0-5]?\d)[,.](\d{1,3})"
|
||||
# Horizontal whitespace only — NEVER plain `\s`, which matches newlines.
|
||||
# A timing line lives on ONE line, so `\s*` bought nothing but catastrophic
|
||||
# backtracking: under re.MULTILINE the engine restarts at every line start,
|
||||
@@ -41,7 +41,19 @@ _TIMING_RE = re.compile(rf"^{_H}{_TS}{_H}-->{_H}{_TS}.*$", re.MULTILINE)
|
||||
def _ts_to_seconds(h: str, m: str, s: str, ms: str) -> float:
|
||||
# Pad ms to 3 digits so "5" -> 0.005, "50" -> 0.050.
|
||||
ms_padded = (ms + "000")[:3]
|
||||
return int(h) * 3600 + int(m) * 60 + int(s) + int(ms_padded) / 1000.0
|
||||
return int(h or 0) * 3600 + int(m) * 60 + int(s) + int(ms_padded) / 1000.0
|
||||
|
||||
|
||||
def _is_index_line(line: str) -> bool:
|
||||
"""True when `line` is a bare SubRip cue number.
|
||||
|
||||
Stricter than `str.isdigit()` on purpose: that also accepts non-ASCII
|
||||
numerals (Arabic-Indic "١٩٩٩", Devanagari "२०२६", and the full-width
|
||||
forms), which in a 646-language dubbing app are dialogue, never the
|
||||
ASCII cue indices SubRip actually writes.
|
||||
"""
|
||||
stripped = line.strip()
|
||||
return stripped.isascii() and stripped.isdigit()
|
||||
|
||||
|
||||
@dataclass
|
||||
@@ -68,13 +80,62 @@ def parse_srt(content: str) -> SrtParseResult:
|
||||
# Strip BOM and normalise line endings; many editors save SRTs as CRLF.
|
||||
text = content.lstrip("").replace("\r\n", "\n").replace("\r", "\n")
|
||||
|
||||
is_webvtt = bool(re.match(r"WEBVTT(?:[ \t]|\n|$)", text.lstrip()))
|
||||
if is_webvtt:
|
||||
# Metadata is block-scoped. Filter it BEFORE scanning timings so an
|
||||
# example timestamp inside a NOTE/STYLE/REGION cannot become speech.
|
||||
blocks = []
|
||||
for block in re.split(r"\n[^\S\n]*\n", text):
|
||||
lines = block.strip().split("\n")
|
||||
first = lines[0].strip()
|
||||
# WebVTT's block parser gives a timing line in position two
|
||||
# precedence over the identifier (including STYLE/REGION/NOTE).
|
||||
# https://www.w3.org/TR/webvtt1/#file-parsing
|
||||
identifies_cue = len(lines) > 1 and _TIMING_RE.match(lines[1])
|
||||
metadata = first in {"STYLE", "REGION"} or re.match(r"NOTE(?:[ \t]|$)", first)
|
||||
if metadata and not identifies_cue:
|
||||
continue
|
||||
blocks.append(block)
|
||||
text = "\n\n".join(blocks)
|
||||
raw: list[dict] = []
|
||||
skipped = 0
|
||||
# Find every timing line, slice the cue text from there to the next
|
||||
# timing line (or end of file). This is robust to missing index
|
||||
# numbers and to spec deviations in the blank-line separator.
|
||||
matches = list(_TIMING_RE.finditer(text))
|
||||
# A mixed file can stop numbering at any cue. Track each boundary;
|
||||
# never treat an initial index as permission to discard later numbers.
|
||||
head = text[:matches[0].start()].strip() if matches else ""
|
||||
first_marker = head.split("\n")[-1].strip() if head else ""
|
||||
cue_index = int(first_marker) if _is_index_line(first_marker) and len(first_marker) <= 12 else None
|
||||
for i, m in enumerate(matches):
|
||||
body_start = m.end()
|
||||
has_next = i + 1 < len(matches)
|
||||
body_end = matches[i + 1].start() if has_next else len(text)
|
||||
body = text[body_start:body_end]
|
||||
if is_webvtt:
|
||||
# The blank separator ends WebVTT dialogue; following identifiers,
|
||||
# NOTE/STYLE blocks belong outside the cue, even when numeric.
|
||||
body = re.split(r"\n[^\S\n]*\n", body, maxsplit=1)[0]
|
||||
# An index must directly precede the next timing line. A blank line
|
||||
# AFTER a number instead marks that number as preceding dialogue.
|
||||
next_index = None
|
||||
if has_next and not is_webvtt:
|
||||
# Inspect lines rather than a backtracking regex on uploaded text.
|
||||
# One newline terminates the marker; a second means it is dialogue.
|
||||
marker_lines = body.split("\n")
|
||||
if marker_lines and not marker_lines[-1].strip(" \t"):
|
||||
marker_lines.pop()
|
||||
marker = marker_lines[-1].strip(" \t") if marker_lines else ""
|
||||
numeric = bool(marker) and marker.isascii() and marker.isdecimal()
|
||||
separated = len(marker_lines) > 2 and not marker_lines[-2].strip()
|
||||
expected_index = cue_index + 1 if cue_index is not None else i + 2
|
||||
expected = marker.lstrip("0") == str(expected_index)
|
||||
has_dialogue = any(line.strip() for line in marker_lines[:-1])
|
||||
if numeric and expected and (has_dialogue or separated) and (cue_index is not None or separated):
|
||||
body = "\n".join(marker_lines[:-1])
|
||||
next_index = expected_index
|
||||
cue_index = next_index
|
||||
try:
|
||||
start = _ts_to_seconds(m.group(1), m.group(2), m.group(3), m.group(4))
|
||||
end = _ts_to_seconds(m.group(5), m.group(6), m.group(7), m.group(8))
|
||||
@@ -84,14 +145,7 @@ def parse_srt(content: str) -> SrtParseResult:
|
||||
if end <= start:
|
||||
skipped += 1
|
||||
continue
|
||||
body_start = m.end()
|
||||
body_end = matches[i + 1].start() if i + 1 < len(matches) else len(text)
|
||||
body = text[body_start:body_end].strip("\n")
|
||||
# Drop the trailing index number of the NEXT cue (which got eaten
|
||||
# into our body) by trimming trailing digit-only lines.
|
||||
lines = body.split("\n")
|
||||
while lines and lines[-1].strip().isdigit():
|
||||
lines.pop()
|
||||
lines = body.strip("\n").split("\n")
|
||||
cue_text = "\n".join(line.strip() for line in lines if line.strip())
|
||||
if not cue_text:
|
||||
skipped += 1
|
||||
@@ -126,3 +180,19 @@ def parse_srt(content: str) -> SrtParseResult:
|
||||
for i, seg in enumerate(out)
|
||||
]
|
||||
return SrtParseResult(segments=segments, skipped_cues=skipped, dropped_overlaps=dropped)
|
||||
|
||||
|
||||
def format_cue_timestamp(seconds: float, ms_separator: str) -> str:
|
||||
"""`HH:MM:SS<sep>mmm` for `seconds`, rounded to the millisecond.
|
||||
|
||||
Rounds the whole value once, then splits it, so a time that is not exact
|
||||
in binary (2.3 is 2.29999...) stays 2.300 instead of truncating to 2.299,
|
||||
which moved every such cue a millisecond early on export, and 59.9996
|
||||
carries to the next second instead of printing `,1000`. SRT separates
|
||||
the milliseconds with `,`; WebVTT with `.`.
|
||||
"""
|
||||
total_ms = int(round(seconds * 1000))
|
||||
h, rem = divmod(total_ms, 3_600_000)
|
||||
m, rem = divmod(rem, 60_000)
|
||||
s, ms = divmod(rem, 1000)
|
||||
return f"{h:02d}:{m:02d}:{s:02d}{ms_separator}{ms:03d}"
|
||||
|
||||
@@ -22,6 +22,8 @@ only process isolation.
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import math
|
||||
import os
|
||||
import sys
|
||||
import threading
|
||||
from pathlib import Path
|
||||
@@ -34,7 +36,8 @@ from services.subprocess_backend import (
|
||||
logger = logging.getLogger("omnivoice.asr.subprocess")
|
||||
|
||||
# A model load + transcription can take a while on CPU for a long clip; give
|
||||
# the transcribe round-trip more headroom than the TTS default.
|
||||
# the transcribe round-trip more headroom than the TTS default. Configurable via
|
||||
# OMNIVOICE_ASR_RECV_TIMEOUT_S (#2103).
|
||||
ASR_RECV_TIMEOUT_S = 600.0
|
||||
|
||||
|
||||
@@ -78,6 +81,17 @@ class SubprocessASRBackend(SubprocessBackend):
|
||||
pass
|
||||
return "cpu"
|
||||
|
||||
@property
|
||||
def recv_timeout_s(self) -> float:
|
||||
"""Wall-clock timeout in seconds waiting for an ASR sidecar response (#2103)."""
|
||||
try:
|
||||
v = float(os.environ.get("OMNIVOICE_ASR_RECV_TIMEOUT_S", str(ASR_RECV_TIMEOUT_S)))
|
||||
except (ValueError, TypeError):
|
||||
return ASR_RECV_TIMEOUT_S
|
||||
if not math.isfinite(v):
|
||||
return ASR_RECV_TIMEOUT_S
|
||||
return max(30.0, v)
|
||||
|
||||
def transcribe(self, audio_path: str, *, word_timestamps: bool = True) -> dict:
|
||||
"""Transcribe ``audio_path`` in the sidecar. Returns the engine's
|
||||
result dict ({"segments": [...], "language": ...}).
|
||||
@@ -112,6 +126,7 @@ class SubprocessASRBackend(SubprocessBackend):
|
||||
if slot_future is not None:
|
||||
slot_future.cancel()
|
||||
raise TimeoutError("timed out waiting for a free GPU worker")
|
||||
timeout_s = self.recv_timeout_s
|
||||
with self._lock:
|
||||
self._spawn()
|
||||
from services.performance_profiles import asr_decode_defaults
|
||||
@@ -121,17 +136,25 @@ class SubprocessASRBackend(SubprocessBackend):
|
||||
"word_timestamps": bool(word_timestamps),
|
||||
"decode_options": asr_decode_defaults(),
|
||||
})
|
||||
reply = self._recv_with_timeout(ASR_RECV_TIMEOUT_S)
|
||||
if not reply:
|
||||
# EOF can arrive before Windows updates poll(); retire the
|
||||
# stale handle so an immediate retry respawns the sidecar.
|
||||
self.shutdown()
|
||||
# Pipe closed mid-transcription → the child crashed.
|
||||
raise RuntimeError(
|
||||
f"{self.id} ASR sidecar crashed mid-transcription "
|
||||
f"(device={self._device()}); the job failed but the backend "
|
||||
f"stayed up — retry to respawn a fresh sidecar."
|
||||
)
|
||||
reply = self._recv_with_timeout(timeout_s)
|
||||
timed_out = self._last_recv_timed_out
|
||||
if not reply:
|
||||
# EOF can arrive before Windows updates poll(); retire the
|
||||
# stale handle so an immediate retry respawns the sidecar.
|
||||
self.shutdown()
|
||||
if timed_out:
|
||||
raise RuntimeError(
|
||||
f"{self.id} ASR sidecar exceeded receive timeout "
|
||||
f"({timeout_s:g}s); killed mid-transcription "
|
||||
f"(device={self._device()}) — retry or raise "
|
||||
f"OMNIVOICE_ASR_RECV_TIMEOUT_S."
|
||||
)
|
||||
# Pipe closed mid-transcription → the child crashed.
|
||||
raise RuntimeError(
|
||||
f"{self.id} ASR sidecar crashed mid-transcription "
|
||||
f"(device={self._device()}); the job failed but the backend "
|
||||
f"stayed up — retry to respawn a fresh sidecar."
|
||||
)
|
||||
if reply.get("op") == "error":
|
||||
raise RuntimeError(
|
||||
f"{self.id} ASR sidecar error (device={self._device()}): "
|
||||
@@ -165,6 +188,11 @@ class IsolatedFasterWhisperBackend(SubprocessASRBackend):
|
||||
|
||||
@classmethod
|
||||
def is_available(cls) -> tuple[bool, str]:
|
||||
from core.execstack import ensure_ctranslate2_loadable
|
||||
|
||||
ok, detail = ensure_ctranslate2_loadable()
|
||||
if not ok:
|
||||
return False, f"faster-whisper cannot load CTranslate2: {detail}"
|
||||
try:
|
||||
import faster_whisper # noqa: F401
|
||||
except Exception as e:
|
||||
|
||||
@@ -44,6 +44,7 @@ import contextlib
|
||||
import collections
|
||||
import json
|
||||
import logging
|
||||
import math
|
||||
import os
|
||||
import struct
|
||||
import subprocess
|
||||
@@ -103,10 +104,32 @@ _STDERR_TAIL_LINES = 12
|
||||
_STDERR_TAIL_CHARS = 800
|
||||
|
||||
#: Per-frame _recv read timeout (best-effort — applies to header read; body
|
||||
#: read is uninterruptible on a stdlib BufferedReader). Used in health_check
|
||||
#: and generate to bound a hung sidecar.
|
||||
#: read is uninterruptible on a stdlib BufferedReader). Used by health_check
|
||||
#: to bound a hung sidecar: a ping must stay fast, so this stays short.
|
||||
RECV_TIMEOUT_S = 60.0
|
||||
|
||||
#: Floor for the generate() deadline of a sidecar that does not choose its own.
|
||||
#:
|
||||
#: It must not undercut the budget the job was already granted by
|
||||
#: services.model_manager.generate_timeout_s — 300s on an accelerated host,
|
||||
#: 600s on a CPU one — or the watchdog kills a synthesis the caller still
|
||||
#: considers valid, which is #2103. 600s is that CPU floor.
|
||||
#:
|
||||
#: A floor, not the whole answer: that budget also scales with text length, so
|
||||
#: _effective_recv_timeout_s derives the real per-request deadline and falls
|
||||
#: back to this value when the budget cannot be computed.
|
||||
#:
|
||||
#: Every engine that overrode the hook picked somewhere in 300s..900s, i.e. at
|
||||
#: or above the accelerated budget; only the ones that stayed silent got 60s.
|
||||
#:
|
||||
#: A literal rather than an import of model_manager.CPU_JOB_TIMEOUT_S: that
|
||||
#: value is read from the environment at import time and monkeypatched by
|
||||
#: tests, and a class-attribute default that moves with the environment is
|
||||
#: harder to reason about than one that does not. This module also reaches
|
||||
#: model_manager only lazily, from inside functions. The two are held in
|
||||
#: lockstep by backend/tests/test_subprocess_recv_timeout.py instead.
|
||||
GENERATE_RECV_TIMEOUT_S = 600.0
|
||||
|
||||
|
||||
# ── Idle sidecar reaping (parity Action 13) ─────────────────────────────────
|
||||
#
|
||||
@@ -378,12 +401,17 @@ class SubprocessBackend(TTSBackend):
|
||||
|
||||
# Per-engine recv timeout for generate(): how long the parent waits for the
|
||||
# sidecar's audio frame before the watchdog hard-kills the child and reclaims
|
||||
# its VRAM/device. Default is the conservative RECV_TIMEOUT_S (60s). A
|
||||
# subclass whose legitimate generates run longer overrides it (or exposes it
|
||||
# as a property) so a slow-but-valid synth is not falsely killed, while a
|
||||
# genuinely wedged one is still reclaimed. health_check() keeps using
|
||||
# RECV_TIMEOUT_S directly, since a ping must stay fast.
|
||||
recv_timeout_s: float = RECV_TIMEOUT_S
|
||||
# its VRAM/device. A subclass whose legitimate generates run longer overrides
|
||||
# it (or exposes it as a property) so a slow-but-valid synth is not falsely
|
||||
# killed, while a genuinely wedged one is still reclaimed. health_check()
|
||||
# keeps using RECV_TIMEOUT_S directly, since a ping must stay fast.
|
||||
#
|
||||
# The default is GENERATE_RECV_TIMEOUT_S, not RECV_TIMEOUT_S: 60s is a
|
||||
# health-check ping budget, and inheriting it as a *generation* deadline
|
||||
# killed four engines mid-sentence (#2103). Overriding remains the way to
|
||||
# ask for more; inheriting no longer means asking for less than the job's
|
||||
# own budget.
|
||||
recv_timeout_s: float = GENERATE_RECV_TIMEOUT_S
|
||||
|
||||
# ── instance state (initialised in __init__) ───────────────────────────
|
||||
|
||||
@@ -598,6 +626,58 @@ class SubprocessBackend(TTSBackend):
|
||||
tail = self._stderr_tail_text()
|
||||
return f"{reason}. Last stderr: {tail}" if tail else f"{reason} (no stderr output)"
|
||||
|
||||
def _effective_recv_timeout_s(self, text: str) -> float:
|
||||
"""This request's silence deadline: never under the budget it was granted.
|
||||
|
||||
``recv_timeout_s`` is a per-engine constant, but ``generate_timeout_s``
|
||||
scales the wall-clock budget with text length, so only a per-request
|
||||
deadline can satisfy "the watchdog must not fire before the caller's
|
||||
own budget expires".
|
||||
|
||||
An engine that overrode the hook keeps exactly its own value, including
|
||||
a smaller one: #1611 asks for more and #2103 asks that opting down stay
|
||||
possible.
|
||||
"""
|
||||
own = self.recv_timeout_s
|
||||
for klass in type(self).__mro__:
|
||||
if klass is SubprocessBackend:
|
||||
break # reached the base without finding an override
|
||||
if "recv_timeout_s" in klass.__dict__:
|
||||
return own # the engine chose; that choice is the answer
|
||||
try:
|
||||
from services.model_manager import generate_timeout_s
|
||||
|
||||
budget = float(generate_timeout_s(text, engine=self, _include_sidecar_grace=False))
|
||||
except Exception:
|
||||
# Budget probing is advisory: a failure here must not turn a
|
||||
# working generate into an error. Fall back to the class floor.
|
||||
return own
|
||||
if not math.isfinite(budget):
|
||||
return own
|
||||
return max(own, budget)
|
||||
|
||||
def _generate_failure_reason(self, elapsed_s: float, deadline_s: float) -> str:
|
||||
"""Say whether generate() lost the sidecar to the deadline or a crash.
|
||||
|
||||
The spawn handshake already distinguishes these (#2026); generate() did
|
||||
not, so a watchdog kill surfaced as "sidecar closed pipe mid-generate"
|
||||
and every reporter reasonably read it as a crash (#2103). The deadline
|
||||
is the one fact that explains the failure, so it belongs in the message
|
||||
the caller sees, not only in the backend log.
|
||||
"""
|
||||
if not self._last_recv_timed_out:
|
||||
# Unchanged wording for a genuine crash: only the deadline case was
|
||||
# misreported, and #2026 already gave the spawn path its own tail.
|
||||
return f"{self.id} sidecar closed pipe mid-generate"
|
||||
reason = (
|
||||
f"{self.id} sidecar sent nothing for {deadline_s:g}s "
|
||||
f"(elapsed {elapsed_s:.0f}s), so VoiceStudio stopped it. It may "
|
||||
f"simply be slower than that deadline on this host; retry or increase "
|
||||
f"this engine's receive timeout (see Troubleshooting)."
|
||||
)
|
||||
tail = self._stderr_tail_text()
|
||||
return f"{reason}. Last stderr: {tail}" if tail else reason
|
||||
|
||||
def _stderr_tail_text(self) -> str:
|
||||
"""The sidecar's last stderr lines, scrubbed for a user-visible error."""
|
||||
from core.scrub import scrub_text
|
||||
@@ -757,9 +837,11 @@ class SubprocessBackend(TTSBackend):
|
||||
for k, v in kw.items():
|
||||
if _is_jsonable(v):
|
||||
msg[k] = v
|
||||
deadline_s = self._effective_recv_timeout_s(text)
|
||||
started_at = time.monotonic()
|
||||
try:
|
||||
self._send(msg)
|
||||
reply = self._recv_with_timeout(self.recv_timeout_s)
|
||||
reply = self._recv_with_timeout(deadline_s)
|
||||
except (RuntimeError, OSError):
|
||||
# A broken or malformed protocol stream cannot be reused.
|
||||
# Reap it before releasing the request lock so an immediate
|
||||
@@ -790,13 +872,18 @@ class SubprocessBackend(TTSBackend):
|
||||
except Exception:
|
||||
pass # the heartbeat is best-effort; never fail a synth over it
|
||||
try:
|
||||
reply = self._recv_with_timeout(self.recv_timeout_s)
|
||||
reply = self._recv_with_timeout(deadline_s)
|
||||
except (RuntimeError, OSError):
|
||||
self._reap_unusable_process(proc)
|
||||
raise
|
||||
if not reply:
|
||||
# Same EOF for a deadline kill and a crash; _last_recv_timed_out
|
||||
# is what tells them apart (#2026's spawn path does the same).
|
||||
reason = self._generate_failure_reason(
|
||||
time.monotonic() - started_at, deadline_s
|
||||
)
|
||||
self._reap_unusable_process(proc)
|
||||
raise RuntimeError(f"{self.id} sidecar closed pipe mid-generate")
|
||||
raise RuntimeError(reason)
|
||||
if reply.get("op") == "error":
|
||||
stage = str(reply.get("stage") or "unknown")
|
||||
message = str(reply.get("message") or "unknown sidecar error")
|
||||
@@ -896,13 +983,31 @@ class SubprocessBackend(TTSBackend):
|
||||
fired.set()
|
||||
self._timeout_kill(proc)
|
||||
|
||||
watchdog = threading.Timer(timeout_s, _on_timeout)
|
||||
watchdog.daemon = True
|
||||
from services.inference_cancellation import current_cancellation
|
||||
cancellation = current_cancellation()
|
||||
stop_watchdog = threading.Event()
|
||||
|
||||
def _watch() -> None:
|
||||
deadline = time.monotonic() + timeout_s
|
||||
while not stop_watchdog.is_set():
|
||||
remaining = deadline - time.monotonic()
|
||||
if remaining <= 0 or (cancellation and cancellation.cancelled.is_set()):
|
||||
_on_timeout() # captured process only; never a later retry
|
||||
return
|
||||
stop_watchdog.wait(min(remaining, 0.05))
|
||||
|
||||
if cancellation is None:
|
||||
watchdog = threading.Timer(timeout_s, _on_timeout)
|
||||
watchdog.daemon = True
|
||||
else:
|
||||
watchdog = threading.Thread(target=_watch, daemon=True)
|
||||
watchdog.start()
|
||||
try:
|
||||
return self._recv()
|
||||
finally:
|
||||
watchdog.cancel()
|
||||
stop_watchdog.set()
|
||||
if cancellation is None:
|
||||
watchdog.cancel()
|
||||
# cancel() cannot stop an already-running callback. Finish its
|
||||
# bounded reap before another receive or generation starts.
|
||||
watchdog.join()
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
"""Decode an uploaded text file whose encoding nobody declared.
|
||||
|
||||
Subtitle and manuscript uploads arrive as raw bytes. Windows tools commonly
|
||||
save them as UTF-16 with a byte-order mark (Notepad's "Unicode", many subtitle
|
||||
editors) or in the legacy Windows-1252 code page. A UTF-8 decode turns the
|
||||
first into NUL-interleaved text and, lossily, drops every accent, dash and
|
||||
curly quote from the second.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import codecs
|
||||
|
||||
_BOMS = (
|
||||
(codecs.BOM_UTF8, "utf-8"),
|
||||
(codecs.BOM_UTF16_LE, "utf-16-le"),
|
||||
(codecs.BOM_UTF16_BE, "utf-16-be"),
|
||||
)
|
||||
|
||||
_CP1252_UNDEFINED = "voicestudio-cp1252-undefined"
|
||||
|
||||
|
||||
def _undefined_as_latin1(exc: UnicodeDecodeError) -> tuple[str, int]:
|
||||
# Only the offending bytes take their Latin-1 code point — what the
|
||||
# browser's windows-1252 decoder does — so the rest of the file keeps its
|
||||
# curly quotes and dashes.
|
||||
return exc.object[exc.start:exc.end].decode("latin-1"), exc.end
|
||||
|
||||
|
||||
codecs.register_error(_CP1252_UNDEFINED, _undefined_as_latin1)
|
||||
|
||||
|
||||
def decode_text_upload(data: bytes) -> str:
|
||||
"""Return the text of ``data``, without its byte-order mark.
|
||||
|
||||
A BOM names the encoding. Without one, valid UTF-8 is UTF-8; anything else
|
||||
is read as Windows-1252, the legacy code page such files come from. Each of
|
||||
the five bytes Windows-1252 leaves undefined takes its Latin-1 code point,
|
||||
so the decode never raises.
|
||||
"""
|
||||
for bom, encoding in _BOMS:
|
||||
if data.startswith(bom):
|
||||
return data[len(bom):].decode(encoding, errors="replace")
|
||||
try:
|
||||
return data.decode("utf-8")
|
||||
except UnicodeDecodeError:
|
||||
return data.decode("cp1252", errors=_CP1252_UNDEFINED)
|
||||
@@ -29,7 +29,19 @@ logger = logging.getLogger("omnivoice.translation_engines")
|
||||
|
||||
_NLLB_REPO_ID = "facebook/nllb-200-distilled-600M"
|
||||
_ARGOS_INSTALL_LOCK = threading.Lock()
|
||||
_ARGOS_LANG_ALIASES = {"cmn": "zh"}
|
||||
_ARGOS_LANG_ALIASES = {
|
||||
"cmn": "zh",
|
||||
"zho": "zh",
|
||||
"in": "id",
|
||||
"iw": "he",
|
||||
"fil": "tl",
|
||||
}
|
||||
# Human names (and the UI's own labels) that are not ISO 639-1 tokens.
|
||||
_ARGOS_NAME_ALIASES = {
|
||||
"chinese": "zh",
|
||||
"chinese (simplified)": "zh",
|
||||
"mandarin": "zh",
|
||||
}
|
||||
|
||||
|
||||
# Engine ID → registry entry. Keyed by the `provider` string sent from the
|
||||
@@ -39,7 +51,12 @@ REGISTRY: dict[str, dict] = {
|
||||
"id": "argos",
|
||||
"display_name": "Argos (Local, Fast)",
|
||||
"pip_package": "argostranslate",
|
||||
"probe_module": "argostranslate",
|
||||
# `argostranslate.translate`, not the bare package: the translator runs
|
||||
# on CTranslate2, and the bare package imports fine on a host whose
|
||||
# kernel rejects CTranslate2's native library (#692) — so a shallow
|
||||
# probe advertised Argos as ready and every translate 500'd. Probe the
|
||||
# module that actually pulls the native dep (same lesson as #1185).
|
||||
"probe_module": "argostranslate.translate",
|
||||
"category": "offline",
|
||||
"needs_key": False,
|
||||
"builtin": True,
|
||||
@@ -124,6 +141,18 @@ def _probe(entry: dict) -> tuple[bool, str | None]:
|
||||
mod = entry.get("probe_module")
|
||||
if not mod:
|
||||
return True, None
|
||||
if mod.startswith("argostranslate"):
|
||||
# Repair CTranslate2's exec-stack request before the import that would
|
||||
# be rejected by it (#692) — otherwise Argos, the default offline
|
||||
# engine, is unusable on kernels that refuse an executable stack.
|
||||
try:
|
||||
from core.execstack import ensure_ctranslate2_loadable
|
||||
|
||||
ok, detail = ensure_ctranslate2_loadable()
|
||||
if not ok:
|
||||
return False, detail
|
||||
except Exception as e: # noqa: BLE001 — a broken repair must not hide the engine
|
||||
logger.debug("exec-stack repair unavailable (%s) — probing anyway", e)
|
||||
try:
|
||||
importlib.import_module(mod)
|
||||
if entry.get("id") == "nllb":
|
||||
@@ -141,6 +170,11 @@ def _probe(entry: dict) -> tuple[bool, str | None]:
|
||||
return True, None
|
||||
except ImportError as e:
|
||||
return False, f"import {mod!r} failed: {e}"
|
||||
except Exception as e: # noqa: BLE001
|
||||
# A native library that refuses to load raises OSError, not ImportError
|
||||
# (#692). An availability probe must report "unusable here", never take
|
||||
# the engine list down with it.
|
||||
return False, f"import {mod!r} failed ({type(e).__name__}): {e}"
|
||||
|
||||
|
||||
def install_command(engine: "str | dict | None") -> str | None:
|
||||
@@ -299,9 +333,33 @@ def is_ready(engine_id: str) -> bool:
|
||||
|
||||
|
||||
def argos_lang_code(value: str) -> str:
|
||||
"""Return the base language token used by Argos package metadata."""
|
||||
code = str(value or "").strip().lower().split("-", 1)[0]
|
||||
"""Return the base language token used by Argos package metadata.
|
||||
|
||||
Accepts ISO 639-1 codes (e.g. ``"zh"``), BCP-47 tags with a region or script
|
||||
suffix (e.g. ``"zh-CN"``, ``"cmn-Hans"``), human names from the dub UI's own
|
||||
label list (e.g. ``"Chinese"``, ``"Mandarin"``), legacy / deprecated ISO
|
||||
639-1 codes still seen in older corpora (``"in"``→``"id"`` for Indonesian,
|
||||
``"iw"``→``"he"`` for Hebrew), and ISO 639-2/T (e.g. ``"zho"``→``"zh"``,
|
||||
``"fil"``→``"tl"`` for Tagalog). Empty or whitespace-only input raises
|
||||
``ValueError`` so the caller sees an actionable error instead of a
|
||||
silently-empty language token.
|
||||
"""
|
||||
raw = str(value or "").strip()
|
||||
key = raw.lower()
|
||||
parts = key.replace("_", "-").split("-")
|
||||
if key == "chinese (traditional)" or (
|
||||
parts[0] in {"zh", "zho", "cmn"} and
|
||||
any(part in {"hant", "tw", "hk", "mo"} for part in parts[1:])
|
||||
):
|
||||
raise ValueError("Argos does not provide Traditional Chinese; choose NLLB for this script")
|
||||
named = _ARGOS_NAME_ALIASES.get(key)
|
||||
if named:
|
||||
return named
|
||||
code = parts[0]
|
||||
code = _ARGOS_LANG_ALIASES.get(code, code)
|
||||
named = _ARGOS_NAME_ALIASES.get(code)
|
||||
if named:
|
||||
return named
|
||||
if not re.fullmatch(r"[a-z]{2,3}", code):
|
||||
raise ValueError("Choose a valid source and target language")
|
||||
return code
|
||||
|
||||
@@ -1624,6 +1624,7 @@ class KittenTTSBackend(TTSBackend):
|
||||
# resolved unchanged by `resolve_kokoro_lang_code()` below.
|
||||
_KOKORO_ISO_BY_FULL_NAME = {
|
||||
"english": "en",
|
||||
"british english": "en-gb",
|
||||
"spanish": "es",
|
||||
"french": "fr",
|
||||
"hindi": "hi",
|
||||
@@ -1634,6 +1635,31 @@ _KOKORO_ISO_BY_FULL_NAME = {
|
||||
}
|
||||
|
||||
|
||||
def _kokoro_supported_labels(aliases: dict, lang_codes: dict) -> list[str]:
|
||||
"""Human labels for every language Kokoro accepts, read from its own tables.
|
||||
|
||||
Derived from the installed package, never from
|
||||
``_KOKORO_ISO_BY_FULL_NAME``: that map exists to translate full names
|
||||
*into* Kokoro's codes, and reusing it to describe what Kokoro supports
|
||||
understates the model when a newly supported code has no full-name alias.
|
||||
Read every installed code so new languages remain visible without changing
|
||||
the input-name map.
|
||||
"""
|
||||
labels: dict[str, str] = {}
|
||||
# Prefer the full name a caller can actually pass.
|
||||
for name, iso in _KOKORO_ISO_BY_FULL_NAME.items():
|
||||
code = aliases.get(iso, iso)
|
||||
if code in lang_codes:
|
||||
labels.setdefault(code, name.title())
|
||||
# Then whatever the installed table supports that no full name reaches. Its
|
||||
# own description is a display name for some codes ("British English") and
|
||||
# an ISO tag for others ("pt-br"); either names the language better than
|
||||
# dropping it.
|
||||
for code, described in lang_codes.items():
|
||||
labels.setdefault(code, str(described))
|
||||
return sorted(labels.values())
|
||||
|
||||
|
||||
def resolve_kokoro_lang_code(language: str) -> str:
|
||||
"""Map a full language name / ISO code to Kokoro's single-letter
|
||||
`lang_code`, against the AUTHORITATIVE table read from the installed
|
||||
@@ -1651,7 +1677,12 @@ def resolve_kokoro_lang_code(language: str) -> str:
|
||||
iso = _KOKORO_ISO_BY_FULL_NAME.get(key, key)
|
||||
code = ALIASES.get(iso, iso)
|
||||
if code not in LANG_CODES:
|
||||
supported = ", ".join(sorted(name.title() for name in _KOKORO_ISO_BY_FULL_NAME))
|
||||
# Labels from newer installed tables must remain selectable even when
|
||||
# they have no entry in our compatibility map of full names.
|
||||
code = next((candidate for candidate, label in LANG_CODES.items()
|
||||
if str(label).strip().lower() == key), code)
|
||||
if code not in LANG_CODES:
|
||||
supported = ", ".join(_kokoro_supported_labels(ALIASES, LANG_CODES))
|
||||
raise ValueError(
|
||||
f"mlx-audio's Kokoro model (mlx-community/Kokoro-82M-bf16) doesn't "
|
||||
f"support language={language!r}. Kokoro supports: {supported}. "
|
||||
@@ -1840,22 +1871,40 @@ class MLXAudioBackend(TTSBackend):
|
||||
# model is actually active.
|
||||
kwargs["lang_code"] = language[:2].lower()
|
||||
|
||||
pieces = []
|
||||
def collect(results):
|
||||
groups = []
|
||||
rate = None
|
||||
pending = []
|
||||
|
||||
def flush():
|
||||
if not pending:
|
||||
return
|
||||
audio = np.concatenate(pending, axis=-1)
|
||||
if rate != self.sample_rate:
|
||||
import torchaudio
|
||||
audio = torchaudio.functional.resample(
|
||||
torch.from_numpy(audio), rate, self.sample_rate,
|
||||
).numpy()
|
||||
groups.append(audio)
|
||||
pending.clear()
|
||||
|
||||
for result in results:
|
||||
audio = getattr(result, "audio", result)
|
||||
if hasattr(audio, "numpy"):
|
||||
audio = audio.numpy()
|
||||
sr = getattr(result, "sample_rate", self.sample_rate)
|
||||
if sr != rate:
|
||||
flush()
|
||||
rate = sr
|
||||
pending.append(np.asarray(audio, dtype=np.float32))
|
||||
flush()
|
||||
return groups
|
||||
|
||||
try:
|
||||
for result in self._model.generate(**kwargs):
|
||||
audio = getattr(result, "audio", result)
|
||||
if hasattr(audio, "numpy"):
|
||||
audio = audio.numpy()
|
||||
pieces.append(np.asarray(audio, dtype=np.float32))
|
||||
pieces = collect(self._model.generate(**kwargs))
|
||||
except TypeError:
|
||||
# Some engines don't accept lang_code / ref_audio. Retry with
|
||||
# only the universal kwargs.
|
||||
pieces = []
|
||||
for result in self._model.generate(text=text, speed=speed):
|
||||
audio = getattr(result, "audio", result)
|
||||
if hasattr(audio, "numpy"):
|
||||
audio = audio.numpy()
|
||||
pieces.append(np.asarray(audio, dtype=np.float32))
|
||||
# Retry engines that accept only the universal arguments.
|
||||
pieces = collect(self._model.generate(text=text, speed=speed))
|
||||
|
||||
if not pieces:
|
||||
raise RuntimeError(f"mlx-audio ({self._model_id}) produced no audio")
|
||||
|
||||
@@ -41,6 +41,8 @@ def test_cuda_oom_falls_back_to_cpu(monkeypatch):
|
||||
return object() # CPU load succeeds
|
||||
|
||||
monkeypatch.setattr(whisperx, "load_model", fake_load_model)
|
||||
# Exercise load-time OOM, independent of actual free VRAM on this host.
|
||||
monkeypatch.setattr(WhisperXBackend, "_free_vram_gb", staticmethod(lambda: 10.0))
|
||||
|
||||
be = WhisperXBackend()
|
||||
# Force the CUDA starting point regardless of the CI host's hardware.
|
||||
|
||||
@@ -0,0 +1,247 @@
|
||||
"""Regression tests for #692 — the CTranslate2 exec-stack rejection is now
|
||||
*repaired*, not merely routed around.
|
||||
|
||||
ctranslate2 ≤4.4.0 (what whisperx 3.4.5 pins, and 3.4.5 is the last release
|
||||
supporting the Python 3.11 we ship) marks its native library's stack as
|
||||
executable. Kernels that refuse the request fail the dlopen outright, which
|
||||
took out both CTranslate2 ASR engines *and* Argos — the default offline dub
|
||||
translation engine, whose bare `argostranslate` import succeeds without the
|
||||
native dep, so the engine advertised itself as ready and every translate
|
||||
request 500'd with an opaque ImportError.
|
||||
|
||||
`core.execstack` clears the one ELF bit that causes it. These tests pin the
|
||||
patcher on synthetic ELFs (no ctranslate2 needed), the memoization, and the
|
||||
wiring into every consumer probe.
|
||||
"""
|
||||
import struct
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
_PT_GNU_STACK = 0x6474E551
|
||||
_PT_LOAD = 1
|
||||
|
||||
|
||||
def _elf(path, *, bits=64, endian="<", flags=0x7, phdr_type=_PT_GNU_STACK):
|
||||
"""Write a minimal ELF whose second program header is `phdr_type`.
|
||||
|
||||
Only the fields the patcher reads are meaningful — a real linker would emit
|
||||
far more, but the point is to prove the offsets are computed correctly for
|
||||
both ELF classes, and to fail loudly if they ever drift.
|
||||
"""
|
||||
is_64 = bits == 64
|
||||
phentsize = 56 if is_64 else 32
|
||||
phoff = 64 if is_64 else 52
|
||||
ident = b"\x7fELF" + bytes([2 if is_64 else 1, 1 if endian == "<" else 2, 1]) + b"\0" * 9
|
||||
header = bytearray(phoff)
|
||||
header[: len(ident)] = ident
|
||||
if is_64:
|
||||
struct.pack_into(endian + "Q", header, 0x20, phoff)
|
||||
struct.pack_into(endian + "HH", header, 0x36, phentsize, 2)
|
||||
else:
|
||||
struct.pack_into(endian + "I", header, 0x1C, phoff)
|
||||
struct.pack_into(endian + "HH", header, 0x2A, phentsize, 2)
|
||||
|
||||
def _phdr(p_type, p_flags):
|
||||
raw = bytearray(phentsize)
|
||||
struct.pack_into(endian + "I", raw, 0, p_type)
|
||||
struct.pack_into(endian + "I", raw, 4 if is_64 else 24, p_flags)
|
||||
return bytes(raw)
|
||||
|
||||
path.write_bytes(bytes(header) + _phdr(_PT_LOAD, 0x5) + _phdr(phdr_type, flags))
|
||||
return str(path)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("bits", [64, 32])
|
||||
@pytest.mark.parametrize("endian", ["<", ">"])
|
||||
def test_clear_execstack_flips_only_the_x_bit(tmp_path, bits, endian):
|
||||
from core import execstack
|
||||
lib = _elf(tmp_path / "libfake.so", bits=bits, endian=endian, flags=0x7)
|
||||
assert execstack.has_execstack(lib) is True
|
||||
|
||||
changed, detail = execstack.clear_execstack(lib)
|
||||
assert changed is True and "cleared" in detail
|
||||
assert execstack.has_execstack(lib) is False
|
||||
|
||||
# Read + write permissions survive; only PF_X is gone.
|
||||
with open(lib, "rb") as fh:
|
||||
offset, flags, _ = execstack._gnu_stack_flags_offset(fh)
|
||||
assert flags == 0x6
|
||||
|
||||
# Idempotent: a second pass is a no-op, so a restart never rewrites.
|
||||
assert execstack.clear_execstack(lib) == (False, "already non-executable")
|
||||
|
||||
|
||||
def test_non_executable_stack_is_left_alone(tmp_path):
|
||||
from core import execstack
|
||||
lib = _elf(tmp_path / "libok.so", flags=0x6)
|
||||
assert execstack.has_execstack(lib) is False
|
||||
before = (tmp_path / "libok.so").read_bytes()
|
||||
assert execstack.clear_execstack(lib) == (False, "already non-executable")
|
||||
assert (tmp_path / "libok.so").read_bytes() == before
|
||||
|
||||
|
||||
def test_elf_without_gnu_stack_segment(tmp_path):
|
||||
from core import execstack
|
||||
lib = _elf(tmp_path / "libnostack.so", phdr_type=_PT_LOAD, flags=0x7)
|
||||
assert execstack.has_execstack(lib) is None
|
||||
assert execstack.clear_execstack(lib) == (False, "no PT_GNU_STACK segment")
|
||||
|
||||
|
||||
def test_non_elf_and_missing_files_are_not_errors(tmp_path):
|
||||
from core import execstack
|
||||
text = tmp_path / "notelf.so"
|
||||
text.write_bytes(b"#!/bin/sh\necho hi\n")
|
||||
assert execstack.has_execstack(str(text)) is None
|
||||
assert execstack.clear_execstack(str(text))[0] is False
|
||||
|
||||
missing = str(tmp_path / "nope" / "libghost.so")
|
||||
assert execstack.has_execstack(missing) is None
|
||||
changed, detail = execstack.clear_execstack(missing)
|
||||
assert changed is False and "unreadable" in detail
|
||||
|
||||
|
||||
def test_ensure_rechecks_unrepairable_libraries(tmp_path, monkeypatch):
|
||||
from core import execstack
|
||||
lib = _elf(tmp_path / "libctranslate2-test.so.4.4.0", flags=0x7)
|
||||
monkeypatch.setattr(execstack.sys, "platform", "linux")
|
||||
monkeypatch.setattr(execstack, "ctranslate2_library_paths", lambda: [lib])
|
||||
monkeypatch.setattr(
|
||||
execstack, "clear_execstack", lambda p: (False, "not writable (PermissionError)")
|
||||
)
|
||||
execstack.reset_ctranslate2_cache()
|
||||
ok, detail = execstack.ensure_ctranslate2_loadable()
|
||||
assert ok is False
|
||||
# Actionable: names the library, why, and both ways out.
|
||||
assert "executable stack" in detail and "patchelf" in detail and "3.12" in detail
|
||||
|
||||
# An external repair must become visible without a backend restart.
|
||||
_elf(tmp_path / "libctranslate2-test.so.4.4.0", flags=0x6)
|
||||
assert execstack.ensure_ctranslate2_loadable()[0] is True
|
||||
|
||||
|
||||
def test_ensure_repairs_then_reports_ok(tmp_path, monkeypatch):
|
||||
from core import execstack
|
||||
lib = _elf(tmp_path / "libctranslate2-test.so.4.4.0", flags=0x7)
|
||||
monkeypatch.setattr(execstack.sys, "platform", "linux")
|
||||
monkeypatch.setattr(execstack, "ctranslate2_library_paths", lambda: [lib])
|
||||
execstack.reset_ctranslate2_cache()
|
||||
ok, detail = execstack.ensure_ctranslate2_loadable()
|
||||
assert ok is True and "repaired" in detail
|
||||
assert execstack.has_execstack(lib) is False
|
||||
execstack.reset_ctranslate2_cache()
|
||||
|
||||
|
||||
def test_ensure_is_a_noop_off_linux(tmp_path, monkeypatch):
|
||||
"""macOS/Windows never reject an exec-stack request — don't touch signed
|
||||
bundles looking for a problem that cannot exist there."""
|
||||
from core import execstack
|
||||
monkeypatch.setattr(execstack.sys, "platform", "darwin")
|
||||
monkeypatch.setattr(
|
||||
execstack, "ctranslate2_library_paths", lambda: pytest.fail("probed off Linux")
|
||||
)
|
||||
execstack.reset_ctranslate2_cache()
|
||||
ok, detail = execstack.ensure_ctranslate2_loadable()
|
||||
assert ok is True and "off Linux" in detail
|
||||
execstack.reset_ctranslate2_cache()
|
||||
|
||||
|
||||
# ── Wiring: every consumer of the native lib must consult the repair ─────────
|
||||
|
||||
|
||||
def test_asr_probes_report_unavailable_when_repair_impossible(monkeypatch):
|
||||
from services import asr_backend as ab
|
||||
|
||||
monkeypatch.setattr(
|
||||
ab, "_ctranslate2_execstack_ok", lambda: (False, "kernel refuses it")
|
||||
)
|
||||
okx, msgx = ab.WhisperXBackend.is_available()
|
||||
okf, msgf = ab.FasterWhisperBackend.is_available()
|
||||
assert okx is False and "CTranslate2" in msgx and "kernel refuses it" in msgx
|
||||
assert okf is False and "CTranslate2" in msgf and "kernel refuses it" in msgf
|
||||
|
||||
|
||||
def test_argos_probe_module_pulls_the_native_dep():
|
||||
"""`argostranslate` alone imports fine with a broken CTranslate2 — the
|
||||
registry must probe the module that actually loads it, or the Engine
|
||||
selector advertises an engine whose every request fails."""
|
||||
from services.translation_engines import REGISTRY
|
||||
|
||||
assert REGISTRY["argos"]["probe_module"] == "argostranslate.translate"
|
||||
|
||||
|
||||
def test_engine_probe_survives_a_native_load_failure(monkeypatch):
|
||||
from services import translation_engines as te
|
||||
|
||||
def boom(name):
|
||||
raise OSError("libctranslate2-x.so: cannot enable executable stack")
|
||||
|
||||
monkeypatch.setattr(te.importlib, "import_module", boom)
|
||||
ok, detail = te._probe({"probe_module": "deep_translator"})
|
||||
assert ok is False and "OSError" in detail
|
||||
|
||||
|
||||
@pytest.mark.parametrize("bits", [32, 64])
|
||||
@pytest.mark.parametrize("length", [16, 31, 45, 63])
|
||||
def test_truncated_elf_is_not_an_error(tmp_path, bits, length):
|
||||
from core import execstack
|
||||
path = tmp_path / "short.so"
|
||||
_elf(path, bits=bits)
|
||||
path.write_bytes(path.read_bytes()[:length])
|
||||
before = path.read_bytes()
|
||||
assert execstack.has_execstack(str(path)) is None
|
||||
assert execstack.clear_execstack(str(path))[0] is False
|
||||
assert path.read_bytes() == before
|
||||
|
||||
|
||||
def test_install_after_absent_probe_is_detected(tmp_path, monkeypatch):
|
||||
from core import execstack
|
||||
monkeypatch.setattr(execstack.sys, "platform", "linux")
|
||||
libs = []
|
||||
monkeypatch.setattr(execstack, "ctranslate2_library_paths", lambda: libs)
|
||||
assert execstack.ensure_ctranslate2_loadable()[0]
|
||||
libs.append(_elf(tmp_path / "new.so"))
|
||||
assert execstack.ensure_ctranslate2_loadable()[0]
|
||||
assert execstack.has_execstack(libs[0]) is False
|
||||
|
||||
|
||||
def test_concurrent_repairs_remain_available(tmp_path, monkeypatch):
|
||||
from core import execstack
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
lib = _elf(tmp_path / "parallel.so")
|
||||
monkeypatch.setattr(execstack.sys, "platform", "linux")
|
||||
monkeypatch.setattr(execstack, "ctranslate2_library_paths", lambda: [lib])
|
||||
with ThreadPoolExecutor(max_workers=8) as pool:
|
||||
results = list(pool.map(lambda _: execstack.ensure_ctranslate2_loadable(), range(24)))
|
||||
assert all(ok for ok, _ in results)
|
||||
assert execstack.has_execstack(lib) is False
|
||||
|
||||
|
||||
def test_isolated_probe_checks_repair_before_import(monkeypatch):
|
||||
from core import execstack
|
||||
from services.subprocess_asr import IsolatedFasterWhisperBackend
|
||||
monkeypatch.setattr(execstack, "ensure_ctranslate2_loadable", lambda: (False, "repair blocked"))
|
||||
ok, detail = IsolatedFasterWhisperBackend.is_available()
|
||||
assert not ok and "repair blocked" in detail
|
||||
|
||||
|
||||
def test_repair_uses_host_locking_capability_when_target_platform_is_emulated(tmp_path, monkeypatch):
|
||||
from core import execstack
|
||||
import builtins
|
||||
import os
|
||||
from types import SimpleNamespace
|
||||
|
||||
lib = _elf(tmp_path / 'libctranslate2-test.so', flags=0x7)
|
||||
monkeypatch.setattr(execstack.sys, 'platform', 'linux')
|
||||
monkeypatch.setattr(execstack, 'os', SimpleNamespace(**{**vars(os), 'name': 'nt'}))
|
||||
original_import = builtins.__import__
|
||||
|
||||
def windows_import(name, *args, **kwargs):
|
||||
if name == 'fcntl':
|
||||
raise ModuleNotFoundError('fcntl is unavailable on Windows')
|
||||
return original_import(name, *args, **kwargs)
|
||||
|
||||
monkeypatch.setattr(builtins, '__import__', windows_import)
|
||||
changed, _ = execstack.clear_execstack(lib)
|
||||
assert changed
|
||||
assert execstack.has_execstack(lib) is False
|
||||
@@ -27,10 +27,7 @@ from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from services.subprocess_backend import (
|
||||
RECV_TIMEOUT_S,
|
||||
SubprocessBackend,
|
||||
)
|
||||
from services.subprocess_backend import RECV_TIMEOUT_S, SubprocessBackend
|
||||
from services.tts_backend import OmniVoiceBackend, get_backend_class, list_backends
|
||||
from engines.omnivoice_subprocess import (
|
||||
OmniVoiceMPSSubprocessBackend,
|
||||
@@ -321,11 +318,14 @@ class _PlainBackend(SubprocessBackend):
|
||||
return ["multi"]
|
||||
|
||||
|
||||
def test_base_default_recv_timeout_is_60s():
|
||||
# A subclass that does NOT override keeps the conservative default, so the
|
||||
# existing subprocess engines (IndexTTS, dots.tts, ...) are byte-identical.
|
||||
assert SubprocessBackend.recv_timeout_s == RECV_TIMEOUT_S == 60.0
|
||||
assert _PlainBackend().recv_timeout_s == 60.0
|
||||
def test_base_default_recv_timeout_covers_a_generation():
|
||||
# A sidecar that does not choose gets a deadline that outlasts the
|
||||
# wall-clock budget its own job was granted (#2103).
|
||||
from services.subprocess_backend import GENERATE_RECV_TIMEOUT_S
|
||||
assert SubprocessBackend.recv_timeout_s == GENERATE_RECV_TIMEOUT_S == 600.0
|
||||
assert _PlainBackend().recv_timeout_s == 600.0
|
||||
# The ping budget itself is unchanged: health_check() still wants 60s.
|
||||
assert RECV_TIMEOUT_S == 60.0
|
||||
|
||||
|
||||
def test_sidecar_spawn_delegates_all_containment_to_nested_owner(monkeypatch, tmp_path):
|
||||
@@ -414,6 +414,189 @@ def test_omnivoice_subprocess_recv_timeout_floors_at_30s(monkeypatch):
|
||||
assert OmniVoiceSubprocessBackend().recv_timeout_s == 30.0
|
||||
|
||||
|
||||
def test_subprocess_sidecar_timeout_error_message(monkeypatch):
|
||||
"""When a sidecar hits its receive timeout, generate() must report the
|
||||
timeout and deadline rather than describing a generic pipe-closed crash (#2103)."""
|
||||
b = _PlainBackend()
|
||||
|
||||
class _FakeProc:
|
||||
def poll(self):
|
||||
return None
|
||||
|
||||
b._proc = _FakeProc()
|
||||
monkeypatch.setattr(b, "_send", lambda msg: None)
|
||||
|
||||
def fake_recv_timeout(timeout_s):
|
||||
b._last_recv_timed_out = True
|
||||
return None
|
||||
|
||||
monkeypatch.setattr(b, "_recv_with_timeout", fake_recv_timeout)
|
||||
monkeypatch.setattr(b, "_reap_unusable_process", lambda proc: None)
|
||||
|
||||
with pytest.raises(RuntimeError) as exc:
|
||||
b.generate("hello")
|
||||
assert "600s" in str(exc.value)
|
||||
assert "stopped it" in str(exc.value)
|
||||
b._proc = None
|
||||
|
||||
|
||||
def test_subprocess_sidecar_initial_empty_crash_reports_pipe_closed(monkeypatch):
|
||||
"""When a sidecar process terminates without timing out, generate() reports pipe closure (#2103)."""
|
||||
b = _PlainBackend()
|
||||
|
||||
class _FakeProc:
|
||||
def poll(self):
|
||||
return None
|
||||
|
||||
b._proc = _FakeProc()
|
||||
monkeypatch.setattr(b, "_send", lambda msg: None)
|
||||
|
||||
def fake_recv_crash(timeout_s):
|
||||
b._last_recv_timed_out = False
|
||||
return None
|
||||
|
||||
monkeypatch.setattr(b, "_recv_with_timeout", fake_recv_crash)
|
||||
monkeypatch.setattr(b, "_reap_unusable_process", lambda proc: None)
|
||||
|
||||
with pytest.raises(RuntimeError) as exc:
|
||||
b.generate("hello")
|
||||
assert "sidecar closed pipe mid-generate" in str(exc.value)
|
||||
b._proc = None
|
||||
|
||||
|
||||
def test_subprocess_engine_timeouts_raised():
|
||||
"""All large subprocess TTS engines must declare generous timeouts rather
|
||||
than inheriting the 60s class default (#2103)."""
|
||||
from engines.confucius4 import Confucius4Backend
|
||||
from engines.dots_tts import DotsTTSBackend
|
||||
from engines.moss_tts_v15 import MossTTSV15Backend
|
||||
from engines.supertonic3.backend import Supertonic3Backend
|
||||
|
||||
assert Confucius4Backend().recv_timeout_s >= 300.0
|
||||
assert DotsTTSBackend().recv_timeout_s >= 300.0
|
||||
assert MossTTSV15Backend().recv_timeout_s >= 300.0
|
||||
assert Supertonic3Backend().recv_timeout_s >= 300.0
|
||||
|
||||
|
||||
def test_subprocess_engine_timeout_env_overrides(monkeypatch):
|
||||
"""Subprocess TTS engines honor their engine-specific receive timeout env overrides (#2103)."""
|
||||
from engines.confucius4 import Confucius4Backend
|
||||
from engines.dots_tts import DotsTTSBackend
|
||||
from engines.moss_tts_v15 import MossTTSV15Backend
|
||||
from engines.supertonic3.backend import Supertonic3Backend
|
||||
|
||||
monkeypatch.setenv("OMNIVOICE_CONFUCIUS4_RECV_TIMEOUT_S", "1200")
|
||||
monkeypatch.setenv("OMNIVOICE_DOTS_TTS_RECV_TIMEOUT_S", "1000")
|
||||
monkeypatch.setenv("OMNIVOICE_MOSS_TTS_V15_RECV_TIMEOUT_S", "1100")
|
||||
monkeypatch.setenv("OMNIVOICE_SUPERTONIC3_RECV_TIMEOUT_S", "500")
|
||||
|
||||
assert Confucius4Backend().recv_timeout_s == 1200.0
|
||||
assert DotsTTSBackend().recv_timeout_s == 1000.0
|
||||
assert MossTTSV15Backend().recv_timeout_s == 1100.0
|
||||
assert Supertonic3Backend().recv_timeout_s == 500.0
|
||||
|
||||
|
||||
def test_subprocess_asr_recv_timeout_env_override(monkeypatch):
|
||||
"""SubprocessASRBackend.recv_timeout_s honors OMNIVOICE_ASR_RECV_TIMEOUT_S
|
||||
and safely rejects non-finite, malformed, zero, or negative inputs (#2103)."""
|
||||
from pathlib import Path
|
||||
from services.subprocess_asr import SubprocessASRBackend
|
||||
|
||||
class _FakeASR(SubprocessASRBackend):
|
||||
id = "fake-asr"
|
||||
display_name = "fake-asr"
|
||||
gpu_compat = ("cuda", "mps", "cpu")
|
||||
|
||||
@classmethod
|
||||
def is_available(cls): return True, "ok"
|
||||
@classmethod
|
||||
def venv_python(cls): return Path(sys.executable)
|
||||
@classmethod
|
||||
def sidecar_script(cls): return Path("fake")
|
||||
|
||||
b = _FakeASR()
|
||||
assert b.recv_timeout_s == 600.0
|
||||
|
||||
# Valid override
|
||||
monkeypatch.setenv("OMNIVOICE_ASR_RECV_TIMEOUT_S", "750")
|
||||
assert b.recv_timeout_s == 750.0
|
||||
|
||||
# Malformed value falls back to default
|
||||
monkeypatch.setenv("OMNIVOICE_ASR_RECV_TIMEOUT_S", "not-a-number")
|
||||
assert b.recv_timeout_s == 600.0
|
||||
|
||||
# Non-finite values fall back to default
|
||||
for invalid in ("nan", "inf", "-inf"):
|
||||
monkeypatch.setenv("OMNIVOICE_ASR_RECV_TIMEOUT_S", invalid)
|
||||
assert b.recv_timeout_s == 600.0
|
||||
|
||||
# Zero or negative values clamped to 30.0 minimum
|
||||
for low in ("0", "-10", "15"):
|
||||
monkeypatch.setenv("OMNIVOICE_ASR_RECV_TIMEOUT_S", low)
|
||||
assert b.recv_timeout_s == 30.0
|
||||
|
||||
|
||||
def test_subprocess_asr_timeout_error_message(monkeypatch):
|
||||
"""SubprocessASRBackend.transcribe() raises actionable timeout guidance when watchdog fires (#2103)."""
|
||||
from pathlib import Path
|
||||
from services.subprocess_asr import SubprocessASRBackend
|
||||
|
||||
class _FakeASR(SubprocessASRBackend):
|
||||
id = "fake-asr"
|
||||
display_name = "fake-asr"
|
||||
gpu_compat = ("cuda", "mps", "cpu")
|
||||
|
||||
@classmethod
|
||||
def is_available(cls): return True, "ok"
|
||||
@classmethod
|
||||
def venv_python(cls): return Path(sys.executable)
|
||||
@classmethod
|
||||
def sidecar_script(cls): return Path("fake")
|
||||
|
||||
class _FakeProc:
|
||||
def poll(self): return None
|
||||
def wait(self, timeout=None): return 0
|
||||
def kill(self): pass
|
||||
def terminate(self): pass
|
||||
|
||||
b = _FakeASR()
|
||||
b._proc = _FakeProc()
|
||||
monkeypatch.setattr(b, "_spawn", lambda: None)
|
||||
monkeypatch.setattr(b, "_send", lambda msg: None)
|
||||
|
||||
def fake_recv_timeout(timeout_s):
|
||||
b._last_recv_timed_out = True
|
||||
return None
|
||||
|
||||
monkeypatch.setattr(b, "_recv_with_timeout", fake_recv_timeout)
|
||||
monkeypatch.setattr(b, "shutdown", lambda: None)
|
||||
monkeypatch.setattr("services.model_manager.running_on_gpu_pool", lambda: True)
|
||||
|
||||
with pytest.raises(RuntimeError) as exc:
|
||||
b.transcribe("test.wav")
|
||||
assert "fake-asr ASR sidecar exceeded receive timeout" in str(exc.value)
|
||||
assert "OMNIVOICE_ASR_RECV_TIMEOUT_S" in str(exc.value)
|
||||
|
||||
|
||||
def test_generate_timeout_s_coordinates_with_engine_recv_timeout():
|
||||
"""Outer generation timeout must coordinate with engine sidecar timeout with bounded grace (#2103)."""
|
||||
from services.model_manager import generate_timeout_s
|
||||
|
||||
class _SlowEngine:
|
||||
recv_timeout_s = 900.0
|
||||
|
||||
# With 900s sidecar timeout, outer budget must include at least 5s grace (>= 905s).
|
||||
budget = generate_timeout_s("short text", engine=_SlowEngine())
|
||||
assert budget >= 905.0
|
||||
|
||||
class _FastEngine:
|
||||
recv_timeout_s = 60.0
|
||||
|
||||
# For fast engines, the default GPU/CPU budget still applies as the floor.
|
||||
budget_fast = generate_timeout_s("short text", engine=_FastEngine(), execution_device="cuda")
|
||||
assert budget_fast >= 300.0
|
||||
|
||||
|
||||
# ── roundtrip via the stub sidecar ─────────────────────────────────────────
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,118 @@
|
||||
"""The pytorch-whisper fallback must survive a CUDA OOM instead of losing a chunk.
|
||||
|
||||
The VRAM preflight sizes the *weights*; generation adds a workspace that scales
|
||||
with the batch, and word timestamps keep every layer's cross-attention for the
|
||||
whole batch. So a card with room for the model still OOMs at the first
|
||||
transcribe — and the dub path merely retried the identical call, gave up, and
|
||||
emitted nothing for that chunk: a silent hole in the transcript, indistinguishable
|
||||
from silence. Step the batch down, then finish on CPU.
|
||||
"""
|
||||
import pytest
|
||||
|
||||
from services.asr_backend import PyTorchWhisperBackend
|
||||
|
||||
_OOM = RuntimeError("CUDA out of memory. Tried to allocate 2.00 GiB")
|
||||
|
||||
|
||||
class _FakePipe:
|
||||
"""Stands in for a transformers ASR pipeline on CUDA."""
|
||||
|
||||
def __init__(self, oom_below_batch):
|
||||
self.device = "cuda:0"
|
||||
self.oom_below_batch = oom_below_batch
|
||||
self.calls = []
|
||||
|
||||
def __call__(self, _audio, *, return_timestamps, chunk_length_s, batch_size):
|
||||
self.calls.append(batch_size)
|
||||
if batch_size > self.oom_below_batch:
|
||||
raise _OOM
|
||||
return {"text": "ok", "chunks": [{"text": "ok", "timestamp": (0.0, 1.0)}]}
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def audio(tmp_path):
|
||||
import numpy as np
|
||||
import soundfile as sf
|
||||
|
||||
path = tmp_path / "a.wav"
|
||||
sf.write(path, np.zeros(16000, dtype="float32"), 16000)
|
||||
return str(path)
|
||||
|
||||
|
||||
def test_oom_steps_down_the_batch_and_still_returns_text(audio):
|
||||
pipe = _FakePipe(oom_below_batch=2)
|
||||
backend = PyTorchWhisperBackend(asr_pipe=pipe)
|
||||
out = backend.transcribe(audio, word_timestamps=True)
|
||||
assert out["text"] == "ok"
|
||||
# Tried the word-timestamp ladder in order, stopping at the first that fits.
|
||||
assert pipe.calls == [8, 2]
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"word_timestamps,expected", [(True, [8, 2, 1]), (False, [16, 4, 1])]
|
||||
)
|
||||
def test_batch_ladder_is_smaller_when_word_timestamps_are_requested(
|
||||
audio, monkeypatch, word_timestamps, expected
|
||||
):
|
||||
"""Word timestamps retain per-layer cross-attention for the whole batch, so
|
||||
the ladder must start lower than for plain transcription."""
|
||||
pipe = _FakePipe(oom_below_batch=0)
|
||||
backend = PyTorchWhisperBackend(asr_pipe=pipe)
|
||||
sentinel = RuntimeError("cpu rebuild reached")
|
||||
monkeypatch.setattr(
|
||||
backend, "_rebuild_on_cpu", lambda: (_ for _ in ()).throw(sentinel)
|
||||
)
|
||||
with pytest.raises(RuntimeError, match="cpu rebuild reached"):
|
||||
backend.transcribe(audio, word_timestamps=word_timestamps)
|
||||
assert pipe.calls == expected
|
||||
|
||||
|
||||
def test_exhausted_ladder_falls_back_to_cpu(audio, monkeypatch):
|
||||
pipe = _FakePipe(oom_below_batch=0)
|
||||
backend = PyTorchWhisperBackend(asr_pipe=pipe)
|
||||
|
||||
cpu = _FakePipe(oom_below_batch=99)
|
||||
cpu.device = "cpu"
|
||||
|
||||
def _rebuild():
|
||||
backend._pipe = cpu
|
||||
|
||||
monkeypatch.setattr(backend, "_rebuild_on_cpu", _rebuild)
|
||||
out = backend.transcribe(audio, word_timestamps=True)
|
||||
assert out["text"] == "ok"
|
||||
assert pipe.calls == [8, 2, 1] and cpu.calls == [2]
|
||||
|
||||
|
||||
def test_non_oom_errors_are_not_retried(audio):
|
||||
class _Boom(_FakePipe):
|
||||
def __call__(self, *a, **kw):
|
||||
self.calls.append(kw["batch_size"])
|
||||
raise ValueError("bad audio")
|
||||
|
||||
pipe = _Boom(oom_below_batch=99)
|
||||
with pytest.raises(ValueError):
|
||||
PyTorchWhisperBackend(asr_pipe=pipe).transcribe(audio)
|
||||
assert pipe.calls == [8] # one attempt, no ladder
|
||||
|
||||
|
||||
def test_cpu_pipeline_keeps_the_small_batch(audio):
|
||||
pipe = _FakePipe(oom_below_batch=99)
|
||||
pipe.device = "cpu"
|
||||
PyTorchWhisperBackend(asr_pipe=pipe).transcribe(audio)
|
||||
assert pipe.calls == [2]
|
||||
|
||||
|
||||
def test_cpu_fallback_preserves_injected_model_and_processor(monkeypatch):
|
||||
import torch
|
||||
from types import SimpleNamespace
|
||||
calls = []
|
||||
model = SimpleNamespace(to=lambda **kw: calls.append(kw))
|
||||
processor = object()
|
||||
pipe = SimpleNamespace(model=model, tokenizer=processor, feature_extractor=processor, device="cuda:0")
|
||||
backend = PyTorchWhisperBackend(asr_pipe=pipe)
|
||||
monkeypatch.setattr(backend, "_model_name", lambda: pytest.fail("must not resolve another checkpoint"))
|
||||
backend._rebuild_on_cpu()
|
||||
assert backend._pipe is pipe and pipe.model is model
|
||||
assert pipe.tokenizer is processor and pipe.feature_extractor is processor
|
||||
assert str(pipe.device) == "cpu"
|
||||
assert calls == [{"device": "cpu", "dtype": torch.float32}]
|
||||
@@ -0,0 +1,249 @@
|
||||
"""Every sidecar's generate deadline outlasts the job budget it was granted (#2103).
|
||||
|
||||
#1611 raised IndexTTS's deadline because a healthy synthesis was being killed at
|
||||
60s. That fixed the reported engine and left the class default alone, so four
|
||||
more engines — confucius4, dots_tts, moss_tts_v15, supertonic3 — inherited the
|
||||
same 60s and were killed the same way.
|
||||
|
||||
60s is the ``health_check`` ping budget. Inheriting it as a *generation*
|
||||
deadline puts the sidecar watchdog five to ten times below
|
||||
``model_manager.generate_timeout_s`` (300s accelerated, 600s CPU), so the
|
||||
watchdog reclaims a sidecar the caller still considers well inside its budget.
|
||||
Every engine that overrode the hook picked 300s..900s, i.e. at or above the
|
||||
accelerated budget; the four that stayed silent are the whole bug.
|
||||
|
||||
The invariant below is what keeps a new engine from re-entering that state by
|
||||
omission, which is the part #1611 could not do by fixing one engine.
|
||||
"""
|
||||
import pytest
|
||||
|
||||
|
||||
|
||||
# The engines named in #2103 that inherited the ping budget. Listed explicitly
|
||||
# so the regression is legible even if the registry is reorganised later.
|
||||
REGRESSED_ENGINE_IDS = ("confucius4-tts", "dots-tts", "moss-tts-v15", "supertonic3")
|
||||
|
||||
|
||||
def _subprocess_backend_classes():
|
||||
"""Every SubprocessBackend the registry can hand a user, by id."""
|
||||
from services.tts_backend import get_backend_class
|
||||
from services.tts_backend import list_backends
|
||||
found = {}
|
||||
for row in list_backends(include_hidden=True):
|
||||
try:
|
||||
cls = get_backend_class(row["id"])
|
||||
except Exception:
|
||||
continue # an engine whose optional import is absent cannot be dispatched
|
||||
if isinstance(cls, type) and getattr(cls, "_is_subprocess_isolated", False) and hasattr(cls, "recv_timeout_s"):
|
||||
found[row["id"]] = cls
|
||||
return found
|
||||
|
||||
|
||||
def test_ping_budget_and_generate_budget_are_separate_constants():
|
||||
# A ping must stay fast; a generation must not be cut off at a ping's deadline.
|
||||
from services.subprocess_backend import GENERATE_RECV_TIMEOUT_S
|
||||
from services.subprocess_backend import RECV_TIMEOUT_S
|
||||
assert RECV_TIMEOUT_S == 60.0
|
||||
assert GENERATE_RECV_TIMEOUT_S > RECV_TIMEOUT_S
|
||||
|
||||
|
||||
def test_default_generate_deadline_covers_the_cpu_job_budget():
|
||||
# Lockstep with model_manager: raising either budget there without raising
|
||||
# this one re-opens #2103 for every engine that does not override.
|
||||
# Imported rather than duplicated so the two cannot drift silently.
|
||||
from services.subprocess_backend import GENERATE_RECV_TIMEOUT_S
|
||||
from services.subprocess_backend import SubprocessBackend
|
||||
assert GENERATE_RECV_TIMEOUT_S >= 600.0
|
||||
assert SubprocessBackend.recv_timeout_s == GENERATE_RECV_TIMEOUT_S
|
||||
|
||||
|
||||
@pytest.mark.parametrize("engine_id", REGRESSED_ENGINE_IDS)
|
||||
def test_regressed_engines_no_longer_inherit_the_ping_budget(engine_id):
|
||||
from services.subprocess_backend import RECV_TIMEOUT_S
|
||||
cls = _subprocess_backend_classes().get(engine_id)
|
||||
if cls is None:
|
||||
pytest.fail(f"{engine_id} is not registered in this build")
|
||||
# Read through an instance: several engines expose the hook as a property.
|
||||
assert cls.__new__(cls).recv_timeout_s > RECV_TIMEOUT_S
|
||||
|
||||
|
||||
def test_no_registered_sidecar_undercuts_the_accelerated_job_budget():
|
||||
"""The class-level guard #1611 was missing.
|
||||
|
||||
A new SubprocessBackend that simply does not think about ``recv_timeout_s``
|
||||
now inherits a deadline that already satisfies this; one that overrides it
|
||||
with something too small fails here rather than in a user's generation.
|
||||
"""
|
||||
from services.model_manager import GPU_JOB_TIMEOUT_S
|
||||
too_short = {}
|
||||
for engine_id, cls in _subprocess_backend_classes().items():
|
||||
deadline = cls.__new__(cls).recv_timeout_s
|
||||
if deadline < GPU_JOB_TIMEOUT_S:
|
||||
too_short[engine_id] = deadline
|
||||
assert not too_short, (
|
||||
"these sidecars would be killed before their own job budget expires: "
|
||||
f"{too_short} (accelerated budget is {GPU_JOB_TIMEOUT_S:g}s)"
|
||||
)
|
||||
|
||||
|
||||
# ── a constant is not enough: the budget scales with the text (#2109 review) ─
|
||||
|
||||
|
||||
def _SilentBackend():
|
||||
from services.subprocess_backend import SubprocessBackend
|
||||
from services.subprocess_backend import SubprocessBackend
|
||||
class SilentBackend(SubprocessBackend):
|
||||
"""A sidecar with no custom deadline."""
|
||||
id = "silent"
|
||||
@classmethod
|
||||
def is_available(cls):
|
||||
return True, "ok"
|
||||
@property
|
||||
def sample_rate(self):
|
||||
return 24000
|
||||
@property
|
||||
def supported_languages(self):
|
||||
return ["multi"]
|
||||
return SilentBackend()
|
||||
|
||||
|
||||
def _OpinionatedBackend():
|
||||
backend = _SilentBackend()
|
||||
type(backend).id = "opinionated"
|
||||
type(backend).recv_timeout_s = 45.0
|
||||
return backend
|
||||
|
||||
|
||||
def test_a_long_passage_raises_the_deadline_past_the_flat_default():
|
||||
# generate_timeout_s adds 1s per 40 characters past a 1200-char allowance,
|
||||
# so a long passage is granted more than the flat floor.
|
||||
from services.subprocess_backend import GENERATE_RECV_TIMEOUT_S
|
||||
backend = _SilentBackend()
|
||||
short = backend._effective_recv_timeout_s("hello")
|
||||
long_text = "x" * 200_000
|
||||
long_deadline = backend._effective_recv_timeout_s(long_text)
|
||||
|
||||
assert short == GENERATE_RECV_TIMEOUT_S
|
||||
assert long_deadline > short
|
||||
# And it tracks the budget itself, not some second guess at it.
|
||||
from services.model_manager import generate_timeout_s
|
||||
assert generate_timeout_s(long_text, engine=backend) >= long_deadline + 5.0
|
||||
|
||||
|
||||
def test_an_engine_that_opts_down_keeps_its_own_deadline():
|
||||
# #2103 asks that fast engines stay able to opt down, so deriving from the
|
||||
# budget must not overrule an override in either direction.
|
||||
backend = _OpinionatedBackend()
|
||||
assert backend._effective_recv_timeout_s("hello") == 45.0
|
||||
assert backend._effective_recv_timeout_s("x" * 200_000) == 45.0
|
||||
|
||||
|
||||
def test_budget_probe_failure_falls_back_instead_of_failing_the_generate(monkeypatch):
|
||||
from services.subprocess_backend import GENERATE_RECV_TIMEOUT_S
|
||||
import services.model_manager as mm
|
||||
|
||||
def _boom(*a, **kw):
|
||||
raise RuntimeError("device probe unavailable")
|
||||
|
||||
monkeypatch.setattr(mm, "generate_timeout_s", _boom)
|
||||
assert _SilentBackend()._effective_recv_timeout_s("hello") == GENERATE_RECV_TIMEOUT_S
|
||||
|
||||
|
||||
# ── the deadline has to appear in the error the caller sees (#2103) ─────────
|
||||
|
||||
# Wedges on the first synthesize, so the parent's watchdog is the only thing
|
||||
# that can end the request — the exact shape the #1611 and #2103 reporters hit.
|
||||
WEDGING_SIDECAR = r'''
|
||||
import sys, json, struct, time
|
||||
|
||||
def _send(o):
|
||||
b = json.dumps(o, separators=(",", ":")).encode()
|
||||
sys.stdout.buffer.write(struct.pack("!I", len(b)) + b)
|
||||
sys.stdout.buffer.flush()
|
||||
|
||||
_send({"op": "ready", "engine": "omnivoice-subprocess", "sample_rate": 24000})
|
||||
print("sidecar still alive, just slow", file=sys.stderr, flush=True)
|
||||
while True:
|
||||
time.sleep(1)
|
||||
'''
|
||||
|
||||
|
||||
def test_timeout_error_names_the_deadline_instead_of_blaming_the_pipe(
|
||||
tmp_path, monkeypatch,
|
||||
):
|
||||
"""#2103's second half: the watchdog's own deadline reached the user.
|
||||
|
||||
Before this, a kill and a crash both raised "sidecar closed pipe
|
||||
mid-generate", so the one fact that explains the failure — that
|
||||
VoiceStudio stopped the sidecar on its own deadline — appeared only in the
|
||||
backend log, and reporters reasonably concluded the engine had crashed.
|
||||
"""
|
||||
from engines.omnivoice_subprocess import OmniVoiceSubprocessBackend
|
||||
script = tmp_path / "wedging_sidecar.py"
|
||||
script.write_text(WEDGING_SIDECAR)
|
||||
monkeypatch.setattr(
|
||||
OmniVoiceSubprocessBackend, "sidecar_script", classmethod(lambda cls: script),
|
||||
)
|
||||
# 2s so the test is fast; the property floors env overrides at 30s, so set
|
||||
# the attribute the base actually reads (as the existing wedge test does).
|
||||
monkeypatch.setattr(
|
||||
OmniVoiceSubprocessBackend, "recv_timeout_s", property(lambda self: 2.0),
|
||||
)
|
||||
backend = OmniVoiceSubprocessBackend()
|
||||
try:
|
||||
with pytest.raises(RuntimeError) as excinfo:
|
||||
backend.generate("anything")
|
||||
finally:
|
||||
backend.shutdown()
|
||||
|
||||
message = str(excinfo.value)
|
||||
assert "2s" in message, message # the deadline that ended it
|
||||
assert "stopped it" in message, message # who ended it, not "it closed"
|
||||
assert "closed pipe" not in message, message
|
||||
# #2026's stderr tail is carried on this path too, so a sidecar that did
|
||||
# say something before the kill is not silenced by the timeout.
|
||||
assert "still alive" in message, message
|
||||
|
||||
|
||||
@pytest.mark.parametrize("text", ["short", "x" * 200000], ids=["short", "long"])
|
||||
@pytest.mark.parametrize("engine_type", [_SilentBackend, _OpinionatedBackend])
|
||||
def test_outer_guard_outlasts_sidecar_watchdog(text, engine_type):
|
||||
from services.model_manager import generate_timeout_s
|
||||
backend = engine_type()
|
||||
assert generate_timeout_s(text, engine=backend) >= backend._effective_recv_timeout_s(text) + 5.0
|
||||
|
||||
|
||||
def test_explicit_generation_budget_is_authoritative(monkeypatch):
|
||||
import services.model_manager as mm
|
||||
monkeypatch.setattr(mm, 'GPU_JOB_TIMEOUT_S', 12.0)
|
||||
monkeypatch.setattr(mm, '_GENERATE_TIMEOUT_EXPLICIT', True)
|
||||
assert mm.generate_timeout_s('short', engine=_SilentBackend(), execution_device='cuda') == 12.0
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize('guard_kind', ['asr', 'tts'])
|
||||
async def test_outer_abandonment_terminates_owned_sidecar(tmp_path, monkeypatch, guard_kind):
|
||||
from engines.omnivoice_subprocess import OmniVoiceSubprocessBackend
|
||||
import asyncio
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from services.model_manager import run_on_gpu_pool_guarded
|
||||
from services.asr_backend import run_transcribe_guarded
|
||||
script = tmp_path / 'wedging_sidecar.py'
|
||||
script.write_text(WEDGING_SIDECAR)
|
||||
monkeypatch.setattr(OmniVoiceSubprocessBackend, 'sidecar_script', classmethod(lambda cls: script))
|
||||
monkeypatch.setattr(OmniVoiceSubprocessBackend, 'recv_timeout_s', property(lambda self: 600.0))
|
||||
backend = OmniVoiceSubprocessBackend()
|
||||
# Guard lifetime is independent of which protocol operation is waiting.
|
||||
with ThreadPoolExecutor(max_workers=1) as executor:
|
||||
try:
|
||||
if guard_kind == 'asr':
|
||||
work = run_transcribe_guarded(executor, lambda: backend.generate('hang'), timeout=0.5)
|
||||
else:
|
||||
work = run_on_gpu_pool_guarded(lambda: backend.generate('hang'), executor=executor, timeout=0.5)
|
||||
with pytest.raises(TimeoutError):
|
||||
await work
|
||||
assert backend._proc is not None
|
||||
await asyncio.wait_for(asyncio.to_thread(backend._proc.wait, timeout=5), timeout=6)
|
||||
assert backend._proc.poll() is not None
|
||||
finally:
|
||||
backend.shutdown()
|
||||
+6
-7
@@ -21,13 +21,12 @@ The pinned commit SHA for `omnivoice.cpp` lives in
|
||||
scripts/build-omnivoice-tts.sh --platform <slug> --commit-sha <40hex>
|
||||
```
|
||||
|
||||
See `.github/workflows/ci.yml` `build-omnivoice-tts` job for the CI
|
||||
matrix that produces these artifacts on every PR. The Apple Silicon
|
||||
slot (`macos-14`) is marked `continue-on-error: true` because the
|
||||
upstream `omnivoice.cpp` README does not publish a `buildmetal.sh`
|
||||
(Pitfall 1 in `04-RESEARCH.md`); a failed Metal build is documented
|
||||
and the macOS Apple Silicon cloning default falls back to the
|
||||
in-process `VoiceStudioBackend`.
|
||||
See `.github/workflows/build-omnivoice-tts.yml` `build-omnivoice-tts` job
|
||||
for the CI matrix that produces these artifacts. Apple Silicon (`macos-14`)
|
||||
builds cleanly with `-DGGML_METAL=ON` at the pinned SHA (#2105), enabling
|
||||
hardware-accelerated Metal inference when the packaged artifact passes binary
|
||||
preflight and macOS permits execution. Missing or blocked binaries retain the
|
||||
in-process `VoiceStudioBackend` fallback.
|
||||
|
||||
## Placeholder note
|
||||
|
||||
|
||||
@@ -75,7 +75,7 @@
|
||||
},
|
||||
"frontend": {
|
||||
"name": "omnivoice-studio",
|
||||
"version": "0.5.3",
|
||||
"version": "0.5.4",
|
||||
"dependencies": {
|
||||
"@fontsource-variable/inter": "^5.3.0",
|
||||
"@fontsource-variable/source-serif-4": "^5.3.0",
|
||||
|
||||
@@ -103,6 +103,27 @@ RUN python3 -c "import os, torch, torchaudio, torchvision; \
|
||||
import torchvision.ops; torchvision.ops.nms; \
|
||||
print('torchvision C++ ops resolve against this torch')"
|
||||
|
||||
# CTranslate2 (WhisperX, faster-whisper) links cuDNN 8, but the CUDA base image
|
||||
# ships cuDNN 9, so libcudnn_ops_infer.so.8 is absent and loading it aborts the
|
||||
# backend process outright rather than raising (#1371). scripts/setup.py
|
||||
# side-loads the cuDNN 8 libraries for source installs; the image needs the same
|
||||
# shim or Docker users lose every CTranslate2 ASR engine (#2050).
|
||||
#
|
||||
# The target is derived from sys.prefix rather than hardcoded: backend/core/
|
||||
# cudnn8.py looks for <sys.prefix>/lib/pythonX.Y/site-packages/cudnn8_compat,
|
||||
# and sys.prefix differs between the conda-based CUDA image and the ROCm venv.
|
||||
# --no-deps keeps this to the cuDNN wheels alone, leaving the base image's torch
|
||||
# stack untouched. Skipped for ROCm, which does not use cuDNN.
|
||||
RUN if [ "$GPU_FLAVOR" = "cuda" ]; then \
|
||||
target="$(python3 -c "import os, sys; print(os.path.join(sys.prefix, 'lib', 'python%d.%d' % sys.version_info[:2], 'site-packages', 'cudnn8_compat'))")" && \
|
||||
uv pip install --python "$(command -v python3)" --no-cache --no-deps \
|
||||
--target "$target" nvidia-cudnn-cu12==8.9.7.29 && \
|
||||
python3 -c "import os, sys; d = os.path.join(sys.prefix, 'lib', 'python%d.%d' % sys.version_info[:2], 'site-packages', 'cudnn8_compat', 'nvidia', 'cudnn', 'lib'); \
|
||||
libs = [f for f in os.listdir(d) if '.so.8' in f]; \
|
||||
assert libs, 'cudnn8_compat installed but no .so.8 libraries in ' + d; \
|
||||
print('cuDNN 8 compat libraries: %d' % len(libs))"; \
|
||||
fi
|
||||
|
||||
# Copy application source
|
||||
COPY backend/ ./backend/
|
||||
COPY omnivoice/ ./omnivoice/
|
||||
|
||||
+29
-3
@@ -58,6 +58,20 @@ does not update clients already on `v0.2.1`.
|
||||
|
||||
## 5. Cutting a release
|
||||
|
||||
Release notes lead with the biggest user-visible change, not README edits or
|
||||
packaging internals. For a desktop redesign or migration, include a real UI
|
||||
screenshot pinned to the release tag, explain what changed, and give direct
|
||||
installer links and migration steps. Keep Highlights to 3–5 bullets, followed
|
||||
by concise themed entries. Preserve any installer-trust disclosures.
|
||||
|
||||
Verify credits against the previous-tag-to-new-tag commit comparison and the
|
||||
included PRs, including contributor branches merged through maintainer branches.
|
||||
Add a Contributors section naming every human author and their contribution;
|
||||
thank verified bug reporters separately and identify dependency bots separately.
|
||||
Do not infer contributors from the existing changelog's `thanks` entries alone.
|
||||
Electron publishes the authored version section verbatim, with any required
|
||||
installer-trust disclosure appended by the workflow.
|
||||
|
||||
1. **CHANGELOG first (hard rule):** make sure `CHANGELOG.md` has a complete,
|
||||
user-facing `## [X.Y.Z] — DATE` section (rename `## [Unreleased]`).
|
||||
`release.yml` extracts that section verbatim as the GitHub Release body —
|
||||
@@ -81,7 +95,7 @@ on that tag with `draft=true`, and wait for its final Tauri installers and signe
|
||||
updater feeds. Then dispatch `electron-release.yml` on the same tag. Automatic
|
||||
Electron builds are skipped for this tag to avoid racing the Tauri draft.
|
||||
Keep the release draft until both builds and their checks have passed.
|
||||
See [Electron transition](#electron-transition-next-desktop-release) below for
|
||||
See [Electron transition](#electron-desktop-releases) below for
|
||||
signing requirements and the explicit owner-only unsigned exception.
|
||||
|
||||
## 5b. Deployment channels — all must ship (hard rule, owner-set 2026-07-16)
|
||||
@@ -152,7 +166,7 @@ versionless updater archive. Other versions, sibling platforms, and updater
|
||||
manifests remain intact. Inventory or deletion permission/network failures stop
|
||||
the job instead of hiding an upload collision.
|
||||
|
||||
## Electron transition (next desktop release)
|
||||
## Electron desktop releases
|
||||
|
||||
Electron is the primary desktop distribution. electron-release.yml builds Linux
|
||||
x64, Windows x64, macOS arm64 and macOS x64, checks packaged startup and updater
|
||||
@@ -167,7 +181,10 @@ same tag completes. Electron requires the final signed latest.json and
|
||||
latest-user.json assets; subsequent releases copy those feeds without changing
|
||||
their immutable sunset payload URLs. Retain the sunset release and its assets.
|
||||
|
||||
Write versioned CHANGELOG notes before release. Review all four platform builds,
|
||||
Write versioned CHANGELOG notes before release. The Electron release body uses
|
||||
that authored section verbatim, including its introduction and contributor
|
||||
credits; describe Tauri migration only when relevant, without announcing another
|
||||
Tauri release. Review all four platform builds,
|
||||
checksums, signing requirements and docs/electron-migration.md. Existing Electron
|
||||
artifact names and app IDs remain stable for updater compatibility. This pipeline
|
||||
ships stable releases; rolling preview publication is paused during transition.
|
||||
@@ -188,3 +205,12 @@ choice. Tauri's signing keys do not sign Electron packages.
|
||||
For the transition tag, automatic Electron release jobs are skipped. Build the
|
||||
manual Tauri sunset draft first, then dispatch Electron on the same tag after
|
||||
its signed updater feeds exist. Later tags build Electron automatically.
|
||||
|
||||
|
||||
If a packaging-workflow fix is needed after tagging, keep the release tag
|
||||
immutable. Merge and validate the workflow fix on main, then dispatch
|
||||
`electron-release.yml` from main with `release_tag=vX.Y.Z`. Validation and every
|
||||
packaging/release job check out that exact tag; only the workflow comes from
|
||||
main. Empty signing secrets are omitted from the builder environment so drafts
|
||||
and explicitly accepted unsigned builds do not interpret the working directory
|
||||
as a certificate. Publication still requires `publish=true` and the same guards.
|
||||
|
||||
+1
-1
@@ -39,7 +39,7 @@ VoiceStudio/
|
||||
│ │ └── setup/ first-run wizard, model download
|
||||
│ ├── core/ config, db, job queue, event bus, auth/CSRF, path security,
|
||||
│ │ opt-in analytics, version, diagnostics
|
||||
│ ├── services/ 86 modules of business logic — TTS, dubbing pipeline,
|
||||
│ ├── services/ 88 modules of business logic — TTS, dubbing pipeline,
|
||||
│ │ audio DSP, GPU gateway, engine routing, model lifecycle
|
||||
│ ├── engines/ per-engine adapters: indextts, supertonic3, confucius4,
|
||||
│ │ dots_tts, moss_tts_v15, pockettts, audiocpp,
|
||||
|
||||
@@ -30,7 +30,7 @@ The integration shape is `VoiceStudioGGUFBackend(TTSBackend)` wrapping Phase 2's
|
||||
| License compatible with v0.3.x ship? | YES — Apache-2.0 (model) + MIT (runtime) | Both verified via HF model card + GitHub README. Same Apache-2.0 chain as the upstream model already shipping in v0.2.7. |
|
||||
| Runtime: llama.cpp / candle / custom? | CUSTOM (`omnivoice.cpp`, MIT) — does NOT load in vanilla llama.cpp | `gguf.architecture = "omnivoice-lm"` from HF API; README states "GGUF weights for omnivoice.cpp, a C++17/GGML port of VoiceStudio". |
|
||||
| Quant variants and footprints? | 4 quants × 2 files each (base + tokenizer): Q4_K_M (659 MB), Q8_0 (945 MB), BF16 (1.60 GB), F32 (3.19 GB) | HF `siblings` list confirms all 8 files; sizes from model card table. |
|
||||
| Cross-platform runtime fit? | Linux + Windows + macOS Intel YES via documented build scripts; macOS Apple Silicon Metal CONDITIONAL (no `buildmetal.sh` published, only feature mention) | `buildcpu.sh`, `buildcuda.sh`, `buildvulkan.sh`, `buildall.sh` listed; Metal in description only — Wave 1 Task 3 builds and verifies via `cmake -DGGML_METAL=ON` per A1. |
|
||||
| Cross-platform runtime fit? | Linux + Windows + macOS Intel YES via documented build scripts; macOS Apple Silicon Metal YES (builds clean via `cmake -DGGML_METAL=ON` at pinned SHA, #2105) | `buildcpu.sh`, `buildcuda.sh`, `buildvulkan.sh`, `buildall.sh` listed; Metal builds cleanly via `cmake -DGGML_METAL=ON` in CI and locally (#2105). |
|
||||
| Subprocess CLI fits Phase 2 `SubprocessBackend`? | YES | README shows `echo "Hello world." | ./build/omnivoice-tts --model … --codec … --lang … -o …` — line-oriented stdin + argv + output-file pattern is exactly what `SubprocessBackend` is designed for. |
|
||||
|
||||
### Pinned SHAs (filled in Wave 1 by Task 1)
|
||||
@@ -52,13 +52,13 @@ Both SHAs are mirrored in `backend/engines/omnivoice_gguf/quant_map.json` `_meta
|
||||
- Adds a maintained-by-others C++ runtime to the dependency graph (`omnivoice.cpp`, 42 stars at decision time).
|
||||
- Adds ~12-16 MB of platform binaries to the installer (must verify against Phase 3 mirror-timing baseline per Pitfall 6).
|
||||
- macOS code signing scope expands by 4 binaries (track via REL-05; same `xattr -cr` workaround as #54 applies in v0.3.x).
|
||||
- `omnivoice.cpp` README does not publish a macOS Metal build script — only `buildcpu.sh`, `buildcuda.sh`, `buildvulkan.sh`, `buildall.sh`. Apple Silicon Metal must be verified in Wave 1.
|
||||
- `omnivoice.cpp` README does not publish a standalone `buildmetal.sh` script; Apple Silicon Metal is built directly via `cmake -DGGML_METAL=ON` (#2105).
|
||||
|
||||
**Mitigations:**
|
||||
- Pin `omnivoice.cpp` by commit SHA (`886fc079838ca7400cb2b42b36e2a65aa1daabe8`); rebuild from pinned SHA in CI for all 4 target platforms.
|
||||
- Pin every quant file by commit SHA in `quant_map.json` (`361609388ae572a820d085185bbbe2a2aac4b30e`); shippable JSON so the table can update without an app release.
|
||||
- In-process `VoiceStudioBackend` remains as fallback if any GGUF step fails (probe, download, load, generate).
|
||||
- macOS Apple Silicon Metal build is verified in Wave 1 with explicit acceptance criteria; if blocked, downgrade SPIKE-01 default on macOS to in-process path and document in this ADR's "Status" line.
|
||||
- macOS Apple Silicon Metal build is verified via `cmake -DGGML_METAL=ON` (#2105); in-process `VoiceStudioBackend` remains available as general fallback.
|
||||
- SHA-256 checksums on bundled binaries (per GATE-05); verify at first launch and on every quant load.
|
||||
- Subprocess arg composition uses typed `Path` objects rooted in app directories; quant override UI is a dropdown over `quant_map.json` entries only (no freeform path input — supply-chain control analogous to INST-09).
|
||||
|
||||
|
||||
@@ -234,3 +234,33 @@ panels instead so neither editor becomes unusably small.
|
||||
from-source checkout.
|
||||
- **Installed it but still "needs install"** — restart the backend so Python
|
||||
picks up the newly-installed module.
|
||||
- **"The 'argos' engine's CTranslate2 runtime could not be loaded…"** — Argos
|
||||
translates on CTranslate2, and on Linux kernels that refuse an executable
|
||||
stack the CTranslate2 library shipped with Python 3.11 installs (4.4.0) is
|
||||
rejected outright. VoiceStudio repairs that library in place on first use; if
|
||||
it cannot (read-only install), the message names the fix — reinstall the
|
||||
backend on Python 3.12+, or run `patchelf --clear-execstack` on the library
|
||||
once — and NLLB stays available in the meantime
|
||||
([#692](https://github.com/debpalash/VoiceStudio/issues/692)).
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
SRT and WebVTT exports round each cue timestamp once to the nearest millisecond, including carry into the next second or minute. The OpenAI-compatible transcription exports use the same formatter.
|
||||
|
||||
|
||||
Paste translation accepts WebVTT files with hourless timestamps. Only timing records at line starts activate timestamp matching; timestamp-like text inside a sentence remains dialogue.
|
||||
|
||||
Subtitle import preserves numeric dialogue such as years and countdowns, including files mixing numbered and unnumbered cues. Cue numbers are removed only at identified cue boundaries.
|
||||
|
||||
|
||||
Argos accepts Chinese/Simplified Chinese names, Mandarin aliases, and language tags such as `zh-CN`. Traditional Chinese requests (`zh-TW`, `zh-Hant`, and the display name) are rejected explicitly; select NLLB for Traditional Chinese rather than silently receiving a different script.
|
||||
|
||||
Mixed or malformed SRT files can make a bare number indistinguishable from spoken dialogue. The importer removes numbering only when cue boundaries and sequential numbering support it; ambiguous nonsequential numbers are retained as text to avoid silent data loss. Standard indexed SRT and WebVTT exports avoid this ambiguity.
|
||||
|
||||
If the Argos native runtime cannot load, both desktop and browser clients show localized recovery guidance: reinstall the backend or select NLLB. The API returns the stable `argos_runtime_unavailable` error code without exposing native library paths.
|
||||
|
||||
WebVTT import separates metadata blocks from cue identifiers using the [WebVTT block-parsing rules](https://www.w3.org/TR/webvtt1/#file-parsing): a timing line immediately after an identifier makes a cue, even when that identifier is NOTE, STYLE, or REGION. Later timing examples inside metadata are ignored, and empty cues never borrow the next cue’s identifier as dialogue.
|
||||
|
||||
Dubbing transcription emits keepalives during quiet diarization, reference-refinement, and cleanup steps. Disconnecting stops queued model work; native calls already running retain their model until they finish, then cleanup restores TTS. Task streams also request that proxies disable buffering so keepalives reach the client promptly.
|
||||
|
||||
@@ -18,3 +18,21 @@ permissions. No automatic installer-to-installer migration is provided.
|
||||
|
||||
The final Tauri updater feeds retain signed Tauri payloads at immutable URLs.
|
||||
A Tauri updater must never receive an Electron installer.
|
||||
|
||||
Electron checks required Python imports before reusing an existing runtime. An
|
||||
incomplete environment opens setup instead of repeatedly crashing; installation
|
||||
still requires your explicit action. Automatic selection skips broken legacy
|
||||
runtimes and uses the Electron runtime location, leaving Tauri data intact.
|
||||
|
||||
An explicitly selected runtime is not silently replaced during startup. If that
|
||||
location contains an incomplete environment Electron does not own, choosing
|
||||
setup creates a separate runtime in Electron’s default location instead of
|
||||
modifying or taking ownership of the existing environment.
|
||||
|
||||
A new custom runtime destination remains selectable. Setup creates and owns it
|
||||
only when that directory does not already exist; existing unowned roots stay
|
||||
untouched even if their `project` subdirectory is missing.
|
||||
|
||||
Dependency checks ignore inherited `PYTHONPATH` and `PYTHONHOME`, matching backend
|
||||
startup. If repair switches away from an unowned environment and a healthy
|
||||
Electron runtime already exists, it is reused without reinstalling dependencies.
|
||||
|
||||
@@ -55,10 +55,12 @@ transcription.
|
||||
process, so the engine checks up front and reports itself unavailable
|
||||
instead ([#1371](https://github.com/debpalash/VoiceStudio/issues/1371)).
|
||||
pytorch-whisper covers that case on torch's bundled cuDNN 9.
|
||||
- On some hardened Linux kernels the CTranslate2 native library is rejected
|
||||
with "cannot enable executable stack" (an OSError, not an ImportError) —
|
||||
reported as unavailable rather than crashing engine selection
|
||||
([#692](https://github.com/debpalash/VoiceStudio/issues/692)).
|
||||
- On Linux kernels that refuse an executable stack, the CTranslate2 native
|
||||
library (4.4.0 and older) is rejected with "cannot enable executable stack"
|
||||
(an OSError, not an ImportError). VoiceStudio clears that ELF flag in place
|
||||
on first probe so the engine loads; if the file cannot be written it reports
|
||||
itself unavailable with the repair command rather than crashing engine
|
||||
selection ([#692](https://github.com/debpalash/VoiceStudio/issues/692)).
|
||||
- CTranslate2's GPU teardown can rarely segfault the process at unload. If
|
||||
you hit that, switch to the crash-isolated variant —
|
||||
[faster-whisper-isolated](faster-whisper-isolated.md)
|
||||
|
||||
@@ -43,7 +43,8 @@ HF repo id. The env var overrides the persisted UI choice.
|
||||
|
||||
## Behaviour notes
|
||||
|
||||
- Output is 24 kHz mono for most hosted models.
|
||||
- Output is 24 kHz mono. Results from models with a different native rate
|
||||
(such as Dia at 44.1 kHz) are resampled before stitching and export.
|
||||
- **Cloning works only with the `csm` model** — it is the only curated model
|
||||
confirmed to accept a reference clip. Other models silently ignore
|
||||
reference audio, so the engine reports cloning support only when CSM is
|
||||
@@ -53,7 +54,14 @@ HF repo id. The env var overrides the persisted UI choice.
|
||||
- Language support is per-model (Kokoro ~8 languages, others vary). An
|
||||
unsupported language for Kokoro produces a clear error naming what it
|
||||
does support ([#977](https://github.com/debpalash/VoiceStudio/issues/977))
|
||||
— leave language on Auto or switch to a multilingual engine.
|
||||
— pick a language it supports or switch to a multilingual engine.
|
||||
- Auto is not an escape hatch from that. With the picker on Auto the request
|
||||
carries no language, and a selected voice profile's saved language fills the
|
||||
gap ([#533](https://github.com/debpalash/VoiceStudio/issues/533)) — so a
|
||||
profile saved as, say, Persian still reaches Kokoro and is still refused. The
|
||||
error names the profile as the source in that case
|
||||
([#2156](https://github.com/debpalash/VoiceStudio/issues/2156)); change the
|
||||
profile's language, or pick a supported one explicitly for the render.
|
||||
|
||||
## Platform notes
|
||||
|
||||
@@ -74,3 +82,15 @@ See also: [benchmarks.md](../benchmarks.md),
|
||||
[languages.md](../languages.md),
|
||||
[downloading-models.md](../downloading-models.md),
|
||||
[disk usage](disk-usage.md).
|
||||
|
||||
Consecutive chunks with the same native sample rate are resampled together to
|
||||
preserve filter context at chunk boundaries; rate changes start a new group.
|
||||
|
||||
Profile language refusals are terminal request errors on both local and remote
|
||||
rendering, including streaming. Electron and web show localized guidance that
|
||||
Auto inherits the profile language; neither silently retries the same refusal.
|
||||
|
||||
Kokoro language errors list every language in the installed model’s table.
|
||||
“British English” and `en-gb` both select its British English voice pipeline.
|
||||
Display names from newer installed Kokoro tables are accepted too, so a language
|
||||
advertised by the error message can be selected without updating a hardcoded map.
|
||||
|
||||
@@ -61,6 +61,12 @@ A 6 GB card with nothing else loaded runs the default model on the GPU
|
||||
([#2041](https://github.com/debpalash/VoiceStudio/issues/2041)). Disable
|
||||
the check with `OMNIVOICE_ASR_VRAM_PREFLIGHT=0`.
|
||||
|
||||
The preflight sizes the weights; the generation workspace grows with the
|
||||
batch on top. If a transcribe still hits a CUDA out-of-memory, the engine
|
||||
steps the batch down (16 → 4 → 1, or 8 → 2 → 1 with word timestamps) and
|
||||
finishes on the CPU rather than dropping the chunk — no silent holes in a
|
||||
dub transcript.
|
||||
|
||||
## Quirks
|
||||
|
||||
- If the pipeline fails to import (`AutoFeatureExtractor` errors), the cause
|
||||
|
||||
@@ -62,8 +62,12 @@ Two more fallback chains run at load time:
|
||||
process fast-fails with no traceback, so the engine is reported unavailable
|
||||
up front and selection falls through to pytorch-whisper, which uses torch's
|
||||
own cuDNN 9 ([#1371](https://github.com/debpalash/VoiceStudio/issues/1371)).
|
||||
- On some hardened Linux kernels CTranslate2's native library is rejected with
|
||||
"cannot enable executable stack" — reported as unavailable, not a crash
|
||||
- On Linux kernels that refuse an executable stack, CTranslate2's native
|
||||
library (4.4.0 and older — what whisperx 3.4.5 pins on Python 3.11) is
|
||||
rejected with "cannot enable executable stack". VoiceStudio now clears that
|
||||
one ELF flag in place on first probe and the engine loads normally; if the
|
||||
library cannot be written (a read-only bundle), the engine reports itself
|
||||
unavailable with the repair command instead of crashing
|
||||
([#692](https://github.com/debpalash/VoiceStudio/issues/692)).
|
||||
- A partially-installed environment (interrupted sync, antivirus quarantine)
|
||||
can break WhisperX's deep import chain (whisperx → pyannote →
|
||||
|
||||
@@ -419,3 +419,7 @@ Two paths are worth persisting across container restarts:
|
||||
[Pull and run (AMD GPU / ROCm)](#pull-and-run-amd-gpu--rocm) above for when
|
||||
to set one by hand.
|
||||
- More entries: [docs/install/troubleshooting.md](troubleshooting.md).
|
||||
|
||||
### CTranslate2 compatibility
|
||||
|
||||
CUDA images include an isolated cuDNN 8 compatibility library directory for WhisperX and faster-whisper alongside the base PyTorch cuDNN 9 runtime. The image build verifies the compatibility libraries exist. ROCm images skip this NVIDIA-only dependency.
|
||||
|
||||
@@ -188,7 +188,13 @@ before the token works for downloads.
|
||||
3. Retry the job. The token state in **Settings → API Keys** should now show
|
||||
the "App" row with a green check next to your username.
|
||||
|
||||
**Linked issue:** [#35](https://github.com/debpalash/VoiceStudio/issues/35)
|
||||
If the token and license are already valid but an older build still reports
|
||||
401 during installation, update VoiceStudio and retry. Settings tokens now
|
||||
reach both the fast download path and engine-weight installers; no token
|
||||
rotation is needed for that fixed client bug.
|
||||
|
||||
**Linked issues:** [#35](https://github.com/debpalash/VoiceStudio/issues/35),
|
||||
[#2163](https://github.com/debpalash/VoiceStudio/issues/2163)
|
||||
|
||||
### PocketTTS gated weights
|
||||
|
||||
@@ -291,15 +297,29 @@ ready. The desktop app sits on "starting backend", `/health` returns 503, and
|
||||
`/startup/progress` shows `ml_imports` active. From source you see `import
|
||||
torch` die with a native access violation rather than a Python traceback.
|
||||
|
||||
**Cause:** VoiceStudio pins `torch 2.8.0`. That build carries no `sm_120`
|
||||
kernels, so on a Blackwell card the CUDA initializer faults inside the native
|
||||
library. This is not a VoiceStudio bug and no setting works around it — the
|
||||
wheel does not contain code for the GPU.
|
||||
**Cause:** not established. The pinned build is not missing Blackwell code:
|
||||
`torch 2.8.0+cu128` lists `sm_120` in `torch.cuda.get_arch_list()`, and that
|
||||
build imports and runs CUDA normally on some Blackwell cards. What is confirmed
|
||||
is that on the Windows setups in
|
||||
[#1931](https://github.com/debpalash/VoiceStudio/issues/1931) `import torch`
|
||||
faults inside the native library before Python can raise an error, and moving
|
||||
the torch trio to 2.9.x clears it.
|
||||
|
||||
**Fix:** move the whole torch trio to a build with `sm_120` kernels. They must
|
||||
move together — upgrading one past the ABI the others were built against gives
|
||||
you `RuntimeError: operator torchvision::nms does not exist`, which is the
|
||||
next section's problem instead.
|
||||
If `import torch` crashes for you, that crash *is* the symptom — skip straight
|
||||
to the fix below. Where torch does import, this shows what the build actually
|
||||
contains:
|
||||
|
||||
```bash
|
||||
uv run python -c "import torch; print(torch.__version__, torch.cuda.get_arch_list())"
|
||||
```
|
||||
|
||||
`sm_120` in that list means the kernels are present and the crash is elsewhere
|
||||
in the native init path. Either way the upgrade below is the known workaround.
|
||||
|
||||
**Fix:** move the whole torch trio to 2.9.x. They must move together —
|
||||
upgrading one past the ABI the others were built against gives you
|
||||
`RuntimeError: operator torchvision::nms does not exist`, which is the next
|
||||
section's problem instead.
|
||||
|
||||
Edit **both** pin lists, keeping them identical:
|
||||
|
||||
@@ -341,8 +361,10 @@ guarded now, so the upgrade path above is clean on a current checkout.
|
||||
**Keeping the change:** these are the repo's own pins, so a `git pull` that
|
||||
touches them will conflict or overwrite. Re-apply after updating until the
|
||||
default pin moves — the default cannot move for everyone until the newer torch
|
||||
is verified across the older GPUs VoiceStudio supports, since a build that adds
|
||||
`sm_120` can drop older architectures.
|
||||
is verified across the older GPUs VoiceStudio supports, since a newer build can
|
||||
drop older architectures. Which ones varies by torch release, not by the CUDA
|
||||
variant alone: the pinned 2.8.0+cu128 build reports `sm_70` first, while the
|
||||
cu128 arch list captured in #1285 still carried `sm_61`.
|
||||
|
||||
**Linked issue:** [#1931](https://github.com/debpalash/VoiceStudio/issues/1931)
|
||||
— thanks to the reporter for the full diagnosis, including the verification
|
||||
@@ -1091,14 +1113,16 @@ retrying.
|
||||
and the backend log ends inside the `ml_imports` phase — often with a native
|
||||
crash (exit code `0xffffffff` / `-1073741819`) rather than a Python traceback.
|
||||
|
||||
**Cause.** VoiceStudio pins `torch 2.8.0+cu128`, which ships no `sm_120`
|
||||
kernels. On an RTX 50-series card `import torch` dies natively, before any
|
||||
VoiceStudio code can classify it — which is why the app can only say the
|
||||
backend did not start. This is a property of the pinned build, not of your
|
||||
driver or your install.
|
||||
**Cause.** Not established. `torch 2.8.0+cu128` does contain Blackwell code:
|
||||
`sm_120` is in `torch.cuda.get_arch_list()`, and that build imports and runs
|
||||
CUDA normally on some Blackwell cards, so this is not simply a wheel without
|
||||
kernels for your GPU. What is confirmed is that on the Windows setups in
|
||||
[#1931](https://github.com/debpalash/VoiceStudio/issues/1931) `import torch`
|
||||
dies natively before any VoiceStudio code can classify it, which is why the app
|
||||
can only say the backend did not start. Moving to torch 2.9.x clears it for the
|
||||
users who hit it.
|
||||
|
||||
**Fix.** Move to a torch build that has Blackwell kernels. From a source
|
||||
checkout, in the project folder:
|
||||
**Fix.** Move to torch 2.9.x. From a source checkout, in the project folder:
|
||||
|
||||
1. Edit `pyproject.toml` → `[tool.uv] constraint-dependencies` and raise the
|
||||
torch constraint to `torch==2.9.1+cu128` (matching `torchaudio` /
|
||||
@@ -1148,3 +1172,48 @@ remove the app binary itself are in
|
||||
[docs/install/uninstall.md](uninstall.md).
|
||||
|
||||
**Linked issue:** [#1089](https://github.com/debpalash/VoiceStudio/issues/1089)
|
||||
|
||||
### Pedalboard illegal-instruction crashes
|
||||
|
||||
VoiceStudio pins pedalboard to `>=0.9.14,<0.9.21` while [upstream portable-wheel repair #466](https://github.com/spotify/pedalboard/pull/466) remains open. Re-sync the locked environment after updating from a build with newer affected wheels. This keeps the existing effects API floor while avoiding the reported Linux CPU import crash.
|
||||
|
||||
### TorchCodec unavailable
|
||||
|
||||
When torchaudio requires an unavailable TorchCodec installation, VoiceStudio writes through soundfile and reads reference audio through its FFmpeg fallback. Reference amplitude is normalized using the decoded sample representation, including 8-, 24-, and 32-bit PCM.
|
||||
|
||||
### Isolated engine timeouts
|
||||
|
||||
Generation has a separate deadline from health checks. Default sidecar deadlines scale with text length and the host execution budget; per-engine timeout overrides remain supported. The outer job guard includes time for sidecar termination and error reporting. A timeout identifies the deadline, while a closed pipe without a timeout indicates a crash.
|
||||
|
||||
Per-engine receive overrides include `OMNIVOICE_CONFUCIUS4_RECV_TIMEOUT_S`, `OMNIVOICE_DOTS_TTS_RECV_TIMEOUT_S`, `OMNIVOICE_MOSS_TTS_V15_RECV_TIMEOUT_S`, and `OMNIVOICE_SUPERTONIC3_RECV_TIMEOUT_S` (seconds). Invalid or non-finite values use the default; values below 30 seconds are raised to 30.
|
||||
|
||||
### FFprobe alongside FFmpeg
|
||||
|
||||
When FFprobe is not on PATH, VoiceStudio also checks beside the selected FFmpeg binary. Parent folders named `ffmpeg` remain unchanged; only the executable name becomes `ffprobe` (or `ffprobe.exe` on Windows).
|
||||
|
||||
### Subtitle and manuscript encodings
|
||||
|
||||
Electron and web imports accept UTF-8, UTF-16 with a byte-order mark, and Windows-1252 text. The same decoder rules apply to uploaded subtitles and audiobook manuscripts; legacy punctuation is preserved.
|
||||
|
||||
|
||||
### Compile fallback after startup
|
||||
|
||||
An architecture accepted by the torch.compile preflight may still encounter independent Dynamo, Inductor, Triton, or CUDA-graph runtime errors. VoiceStudio distinguishes those from GPU memory exhaustion and retries with eager execution; architecture support alone does not guarantee compilation succeeds.
|
||||
|
||||
### Native engine installation from a remote client
|
||||
|
||||
Sidecar and audio.cpp runtime installation is restricted to requests from the backend computer's loopback interface. An API key does not bypass this restriction. The catalogue now shows local setup guidance instead of offering a remote install that will be rejected. Open the backend through `localhost` on that computer, or follow the engine's setup guide there. In Docker, bridge-network requests may not be loopback even when the browser runs on the host; use the documented container setup rather than weakening the native-install gate.
|
||||
|
||||
### Dubbing extraction fails
|
||||
|
||||
Extraction errors show the FFmpeg exit code and the end of its diagnostics, with private paths scrubbed. Use the final error line to distinguish missing audio streams, unsupported inputs, permissions, or disk errors. A version banner alone does not identify the cause; include the final diagnostic and source format when reporting a failure.
|
||||
|
||||
Explicit generation budgets remain authoritative. If an outer TTS/ASR guard times out or its caller disconnects, the active sidecar receive kills and reaps its captured child; it cannot terminate a later retry. In-process inference keeps its existing lifetime accounting until the native call returns.
|
||||
|
||||
### Voice conversion requires a cloning model
|
||||
|
||||
In Electron, voice conversion stays disabled until the active text-to-speech
|
||||
model is ready and supports voice cloning. Use the Models link to choose one;
|
||||
the source recording and target voice are preserved when returning. Preset-only
|
||||
models such as MLX Kokoro cannot clone a target voice. This capability check does
|
||||
not download or load model weights.
|
||||
|
||||
@@ -48,7 +48,7 @@ Everything above, plus the toolchain:
|
||||
with the **"Desktop development with C++"** workload checked.
|
||||
- **Bun** — `powershell -c "irm bun.sh/install.ps1 | iex"`.
|
||||
- **FFmpeg** — `winget install Gyan.FFmpeg`.
|
||||
- **Rust / Cargo** — `winget install Rust.Rustup` or download `rustup-init.exe` from [rustup.rs](https://rustup.rs/).
|
||||
- **Rust / Cargo** — `winget install Rustlang.Rustup` or download `rustup-init.exe` from [rustup.rs](https://rustup.rs/).
|
||||
After installing Rustup, close and reopen PowerShell before running `bun run tauri:desktop-prod`.
|
||||
|
||||
## GPU support on Windows
|
||||
@@ -350,3 +350,6 @@ This test-host preparation does not change installer privileges or user machines
|
||||
Verbose MSI logs are printed if installation or removal fails. The Windows CI
|
||||
job also rejects an invalid MSI and verifies policy absence, value types, account
|
||||
cleanup, and verbose failure logs using Windows PowerShell 5.1.
|
||||
|
||||
|
||||
Migration configuration is kept ASCII so Alembic can read it under Windows locale code pages as well as UTF-8. This applies to source installs and direct Alembic commands.
|
||||
|
||||
@@ -2,6 +2,10 @@
|
||||
import { afterEach, expect, it, vi } from 'vitest';
|
||||
import { EventEmitter } from 'node:events';
|
||||
const mocks = vi.hoisted(() => ({
|
||||
runtimeConfig: null as { root: string; owned: boolean } | null,
|
||||
existingProject: false,
|
||||
existingRoot: false,
|
||||
dependencies: vi.fn(async (_project?: string) => true),
|
||||
ready: vi.fn(async () => false),
|
||||
compatible: vi.fn(async () => false),
|
||||
interrupted: vi.fn(async () => false),
|
||||
@@ -13,8 +17,34 @@ const mocks = vi.hoisted(() => ({
|
||||
}));
|
||||
vi.mock('electron', () => ({ app: { isPackaged: true, getPath: () => '/private/voicestudio' } }));
|
||||
vi.mock('node:child_process', () => ({ spawn: mocks.spawn, spawnSync: vi.fn() }));
|
||||
vi.mock('node:fs', async (importOriginal) => {
|
||||
const original = await importOriginal<typeof import('node:fs')>();
|
||||
return {
|
||||
...original,
|
||||
readFileSync: (...args: Parameters<typeof original.readFileSync>) =>
|
||||
String(args[0]).endsWith('runtime-location.json') && mocks.runtimeConfig
|
||||
? JSON.stringify(mocks.runtimeConfig)
|
||||
: original.readFileSync(...args),
|
||||
writeFileSync: (...args: Parameters<typeof original.writeFileSync>) => {
|
||||
if (String(args[0]).endsWith('runtime-location.json')) {
|
||||
mocks.runtimeConfig = JSON.parse(String(args[1]));
|
||||
return;
|
||||
}
|
||||
return original.writeFileSync(...args);
|
||||
},
|
||||
mkdirSync: (...args: Parameters<typeof original.mkdirSync>) =>
|
||||
String(args[0]).includes('private') ? undefined : original.mkdirSync(...args),
|
||||
existsSync: (path: Parameters<typeof original.existsSync>[0]) =>
|
||||
String(path).includes('selected')
|
||||
? String(path).endsWith('project')
|
||||
? mocks.existingProject
|
||||
: mocks.existingRoot
|
||||
: original.existsSync(path),
|
||||
};
|
||||
});
|
||||
vi.mock('node:fs/promises', () => ({ rm: mocks.rm }));
|
||||
vi.mock('./runtime-project', () => ({
|
||||
runtimeDependenciesReady: mocks.dependencies,
|
||||
runtimeReady: mocks.ready,
|
||||
runtimeCompatible: mocks.compatible,
|
||||
runtimeInstallInterrupted: mocks.interrupted,
|
||||
@@ -34,6 +64,12 @@ afterEach(() => {
|
||||
vi.unstubAllGlobals();
|
||||
vi.unstubAllEnvs();
|
||||
vi.clearAllMocks();
|
||||
mocks.runtimeConfig = null;
|
||||
mocks.existingProject = false;
|
||||
mocks.existingRoot = false;
|
||||
mocks.dependencies.mockResolvedValue(true);
|
||||
mocks.ready.mockResolvedValue(false);
|
||||
mocks.compatible.mockResolvedValue(false);
|
||||
mocks.interrupted.mockResolvedValue(false);
|
||||
mocks.promoteCaches.mockResolvedValue(undefined);
|
||||
});
|
||||
@@ -364,3 +400,110 @@ it('reserves setup before asynchronous checks and cancels stale preflight', asyn
|
||||
expect(mocks.install).not.toHaveBeenCalled();
|
||||
expect(supervisor.status.stage).toBe('idle');
|
||||
});
|
||||
|
||||
it.each(['ready', 'compatible'] as const)(
|
||||
'offers setup instead of spawning a %s runtime with missing dependencies',
|
||||
async (kind) => {
|
||||
vi.stubGlobal(
|
||||
'fetch',
|
||||
vi.fn(async () => {
|
||||
throw new Error('no backend');
|
||||
}),
|
||||
);
|
||||
vi.stubEnv('OMNIVOICE_BACKEND_CMD', '');
|
||||
vi.stubEnv('VOICESTUDIO_SKIP_BACKEND', '');
|
||||
mocks[kind].mockResolvedValue(true);
|
||||
mocks.dependencies.mockResolvedValue(false);
|
||||
const supervisor = new BackendSupervisor();
|
||||
await supervisor.start();
|
||||
expect(supervisor.status.stage).toBe('setup_required');
|
||||
expect(mocks.spawn).not.toHaveBeenCalled();
|
||||
expect(mocks.install).not.toHaveBeenCalled();
|
||||
expect(mocks.stage).not.toHaveBeenCalled();
|
||||
await supervisor.shutdown();
|
||||
},
|
||||
);
|
||||
|
||||
it.each([true, false])(
|
||||
'preserves a selected unowned runtime (project exists: %s)',
|
||||
async (projectExists) => {
|
||||
const { resolve, join } = await import('node:path');
|
||||
const selected = resolve('/selected/VoiceStudio');
|
||||
mocks.runtimeConfig = { root: selected, owned: false };
|
||||
mocks.existingProject = projectExists;
|
||||
mocks.existingRoot = true;
|
||||
mocks.ready.mockResolvedValue(true);
|
||||
mocks.dependencies.mockResolvedValue(false);
|
||||
mocks.install.mockRejectedValue(new Error('offline'));
|
||||
vi.stubGlobal(
|
||||
'fetch',
|
||||
vi.fn(async () => {
|
||||
throw new Error('no backend');
|
||||
}),
|
||||
);
|
||||
vi.stubEnv('OMNIVOICE_BACKEND_CMD', '');
|
||||
vi.stubEnv('VOICESTUDIO_SKIP_BACKEND', '');
|
||||
const supervisor = new BackendSupervisor();
|
||||
await supervisor.start();
|
||||
expect(supervisor.status.stage).toBe('setup_required');
|
||||
expect(
|
||||
mocks.dependencies.mock.calls.every(([project]) => String(project).includes('selected')),
|
||||
).toBe(true);
|
||||
expect(mocks.install).not.toHaveBeenCalled();
|
||||
await supervisor.setupRuntime();
|
||||
expect(mocks.install.mock.calls[0][1]).toBe(join('/private/voicestudio', 'runtime', 'project'));
|
||||
expect(mocks.runtimeConfig?.root).not.toBe(selected);
|
||||
expect(mocks.rm).not.toHaveBeenCalled();
|
||||
expect(mocks.stage).not.toHaveBeenCalled();
|
||||
await supervisor.shutdown();
|
||||
},
|
||||
);
|
||||
|
||||
it('installs into a newly selected custom destination that does not yet exist', async () => {
|
||||
const { resolve, join } = await import('node:path');
|
||||
const selected = resolve('/selected/VoiceStudio');
|
||||
mocks.runtimeConfig = { root: selected, owned: false };
|
||||
mocks.install.mockRejectedValue(new Error('offline'));
|
||||
vi.stubGlobal(
|
||||
'fetch',
|
||||
vi.fn(async () => {
|
||||
throw new Error('no backend');
|
||||
}),
|
||||
);
|
||||
vi.stubEnv('OMNIVOICE_BACKEND_CMD', '');
|
||||
vi.stubEnv('VOICESTUDIO_SKIP_BACKEND', '');
|
||||
const supervisor = new BackendSupervisor();
|
||||
await supervisor.start();
|
||||
await supervisor.setupRuntime();
|
||||
expect(mocks.install.mock.calls[0][1]).toBe(join(selected, 'project'));
|
||||
expect(mocks.runtimeConfig).toEqual({ root: selected, owned: true });
|
||||
await supervisor.shutdown();
|
||||
});
|
||||
|
||||
it('reuses a healthy default runtime after explicit setup leaves an unowned environment', async () => {
|
||||
const { resolve, join } = await import('node:path');
|
||||
mocks.runtimeConfig = { root: resolve('/selected/VoiceStudio'), owned: false };
|
||||
mocks.existingRoot = true;
|
||||
mocks.ready.mockResolvedValue(true);
|
||||
mocks.dependencies.mockImplementation(async (project?: string) => !project?.includes('selected'));
|
||||
vi.stubGlobal(
|
||||
'fetch',
|
||||
vi.fn(async () => {
|
||||
throw new Error('no backend');
|
||||
}),
|
||||
);
|
||||
vi.stubEnv('OMNIVOICE_BACKEND_CMD', '');
|
||||
vi.stubEnv('VOICESTUDIO_SKIP_BACKEND', '');
|
||||
const supervisor = new BackendSupervisor();
|
||||
await supervisor.start();
|
||||
expect(supervisor.status.stage).toBe('setup_required');
|
||||
const restart = vi.spyOn(supervisor, 'start').mockResolvedValue();
|
||||
await supervisor.setupRuntime();
|
||||
expect(mocks.install).not.toHaveBeenCalled();
|
||||
expect(mocks.dependencies).toHaveBeenCalledWith(
|
||||
join('/private/voicestudio', 'runtime', 'project'),
|
||||
);
|
||||
expect(restart).toHaveBeenCalledOnce();
|
||||
restart.mockRestore();
|
||||
await supervisor.shutdown();
|
||||
});
|
||||
|
||||
@@ -2,6 +2,7 @@ import {
|
||||
installRuntime,
|
||||
promoteLegacyRuntimeCaches,
|
||||
runtimeCompatible,
|
||||
runtimeDependenciesReady,
|
||||
runtimeInstallInterrupted,
|
||||
runtimeReady,
|
||||
runtimePython,
|
||||
@@ -482,11 +483,8 @@ export class BackendSupervisor extends EventEmitter<{
|
||||
}
|
||||
|
||||
if (app.isPackaged && !parseBackendCmdOverride(process.env.OMNIVOICE_BACKEND_CMD)) {
|
||||
const project = await this.resolveRuntimeProject();
|
||||
if (
|
||||
!(await runtimeReady(backendRoot(), project)) &&
|
||||
!(await runtimeCompatible(backendRoot(), project))
|
||||
) {
|
||||
const { project, ready } = await this.resolveRuntimeProject();
|
||||
if (!ready) {
|
||||
if (gen === this.generation) {
|
||||
this.runtimeInterrupted = await runtimeInstallInterrupted(project);
|
||||
this.setStage('setup_required');
|
||||
@@ -522,7 +520,7 @@ export class BackendSupervisor extends EventEmitter<{
|
||||
this.stage !== 'setup_required'
|
||||
)
|
||||
return;
|
||||
const project =
|
||||
let project =
|
||||
this.runtimeProject ?? join(storedRuntimeRoot() ?? defaultRuntimeRoot(), 'project');
|
||||
this.runtimeProject = project;
|
||||
const controller = new AbortController();
|
||||
@@ -537,15 +535,44 @@ export class BackendSupervisor extends EventEmitter<{
|
||||
this.setStage('installing', { message: undefined });
|
||||
try {
|
||||
const reusable =
|
||||
(await runtimeReady(backendRoot(), project)) ||
|
||||
(await runtimeCompatible(backendRoot(), project));
|
||||
((await runtimeReady(backendRoot(), project)) ||
|
||||
(await runtimeCompatible(backendRoot(), project))) &&
|
||||
(await runtimeDependenciesReady(project));
|
||||
if (gen !== this.generation || controller.signal.aborted) return;
|
||||
if (reusable) {
|
||||
await this.start();
|
||||
return;
|
||||
}
|
||||
const runtimeRoot = dirname(project);
|
||||
let runtimeRoot = dirname(project);
|
||||
const configured = storedRuntimeLocation();
|
||||
if (
|
||||
configured &&
|
||||
!configured.owned &&
|
||||
samePath(configured.root, runtimeRoot) &&
|
||||
!samePath(runtimeRoot, defaultRuntimeRoot()) &&
|
||||
existsSync(runtimeRoot)
|
||||
) {
|
||||
// An explicit setup action may create a new runtime, but must never
|
||||
// take ownership of (or repair in place) another installation's files.
|
||||
runtimeRoot = defaultRuntimeRoot();
|
||||
project = join(runtimeRoot, 'project');
|
||||
this.runtimeProject = project;
|
||||
writeRuntimeLocation(runtimeRoot, true);
|
||||
this.pushLog(
|
||||
'out',
|
||||
'Creating a separate Electron runtime; existing environment preserved.',
|
||||
);
|
||||
this.emitStatus();
|
||||
const fallbackReusable =
|
||||
((await runtimeReady(backendRoot(), project)) ||
|
||||
(await runtimeCompatible(backendRoot(), project))) &&
|
||||
(await runtimeDependenciesReady(project));
|
||||
if (gen !== this.generation || controller.signal.aborted) return;
|
||||
if (fallbackReusable) {
|
||||
await this.start();
|
||||
return;
|
||||
}
|
||||
}
|
||||
if (
|
||||
configured &&
|
||||
samePath(configured.root, runtimeRoot) &&
|
||||
@@ -797,27 +824,28 @@ export class BackendSupervisor extends EventEmitter<{
|
||||
this.emitStatus();
|
||||
}
|
||||
|
||||
private async resolveRuntimeProject(): Promise<string> {
|
||||
private async resolveRuntimeProject(): Promise<{ project: string; ready: boolean }> {
|
||||
const bundle = backendRoot();
|
||||
const own = join(defaultRuntimeRoot(), 'project');
|
||||
const configuredRoot = storedRuntimeRoot();
|
||||
const configured = configuredRoot ? join(configuredRoot, 'project') : null;
|
||||
const candidates = [
|
||||
this.runtimeProject,
|
||||
configured,
|
||||
own,
|
||||
...legacyTauriRuntimeProjects(),
|
||||
].filter((candidate): candidate is string => Boolean(candidate));
|
||||
// Explicit selection is authoritative, including when it needs setup.
|
||||
const candidates = (
|
||||
configured ? [configured] : [this.runtimeProject, own, ...legacyTauriRuntimeProjects()]
|
||||
).filter((candidate): candidate is string => Boolean(candidate));
|
||||
for (const project of new Set(candidates.map((candidate) => resolve(candidate)))) {
|
||||
if ((await runtimeReady(bundle, project)) || (await runtimeCompatible(bundle, project))) {
|
||||
if (
|
||||
((await runtimeReady(bundle, project)) || (await runtimeCompatible(bundle, project))) &&
|
||||
(await runtimeDependenciesReady(project))
|
||||
) {
|
||||
this.runtimeProject = project;
|
||||
if (project !== own && project !== configured)
|
||||
this.pushLog('out', `Reusing compatible Tauri runtime: ${project}`);
|
||||
return project;
|
||||
return { project, ready: true };
|
||||
}
|
||||
}
|
||||
this.runtimeProject = configured ?? own;
|
||||
return this.runtimeProject;
|
||||
return { project: this.runtimeProject, ready: false };
|
||||
}
|
||||
|
||||
private emitStatus(): void {
|
||||
@@ -889,7 +917,10 @@ export class BackendSupervisor extends EventEmitter<{
|
||||
// Python passes this child-side descriptor to every nested operation.
|
||||
// Reading the parent side keeps the ownership channel live and lets
|
||||
// Node observe EOF only after the complete backend subtree releases it.
|
||||
const drain = child.stdio?.[processOptions.drainFd] as NodeJS.ReadableStream | null | undefined;
|
||||
const drain = child.stdio?.[processOptions.drainFd] as
|
||||
| NodeJS.ReadableStream
|
||||
| null
|
||||
| undefined;
|
||||
drain?.on('error', (error: unknown) => {
|
||||
if (!isExpectedPipeClose(error)) {
|
||||
this.pushLog('err', `Backend drain stream failed: ${errorMessage(error)}`);
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
// @vitest-environment node
|
||||
import { afterEach, expect, it, vi } from 'vitest';
|
||||
import { execFile } from 'node:child_process';
|
||||
import { runtimeDependenciesReady, runtimePython } from './runtime-project';
|
||||
vi.mock('node:child_process', () => ({ execFile: vi.fn() }));
|
||||
afterEach(() => {
|
||||
vi.clearAllMocks();
|
||||
vi.unstubAllEnvs();
|
||||
});
|
||||
it.each([null, new Error('No module named uvicorn'), new Error('ETIMEDOUT'), new Error('ENOENT')])(
|
||||
'validates imports using the selected interpreter and fails closed (%s)',
|
||||
async (error) => {
|
||||
vi.mocked(execFile).mockImplementation(((
|
||||
_command: unknown,
|
||||
_args: unknown,
|
||||
_options: unknown,
|
||||
callback: (error: Error | null) => void,
|
||||
) => callback(error)) as never);
|
||||
const project = '/runtime with spaces';
|
||||
expect(await runtimeDependenciesReady(project)).toBe(error === null);
|
||||
expect(execFile).toHaveBeenCalledWith(
|
||||
runtimePython(project),
|
||||
['-c', 'import fastapi, uvicorn, omnivoice, faster_whisper'],
|
||||
expect.objectContaining({
|
||||
cwd: project,
|
||||
timeout: 30_000,
|
||||
windowsHide: true,
|
||||
env: expect.objectContaining({
|
||||
HF_HUB_OFFLINE: '1',
|
||||
TRANSFORMERS_OFFLINE: '1',
|
||||
PYTHONNOUSERSITE: '1',
|
||||
}),
|
||||
}),
|
||||
expect.any(Function),
|
||||
);
|
||||
},
|
||||
);
|
||||
|
||||
it.each(['PYTHONPATH', 'PYTHONHOME'] as const)(
|
||||
'isolates imports from inherited %s',
|
||||
async (variable) => {
|
||||
vi.stubEnv(variable, '/unrelated-python');
|
||||
vi.mocked(execFile).mockImplementation(((
|
||||
_command: unknown,
|
||||
_args: unknown,
|
||||
options: { env: NodeJS.ProcessEnv },
|
||||
callback: (error: Error | null) => void,
|
||||
) => {
|
||||
const contaminated = Boolean(options.env[variable]);
|
||||
callback(
|
||||
variable === 'PYTHONPATH'
|
||||
? contaminated
|
||||
? null
|
||||
: new Error('No module named uvicorn')
|
||||
: contaminated
|
||||
? new Error('invalid Python home')
|
||||
: null,
|
||||
);
|
||||
}) as never);
|
||||
expect(await runtimeDependenciesReady('/selected-runtime')).toBe(variable === 'PYTHONHOME');
|
||||
expect(process.env[variable]).toBe('/unrelated-python');
|
||||
},
|
||||
);
|
||||
@@ -1,3 +1,4 @@
|
||||
import { execFile } from 'node:child_process';
|
||||
import { createHash, randomUUID } from 'node:crypto';
|
||||
import {
|
||||
cp,
|
||||
@@ -247,6 +248,30 @@ export async function runtimeInstallInterrupted(project: string): Promise<boolea
|
||||
}
|
||||
}
|
||||
|
||||
/** Check the selected interpreter locally before trusting runtime metadata. */
|
||||
export async function runtimeDependenciesReady(project: string): Promise<boolean> {
|
||||
const env: NodeJS.ProcessEnv = { ...process.env };
|
||||
delete env.PYTHONHOME;
|
||||
delete env.PYTHONPATH;
|
||||
env.HF_HUB_OFFLINE = '1';
|
||||
env.TRANSFORMERS_OFFLINE = '1';
|
||||
env.PYTHONNOUSERSITE = '1';
|
||||
return new Promise((resolve) => {
|
||||
execFile(
|
||||
runtimePython(project),
|
||||
['-c', 'import fastapi, uvicorn, omnivoice, faster_whisper'],
|
||||
{
|
||||
cwd: project,
|
||||
windowsHide: true,
|
||||
timeout: 30_000,
|
||||
maxBuffer: 256 * 1024,
|
||||
env,
|
||||
},
|
||||
(error) => resolve(!error),
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
export async function runtimeReady(bundle: string, project: string): Promise<boolean> {
|
||||
if (await runtimeIncomplete(project)) return false;
|
||||
try {
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { readTextFile } from '../../../../../../frontend/src/utils/readTextFile';
|
||||
import { useEffect, useMemo, useRef, useState } from 'react';
|
||||
import { AlertTriangleIcon, ClipboardPasteIcon, FileTextIcon, XIcon } from 'lucide-react';
|
||||
import { useTranslation } from 'react-i18next';
|
||||
@@ -73,8 +74,7 @@ export function PasteTranslation({ segments, disabled, onApply, onClose }: Paste
|
||||
const readFile = (file?: File) => {
|
||||
if (!file) return;
|
||||
setReadFailed(false);
|
||||
void file
|
||||
.text()
|
||||
void readTextFile(file)
|
||||
.then(setText)
|
||||
.catch(() => setReadFailed(true));
|
||||
};
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { readTextFile } from '../../../../../../frontend/src/utils/readTextFile';
|
||||
import { ProfileAvatar } from '@/components/profile-avatar';
|
||||
import {
|
||||
BookOpenTextIcon,
|
||||
@@ -119,7 +120,7 @@ export function LongformPage({ mode }: { mode: Mode }) {
|
||||
try {
|
||||
let text: string;
|
||||
if (mode === 'stories' && /\.(txt|md|srt)$/i.test(file.name))
|
||||
text = importToText(file.name, await file.text());
|
||||
text = importToText(file.name, await readTextFile(file));
|
||||
else {
|
||||
const body = new FormData();
|
||||
body.set('file', file);
|
||||
|
||||
@@ -16,3 +16,14 @@ it('splits locally with bounded chunks and stable text order', () => {
|
||||
expect(chunks.join(' ')).toBe(text);
|
||||
expect(splitIntoChunks('', 500)).toEqual([]);
|
||||
});
|
||||
|
||||
it('decodes a UTF-16 manuscript before extracting subtitle speech', async () => {
|
||||
const { readTextFile } = await import('../../../../../../frontend/src/utils/readTextFile');
|
||||
const text = '1\n00:00:01,000 --> 00:00:02,000\nCafé — hello';
|
||||
const bytes = new Uint8Array(2 + text.length * 2);
|
||||
bytes.set([0xff, 0xfe]);
|
||||
const view = new DataView(bytes.buffer);
|
||||
for (let i = 0; i < text.length; i++) view.setUint16(2 + i * 2, text.charCodeAt(i), true);
|
||||
const file = { arrayBuffer: async () => bytes.buffer };
|
||||
expect(importToText('captions.srt', await readTextFile(file))).toBe('Café — hello');
|
||||
});
|
||||
|
||||
@@ -18,6 +18,7 @@ export function EngineInstall({ id }: { id: string }) {
|
||||
queryFn: () =>
|
||||
apiJson<{
|
||||
installed: boolean;
|
||||
install_allowed?: boolean;
|
||||
job: null | {
|
||||
state: string;
|
||||
steps: { name?: string; state: string }[];
|
||||
@@ -36,8 +37,15 @@ export function EngineInstall({ id }: { id: string }) {
|
||||
<Button
|
||||
size="sm"
|
||||
variant="outline"
|
||||
disabled={running || status.data?.installed}
|
||||
disabled={
|
||||
running ||
|
||||
status.isPending ||
|
||||
status.isError ||
|
||||
status.data?.installed ||
|
||||
status.data?.install_allowed === false
|
||||
}
|
||||
onClick={async () => {
|
||||
if (status.data?.install_allowed === false) return;
|
||||
setStarting(true);
|
||||
setFailed(null);
|
||||
setDismissedFailure(false);
|
||||
@@ -54,6 +62,11 @@ export function EngineInstall({ id }: { id: string }) {
|
||||
<DownloadIcon />
|
||||
{t(running ? 'modelMaintenance.installing' : 'modelMaintenance.install')}
|
||||
</Button>
|
||||
{status.data?.install_allowed === false && (
|
||||
<p className="max-w-sm text-xs text-muted-foreground">
|
||||
{t('engines.localInstallRequired')}
|
||||
</p>
|
||||
)}
|
||||
{!dismissedFailure && (failed || status.isError || status.data?.job?.state === 'failed') && (
|
||||
<SettingsActionError
|
||||
className="max-w-sm text-left"
|
||||
|
||||
@@ -107,7 +107,41 @@ it('confirms a diarisation engine selection after the backend accepts it', async
|
||||
);
|
||||
|
||||
fireEvent.click(await screen.findByRole('button', { name: 'modelSettings.select' }));
|
||||
await waitFor(() =>
|
||||
expect(mock.toast.success).toHaveBeenCalledWith('settings.engine_switched'),
|
||||
await waitFor(() => expect(mock.toast.success).toHaveBeenCalledWith('settings.engine_switched'));
|
||||
});
|
||||
|
||||
it('explains remote native installation restrictions before a POST', async () => {
|
||||
mock.api.mockImplementation((path: string) =>
|
||||
Promise.resolve(
|
||||
path === '/engines/diarisation'
|
||||
? {
|
||||
active: 'pyannote',
|
||||
options: [
|
||||
{
|
||||
id: 'audiocpp-sortformer',
|
||||
label: 'Sortformer',
|
||||
model_installed: true,
|
||||
runtime_installed: false,
|
||||
installed: false,
|
||||
},
|
||||
],
|
||||
}
|
||||
: {
|
||||
installed: false,
|
||||
supported: true,
|
||||
install_allowed: false,
|
||||
job: { state: 'idle', progress: 0 },
|
||||
},
|
||||
),
|
||||
);
|
||||
render(
|
||||
<QueryClientProvider client={new QueryClient()}>
|
||||
<DiarisationSettings />
|
||||
</QueryClientProvider>,
|
||||
);
|
||||
expect(await screen.findByText('engines.localInstallRequired')).toBeInTheDocument();
|
||||
const button = screen.getByRole('button', { name: 'modelMaintenance.install' });
|
||||
expect(button).toBeDisabled();
|
||||
fireEvent.click(button);
|
||||
expect(mock.api.mock.calls.some(([, init]) => init?.method === 'POST')).toBe(false);
|
||||
});
|
||||
|
||||
@@ -61,6 +61,7 @@ export function DiarisationSettings() {
|
||||
apiJson<{
|
||||
installed: boolean;
|
||||
supported: boolean;
|
||||
install_allowed?: boolean;
|
||||
job: { state: string; progress: number; error?: string | null };
|
||||
}>('/engines/audiocpp/runtime/install/status'),
|
||||
refetchInterval: (state) => (state.state.data?.job.state === 'running' ? 1_500 : 10_000),
|
||||
@@ -75,6 +76,7 @@ export function DiarisationSettings() {
|
||||
}, [client, runtime.data?.installed]);
|
||||
const runtimeRunning = busy || runtime.data?.job.state === 'running';
|
||||
const installRuntime = async () => {
|
||||
if (runtime.data?.install_allowed === false) return;
|
||||
setBusy(true);
|
||||
setFailed(null);
|
||||
try {
|
||||
@@ -152,7 +154,12 @@ export function DiarisationSettings() {
|
||||
<Button
|
||||
size="sm"
|
||||
variant="outline"
|
||||
disabled={runtimeRunning}
|
||||
disabled={runtimeRunning || runtime.data?.install_allowed === false}
|
||||
title={
|
||||
runtime.data?.install_allowed === false
|
||||
? t('engines.localInstallRequired')
|
||||
: undefined
|
||||
}
|
||||
onClick={() => void installRuntime()}
|
||||
>
|
||||
<DownloadIcon />
|
||||
@@ -163,6 +170,13 @@ export function DiarisationSettings() {
|
||||
: t('modelMaintenance.install')}
|
||||
</Button>
|
||||
)}
|
||||
{option.id === 'audiocpp-sortformer' &&
|
||||
!option.runtime_installed &&
|
||||
runtime.data?.install_allowed === false && (
|
||||
<p className="max-w-sm text-xs text-muted-foreground">
|
||||
{t('engines.localInstallRequired')}
|
||||
</p>
|
||||
)}
|
||||
{option.installed && (
|
||||
<Button
|
||||
size="sm"
|
||||
@@ -380,11 +394,13 @@ export function ModelSettings({
|
||||
name: engine.display_name,
|
||||
available: engine.available,
|
||||
detail:
|
||||
engine.id === selected
|
||||
? engineState.active_model
|
||||
: !engine.available
|
||||
? engine.install_hint || engine.reason || engine.hint || undefined
|
||||
: undefined,
|
||||
engine.local_install_required && !engine.available
|
||||
? t('engines.localInstallRequired')
|
||||
: engine.id === selected
|
||||
? engineState.active_model
|
||||
: !engine.available
|
||||
? engine.install_hint || engine.reason || engine.hint || undefined
|
||||
: undefined,
|
||||
models: engine.curated_models,
|
||||
installable: engine.one_click_install,
|
||||
setupSnippet: !engine.available ? engine.setup_snippet || undefined : undefined,
|
||||
|
||||
@@ -2,7 +2,26 @@ import { clearConversion } from './conversion-state';
|
||||
import { cleanup, fireEvent, render, screen, act } from '@testing-library/react';
|
||||
import { QueryClient, QueryClientProvider } from '@tanstack/react-query';
|
||||
import { afterEach, expect, it, vi } from 'vitest';
|
||||
const mock = vi.hoisted(() => ({ convert: vi.fn() }));
|
||||
const mock = vi.hoisted(() => ({
|
||||
queryError: false,
|
||||
retry: vi.fn(),
|
||||
convert: vi.fn(),
|
||||
ready: true,
|
||||
cloning: true as boolean | null,
|
||||
}));
|
||||
vi.mock('@/hooks/use-engines', () => ({
|
||||
useEngines: () => ({
|
||||
isError: mock.queryError,
|
||||
error: new Error('Engine query failed'),
|
||||
retry: mock.retry,
|
||||
data: mock.queryError ? undefined : {},
|
||||
activeTtsReady: mock.ready,
|
||||
activeTts: { supports_cloning: mock.cloning },
|
||||
}),
|
||||
}));
|
||||
vi.mock('@tanstack/react-router', () => ({
|
||||
Link: ({ children }: { children: React.ReactNode }) => <a>{children}</a>,
|
||||
}));
|
||||
vi.mock('@/lib/api/convert', () => ({ convertSpeech: mock.convert }));
|
||||
vi.mock('@/hooks/use-recording', () => ({
|
||||
useRecording: () => ({ isRecording: false, isStarting: false, isCleaning: false }),
|
||||
@@ -22,6 +41,9 @@ import { ConvertVoice } from './convert-voice';
|
||||
afterEach(() => {
|
||||
cleanup();
|
||||
clearConversion();
|
||||
mock.queryError = false;
|
||||
mock.ready = true;
|
||||
mock.cloning = true;
|
||||
vi.clearAllMocks();
|
||||
});
|
||||
function mount() {
|
||||
@@ -79,3 +101,50 @@ it('retains source and target when returning from model settings', () => {
|
||||
expect(screen.getByRole('button', { name: 'source.wav' })).toBeInTheDocument();
|
||||
expect(screen.getByRole('button', { name: 'convert.convert' })).toBeEnabled();
|
||||
});
|
||||
|
||||
it.each([false, null])(
|
||||
'blocks conversion for unsupported or unknown cloning capability (%s)',
|
||||
(capability) => {
|
||||
mock.cloning = capability;
|
||||
const { upload } = mount();
|
||||
upload();
|
||||
fireEvent.click(screen.getByRole('button', { name: 'Alpha' }));
|
||||
const button = screen.getByRole('button', { name: 'convert.convert' });
|
||||
expect(button).toBeDisabled();
|
||||
expect(screen.getByText('convert.cloning_required')).toBeInTheDocument();
|
||||
fireEvent.click(button);
|
||||
expect(mock.convert).not.toHaveBeenCalled();
|
||||
},
|
||||
);
|
||||
it('rechecks engine capability when returning from settings without losing inputs', () => {
|
||||
const first = mount();
|
||||
first.upload();
|
||||
fireEvent.click(screen.getByRole('button', { name: 'Alpha' }));
|
||||
first.unmount();
|
||||
mock.cloning = false;
|
||||
const second = mount();
|
||||
expect(screen.getByRole('button', { name: 'convert.convert' })).toBeDisabled();
|
||||
second.unmount();
|
||||
mock.cloning = true;
|
||||
mock.ready = false;
|
||||
const third = mount();
|
||||
expect(screen.getByRole('button', { name: 'convert.convert' })).toBeDisabled();
|
||||
third.unmount();
|
||||
mock.ready = true;
|
||||
mount();
|
||||
expect(screen.getByRole('button', { name: 'convert.convert' })).toBeEnabled();
|
||||
expect(screen.getByRole('button', { name: 'source.wav' })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('offers retry instead of model guidance when engine lookup fails', () => {
|
||||
mock.queryError = true;
|
||||
mock.ready = false;
|
||||
const { upload } = mount();
|
||||
upload();
|
||||
fireEvent.click(screen.getByRole('button', { name: 'Alpha' }));
|
||||
expect(screen.getByRole('button', { name: 'convert.convert' })).toBeDisabled();
|
||||
expect(screen.queryByText('convert.cloning_required')).not.toBeInTheDocument();
|
||||
expect(screen.getByText('Engine query failed')).toBeInTheDocument();
|
||||
fireEvent.click(screen.getByRole('button', { name: 'common.retry' }));
|
||||
expect(mock.retry).toHaveBeenCalledOnce();
|
||||
});
|
||||
|
||||
@@ -11,6 +11,7 @@ import { PipelineFailure } from '@/components/pipeline-failure';
|
||||
import { AgentFixButton } from '@/components/agent-fix-button';
|
||||
import { ProfileAvatar } from '@/components/profile-avatar';
|
||||
import { WaveformPlayer } from '@/components/waveform-player';
|
||||
import { useEngines } from '@/hooks/use-engines';
|
||||
import { useProfiles } from '@/hooks/use-profiles';
|
||||
import { useRecording } from '@/hooks/use-recording';
|
||||
import { ApiError, apiPath, describeError } from '@/lib/api/client';
|
||||
@@ -21,6 +22,8 @@ import { beginAppActivity } from '@/lib/app-activity';
|
||||
export function ConvertVoice() {
|
||||
const { t } = useTranslation();
|
||||
const profiles = useProfiles();
|
||||
const engines = useEngines();
|
||||
const canClone = engines.activeTtsReady && engines.activeTts?.supports_cloning === true;
|
||||
const client = useQueryClient();
|
||||
const { file, voice, search, match, result } = useConversion();
|
||||
const setFile = (value: File | null) => setConversion('file', value);
|
||||
@@ -64,7 +67,7 @@ export function ConvertVoice() {
|
||||
}, [file]);
|
||||
useEffect(() => () => request.current?.abort(), []);
|
||||
const voices = (profiles.data ?? []).filter((p) => p.kind === 'clone' && p.ref_audio_path);
|
||||
const valid = file && voices.some((p) => p.id === voice) && !recordingBusy;
|
||||
const valid = canClone && file && voices.some((p) => p.id === voice) && !recordingBusy;
|
||||
const run = async () => {
|
||||
if (!valid || request.current) return;
|
||||
const controller = new AbortController();
|
||||
@@ -210,6 +213,32 @@ export function ConvertVoice() {
|
||||
</label>
|
||||
<p className="text-xs text-muted-foreground">{t('convert.match_duration_hint')}</p>
|
||||
</div>
|
||||
{engines.isError && !engines.data && !busy && (
|
||||
<div
|
||||
role="alert"
|
||||
className="flex flex-wrap items-center gap-2 text-sm text-muted-foreground"
|
||||
>
|
||||
<span>{describeError(engines.error)}</span>
|
||||
<Button variant="outline" size="sm" onClick={engines.retry}>
|
||||
{t('common.retry')}
|
||||
</Button>
|
||||
</div>
|
||||
)}
|
||||
{!canClone && !(engines.isError && !engines.data) && !busy && (
|
||||
<div
|
||||
role="status"
|
||||
className="flex flex-wrap items-center gap-2 text-sm text-muted-foreground"
|
||||
>
|
||||
<span>{t('convert.cloning_required')}</span>
|
||||
<Link
|
||||
to="/settings/models/$family"
|
||||
params={{ family: 'tts' }}
|
||||
className="font-medium text-primary hover:underline"
|
||||
>
|
||||
{t('modelSettings.models')}
|
||||
</Link>
|
||||
</div>
|
||||
)}
|
||||
{error && (
|
||||
<PipelineFailure
|
||||
fallback={error}
|
||||
|
||||
@@ -2245,6 +2245,7 @@
|
||||
"tailscale_qr_alt": "رمز الاستجابة السريعة لعنوان URL لـ Tailscale"
|
||||
},
|
||||
"convert": {
|
||||
"cloning_required": "اختر نموذجًا جاهزًا لتحويل النص إلى كلام يدعم استنساخ الصوت.",
|
||||
"source_kicker": "المقطع المصدر",
|
||||
"drop_audio": "أسقط المقطع المراد تحويله — أو انقر. WAV، MP3، M4A…",
|
||||
"target_voice": "الصوت الهدف",
|
||||
@@ -2492,7 +2493,9 @@
|
||||
"none_ready_title": "تحويل النص إلى كلام غير جاهز",
|
||||
"none_ready_body": "يمكن لـ VoiceStudio إعداد أفضل خيار متوافق مع هذا الجهاز تلقائيًا.",
|
||||
"active": "المحرك: {{name}}",
|
||||
"none": "لا يوجد محرك نشط"
|
||||
"none": "لا يوجد محرك نشط",
|
||||
"localInstallRequired": "ثبّت هذا المحرك على جهاز الخادم باستخدام localhost أو دليل الإعداد. التثبيت عن بُعد معطّل.",
|
||||
"argosRuntimeUnavailable": "بيئة تشغيل Argos غير متاحة. أعد تثبيت الخادم الخلفي أو اختر NLLB."
|
||||
},
|
||||
"performanceProfile": {
|
||||
"requiresEngine": "ثبّت {{engine}} وحدده لضبط السرعة والجودة.",
|
||||
@@ -2527,6 +2530,7 @@
|
||||
"recorder_unsupported": "لا يستطيع هذا النظام ترميز صوت الميكروفون."
|
||||
},
|
||||
"tts_errors": {
|
||||
"profile_language_rejected": "يستخدم الوضع التلقائي لغة ملف الصوت ({{language}})، التي لا يدعمها هذا المحرك. غيّر لغة الملف أو اختر لغة مدعومة أو بدّل المحرك.",
|
||||
"generation_in_progress": "هناك عملية توليد قيد التشغيل بالفعل — انتظر حتى تنتهي قبل بدء عملية أخرى.",
|
||||
"error_prefix": "خطأ: {{message}}",
|
||||
"ignored_duplicate": "تم التجاهل (تم تعيين الفئة بالفعل): {{items}}",
|
||||
|
||||
@@ -2237,6 +2237,7 @@
|
||||
"tailscale_qr_alt": "QR-Code für die Tailscale-URL"
|
||||
},
|
||||
"convert": {
|
||||
"cloning_required": "Wähle ein einsatzbereites Sprachsynthesemodell, das Stimmenklonen unterstützt.",
|
||||
"source_kicker": "Quellclip",
|
||||
"drop_audio": "Clip zum Konvertieren hier ablegen — oder klicken. WAV, MP3, M4A…",
|
||||
"target_voice": "Zielstimme",
|
||||
@@ -2484,7 +2485,9 @@
|
||||
"none_ready_title": "Sprachausgabe ist nicht bereit",
|
||||
"none_ready_body": "VoiceStudio kann automatisch die beste kompatible Option für dieses Gerät einrichten.",
|
||||
"active": "Engine: {{name}}",
|
||||
"none": "Keine Engine aktiv"
|
||||
"none": "Keine Engine aktiv",
|
||||
"localInstallRequired": "Installiere diese Engine auf dem Backend-Rechner über localhost oder die Einrichtungsanleitung. Die Ferninstallation ist deaktiviert.",
|
||||
"argosRuntimeUnavailable": "Die Argos-Laufzeit ist nicht verfügbar. Installiere das Backend neu oder wähle NLLB."
|
||||
},
|
||||
"performanceProfile": {
|
||||
"requiresEngine": "Installiere und wähle {{engine}}, um Geschwindigkeit und Qualität anzupassen.",
|
||||
@@ -2519,6 +2522,7 @@
|
||||
"channels_stereo": "Stereo"
|
||||
},
|
||||
"tts_errors": {
|
||||
"profile_language_rejected": "Automatisch verwendet die Sprache des Stimmprofils ({{language}}), die diese Engine nicht unterstützt. Ändere die Profilsprache, wähle eine unterstützte Sprache oder wechsle die Engine.",
|
||||
"generation_in_progress": "Eine Generierung läuft bereits – warte, bis sie abgeschlossen ist, bevor du eine weitere startest.",
|
||||
"error_prefix": "Fehler: {{message}}",
|
||||
"ignored_duplicate": "Ignoriert (Kategorie bereits festgelegt): {{items}}",
|
||||
|
||||
@@ -114,7 +114,9 @@
|
||||
"recheck": "Re-check",
|
||||
"rechecking": "Re-checking…",
|
||||
"failed": "failed",
|
||||
"latencyMs": "{{ms}} ms"
|
||||
"latencyMs": "{{ms}} ms",
|
||||
"localInstallRequired": "Install this engine on the backend computer using localhost or its setup guide. Remote installation is disabled.",
|
||||
"argosRuntimeUnavailable": "Argos runtime unavailable. Reinstall the backend or select NLLB."
|
||||
},
|
||||
"clone": {
|
||||
"title": "Voice cloning",
|
||||
@@ -290,6 +292,7 @@
|
||||
"recorder_unsupported": "This system cannot encode microphone audio."
|
||||
},
|
||||
"tts_errors": {
|
||||
"profile_language_rejected": "Auto uses the voice profile’s language ({{language}}), which this engine does not support. Change the profile language, choose a supported language, or switch engines.",
|
||||
"enter_text": "Enter some text first.",
|
||||
"upload_or_select": "Upload a reference clip or pick a saved voice.",
|
||||
"trim_hint": "The clip is {{duration}}s — trim to ≤{{max}}s for best cloning.",
|
||||
@@ -2286,6 +2289,7 @@
|
||||
"tailscale_qr_alt": "QR code for the Tailscale URL"
|
||||
},
|
||||
"convert": {
|
||||
"cloning_required": "Choose a ready text-to-speech model that supports voice cloning.",
|
||||
"source_kicker": "Source clip",
|
||||
"drop_audio": "Drop the clip to convert — or click. WAV, MP3, M4A…",
|
||||
"target_voice": "Target voice",
|
||||
|
||||
@@ -2239,6 +2239,7 @@
|
||||
"tailscale_qr_alt": "Código QR para la URL de Tailscale"
|
||||
},
|
||||
"convert": {
|
||||
"cloning_required": "Elige un modelo de texto a voz listo para usar que permita clonar voces.",
|
||||
"source_kicker": "Clip de origen",
|
||||
"drop_audio": "Suelta el clip a convertir — o haz clic. WAV, MP3, M4A…",
|
||||
"target_voice": "Voz de destino",
|
||||
@@ -2486,7 +2487,9 @@
|
||||
"none_ready_title": "La síntesis de voz no está lista",
|
||||
"none_ready_body": "VoiceStudio puede configurar automáticamente la mejor opción compatible con este dispositivo.",
|
||||
"active": "Motor: {{name}}",
|
||||
"none": "Ningún motor activo"
|
||||
"none": "Ningún motor activo",
|
||||
"localInstallRequired": "Instala este motor en el equipo del servidor mediante localhost o su guía de configuración. La instalación remota está desactivada.",
|
||||
"argosRuntimeUnavailable": "El entorno de ejecución de Argos no está disponible. Reinstala el backend o selecciona NLLB."
|
||||
},
|
||||
"performanceProfile": {
|
||||
"requiresEngine": "Instala y selecciona {{engine}} para ajustar la velocidad y la calidad.",
|
||||
@@ -2521,6 +2524,7 @@
|
||||
"channels_mono": "Mono"
|
||||
},
|
||||
"tts_errors": {
|
||||
"profile_language_rejected": "Automático usa el idioma del perfil de voz ({{language}}), que este motor no admite. Cambia el idioma del perfil, elige uno compatible o cambia de motor.",
|
||||
"generation_in_progress": "Ya hay una generación en curso; espera a que termine antes de iniciar otra.",
|
||||
"ignored_duplicate": "Ignorado (categoría ya establecida): {{items}}",
|
||||
"ignored_conflict": "Se ignoró {{items}} — un dialecto chino y un acento inglés no se pueden combinar.",
|
||||
|
||||
@@ -2239,6 +2239,7 @@
|
||||
"tailscale_qr_alt": "Code QR pour l'URL Tailscale"
|
||||
},
|
||||
"convert": {
|
||||
"cloning_required": "Choisissez un modèle de synthèse vocale prêt à utiliser qui prend en charge le clonage vocal.",
|
||||
"source_kicker": "Clip source",
|
||||
"drop_audio": "Déposez le clip à convertir — ou cliquez. WAV, MP3, M4A…",
|
||||
"target_voice": "Voix cible",
|
||||
@@ -2486,7 +2487,9 @@
|
||||
"none_ready_title": "La synthèse vocale n’est pas prête",
|
||||
"none_ready_body": "VoiceStudio peut configurer automatiquement la meilleure option compatible avec cet appareil.",
|
||||
"active": "Moteur : {{name}}",
|
||||
"none": "Aucun moteur actif"
|
||||
"none": "Aucun moteur actif",
|
||||
"localInstallRequired": "Installez ce moteur sur l’ordinateur du serveur via localhost ou son guide de configuration. L’installation à distance est désactivée.",
|
||||
"argosRuntimeUnavailable": "Le moteur Argos est indisponible. Réinstallez le backend ou sélectionnez NLLB."
|
||||
},
|
||||
"performanceProfile": {
|
||||
"requiresEngine": "Installez et sélectionnez {{engine}} pour régler la vitesse et la qualité.",
|
||||
@@ -2521,6 +2524,7 @@
|
||||
"microphone_number": "Microphone {{number}}"
|
||||
},
|
||||
"tts_errors": {
|
||||
"profile_language_rejected": "Auto utilise la langue du profil vocal ({{language}}), non prise en charge par ce moteur. Modifiez la langue du profil, choisissez une langue compatible ou changez de moteur.",
|
||||
"generation_in_progress": "Une génération est déjà en cours — attendez sa fin avant d’en lancer une autre.",
|
||||
"error_prefix": "Erreur : {{message}}",
|
||||
"ignored_duplicate": "Ignoré (catégorie déjà définie) : {{items}}",
|
||||
|
||||
@@ -2237,6 +2237,7 @@
|
||||
"tailscale_qr_alt": "टेलस्केल यूआरएल के लिए क्यूआर कोड"
|
||||
},
|
||||
"convert": {
|
||||
"cloning_required": "वॉइस क्लोनिंग का समर्थन करने वाला तैयार टेक्स्ट-टू-स्पीच मॉडल चुनें।",
|
||||
"source_kicker": "स्रोत क्लिप",
|
||||
"drop_audio": "कन्वर्ट करने के लिए क्लिप यहाँ छोड़ें — या क्लिक करें। WAV, MP3, M4A…",
|
||||
"target_voice": "लक्ष्य आवाज़",
|
||||
@@ -2484,7 +2485,9 @@
|
||||
"none_ready_title": "वाक् संश्लेषण तैयार नहीं है",
|
||||
"none_ready_body": "VoiceStudio इस डिवाइस के लिए सबसे अच्छा संगत विकल्प अपने आप सेट कर सकता है।",
|
||||
"active": "इंजन: {{name}}",
|
||||
"none": "कोई इंजन सक्रिय नहीं है"
|
||||
"none": "कोई इंजन सक्रिय नहीं है",
|
||||
"localInstallRequired": "इस इंजन को बैकएंड कंप्यूटर पर localhost या सेटअप गाइड से इंस्टॉल करें। रिमोट इंस्टॉलेशन बंद है।",
|
||||
"argosRuntimeUnavailable": "Argos रनटाइम उपलब्ध नहीं है। बैकएंड फिर से इंस्टॉल करें या NLLB चुनें।"
|
||||
},
|
||||
"performanceProfile": {
|
||||
"requiresEngine": "गति और गुणवत्ता समायोजित करने के लिए {{engine}} इंस्टॉल करके चुनें।",
|
||||
@@ -2519,6 +2522,7 @@
|
||||
"recorder_unsupported": "यह सिस्टम माइक्रोफ़ोन ऑडियो एन्कोड नहीं कर सकता।"
|
||||
},
|
||||
"tts_errors": {
|
||||
"profile_language_rejected": "ऑटो वॉइस प्रोफ़ाइल की भाषा ({{language}}) का उपयोग करता है, जिसे यह इंजन समर्थन नहीं करता। प्रोफ़ाइल की भाषा बदलें, समर्थित भाषा चुनें या इंजन बदलें।",
|
||||
"generation_in_progress": "एक जनरेशन पहले से चल रहा है — दूसरा शुरू करने से पहले उसके पूरा होने की प्रतीक्षा करें।",
|
||||
"error_prefix": "त्रुटि: {{message}}",
|
||||
"ignored_duplicate": "अनदेखा (श्रेणी पहले से ही सेट): {{items}}",
|
||||
|
||||
@@ -2237,6 +2237,7 @@
|
||||
"tailscale_qr_alt": "Kode QR untuk URL Tailscale"
|
||||
},
|
||||
"convert": {
|
||||
"cloning_required": "Pilih model teks ke suara yang siap digunakan dan mendukung kloning suara.",
|
||||
"source_kicker": "Klip sumber",
|
||||
"drop_audio": "Letakkan klip yang akan dikonversi — atau klik. WAV, MP3, M4A…",
|
||||
"target_voice": "Suara target",
|
||||
@@ -2484,7 +2485,9 @@
|
||||
"none_ready_title": "Sintesis ucapan belum siap",
|
||||
"none_ready_body": "VoiceStudio dapat menyiapkan opsi kompatibel terbaik untuk perangkat ini secara otomatis.",
|
||||
"active": "Mesin: {{name}}",
|
||||
"none": "Tidak ada mesin aktif"
|
||||
"none": "Tidak ada mesin aktif",
|
||||
"localInstallRequired": "Instal mesin ini di komputer backend melalui localhost atau panduan penyiapannya. Instalasi jarak jauh dinonaktifkan.",
|
||||
"argosRuntimeUnavailable": "Runtime Argos tidak tersedia. Instal ulang backend atau pilih NLLB."
|
||||
},
|
||||
"performanceProfile": {
|
||||
"requiresEngine": "Instal dan pilih {{engine}} untuk menyesuaikan kecepatan dan kualitas.",
|
||||
@@ -2519,6 +2522,7 @@
|
||||
"channels_stereo": "Stereo"
|
||||
},
|
||||
"tts_errors": {
|
||||
"profile_language_rejected": "Otomatis menggunakan bahasa profil suara ({{language}}), yang tidak didukung mesin ini. Ubah bahasa profil, pilih bahasa yang didukung, atau ganti mesin.",
|
||||
"generation_in_progress": "Pembuatan sedang berjalan — tunggu hingga selesai sebelum memulai yang lain.",
|
||||
"error_prefix": "Kesalahan: {{message}}",
|
||||
"ignored_duplicate": "Diabaikan (kategori sudah ditetapkan): {{items}}",
|
||||
|
||||
@@ -2239,6 +2239,7 @@
|
||||
"tailscale_qr_alt": "Codice QR per l'URL Tailscale"
|
||||
},
|
||||
"convert": {
|
||||
"cloning_required": "Scegli un modello di sintesi vocale pronto che supporti la clonazione della voce.",
|
||||
"source_kicker": "Clip sorgente",
|
||||
"drop_audio": "Trascina qui il clip da convertire — o fai clic. WAV, MP3, M4A…",
|
||||
"target_voice": "Voce di destinazione",
|
||||
@@ -2486,7 +2487,9 @@
|
||||
"none_ready_title": "La sintesi vocale non è pronta",
|
||||
"none_ready_body": "VoiceStudio può configurare automaticamente l’opzione compatibile migliore per questo dispositivo.",
|
||||
"active": "Motore: {{name}}",
|
||||
"none": "Nessun motore attivo"
|
||||
"none": "Nessun motore attivo",
|
||||
"localInstallRequired": "Installa questo motore sul computer del backend tramite localhost o la guida alla configurazione. L’installazione remota è disabilitata.",
|
||||
"argosRuntimeUnavailable": "Il runtime di Argos non è disponibile. Reinstalla il backend o seleziona NLLB."
|
||||
},
|
||||
"performanceProfile": {
|
||||
"requiresEngine": "Installa e seleziona {{engine}} per regolare velocità e qualità.",
|
||||
@@ -2521,6 +2524,7 @@
|
||||
"channels_stereo": "Stereo"
|
||||
},
|
||||
"tts_errors": {
|
||||
"profile_language_rejected": "Auto usa la lingua del profilo vocale ({{language}}), non supportata da questo motore. Modifica la lingua del profilo, scegli una lingua supportata o cambia motore.",
|
||||
"generation_in_progress": "È già in corso una generazione: attendi che termini prima di avviarne un’altra.",
|
||||
"error_prefix": "Errore: {{message}}",
|
||||
"ignored_duplicate": "Ignorato (categoria già impostata): {{items}}",
|
||||
|
||||
@@ -2237,6 +2237,7 @@
|
||||
"tailscale_qr_alt": "Tailscale URL の QR コード"
|
||||
},
|
||||
"convert": {
|
||||
"cloning_required": "音声クローンに対応した、使用可能な音声合成モデルを選択してください。",
|
||||
"source_kicker": "ソースクリップ",
|
||||
"drop_audio": "変換するクリップをドロップ — またはクリック。WAV、MP3、M4A…",
|
||||
"target_voice": "変換先の声",
|
||||
@@ -2484,7 +2485,9 @@
|
||||
"none_ready_title": "音声合成の準備ができていません",
|
||||
"none_ready_body": "VoiceStudio がこのデバイスに最適な互換オプションを自動で設定できます。",
|
||||
"active": "エンジン:{{name}}",
|
||||
"none": "有効なエンジンがありません"
|
||||
"none": "有効なエンジンがありません",
|
||||
"localInstallRequired": "バックエンドのコンピューターで localhost またはセットアップガイドを使ってこのエンジンをインストールしてください。リモートインストールは無効です。",
|
||||
"argosRuntimeUnavailable": "Argos ランタイムを利用できません。バックエンドを再インストールするか、NLLB を選択してください。"
|
||||
},
|
||||
"performanceProfile": {
|
||||
"requiresEngine": "速度と品質を調整するには、{{engine}} をインストールして選択してください。",
|
||||
@@ -2519,6 +2522,7 @@
|
||||
"recorder_unsupported": "このシステムではマイク音声をエンコードできません。"
|
||||
},
|
||||
"tts_errors": {
|
||||
"profile_language_rejected": "自動では音声プロファイルの言語({{language}})が使われますが、このエンジンは対応していません。プロファイルの言語を変更するか、対応言語または別のエンジンを選択してください。",
|
||||
"generation_in_progress": "生成がすでに実行中です。完了してから次の生成を開始してください。",
|
||||
"error_prefix": "エラー: {{message}}",
|
||||
"ignored_duplicate": "無視 (カテゴリはすでに設定されています): {{items}}",
|
||||
|
||||
@@ -2237,6 +2237,7 @@
|
||||
"tailscale_qr_alt": "Tailscale URL의 QR 코드"
|
||||
},
|
||||
"convert": {
|
||||
"cloning_required": "음성 복제를 지원하는 사용 가능한 음성 합성 모델을 선택하세요.",
|
||||
"source_kicker": "소스 클립",
|
||||
"drop_audio": "변환할 클립을 끌어다 놓거나 클릭하세요. WAV, MP3, M4A…",
|
||||
"target_voice": "대상 목소리",
|
||||
@@ -2484,7 +2485,9 @@
|
||||
"none_ready_title": "음성 합성이 준비되지 않았습니다",
|
||||
"none_ready_body": "VoiceStudio가 이 기기에 가장 적합한 호환 옵션을 자동으로 설정할 수 있습니다.",
|
||||
"active": "엔진: {{name}}",
|
||||
"none": "활성 엔진 없음"
|
||||
"none": "활성 엔진 없음",
|
||||
"localInstallRequired": "백엔드 컴퓨터에서 localhost 또는 설정 가이드를 사용하여 이 엔진을 설치하세요. 원격 설치는 비활성화되어 있습니다.",
|
||||
"argosRuntimeUnavailable": "Argos 런타임을 사용할 수 없습니다. 백엔드를 다시 설치하거나 NLLB를 선택하세요."
|
||||
},
|
||||
"performanceProfile": {
|
||||
"requiresEngine": "속도와 품질을 조절하려면 {{engine}}을 설치하고 선택하세요.",
|
||||
@@ -2519,6 +2522,7 @@
|
||||
"recorder_unsupported": "이 시스템에서는 마이크 오디오를 인코딩할 수 없습니다."
|
||||
},
|
||||
"tts_errors": {
|
||||
"profile_language_rejected": "자동은 이 엔진이 지원하지 않는 음성 프로필의 언어({{language}})를 사용합니다. 프로필 언어를 변경하거나 지원되는 언어 또는 다른 엔진을 선택하세요.",
|
||||
"generation_in_progress": "이미 생성 작업이 실행 중입니다. 완료된 후 새 작업을 시작하세요.",
|
||||
"error_prefix": "오류: {{message}}",
|
||||
"ignored_duplicate": "무시됨(카테고리가 이미 설정됨): {{items}}",
|
||||
|
||||
@@ -2237,6 +2237,7 @@
|
||||
"tailscale_qr_alt": "QR-code voor de Tailscale-URL"
|
||||
},
|
||||
"convert": {
|
||||
"cloning_required": "Kies een gebruiksklaar tekst-naar-spraakmodel dat stemmen klonen ondersteunt.",
|
||||
"source_kicker": "Bronclip",
|
||||
"drop_audio": "Sleep de te converteren clip hierheen — of klik. WAV, MP3, M4A…",
|
||||
"target_voice": "Doelstem",
|
||||
@@ -2484,7 +2485,9 @@
|
||||
"none_ready_title": "Spraaksynthese is niet gereed",
|
||||
"none_ready_body": "VoiceStudio kan automatisch de beste compatibele optie voor dit apparaat instellen.",
|
||||
"active": "Engine: {{name}}",
|
||||
"none": "Geen engine actief"
|
||||
"none": "Geen engine actief",
|
||||
"localInstallRequired": "Installeer deze engine op de backendcomputer via localhost of de installatiehandleiding. Installatie op afstand is uitgeschakeld.",
|
||||
"argosRuntimeUnavailable": "De Argos-runtime is niet beschikbaar. Installeer de backend opnieuw of kies NLLB."
|
||||
},
|
||||
"performanceProfile": {
|
||||
"requiresEngine": "Installeer en selecteer {{engine}} om snelheid en kwaliteit af te stellen.",
|
||||
@@ -2519,6 +2522,7 @@
|
||||
"channels_stereo": "Stereo"
|
||||
},
|
||||
"tts_errors": {
|
||||
"profile_language_rejected": "Automatisch gebruikt de taal van het stemprofiel ({{language}}), die deze engine niet ondersteunt. Wijzig de profieltaal, kies een ondersteunde taal of wissel van engine.",
|
||||
"generation_in_progress": "Er wordt al audio gegenereerd — wacht tot dit klaar is voordat je opnieuw begint.",
|
||||
"error_prefix": "Fout: {{message}}",
|
||||
"ignored_duplicate": "Genegeerd (categorie al ingesteld): {{items}}",
|
||||
|
||||
@@ -2241,6 +2241,7 @@
|
||||
"tailscale_qr_alt": "Kod QR dla adresu URL Tailscale"
|
||||
},
|
||||
"convert": {
|
||||
"cloning_required": "Wybierz gotowy do użycia model syntezy mowy obsługujący klonowanie głosu.",
|
||||
"source_kicker": "Klip źródłowy",
|
||||
"drop_audio": "Upuść klip do konwersji — lub kliknij. WAV, MP3, M4A…",
|
||||
"target_voice": "Głos docelowy",
|
||||
@@ -2488,7 +2489,9 @@
|
||||
"none_ready_title": "Synteza mowy nie jest gotowa",
|
||||
"none_ready_body": "VoiceStudio może automatycznie skonfigurować najlepszą zgodną opcję dla tego urządzenia.",
|
||||
"active": "Silnik: {{name}}",
|
||||
"none": "Brak aktywnego silnika"
|
||||
"none": "Brak aktywnego silnika",
|
||||
"localInstallRequired": "Zainstaluj ten silnik na komputerze zaplecza przez localhost lub zgodnie z instrukcją konfiguracji. Instalacja zdalna jest wyłączona.",
|
||||
"argosRuntimeUnavailable": "Środowisko Argos jest niedostępne. Zainstaluj ponownie backend lub wybierz NLLB."
|
||||
},
|
||||
"performanceProfile": {
|
||||
"requiresEngine": "Zainstaluj i wybierz {{engine}}, aby dostosować szybkość i jakość.",
|
||||
@@ -2523,6 +2526,7 @@
|
||||
"channels_stereo": "Stereo"
|
||||
},
|
||||
"tts_errors": {
|
||||
"profile_language_rejected": "Tryb automatyczny używa języka profilu głosu ({{language}}), którego ten silnik nie obsługuje. Zmień język profilu, wybierz obsługiwany język lub zmień silnik.",
|
||||
"generation_in_progress": "Generowanie już trwa — poczekaj na jego zakończenie przed rozpoczęciem kolejnego.",
|
||||
"error_prefix": "Błąd: {{message}}",
|
||||
"ignored_duplicate": "Ignorowane (kategoria już ustawiona): {{items}}",
|
||||
|
||||
@@ -2239,6 +2239,7 @@
|
||||
"tailscale_qr_alt": "Código QR para o URL Tailscale"
|
||||
},
|
||||
"convert": {
|
||||
"cloning_required": "Escolha um modelo de síntese de voz pronto para uso que permita clonar vozes.",
|
||||
"source_kicker": "Clipe de origem",
|
||||
"drop_audio": "Solte o clipe a converter — ou clique. WAV, MP3, M4A…",
|
||||
"target_voice": "Voz de destino",
|
||||
@@ -2486,7 +2487,9 @@
|
||||
"none_ready_title": "A síntese de voz não está pronta",
|
||||
"none_ready_body": "O VoiceStudio pode configurar automaticamente a melhor opção compatível com este dispositivo.",
|
||||
"active": "Motor: {{name}}",
|
||||
"none": "Nenhum motor ativo"
|
||||
"none": "Nenhum motor ativo",
|
||||
"localInstallRequired": "Instale este motor no computador do backend pelo localhost ou pelo guia de configuração. A instalação remota está desativada.",
|
||||
"argosRuntimeUnavailable": "O ambiente de execução do Argos está indisponível. Reinstale o backend ou selecione NLLB."
|
||||
},
|
||||
"performanceProfile": {
|
||||
"requiresEngine": "Instale e selecione {{engine}} para ajustar a velocidade e a qualidade.",
|
||||
@@ -2521,6 +2524,7 @@
|
||||
"channels_mono": "Mono"
|
||||
},
|
||||
"tts_errors": {
|
||||
"profile_language_rejected": "Automático usa o idioma do perfil de voz ({{language}}), não suportado por este motor. Altere o idioma do perfil, escolha um idioma compatível ou mude de motor.",
|
||||
"generation_in_progress": "Já existe uma geração em andamento — aguarde a conclusão antes de iniciar outra.",
|
||||
"error_prefix": "Erro: {{message}}",
|
||||
"ignored_duplicate": "Ignorado (categoria já definida): {{items}}",
|
||||
|
||||
@@ -2241,6 +2241,7 @@
|
||||
"tailscale_qr_alt": "QR-код для URL-адреса Tailscale"
|
||||
},
|
||||
"convert": {
|
||||
"cloning_required": "Выберите готовую к работе модель синтеза речи с поддержкой клонирования голоса.",
|
||||
"source_kicker": "Исходный клип",
|
||||
"drop_audio": "Перетащите клип для преобразования — или нажмите. WAV, MP3, M4A…",
|
||||
"target_voice": "Целевой голос",
|
||||
@@ -2488,7 +2489,9 @@
|
||||
"none_ready_title": "Синтез речи не готов",
|
||||
"none_ready_body": "VoiceStudio может автоматически настроить лучший совместимый вариант для этого устройства.",
|
||||
"active": "Движок: {{name}}",
|
||||
"none": "Нет активного движка"
|
||||
"none": "Нет активного движка",
|
||||
"localInstallRequired": "Установите этот движок на компьютере сервера через localhost или по инструкции. Удалённая установка отключена.",
|
||||
"argosRuntimeUnavailable": "Среда выполнения Argos недоступна. Переустановите серверную часть или выберите NLLB."
|
||||
},
|
||||
"performanceProfile": {
|
||||
"requiresEngine": "Установите и выберите {{engine}}, чтобы настроить скорость и качество.",
|
||||
@@ -2523,6 +2526,7 @@
|
||||
"recorder_unsupported": "Эта система не может кодировать звук с микрофона."
|
||||
},
|
||||
"tts_errors": {
|
||||
"profile_language_rejected": "Авто использует язык голосового профиля ({{language}}), который этот движок не поддерживает. Измените язык профиля, выберите поддерживаемый язык или другой движок.",
|
||||
"generation_in_progress": "Генерация уже выполняется — дождитесь её завершения, прежде чем запускать следующую.",
|
||||
"error_prefix": "Ошибка: {{message}}",
|
||||
"ignored_duplicate": "Игнорируется (категория уже установлена): {{items}}",
|
||||
|
||||
@@ -2237,6 +2237,7 @@
|
||||
"tailscale_qr_alt": "QR-kod för Tailscale URL"
|
||||
},
|
||||
"convert": {
|
||||
"cloning_required": "Välj en användningsklar talsyntesmodell som stöder röstkloning.",
|
||||
"source_kicker": "Källklipp",
|
||||
"drop_audio": "Släpp klippet som ska konverteras — eller klicka. WAV, MP3, M4A…",
|
||||
"target_voice": "Målröst",
|
||||
@@ -2484,7 +2485,9 @@
|
||||
"none_ready_title": "Talsyntesen är inte redo",
|
||||
"none_ready_body": "VoiceStudio kan automatiskt konfigurera det bästa kompatibla alternativet för den här enheten.",
|
||||
"active": "Motor: {{name}}",
|
||||
"none": "Ingen motor aktiv"
|
||||
"none": "Ingen motor aktiv",
|
||||
"localInstallRequired": "Installera motorn på backenddatorn via localhost eller installationsguiden. Fjärrinstallation är avstängd.",
|
||||
"argosRuntimeUnavailable": "Argos körmiljö är inte tillgänglig. Installera om backend eller välj NLLB."
|
||||
},
|
||||
"performanceProfile": {
|
||||
"requiresEngine": "Installera och välj {{engine}} för att justera hastighet och kvalitet.",
|
||||
@@ -2519,6 +2522,7 @@
|
||||
"channels_stereo": "Stereo"
|
||||
},
|
||||
"tts_errors": {
|
||||
"profile_language_rejected": "Auto använder röstprofilens språk ({{language}}), som denna motor inte stöder. Ändra profilens språk, välj ett språk som stöds eller byt motor.",
|
||||
"generation_in_progress": "En generering pågår redan – vänta tills den är klar innan du startar en ny.",
|
||||
"error_prefix": "Fel: {{message}}",
|
||||
"ignored_duplicate": "Ignorerad (kategori redan inställd): {{items}}",
|
||||
|
||||
@@ -2237,6 +2237,7 @@
|
||||
"tailscale_qr_alt": "รหัส QR สำหรับ Tailscale URL"
|
||||
},
|
||||
"convert": {
|
||||
"cloning_required": "เลือกโมเดลแปลงข้อความเป็นเสียงที่พร้อมใช้งานและรองรับการโคลนเสียง",
|
||||
"source_kicker": "คลิปต้นฉบับ",
|
||||
"drop_audio": "วางคลิปที่จะแปลงที่นี่ — หรือคลิก WAV, MP3, M4A…",
|
||||
"target_voice": "เสียงปลายทาง",
|
||||
@@ -2484,7 +2485,9 @@
|
||||
"none_ready_title": "การสังเคราะห์เสียงยังไม่พร้อม",
|
||||
"none_ready_body": "VoiceStudio สามารถตั้งค่าตัวเลือกที่เข้ากันได้ดีที่สุดสำหรับอุปกรณ์นี้โดยอัตโนมัติ",
|
||||
"active": "เอนจิน: {{name}}",
|
||||
"none": "ไม่มีเอนจินที่ใช้งานอยู่"
|
||||
"none": "ไม่มีเอนจินที่ใช้งานอยู่",
|
||||
"localInstallRequired": "ติดตั้งเอนจินนี้บนคอมพิวเตอร์แบ็กเอนด์ผ่าน localhost หรือคู่มือการตั้งค่า การติดตั้งจากระยะไกลถูกปิดใช้งาน",
|
||||
"argosRuntimeUnavailable": "ไม่สามารถใช้งานรันไทม์ Argos ได้ โปรดติดตั้งแบ็กเอนด์ใหม่หรือเลือก NLLB"
|
||||
},
|
||||
"performanceProfile": {
|
||||
"requiresEngine": "ติดตั้งและเลือก {{engine}} เพื่อปรับความเร็วและคุณภาพ",
|
||||
@@ -2519,6 +2522,7 @@
|
||||
"recorder_unsupported": "ระบบนี้ไม่สามารถเข้ารหัสเสียงจากไมโครโฟนได้"
|
||||
},
|
||||
"tts_errors": {
|
||||
"profile_language_rejected": "อัตโนมัติใช้ภาษาของโปรไฟล์เสียง ({{language}}) ซึ่งเอนจินนี้ไม่รองรับ เปลี่ยนภาษาโปรไฟล์ เลือกภาษาที่รองรับ หรือเปลี่ยนเอนจิน",
|
||||
"generation_in_progress": "มีการสร้างเสียงกำลังทำงานอยู่ — รอให้เสร็จก่อนเริ่มงานใหม่",
|
||||
"error_prefix": "ข้อผิดพลาด: {{message}}",
|
||||
"ignored_duplicate": "ละเว้น (หมวดหมู่ที่ตั้งไว้แล้ว): {{items}}",
|
||||
|
||||
@@ -2237,6 +2237,7 @@
|
||||
"tailscale_qr_alt": "Kuyruk Ölçeği URL'si için QR kodu"
|
||||
},
|
||||
"convert": {
|
||||
"cloning_required": "Ses klonlamayı destekleyen, kullanıma hazır bir metinden konuşmaya modeli seçin.",
|
||||
"source_kicker": "Kaynak klip",
|
||||
"drop_audio": "Dönüştürülecek klibi bırakın — veya tıklayın. WAV, MP3, M4A…",
|
||||
"target_voice": "Hedef ses",
|
||||
@@ -2484,7 +2485,9 @@
|
||||
"none_ready_title": "Konuşma sentezi hazır değil",
|
||||
"none_ready_body": "VoiceStudio bu cihaz için en iyi uyumlu seçeneği otomatik olarak ayarlayabilir.",
|
||||
"active": "Motor: {{name}}",
|
||||
"none": "Etkin motor yok"
|
||||
"none": "Etkin motor yok",
|
||||
"localInstallRequired": "Bu motoru arka uç bilgisayarında localhost veya kurulum kılavuzu üzerinden yükleyin. Uzaktan kurulum devre dışıdır.",
|
||||
"argosRuntimeUnavailable": "Argos çalışma ortamı kullanılamıyor. Arka ucu yeniden yükleyin veya NLLB seçin."
|
||||
},
|
||||
"performanceProfile": {
|
||||
"requiresEngine": "Hız ve kaliteyi ayarlamak için {{engine}} yükleyip seçin.",
|
||||
@@ -2519,6 +2522,7 @@
|
||||
"channels_stereo": "Stereo"
|
||||
},
|
||||
"tts_errors": {
|
||||
"profile_language_rejected": "Otomatik, bu motorun desteklemediği ses profili dilini ({{language}}) kullanır. Profil dilini değiştirin, desteklenen bir dil seçin veya motoru değiştirin.",
|
||||
"generation_in_progress": "Bir oluşturma işlemi zaten çalışıyor — yenisini başlatmadan önce bitmesini bekleyin.",
|
||||
"error_prefix": "Hata: {{message}}",
|
||||
"ignored_duplicate": "Yoksayıldı (kategori zaten ayarlandı): {{items}}",
|
||||
|
||||
@@ -2241,6 +2241,7 @@
|
||||
"tailscale_qr_alt": "QR-код для URL-адреси Tailscale"
|
||||
},
|
||||
"convert": {
|
||||
"cloning_required": "Виберіть готову до роботи модель синтезу мовлення з підтримкою клонування голосу.",
|
||||
"source_kicker": "Вихідний кліп",
|
||||
"drop_audio": "Перетягніть кліп для перетворення — або натисніть. WAV, MP3, M4A…",
|
||||
"target_voice": "Цільовий голос",
|
||||
@@ -2488,7 +2489,9 @@
|
||||
"none_ready_title": "Синтез мовлення не готовий",
|
||||
"none_ready_body": "VoiceStudio може автоматично налаштувати найкращий сумісний варіант для цього пристрою.",
|
||||
"active": "Рушій: {{name}}",
|
||||
"none": "Немає активного рушія"
|
||||
"none": "Немає активного рушія",
|
||||
"localInstallRequired": "Установіть цей рушій на комп’ютері сервера через localhost або за інструкцією. Віддалене встановлення вимкнено.",
|
||||
"argosRuntimeUnavailable": "Середовище виконання Argos недоступне. Перевстановіть серверну частину або виберіть NLLB."
|
||||
},
|
||||
"performanceProfile": {
|
||||
"requiresEngine": "Установіть і виберіть {{engine}}, щоб налаштувати швидкість і якість.",
|
||||
@@ -2523,6 +2526,7 @@
|
||||
"recorder_unsupported": "Ця система не може кодувати звук із мікрофона."
|
||||
},
|
||||
"tts_errors": {
|
||||
"profile_language_rejected": "Авто використовує мову голосового профілю ({{language}}), яку цей рушій не підтримує. Змініть мову профілю, виберіть підтримувану мову або інший рушій.",
|
||||
"generation_in_progress": "Генерація вже виконується — дочекайтеся її завершення, перш ніж запускати наступну.",
|
||||
"error_prefix": "Помилка: {{message}}",
|
||||
"ignored_duplicate": "Проігноровано (категорію вже встановлено): {{items}}",
|
||||
|
||||
@@ -2237,6 +2237,7 @@
|
||||
"tailscale_qr_alt": "Mã QR cho URL Tailscale"
|
||||
},
|
||||
"convert": {
|
||||
"cloning_required": "Chọn mô hình chuyển văn bản thành giọng nói sẵn sàng sử dụng và hỗ trợ nhân bản giọng nói.",
|
||||
"source_kicker": "Clip nguồn",
|
||||
"drop_audio": "Thả clip cần chuyển đổi vào đây — hoặc nhấp. WAV, MP3, M4A…",
|
||||
"target_voice": "Giọng đích",
|
||||
@@ -2484,7 +2485,9 @@
|
||||
"none_ready_title": "Tổng hợp giọng nói chưa sẵn sàng",
|
||||
"none_ready_body": "VoiceStudio có thể tự động thiết lập tùy chọn tương thích tốt nhất cho thiết bị này.",
|
||||
"active": "Bộ máy: {{name}}",
|
||||
"none": "Không có bộ máy đang hoạt động"
|
||||
"none": "Không có bộ máy đang hoạt động",
|
||||
"localInstallRequired": "Cài đặt công cụ này trên máy chủ qua localhost hoặc hướng dẫn thiết lập. Cài đặt từ xa đã bị tắt.",
|
||||
"argosRuntimeUnavailable": "Không thể sử dụng môi trường chạy Argos. Hãy cài lại backend hoặc chọn NLLB."
|
||||
},
|
||||
"performanceProfile": {
|
||||
"requiresEngine": "Cài đặt và chọn {{engine}} để điều chỉnh tốc độ và chất lượng.",
|
||||
@@ -2519,6 +2522,7 @@
|
||||
"recorder_unsupported": "Hệ thống này không thể mã hóa âm thanh từ micrô."
|
||||
},
|
||||
"tts_errors": {
|
||||
"profile_language_rejected": "Tự động sử dụng ngôn ngữ của hồ sơ giọng nói ({{language}}), nhưng bộ máy này không hỗ trợ. Đổi ngôn ngữ hồ sơ, chọn ngôn ngữ được hỗ trợ hoặc đổi bộ máy.",
|
||||
"generation_in_progress": "Một tác vụ tạo âm thanh đang chạy — hãy đợi hoàn tất trước khi bắt đầu tác vụ khác.",
|
||||
"error_prefix": "Lỗi: {{message}}",
|
||||
"ignored_duplicate": "Đã bỏ qua (danh mục đã được đặt): {{items}}",
|
||||
|
||||
@@ -2241,6 +2241,7 @@
|
||||
"tailscale_qr_alt": "Tailscale URL 的二维码"
|
||||
},
|
||||
"convert": {
|
||||
"cloning_required": "请选择已就绪且支持声音克隆的语音合成模型。",
|
||||
"source_kicker": "源音频",
|
||||
"drop_audio": "拖入要转换的音频 — 或点击。WAV、MP3、M4A…",
|
||||
"target_voice": "目标声音",
|
||||
@@ -2488,7 +2489,9 @@
|
||||
"none_ready_title": "语音合成尚未就绪",
|
||||
"none_ready_body": "VoiceStudio 可以自动为此设备设置最佳兼容选项。",
|
||||
"active": "引擎:{{name}}",
|
||||
"none": "没有启用的引擎"
|
||||
"none": "没有启用的引擎",
|
||||
"localInstallRequired": "请在后端计算机上通过 localhost 或安装指南安装此引擎。远程安装已禁用。",
|
||||
"argosRuntimeUnavailable": "Argos 运行时不可用。请重新安装后端或选择 NLLB。"
|
||||
},
|
||||
"performanceProfile": {
|
||||
"requiresEngine": "安装并选择 {{engine}} 以调整速度和质量。",
|
||||
@@ -2523,6 +2526,7 @@
|
||||
"recorder_unsupported": "此系统无法编码麦克风音频。"
|
||||
},
|
||||
"tts_errors": {
|
||||
"profile_language_rejected": "自动模式使用声音档案的语言({{language}}),但此引擎不支持该语言。请修改档案语言、选择支持的语言或更换引擎。",
|
||||
"generation_in_progress": "已有生成任务正在运行,请等待其完成后再启动新任务。",
|
||||
"error_prefix": "错误:{{message}}",
|
||||
"ignored_duplicate": "忽略(类别已设置):{{items}}",
|
||||
|
||||
@@ -2237,6 +2237,7 @@
|
||||
"tailscale_qr_alt": "Tailscale URL 的二維碼"
|
||||
},
|
||||
"convert": {
|
||||
"cloning_required": "請選擇已就緒且支援聲音複製的語音合成模型。",
|
||||
"source_kicker": "來源音訊",
|
||||
"drop_audio": "拖入要轉換的音訊 — 或點擊。WAV、MP3、M4A…",
|
||||
"target_voice": "目標聲音",
|
||||
@@ -2484,7 +2485,9 @@
|
||||
"none_ready_title": "語音合成尚未就緒",
|
||||
"none_ready_body": "VoiceStudio 可以自動為此裝置設定最佳相容選項。",
|
||||
"active": "引擎:{{name}}",
|
||||
"none": "沒有啟用的引擎"
|
||||
"none": "沒有啟用的引擎",
|
||||
"localInstallRequired": "請在後端電腦上透過 localhost 或安裝指南安裝此引擎。遠端安裝已停用。",
|
||||
"argosRuntimeUnavailable": "Argos 執行環境無法使用。請重新安裝後端或選擇 NLLB。"
|
||||
},
|
||||
"performanceProfile": {
|
||||
"requiresEngine": "安裝並選取 {{engine}} 以調整速度與品質。",
|
||||
@@ -2519,6 +2522,7 @@
|
||||
"recorder_unsupported": "此系統無法編碼麥克風音訊。"
|
||||
},
|
||||
"tts_errors": {
|
||||
"profile_language_rejected": "自動模式使用聲音設定檔的語言({{language}}),但此引擎不支援該語言。請修改設定檔語言、選擇支援的語言或更換引擎。",
|
||||
"generation_in_progress": "已有產生工作正在執行,請等待完成後再啟動新工作。",
|
||||
"error_prefix": "錯誤:{{message}}",
|
||||
"ignored_duplicate": "忽略(類別已設定):{{items}}",
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import i18next from 'i18next';
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest';
|
||||
import {
|
||||
_resetBackendContactForTests,
|
||||
@@ -18,11 +19,20 @@ beforeEach(() => {
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
vi.unstubAllGlobals();
|
||||
_resetBackendContactForTests();
|
||||
});
|
||||
|
||||
describe('errorFromResponse', () => {
|
||||
it('localizes Argos runtime errors and retains diagnostics', async () => {
|
||||
const translate = vi.spyOn(i18next, 't').mockReturnValue('Localized recovery guidance');
|
||||
const detail = { code: 'argos_runtime_unavailable', message: 'Raw native diagnostic' };
|
||||
const err = await errorFromResponse(new Response(JSON.stringify({ detail }), { status: 400 }));
|
||||
expect(err.detail).toBe('Localized recovery guidance');
|
||||
expect(translate).toHaveBeenCalledWith('engines.argosRuntimeUnavailable');
|
||||
expect(err.payload?.detail).toEqual(detail);
|
||||
});
|
||||
it('localizes background preservation errors and retains diagnostics', async () => {
|
||||
const detail = { code: 'dub_background_unavailable', message: 'Raw diagnostic' };
|
||||
const err = await errorFromResponse(new Response(JSON.stringify({ detail }), { status: 409 }));
|
||||
@@ -159,3 +169,23 @@ it('uses a structured error message without losing recovery metadata', async ()
|
||||
expect(error.message).toBe(detail.message);
|
||||
expect(error.payload?.detail).toEqual(detail);
|
||||
});
|
||||
|
||||
it('localizes structured profile language failures', async () => {
|
||||
const translate = vi.spyOn(i18next, 't').mockReturnValue('Localized profile guidance');
|
||||
try {
|
||||
const detail = {
|
||||
code: 'profile_language_rejected',
|
||||
language: 'Persian',
|
||||
message: 'English fallback',
|
||||
};
|
||||
const error = await errorFromResponse(
|
||||
new Response(JSON.stringify({ detail }), { status: 400 }),
|
||||
);
|
||||
expect(error.detail).toBe('Localized profile guidance');
|
||||
expect(translate).toHaveBeenCalledWith('tts_errors.profile_language_rejected', {
|
||||
language: 'Persian',
|
||||
});
|
||||
} finally {
|
||||
translate.mockRestore();
|
||||
}
|
||||
});
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { languageRejectionMessage } from '../../../../../../frontend/src/utils/languageRejection.ts';
|
||||
/**
|
||||
* Same-origin API client. The renderer never talks to 127.0.0.1:<port>
|
||||
* directly (CORS); `/api/*` is proxied by the dev server / the app:// protocol
|
||||
@@ -44,7 +45,21 @@ export function describeError(err: unknown): string {
|
||||
}
|
||||
|
||||
function detailToString(detail: unknown): string {
|
||||
if (detail && typeof detail === 'object' && 'code' in detail && detail.code === 'dub_background_unavailable')
|
||||
const localized = languageRejectionMessage(detail, tr);
|
||||
if (localized) return localized;
|
||||
if (
|
||||
detail &&
|
||||
typeof detail === 'object' &&
|
||||
'code' in detail &&
|
||||
detail.code === 'argos_runtime_unavailable'
|
||||
)
|
||||
return tr('engines.argosRuntimeUnavailable');
|
||||
if (
|
||||
detail &&
|
||||
typeof detail === 'object' &&
|
||||
'code' in detail &&
|
||||
detail.code === 'dub_background_unavailable'
|
||||
)
|
||||
return tr('dubIntegrity.backgroundUnavailable');
|
||||
if (typeof detail === 'string') return detail;
|
||||
if (detail == null) return '';
|
||||
|
||||
@@ -1,8 +1,11 @@
|
||||
import i18n from 'i18next';
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest';
|
||||
import {
|
||||
CLONE_MAX_SECONDS,
|
||||
REF_HARD_MAX_SECONDS,
|
||||
generateClone,
|
||||
generateCloneStreaming,
|
||||
shouldFallbackToClassic,
|
||||
parseGenerateHeaders,
|
||||
sanitizeInstruct,
|
||||
toGenerateForm,
|
||||
@@ -277,3 +280,31 @@ describe('generateClone', () => {
|
||||
await assertion;
|
||||
});
|
||||
});
|
||||
|
||||
it('localizes streamed profile language refusals and prevents classic fallback', async () => {
|
||||
const translate = vi.spyOn(i18n, 't').mockReturnValue('Localized profile guidance');
|
||||
try {
|
||||
vi.stubGlobal(
|
||||
'fetch',
|
||||
vi.fn(
|
||||
async () =>
|
||||
new Response(
|
||||
JSON.stringify({
|
||||
type: 'error',
|
||||
code: 'profile_language_rejected',
|
||||
language: 'Persian',
|
||||
detail: 'English fallback',
|
||||
retryable: false,
|
||||
terminal: true,
|
||||
}) + '\n',
|
||||
{ headers: { 'content-type': 'application/x-ndjson' } },
|
||||
),
|
||||
),
|
||||
);
|
||||
const error = await generateCloneStreaming(BASE_INPUT).catch((e) => e);
|
||||
expect(error.message).toBe('Localized profile guidance');
|
||||
expect(shouldFallbackToClassic(error)).toBe(false);
|
||||
} finally {
|
||||
translate.mockRestore();
|
||||
}
|
||||
});
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
import i18next from 'i18next';
|
||||
import { languageRejectionMessage } from '../../../../../../frontend/src/utils/languageRejection.ts';
|
||||
import { ApiError, apiFetch, isAbortError } from './client';
|
||||
import type { CloneGenerateInput, GenerateResult } from './types';
|
||||
import { beginAppActivity } from '@/lib/app-activity';
|
||||
@@ -255,6 +257,9 @@ interface StreamEvent {
|
||||
count?: number;
|
||||
text?: string[];
|
||||
detail?: string;
|
||||
code?: string;
|
||||
language?: string;
|
||||
terminal?: boolean;
|
||||
retryable?: boolean;
|
||||
percent?: number;
|
||||
}
|
||||
@@ -300,12 +305,15 @@ export async function generateCloneStreaming(
|
||||
} else if (event.type === 'done') {
|
||||
meta = event;
|
||||
} else if (event.type === 'error') {
|
||||
const message = event.detail || 'TTS stream reported an error';
|
||||
const terminal = [
|
||||
'[clone_ref_unusable]',
|
||||
'[clone_ref_too_long]',
|
||||
'[clone_ref_no_speech]',
|
||||
].some((marker) => message.includes(marker));
|
||||
const message =
|
||||
languageRejectionMessage(event, i18next.t) ||
|
||||
event.detail ||
|
||||
'TTS stream reported an error';
|
||||
const terminal =
|
||||
event.terminal === true ||
|
||||
['[clone_ref_unusable]', '[clone_ref_too_long]', '[clone_ref_no_speech]'].some((marker) =>
|
||||
message.includes(marker),
|
||||
);
|
||||
throw new StreamingPreviewError(message, { retryable: event.retryable, terminal });
|
||||
}
|
||||
};
|
||||
|
||||
@@ -66,6 +66,7 @@ export interface EngineBackend {
|
||||
setup_snippet?: string | null;
|
||||
docs_url?: string | null;
|
||||
one_click_install?: boolean;
|
||||
local_install_required?: boolean;
|
||||
license_required?: boolean;
|
||||
license_accepted?: boolean;
|
||||
effective_device?: string;
|
||||
|
||||
@@ -27,6 +27,7 @@
|
||||
"src/preload/index.d.ts",
|
||||
"src/shared/**/*",
|
||||
"src/renderer/env.d.ts",
|
||||
"../frontend/src/utils/languageRejection.ts",
|
||||
"../frontend/src/utils/coalescedJsonStorage.ts",
|
||||
"../frontend/src/utils/indexedDbLongformStore.ts",
|
||||
"../frontend/src/utils/cookieExport.ts",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "omnivoice-studio",
|
||||
"version": "0.5.3",
|
||||
"version": "0.5.4",
|
||||
"private": true,
|
||||
"license": "AGPL-3.0-only",
|
||||
"type": "module",
|
||||
|
||||
Generated
+1
-1
@@ -3093,7 +3093,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "omnivoice-studio"
|
||||
version = "0.5.3"
|
||||
version = "0.5.4"
|
||||
dependencies = [
|
||||
"arboard",
|
||||
"cap-std",
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user