Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7c0b0c2572 | ||
|
|
5e2d314efe | ||
|
|
29269b9cf0 | ||
|
|
c5c57508b3 | ||
|
|
5e0d6826da | ||
|
|
e347f99542 | ||
|
|
38b8c55e52 | ||
|
|
255ac1bad0 | ||
|
|
522bbddccf | ||
|
|
66ad03948b | ||
|
|
18de6d26d3 | ||
|
|
424ab000dd | ||
|
|
28aeacbc39 | ||
|
|
48ccb1da30 | ||
|
|
2345888c4e | ||
|
|
df194e2a93 | ||
|
|
c7b5d99133 | ||
|
|
e97c297113 | ||
|
|
c9843479db | ||
|
|
7264c321f8 | ||
|
|
8541eb75c5 | ||
|
|
ff7d8d08dc | ||
|
|
ce74ac8777 | ||
|
|
94799b4a63 | ||
|
|
b3690e8d31 | ||
|
|
7ccb2da19f | ||
|
|
0f014129ad | ||
|
|
60ac321a1b | ||
|
|
c931d8f4f8 | ||
|
|
b940eb460f | ||
|
|
8a9c2f1ce4 | ||
|
|
b033b8e9da | ||
|
|
bafe9677d9 | ||
|
|
c44bae0d0d | ||
|
|
093e85e47a | ||
|
|
630912df55 | ||
|
|
3d17104228 | ||
|
|
992eb67143 | ||
|
|
b14ad0cc1a | ||
|
|
c7220a45ab | ||
|
|
4e03e363ed | ||
|
|
40a97daa9a | ||
|
|
12f125c77e | ||
|
|
e6067dfaec | ||
|
|
86b30ece98 | ||
|
|
836d69178c | ||
|
|
cb70c2b1af | ||
|
|
200a559183 | ||
|
|
30b3886f9f | ||
|
|
9d79bb8e36 | ||
|
|
ab87ef734d | ||
|
|
3d88779eff | ||
|
|
d88e57675d | ||
|
|
3c17d19b7e | ||
|
|
0b16481633 | ||
|
|
ecd2545fcf | ||
|
|
930b403799 | ||
|
|
4a6da3bf84 | ||
|
|
16a8995a9a | ||
|
|
5673e2e772 | ||
|
|
349a40a261 | ||
|
|
5b6f49ee0c | ||
|
|
740a690644 | ||
|
|
9bb3cc2664 | ||
|
|
a0a4bcc903 | ||
|
|
2b76518c22 | ||
|
|
7cad4633dc | ||
|
|
5c9b7ca313 | ||
|
|
f293f10e7e | ||
|
|
721cb34a9d | ||
|
|
7b4033bc35 | ||
|
|
68d456bb58 | ||
|
|
87f884d7a1 | ||
|
|
294a5db4c1 | ||
|
|
8765f766e1 | ||
|
|
b9707e0d8f | ||
|
|
a7f813b7a9 | ||
|
|
f33bdc731d | ||
|
|
e2f02327c3 | ||
|
|
39385feb78 | ||
|
|
fa66c6a025 | ||
|
|
4a01ecdfae | ||
|
|
cc95f526e0 | ||
|
|
9fc18dbd89 | ||
|
|
de3d83f14b | ||
|
|
0fc9f2afec | ||
|
|
a9948de99e | ||
|
|
7f9a97b0fa | ||
|
|
935b38a962 | ||
|
|
ea4d5e7839 | ||
|
|
d566e5bc8e | ||
|
|
2c2e493df8 | ||
|
|
2fc90c44e6 | ||
|
|
aa17e3319f | ||
|
|
883a06e9c0 | ||
|
|
85fd8b6494 | ||
|
|
598b1bd911 | ||
|
|
ea3b565f9c | ||
|
|
97a020ecf0 | ||
|
|
5e6b23aeae | ||
|
|
5a650eeee7 | ||
|
|
53d00ab76a | ||
|
|
823e222bd0 | ||
|
|
70857bd1ad | ||
|
|
6a64db0b8c | ||
|
|
ee638bda6c | ||
|
|
25105605f1 | ||
|
|
022a3bd6b9 | ||
|
|
e7d358b541 | ||
|
|
2339cc8e85 | ||
|
|
0af744bc3a | ||
|
|
87cce2e2b8 | ||
|
|
8c45d4a9f2 | ||
|
|
dba8934aa2 | ||
|
|
b7cecde57e | ||
|
|
46abe2e29a | ||
|
|
81a007552b | ||
|
|
db016d4675 | ||
|
|
9d2e395437 | ||
|
|
10b9d6950d | ||
|
|
e87c13e919 | ||
|
|
252f0d4fac | ||
|
|
d3cebe58a0 | ||
|
|
1f29e374be | ||
|
|
4eac1b50d5 | ||
|
|
1dc8eb1adb | ||
|
|
41e866b46c | ||
|
|
8e4044bb37 | ||
|
|
d8b059813a | ||
|
|
6819feb8b2 | ||
|
|
79f3e35682 | ||
|
|
b15acbaae9 | ||
|
|
0cf6bb3087 | ||
|
|
a28a19dc63 | ||
|
|
a63c8e851b | ||
|
|
1fe68ba11e | ||
|
|
d21fb765e2 | ||
|
|
de80856cd9 | ||
|
|
1575baca36 | ||
|
|
31ba6d3d27 | ||
|
|
2ad83f37fd | ||
|
|
2767e2995d | ||
|
|
7b70d82322 | ||
|
|
2549d9539e | ||
|
|
6e3e4bcfdd | ||
|
|
c965a7fbdd | ||
|
|
8f2c4bbc5c | ||
|
|
b8e219e7d1 | ||
|
|
7393ae80f5 | ||
|
|
21b0b1f0b2 | ||
|
|
0e17caa52a | ||
|
|
ff168048be | ||
|
|
e14644f77a | ||
|
|
b0c93a598e | ||
|
|
a22f029c2c | ||
|
|
8e6344ee4e | ||
|
|
8ed76b40a6 | ||
|
|
ad78a418bc | ||
|
|
b8cc0de44a | ||
|
|
a7ab148483 | ||
|
|
51ad469ef2 | ||
|
|
7678d4cca3 | ||
|
|
6dccf0390d | ||
|
|
3f23a77eab | ||
|
|
cb8ff16d86 | ||
|
|
915256cb8c | ||
|
|
1ee4b3c2a0 | ||
|
|
9a1d549686 | ||
|
|
3777d3a62c | ||
|
|
80891d3239 | ||
|
|
c0727c261d | ||
|
|
2eac93c645 | ||
|
|
4ec0ece457 | ||
|
|
50fa5c206f | ||
|
|
ef93835e22 | ||
|
|
25a471fe59 | ||
|
|
e02b7941a8 | ||
|
|
ad443e853a | ||
|
|
1fa2dda7c8 | ||
|
|
abb5f00d81 | ||
|
|
69b8c7c0bc | ||
|
|
f4e7a76886 | ||
|
|
74d63f099c | ||
|
|
f448c1c73c | ||
|
|
35e7dbc11c | ||
|
|
05369e748a | ||
|
|
ade951688e | ||
|
|
5a28c04c44 | ||
|
|
8db5feccc7 | ||
|
|
0337e15a2e | ||
|
|
e58106d552 | ||
|
|
364611f7d7 | ||
|
|
4471bc0770 |
+4
-3
@@ -1,6 +1,7 @@
|
||||
# These are supported funding model platforms
|
||||
# GitHub Sponsors isn't set up for this account — fund via Ko-fi or PayPal.
|
||||
|
||||
github: [debpalash]
|
||||
# ko_fi: omnivoice
|
||||
ko_fi: debpalash
|
||||
custom: ["https://paypal.me/palashCoder"]
|
||||
# github: [debpalash] # not available
|
||||
# open_collective: omnivoice-studio
|
||||
# custom: ["https://omnivoice.palash.dev/sponsor"]
|
||||
|
||||
@@ -105,6 +105,19 @@ jobs:
|
||||
working-directory: frontend
|
||||
run: bun run typecheck:ci
|
||||
|
||||
# oxlint gate — fast Rust linter, blocks on errors so lint debt can't
|
||||
# re-accumulate (warnings, incl. the react-compiler advisories in
|
||||
# `lint:hooks`, are non-blocking). See frontend/.oxlintrc.json.
|
||||
- name: Frontend lint (oxlint)
|
||||
working-directory: frontend
|
||||
run: bun run lint
|
||||
|
||||
# oxfmt format gate — JS/TS/JSX only (CSS/JSON/Tauri excluded; see
|
||||
# frontend/.oxfmtrc.json). `bun run format` fixes locally.
|
||||
- name: Frontend format check (oxfmt)
|
||||
working-directory: frontend
|
||||
run: bun run format:check
|
||||
|
||||
- name: Run Vitest (frontend)
|
||||
working-directory: frontend
|
||||
run: bunx vitest run
|
||||
|
||||
+101
-24
@@ -4,12 +4,16 @@
|
||||
# - push of a tag matching `v*` (e.g. `v0.2.0`) → full STABLE release,
|
||||
# publishes artifacts + signed updater manifest (`latest.json`) to the
|
||||
# tag's GH Release. This is the default Stable updater channel.
|
||||
# - workflow_dispatch (publish_preview=true) → builds the selected branch and
|
||||
# publishes a rolling `preview` PRERELEASE with its own signed
|
||||
# `latest.json` at releases/download/preview/. This feeds the opt-in
|
||||
# Preview updater channel (Settings → About → Update channel). The stable
|
||||
# `latest` release is untouched. Run this manually whenever you want to cut
|
||||
# a preview from `main`.
|
||||
# - schedule (nightly, 07:00 UTC) → rolling `preview` PRERELEASE built from
|
||||
# `main` with its own signed `latest.json` at releases/download/preview/.
|
||||
# Feeds the opt-in Preview updater channel (Settings → About → Update
|
||||
# channel). The `preview-gate` job skips the matrix on nights when `main`
|
||||
# didn't move, so an idle day costs only a ~30s gate job — keeping Preview
|
||||
# ≤24h behind `main` at a predictable ~1-matrix/day cost. The stable
|
||||
# `latest` release is untouched.
|
||||
# - workflow_dispatch (publish_preview=true) → the same preview build on
|
||||
# demand from the selected branch (e.g. to preview a feature branch, or to
|
||||
# refresh immediately without waiting for the nightly).
|
||||
# - workflow_dispatch (publish_preview=false) → on-demand build (prior
|
||||
# behavior; draft release named after the branch).
|
||||
#
|
||||
@@ -28,6 +32,10 @@ name: Desktop Release
|
||||
on:
|
||||
push:
|
||||
tags: ['v*']
|
||||
schedule:
|
||||
# 07:00 UTC daily — rolling `preview` prerelease from `main`. The
|
||||
# preview-gate job no-ops the matrix when main hasn't moved in a day.
|
||||
- cron: '0 7 * * *'
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
draft:
|
||||
@@ -121,8 +129,41 @@ jobs:
|
||||
working-directory: frontend
|
||||
run: node --experimental-strip-types --no-warnings --test ../tests/frontend/*.test.mjs
|
||||
|
||||
# Decide preview-vs-stable, and for nightly runs whether `main` actually
|
||||
# moved in the last day. Outputs gate the expensive matrix (`build`) and the
|
||||
# `preview-notes` job, so a no-commit night costs only this ~30s job.
|
||||
preview-gate:
|
||||
name: Preview gate
|
||||
runs-on: ubuntu-22.04
|
||||
outputs:
|
||||
is_preview: ${{ steps.decide.outputs.is_preview }}
|
||||
proceed: ${{ steps.decide.outputs.proceed }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 50
|
||||
- id: decide
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
event="${{ github.event_name }}"
|
||||
if [ "$event" = "schedule" ] || { [ "$event" = "workflow_dispatch" ] && [ "${{ inputs.publish_preview }}" = "true" ]; }; then
|
||||
echo "is_preview=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "is_preview=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
# Nightly: skip the matrix when main hasn't moved in the last day.
|
||||
if [ "$event" = "schedule" ] && [ -z "$(git log --since='25 hours ago' --oneline)" ]; then
|
||||
echo "No new commits on main in the last day — skipping nightly preview."
|
||||
echo "proceed=false" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "proceed=true" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
build:
|
||||
needs: test
|
||||
needs: [test, preview-gate]
|
||||
# Nightly runs with no new commits on main skip the 4-platform matrix.
|
||||
if: needs.preview-gate.outputs.proceed == 'true'
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
@@ -427,11 +468,14 @@ jobs:
|
||||
# the Windows MSI ProductVersion (which strips the prerelease → 0.3.6)
|
||||
# is also correctly above the last stable.
|
||||
- name: Stamp preview version
|
||||
if: github.event_name == 'workflow_dispatch' && inputs.publish_preview
|
||||
if: needs.preview-gate.outputs.is_preview == 'true'
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
CONF=frontend/src-tauri/tauri.conf.json
|
||||
# package.json is the single source of truth; tauri.conf.json reads its
|
||||
# version from it ("version": "../package.json"), so stamping
|
||||
# package.json restamps the whole bundle.
|
||||
CONF=frontend/package.json
|
||||
BASE=$(jq -r .version "$CONF")
|
||||
# MSI/WiX requires the semver pre-release identifier to be numeric-only
|
||||
# (and <= 65535). "preview.N" hard-fails the Windows bundler, so the
|
||||
@@ -471,11 +515,11 @@ jobs:
|
||||
# rolling `preview` prerelease for the updater's Preview channel.
|
||||
# Every other invocation — crucially the `v*` tag-push stable release
|
||||
# — evaluates these expressions to exactly their prior values.
|
||||
tagName: ${{ (github.event_name == 'workflow_dispatch' && inputs.publish_preview) && 'preview' || github.ref_name }}
|
||||
releaseName: ${{ (github.event_name == 'workflow_dispatch' && inputs.publish_preview) && 'OmniVoice Studio (Preview)' || format('OmniVoice Studio {0}', github.ref_name) }}
|
||||
tagName: ${{ (needs.preview-gate.outputs.is_preview == 'true') && 'preview' || github.ref_name }}
|
||||
releaseName: ${{ (needs.preview-gate.outputs.is_preview == 'true') && 'OmniVoice Studio (Preview)' || format('OmniVoice Studio {0}', github.ref_name) }}
|
||||
releaseBody: ${{ steps.changelog.outputs.body }}
|
||||
releaseDraft: ${{ (github.event_name == 'workflow_dispatch' && inputs.publish_preview) && 'false' || (inputs.draft || 'true') }}
|
||||
prerelease: ${{ (github.event_name == 'workflow_dispatch' && inputs.publish_preview) || false }}
|
||||
releaseDraft: ${{ (needs.preview-gate.outputs.is_preview == 'true') && 'false' || (inputs.draft || 'true') }}
|
||||
prerelease: ${{ needs.preview-gate.outputs.is_preview == 'true' }}
|
||||
updaterJsonPreferNsis: false
|
||||
includeUpdaterJson: true
|
||||
|
||||
@@ -661,8 +705,8 @@ jobs:
|
||||
# preview-only — stable `v*` releases keep their CHANGELOG section + the
|
||||
# appended checksums.
|
||||
preview-notes:
|
||||
needs: build
|
||||
if: github.event_name == 'workflow_dispatch' && inputs.publish_preview
|
||||
needs: [build, preview-gate]
|
||||
if: needs.preview-gate.outputs.is_preview == 'true'
|
||||
runs-on: ubuntu-22.04
|
||||
permissions:
|
||||
contents: write
|
||||
@@ -693,9 +737,39 @@ jobs:
|
||||
echo ""
|
||||
echo "$CONTRIB"
|
||||
} > /tmp/preview-notes.md
|
||||
gh release edit preview --repo "$REPO" --notes-file /tmp/preview-notes.md
|
||||
# --prerelease re-asserts the flag every run: a non-prerelease
|
||||
# `preview` release is eligible to become GitHub's "Latest", which is
|
||||
# the exact URL the Stable updater channel reads — so it must never
|
||||
# flip off.
|
||||
gh release edit preview --repo "$REPO" --prerelease --notes-file /tmp/preview-notes.md
|
||||
echo "Applied auto-generated release notes + contributors to the preview release."
|
||||
|
||||
- name: Verify preview updater manifest (prerelease + platform parity)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
REPO: ${{ github.repository }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
# The preview release must stay a prerelease (or it can hijack the
|
||||
# Stable channel's releases/latest endpoint), and its updater manifest
|
||||
# must cover every platform stable does (else those users — e.g. Intel
|
||||
# Mac — silently get no preview updates).
|
||||
is_pre=$(gh release view preview --repo "$REPO" --json isPrerelease -q .isPrerelease)
|
||||
test "$is_pre" = "true" || { echo "::error::preview release is not a prerelease"; exit 1; }
|
||||
curl -fsSL "https://github.com/$REPO/releases/download/preview/latest.json" -o /tmp/preview-latest.json
|
||||
curl -fsSL "https://github.com/$REPO/releases/latest/download/latest.json" -o /tmp/stable-latest.json
|
||||
python3 - <<'PY'
|
||||
import json, re
|
||||
prev = json.load(open("/tmp/preview-latest.json"))
|
||||
stab = json.load(open("/tmp/stable-latest.json"))
|
||||
v = prev.get("version", "")
|
||||
assert re.fullmatch(r"\d+\.\d+\.\d+-\d+", v), f"preview version not X.Y.Z-N: {v!r}"
|
||||
pk, sk = set(prev.get("platforms", {})), set(stab.get("platforms", {}))
|
||||
missing = sk - pk
|
||||
assert not missing, f"preview manifest missing platforms vs stable: {sorted(missing)}"
|
||||
print(f"preview manifest OK: {v} platforms={sorted(pk)}")
|
||||
PY
|
||||
|
||||
# ── Post-release version bump (versioning hard rule, owner-set 2026-06-11) ──
|
||||
# main is always last-release + 1 patch. The moment a stable v* tag is
|
||||
# released, bump the three version sources on main to the next patch so every
|
||||
@@ -718,23 +792,26 @@ jobs:
|
||||
RELEASED="${GITHUB_REF_NAME#v}"
|
||||
IFS=. read -r MAJ MIN PAT <<< "$RELEASED"
|
||||
NEXT="$MAJ.$MIN.$((PAT + 1))"
|
||||
CURRENT=$(jq -r .version frontend/src-tauri/tauri.conf.json)
|
||||
# frontend/package.json is the SINGLE SOURCE OF TRUTH: vite injects
|
||||
# __APP_VERSION__ from it, and tauri.conf.json reads its bundle version
|
||||
# from it ("version": "../package.json"). Read CURRENT from it.
|
||||
CURRENT=$(jq -r .version frontend/package.json)
|
||||
if [ "$(printf '%s\n' "$NEXT" "$CURRENT" | sort -V | tail -1)" = "$CURRENT" ] && [ "$NEXT" != "$CURRENT" ]; then
|
||||
echo "main is already at $CURRENT (>= $NEXT) — nothing to bump"; exit 0
|
||||
fi
|
||||
tmp=$(mktemp)
|
||||
jq --arg v "$NEXT" '.version = $v' frontend/src-tauri/tauri.conf.json > "$tmp"
|
||||
mv "$tmp" frontend/src-tauri/tauri.conf.json
|
||||
# frontend/package.json drives __APP_VERSION__ (vite.config.js) — the
|
||||
# first-run footer + every auto bug report. Keep it in lockstep too,
|
||||
# set absolutely (jq) so any prior drift self-heals. (#248-sweep finding)
|
||||
# Bump the canonical (package.json), set absolutely so any prior drift
|
||||
# self-heals. tauri.conf.json needs no edit — it derives from this.
|
||||
tmp=$(mktemp)
|
||||
jq --arg v "$NEXT" '.version = $v' frontend/package.json > "$tmp"
|
||||
mv "$tmp" frontend/package.json
|
||||
# The remaining files are CI-guarded mirrors (cargo/uv require a
|
||||
# literal; the version.py literal is the frozen-backend last resort) —
|
||||
# bump them in lockstep with the canonical.
|
||||
sed -i "0,/^version = \"$CURRENT\"/s//version = \"$NEXT\"/" frontend/src-tauri/Cargo.toml
|
||||
sed -i "0,/^version = \"$CURRENT\"/s//version = \"$NEXT\"/" pyproject.toml
|
||||
sed -i "0,/_FALLBACK_VERSION = \"$CURRENT\"/s//_FALLBACK_VERSION = \"$NEXT\"/" backend/core/version.py
|
||||
git config user.name "github-actions[bot]"
|
||||
git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
|
||||
git add frontend/src-tauri/tauri.conf.json frontend/package.json frontend/src-tauri/Cargo.toml pyproject.toml
|
||||
git add frontend/package.json frontend/src-tauri/Cargo.toml pyproject.toml backend/core/version.py
|
||||
git commit -m "chore(version): main -> $NEXT after $GITHUB_REF_NAME release"
|
||||
git push origin main
|
||||
|
||||
+862
-1
@@ -6,9 +6,693 @@ The format is loosely based on [Keep a Changelog](https://keepachangelog.com/).
|
||||
Versions track the desktop app (`tauri.conf.json` + `frontend/src-tauri/Cargo.toml`).
|
||||
The bundled TTS model package (`pyproject.toml`) is versioned independently.
|
||||
|
||||
## [Unreleased]
|
||||
## [0.3.8] — 2026-07-01
|
||||
|
||||
A stability-focused release that makes first-run and Windows "just work," ships
|
||||
**live, faster-than-real-time local dictation** and a **user pronunciation
|
||||
dictionary**, and gives **Settings a full redesign**. It clears the wave of
|
||||
**"Can't reach the local backend"** reports at the source — the 8 GB-card OOM
|
||||
crash, the slow-load future-scheduling break, a Windows-only WhisperX load
|
||||
failure, an ASR engine that couldn't load CTranslate2 on newer Linux/WSL, and
|
||||
both transcription **and generation** stalls that *looked* like a dead backend
|
||||
(a wedged GPU job now resets the worker pool and returns an actionable timeout)
|
||||
are all fixed or now fail with a clear, actionable message. **macOS gets native file drag-and-drop back**
|
||||
(including macOS 26 Tahoe). Downloads are faster out of the box (parallel
|
||||
segmented transfer on by default) and the Hugging Face token that speeds them up
|
||||
is front-and-center on setup. Plus multi-voice story casting, faster long-form
|
||||
previews on Windows, and a friendlier, more honest batch of error messages
|
||||
across dub, generate, and design (a corrupt-binary failure no longer poses as
|
||||
"out of memory," a bad model id self-heals, and a stale dub job resets cleanly).
|
||||
|
||||
### Added
|
||||
|
||||
- **"Autofit" translation quality — the dub keeps the video's timing.** A new
|
||||
quality alongside Fast and Cinematic: the LLM rewrites each translated line so
|
||||
its target-language reading time fits *within* the segment's slot (a strict
|
||||
"never overrun" bound, per-language pronunciation-speed aware), so long
|
||||
translations no longer force the audio into a stressed >1.3× time-stretch.
|
||||
Cinematic still applies its reflect/adapt polish; Autofit adds the hard
|
||||
fit-to-slot pass on top. Needs an LLM (below); falls back to Fast with a clear
|
||||
notice if none is set. (#838)
|
||||
- **A new LLM Providers settings page — bring your own high-quality LLM.**
|
||||
Settings → System → **LLM Providers** configures the LLM that powers Cinematic
|
||||
and Autofit translation. One page for **16 providers** — OpenAI, OpenRouter,
|
||||
Groq, Cerebras, Google AI (Gemini), Mistral, Cohere, NVIDIA, GitHub Models,
|
||||
Cloudflare, Hugging Face, SambaNova, SiliconFlow, plus **local Ollama / LM
|
||||
Studio** (fully offline, no key) and a **Custom** OpenAI-compatible endpoint.
|
||||
Paste a key, pick a model, **Test** the connection in one click, and "use for
|
||||
translation" to make it active. Keys are stored **encrypted** (the same
|
||||
at-rest protection as the HF token) and never leave the machine unless you
|
||||
choose a cloud provider; env vars still override for power users. The dub
|
||||
translate menu now routes you straight here when you pick a high-quality
|
||||
style without an LLM, instead of dead-ending on a toast. (#838)
|
||||
- **A dedicated Network pane.** The HTTP/SOCKS proxy and FFmpeg-path controls
|
||||
(previously buried in General → Advanced) are promoted to their own category.
|
||||
- **Factory reset in Storage.** A confirm-dialog-guarded action that clears the
|
||||
locally-saved UI preferences and reloads — without touching your voices,
|
||||
projects, or generated audio on disk.
|
||||
- **Proactive, highlighted "Install" affordance for translation engines.** When
|
||||
you pick a Dub translation engine whose optional package isn't installed yet
|
||||
(e.g. Google / DeepL via `deep_translator`), the Engine selector now surfaces a
|
||||
bright accent **Install** button *before* you hit Translate — no more
|
||||
discovering the missing package only via a translate-time 400. On a from-source
|
||||
install it one-click installs into the backend's own interpreter; on a
|
||||
read-only **packaged build** it opens a popover with the exact `uv pip install …`
|
||||
command (copy-to-clipboard), a one-click **Switch to Argos (bundled, offline)**
|
||||
escape hatch, and a docs link. The install command is single-sourced in the
|
||||
backend registry, so the button and the 400 error can never disagree. New guide:
|
||||
`docs/dubbing/translation-engines.md`.
|
||||
|
||||
- **A user pronunciation dictionary that actually changes the audio.** Settings →
|
||||
General → Pronunciation lets you teach the engine how to say tricky words —
|
||||
each entry replaces a term with a respelling (`GIF` → `jiff`) right before
|
||||
synthesis, so it works on **every** engine, not just one. Scope an entry
|
||||
Global or to a single language (a German rule never fires on an English
|
||||
render), with longest-match-first, word-boundary-aware, case-insensitive
|
||||
substitution. For one-offs, write `[[word|respelling]]` inline in your text —
|
||||
it overrides the dictionary for that occurrence and never persists. A built-in
|
||||
Test field previews the substitution with no model call. Pure text transform,
|
||||
identical on macOS/Windows/Linux; plain text stays byte-identical, existing
|
||||
data upgrades cleanly via an additive migration. (Expressive-TTS Spec 01)
|
||||
|
||||
- **Live, faster-than-real-time dictation via a new sherpa-onnx ASR engine.**
|
||||
Pick one of seven small ONNX speech-to-text models (Parakeet TDT v3/v2,
|
||||
streaming Zipformer EN/ZH/bilingual, streaming Paraformer, multilingual
|
||||
Whisper Tiny) for dictation, and watch text appear *as you speak*. Streaming
|
||||
models emit partials frame-by-frame and commit a sentence on natural silence;
|
||||
offline models surface live partials too by re-decoding a growing buffer.
|
||||
Runs CPU-only and identically on macOS, Windows, and Linux — no GPU, no cloud,
|
||||
no extra setup beyond a ~75–180 MB one-time model download. Parakeet TDT v3 is
|
||||
the recommended default; existing Whisper/MLX/NeMo dictation engines are
|
||||
untouched and still the fallback.
|
||||
|
||||
- **New "Voice" settings panel for live dictation.** Settings → Capture now
|
||||
leads with a Voice card: an Enable Voice Dictation toggle (showing your real
|
||||
registered shortcut), a Toggle/Hold mode switch, and a Speech Model dropdown
|
||||
that lists all seven models with offline/streaming + recommended badges, size,
|
||||
one-line descriptions, the installed checkmark, and inline download/delete —
|
||||
reusing the model-store download progress. Picking an uninstalled model starts
|
||||
its download and switches to it once ready. **Toggle vs Hold** is wired for
|
||||
both the desktop global hotkey and the in-app Ctrl/Cmd+Shift+Space fallback, so
|
||||
the behaviour is identical on macOS, Windows, and Linux. While you speak, the
|
||||
dictation pill shows the transcript building **live**, and words type straight
|
||||
into the focused field *as you speak* — self-correcting with backspaces as the
|
||||
streaming recognizer refines, with clipboard-paste as an automatic fallback.
|
||||
|
||||
- **Tagged scripts auto-cast into a multi-voice podcast/audiobook.** Paste a
|
||||
`[Alice] … [Bob] …` script into Stories and hit Auto-cast: it now recognizes
|
||||
the `[Name]` tag format (alongside the existing `NAME:` screenplay and quoted
|
||||
prose), builds the cast, and assigns a voice per character automatically.
|
||||
Editing one line only re-synthesizes that line on export (the chapter cache
|
||||
is content-addressed), and inline markers like `[pause]` / `[voice:…]` are
|
||||
never mistaken for speakers. (#487)
|
||||
- **A dedicated Contact page.** Discord, email, GitHub issues, and the project
|
||||
website (palash.dev) as clean one-tap rows, reachable from the footer — so
|
||||
reaching the maker is never more than a click away.
|
||||
- **Live download speed, remaining size, and ETA on first-run setup.** The
|
||||
Models & Engines step now shows `38% · 5.2 MB/s · 1.2 GB left · ~3m` while a
|
||||
model downloads, instead of a bare "downloading…". (#657)
|
||||
- **Turn off auto-play of the preview after a render.** New Settings →
|
||||
Appearance toggle, "Auto-play preview" (on by default) — switch it off so a
|
||||
finished clip doesn't start playing on its own, ideal when batch-generating
|
||||
segments. (#666)
|
||||
- **App version in the status bar, one click from updates.** A `v<version>`
|
||||
badge sits by the network icon in the bottom bar; clicking it opens Settings →
|
||||
Updates, and it grows a pulsing dot the moment a new version is ready to
|
||||
install. (#671)
|
||||
|
||||
### Changed
|
||||
|
||||
- **Settings is now a sidebar-nav hub instead of an 11-tab strip.** The whole
|
||||
page was rebuilt from scratch as a grouped left-rail navigator (with a
|
||||
search/filter box) plus a scrollable content pane — the macOS System Settings /
|
||||
VS Code layout. Settings are organized into four groups and sixteen
|
||||
categories: **General** (Appearance · General), **Voice & Engines** (Engines ·
|
||||
Models · Dictation · Pronunciation · Translation), **System** (Performance &
|
||||
Device · Storage · Network · Sharing & Remote · Credentials), and **App**
|
||||
(Updates · Privacy & Reporting · Logs · About). Every existing control keeps
|
||||
its behavior and store/API bindings — this is a reorganization, not a rewrite.
|
||||
Typing in the search box filters the category list and jumps to the first
|
||||
match, and the rail collapses to a dropdown navigator below 760px so the full
|
||||
IA stays reachable on a narrow window. Categories whose changes need a backend
|
||||
restart (Models, Performance & Device, Sharing & Remote) carry a "restart
|
||||
required" badge.
|
||||
|
||||
- **The Settings pages got a full redesign — cleaner, denser, responsive.** A
|
||||
shared design system replaces the old patchwork: a left icon nav-rail,
|
||||
sentence-case section titles (no more debug-log uppercase), exactly one muted
|
||||
description per row, unified toggles/inputs, full-width content with proper
|
||||
padding, and horizontal font/theme pickers. Premium and compact instead of
|
||||
sparse and cluttered, and it adapts cleanly to window width. (#686, #690, #696)
|
||||
- **Adding a Hugging Face token on first-run is now a one-line input right by
|
||||
Continue.** Was a bulky card buried at the bottom of the model list; it's now a
|
||||
compact "paste a token, Save" bar pinned next to the "Waiting for required
|
||||
models…" button, so you can add it (for faster, authenticated downloads)
|
||||
without scrolling. (#687, #688)
|
||||
- **First-run setup is calmer and surfaces the best models for your machine.**
|
||||
Dimmed and tightened the setup descriptions (less wordy, more compact). The
|
||||
"Models & engines" step now shows the **platform-tuned** optional models up-front
|
||||
with a green "recommended" tag and their catalog note — e.g. MLX Whisper on
|
||||
Apple Silicon, CUDA-tuned variants on NVIDIA — instead of burying every optional
|
||||
model behind the fold (the universal long tail still folds).
|
||||
|
||||
- **Donations now go through Ko-fi or PayPal (GitHub Sponsors removed).** GitHub
|
||||
Sponsors isn't available, so the Support page no longer routes there: pick an
|
||||
amount (now $10 / $20 / $50) and then choose Ko-fi or PayPal — PayPal carries
|
||||
the amount straight into checkout. `.github/FUNDING.yml` and the README badges
|
||||
were updated to match.
|
||||
- **Simplified the Commercial License page.** Trimmed the six-tile benefit grid
|
||||
and FAQ down to the three things that actually drive the decision (you own the
|
||||
output, no per-minute cost, direct support) plus one clear "request a quote"
|
||||
contact — less wall-of-text, faster to act on.
|
||||
- **Model downloads are faster out of the box.** The built-in multi-connection
|
||||
(segmented) downloader — parallel byte-ranges with live speed/ETA — is now on
|
||||
by default, so the legacy-LFS path is no longer single-stream and slow. It
|
||||
falls back to the normal download on any error, so it can never compromise a
|
||||
correct install (`OMNIVOICE_SEGMENTED_DOWNLOAD=0` to disable). (#669)
|
||||
- **The Hugging Face token is now front-and-center on first-run.** Was a
|
||||
collapsed "advanced" fold almost nobody opened; it's now a prominent card right
|
||||
above Continue, framed around what it actually buys you — authenticated, faster,
|
||||
more reliable downloads (higher rate limits, fewer stalls) — with a one-click
|
||||
"get a free token" link. (#657, #669)
|
||||
### Fixed
|
||||
|
||||
- **Bug reports redact more secrets and every Windows username casing.** The
|
||||
opt-in bug-report scrubber now catches more credential shapes (JWT/Bearer,
|
||||
Google, Slack, AWS keys, and `?token=`/`?api_key=` URL secrets), redacts
|
||||
Windows home paths regardless of `Users`/`users` casing, and stops a superstring
|
||||
username (`/Users/john` vs `/Users/johnny`) from leaking a fragment. The
|
||||
prefilled-issue URL is now bounded by its *encoded* length so a large report
|
||||
can't silently truncate. Nothing new leaves the machine — this only makes the
|
||||
existing local-first, user-reviewed report stricter. (#856)
|
||||
|
||||
- **A hung TTS generate can no longer brick the backend ("Can't reach the local
|
||||
backend").** A GPU job that wedges on some Windows + CUDA setups occupies its
|
||||
worker forever — Python can't cancel the thread — so on the 1–2 worker pools we
|
||||
ship, one stuck job starved every other request and the next action surfaced as
|
||||
the misleading "Can't reach the local backend" even though the process was
|
||||
alive. ASR/dub/model-load already bounded and reset the pool on hang (#730); but
|
||||
**every generate path** — Studio synthesis, the streaming path, batch, the dub
|
||||
per-segment + preview render, archetype previews, and the OpenAI-compatible
|
||||
`/v1/audio/speech` API — was still an unguarded GPU dispatch, and the residual
|
||||
reports all failed on `generate:start (audio)`. Every one is now bounded by the
|
||||
same wall-clock guard (`OMNIVOICE_GENERATE_TIMEOUT_S`, default 300s) that
|
||||
abandons the wedged worker and rebuilds the pool, so capacity is restored
|
||||
automatically and you get an actionable timeout instead of a dead backend.
|
||||
Closes the whole class of GPU-job-hang reports (#851 — #850, #802, #755, #723,
|
||||
#721, and the 0.3.7 cohort, all tracked in #730).
|
||||
|
||||
- **An unsupported GPU now falls back to CPU instead of 500-ing every generate.**
|
||||
When the installed PyTorch build has no kernels for your GPU's compute
|
||||
capability — a too-old card (Pascal / GTX 10-series) or a too-new one
|
||||
(Blackwell RTX 50-series on pre-cu128 wheels) — CUDA failed at launch with the
|
||||
cryptic `CUDA error: no kernel image is available for execution`. The backend
|
||||
now detects that up front and runs on CPU (slower, but it works), and any raw
|
||||
occurrence is reported as "your GPU isn't supported — switch to CPU or install a
|
||||
matching PyTorch," not a Flush-the-memory dead end. Force the GPU anyway with
|
||||
`OMNIVOICE_FORCE_CUDA=1`. (#756)
|
||||
|
||||
- **The "TRANSLATION FAILED" banner now dismisses and clears itself.** The Dub
|
||||
translation-error banner used to be sticky — it survived a successful re-try and
|
||||
never went away. It now has a close (×), auto-clears on the next corrective
|
||||
action (re-translating, changing the engine, or installing the package), and
|
||||
self-clears after a short timeout — fixing the whole class of translate/pipeline
|
||||
banners that outlived the state that caused them.
|
||||
|
||||
- **Dubbing a video URL no longer fails with "ffmpeg is not installed."** yt-dlp
|
||||
downloads video and audio as separate streams and muxes them with ffmpeg, but
|
||||
it only looked on PATH — so on Windows (where OmniVoice's ffmpeg is a bundled
|
||||
sidecar / `imageio-ffmpeg` binary off PATH) the merge aborted before the dub
|
||||
could start. yt-dlp is now pointed at the same ffmpeg OmniVoice resolves. (#712)
|
||||
- **A synth that succeeded no longer 500s because of a history-logging hiccup.**
|
||||
If the local database somehow missed schema init, recording the clip to
|
||||
generation history failed with *"no such table: generation_history"* and
|
||||
surfaced as a 500 — even though the audio had already been generated and saved.
|
||||
The write now self-heals the schema and retries, and a history-logging failure
|
||||
never fails the generation: you get your audio regardless. (#710)
|
||||
- **Long-video dubs no longer spike RAM during assembly.** Dub generation used
|
||||
to hold every segment's audio in memory until the whole track was mixed, so a
|
||||
50-video batch or a single feature-length dub could exhaust RAM and crash. Each
|
||||
segment now streams to disk as it's rendered and the final track is assembled
|
||||
from those files via a 30s-chunk memmap writer, keeping memory flat regardless
|
||||
of video length. Per-segment download WAVs and the final track stay correctly
|
||||
watermarked (marked once at synthesis, no double-mark), and zero/negative-length
|
||||
segments no longer crash the run. (#639)
|
||||
- **A corrupt or wrong-architecture native component no longer masquerades as
|
||||
"out of memory."** A synth failure caused by a bad `.dll`/`.pyd`/`.exe` on
|
||||
Windows (`[WinError 193] %1 is not a valid Win32 application` — e.g. torch,
|
||||
ffmpeg, or an engine binary) was labelled *"ran out of memory — try Flush,"*
|
||||
sending users down the wrong path. It now says the component is corrupt or
|
||||
built for the wrong architecture and to reinstall/repair it. (#705)
|
||||
- **A "[Errno 32] Broken pipe" mid-generation no longer poses as "out of
|
||||
memory."** When the desktop app that launched the backend closes or relaunches,
|
||||
the backend's output pipe breaks and a synth can fail with `[Errno 32] Broken
|
||||
pipe`. That was labelled *"ran out of memory — try Flush,"* which never helps;
|
||||
it now tells you the backend lost its pipe and to restart the app. (#715)
|
||||
- **Settings content no longer sprawls or spills out of view.** The content
|
||||
column capped at 1280px, so on wide windows rows stretched edge-to-edge with a
|
||||
big empty gap between each label and its control ("too spread out"), and a few
|
||||
panels (API keys, the shared button rows, appearance scale) used rigid pixel
|
||||
widths that pushed controls past the card's padding on narrow content. Now the
|
||||
content sits at a readable measure (a single `--settings-measure` token), the
|
||||
shared button/badge rows wrap instead of overflowing, rigid widths can shrink,
|
||||
and rows decide whether to sit side-by-side or stack based on their **actual**
|
||||
width (a container query) — not the viewport, which the 168px nav rail skews.
|
||||
Everything stays inside its padding, edge to edge, on every width. (#696)
|
||||
- **File drag-and-drop works on macOS again.** The app's drop zones use HTML5
|
||||
file drops, but Tauri intercepts OS drag-and-drop by default (`dragDropEnabled`)
|
||||
and swallowed the files before the webview saw them — most visibly on macOS
|
||||
WKWebView, and fully broken on macOS 26 (Tahoe), where dropping a file did
|
||||
nothing. Disabled the interception so the webview handles native HTML5 drops
|
||||
on every platform. (#700)
|
||||
- **A misconfigured `OMNIVOICE_MODEL` no longer bricks model load with a 500.**
|
||||
A stale or leaked TTS *engine id* (e.g. `omnivoice`) reaching the model loader
|
||||
used to fail every launch with *"omnivoice is not a local folder and is not a
|
||||
valid model identifier."* It now self-heals — only a real HF repo id
|
||||
(`org/repo`) or an explicit local path is honored; anything else falls back to
|
||||
the default with a logged warning. Every consumer of the setting routes through
|
||||
the same resolver, so a bad value also can't silently disable model warm-up,
|
||||
mislabel the Settings checkpoint, or get baked into an exported persona bundle.
|
||||
(#693)
|
||||
- **ASR no longer crashes the dub/transcribe preflight when CTranslate2's native
|
||||
library can't load.** On hardened kernels / newer glibc (e.g. WSL2) the
|
||||
CTranslate2 `.so` is rejected with *"cannot enable executable stack"* — an
|
||||
OSError the WhisperX/faster-whisper checks didn't catch, so it took down the
|
||||
whole preflight. They now report the engine as unavailable and auto-detect
|
||||
falls back to PyTorch-Whisper instead of dead-ending. (#692)
|
||||
- **A wedged transcription can no longer take the whole backend offline ("Can't
|
||||
reach the local backend").** On some Windows + CUDA setups a whisperx/CTranslate2
|
||||
transcribe hangs hard and never returns. Because ASR shares a small (1–2 worker)
|
||||
GPU pool with TTS, one stuck worker starved every other request — so the next
|
||||
thing you did (often a TTS *generate*) failed with "can't reach backend" even
|
||||
though the process was alive. Two fixes: every transcribe path — whole-file
|
||||
(dub whole-file, batch, live dictation) **and** the chunked dub stream — is now
|
||||
wall-clock **bounded** like the dub QC / dictation / OpenAI paths already were;
|
||||
and on timeout the poisoned GPU worker is **abandoned and the pool rebuilt**, so
|
||||
capacity is restored without restarting the app. You still get an actionable
|
||||
message (Flush VRAM / pick a smaller ASR model) for the durable fix. (#730)
|
||||
- **The stale-dub-session recovery now also covers the first upload/ingest, not
|
||||
just retry/import.** A dubbing job that vanished server-side during the initial
|
||||
transcribe flow showed the scary *"Job not found … report a bug"* toast; it
|
||||
now resets gracefully and invites a fresh upload, like the other paths. (#695)
|
||||
- **In-app preview of finished audiobooks/stories now plays on Windows.**
|
||||
The preview decoded the entire render into one in-memory PCM buffer via Web
|
||||
Audio `decodeAudioData`, which fails on long-form `.m4b`/AAC under WebView2
|
||||
(`EncodingError: Unable to decode audio data`), and the blob-URL fallback can't
|
||||
play in a Tauri `<audio>` element — so nothing played. The fallback now uploads
|
||||
to the preview endpoint (ffmpeg-extracts a streamable WAV) and plays the HTTP
|
||||
URL, the same path video previews use. Short TTS previews are unchanged. (#653)
|
||||
|
||||
- **First-run setup splash no longer shows a raw `bootstrap.lines` key in English.**
|
||||
The log-line counter string was present in 4 locales but missing from the `en`
|
||||
reference, so English (and 16 other locales falling back to it) rendered the
|
||||
literal key instead of "{{count}} lines". Added it to `en`. Also removed 160
|
||||
dead `gallery.cat_*` keys (renamed to `archetypes.use_*` long ago) orphaned
|
||||
across 20 non-English locales, clearing the i18n orphan-key advisory.
|
||||
|
||||
- **Backend no longer hangs on startup (unreachable, no error) on Apple-Silicon Macs.**
|
||||
The MCP session manager could hang on its anyio task group during lifespan
|
||||
startup (observed on M1, #632); because that start was awaited before the server
|
||||
began serving, "Application startup complete" never fired and the whole backend
|
||||
was unreachable. The MCP start is now timeout-bounded (`OMNIVOICE_MCP_START_TIMEOUT_S`,
|
||||
default 30s) — a hang becomes a logged warning and the backend serves normally
|
||||
without MCP, instead of wedging. (#632)
|
||||
|
||||
- **Dubbing a URL no longer fails with `[Errno 22] Invalid argument` on Windows.**
|
||||
yt-dlp stamps the downloaded file's modified-time with the video's upload
|
||||
date; an out-of-range/invalid timestamp makes the `os.utime` call raise
|
||||
`[Errno 22]` and aborts the whole URL ingest. OmniVoice downloads to a throwaway
|
||||
file and never uses its mtime, so it now skips the stamp entirely
|
||||
(`updatetime=False`). (#642)
|
||||
|
||||
- **Dubbing a YouTube link that 403s now retries with a different player
|
||||
client.** Some videos serve their formats signature-protected to the default
|
||||
player client, so the media download fails with `HTTP Error 403: Forbidden`
|
||||
even though extraction worked — and a plain retry keeps 403ing. The URL
|
||||
download now escalates the YouTube player client (tv → android → web_safari)
|
||||
on a 403, which commonly bypasses it, before surfacing the actionable error.
|
||||
(#625)
|
||||
- **A synth glitch that produced unreadable audio is now caught instead of a
|
||||
misleading "out of memory".** A numerical glitch in the model (seen on Apple
|
||||
Silicon/MPS) could leave NaN/∞ samples, which wrote a WAV that then failed
|
||||
decoding with an opaque `ffmpeg returned error code: 183 / Invalid data` — and
|
||||
the generic error handler labelled it "ran out of memory". Non-finite samples
|
||||
are now sanitized to silence before any encode (so the WAV is always
|
||||
decodable), and a genuine decode failure is reported as "unreadable audio —
|
||||
Flush and regenerate", not OOM. (#629)
|
||||
- **A silent startup hang now leaves a diagnostic instead of nothing.** On some
|
||||
setups the backend could load all model weights and then hang forever before
|
||||
"Application startup complete" — no error, no crash, an unusable app (reported
|
||||
as a Mac M1 hang after `Loading weights: 527/527`, #632). A startup watchdog
|
||||
now dumps every thread's stack to the error log if startup stalls past a
|
||||
window (default 5 min, `OMNIVOICE_STARTUP_WATCHDOG_S` to tune, `0` to disable),
|
||||
so the deadlock is captured rather than invisible. It's disarmed the instant
|
||||
startup finishes, so a normal (even slow-first-download) boot never trips it.
|
||||
(#632)
|
||||
- **First-run demo voice is back.** The bundled demo clip
|
||||
(`backend/assets/samples/demo_voice.wav`) was a build artifact that never got
|
||||
committed, so it shipped absent — onboarding logged "Demo audio not found" and
|
||||
seeded nothing, leaving a brand-new install with an empty Launchpad and no
|
||||
`/demo_audio` route. The clip is now committed (it's already un-ignored and
|
||||
bundled via the Tauri `backend` resource), so first-run seeds the demo voice
|
||||
on every platform; onboarding still degrades gracefully (with a regenerate
|
||||
hint) if it's ever absent. (#621)
|
||||
- **Multi-speaker dubbing: two speakers' turns merged onto one line are now
|
||||
split apart.** Segmentation groups words into sentences *before* diarization
|
||||
runs, so a back-and-forth exchange could land in a single segment; the speaker
|
||||
pass then only *relabelled* that segment with its majority speaker, losing the
|
||||
turn boundary (the second half of #486; the per-speaker voice auto-assign was
|
||||
fixed earlier in #490). A new post-diarization pass re-splits any segment whose
|
||||
words span more than one speaker at the word-level boundary, assigning each
|
||||
piece its own speaker. Single-speaker segments pass through **byte-for-byte
|
||||
unchanged**, so single-speaker dubs and their timing never move, and a lone
|
||||
mis-attributed word (diarization noise) is smoothed rather than causing a
|
||||
spurious split. (#486)
|
||||
- **Designed voices saved with a bad style no longer render wrong or crash
|
||||
generation.** A designed voice could persist an `instruct` the engine
|
||||
validator rejects — either the literal `"[object Object]"` from an old build,
|
||||
or freeform prose typed into the style field — which made every generation or
|
||||
dub that used the voice fail with `Unsupported instruct items found in …`
|
||||
(surfacing to users as a 400/500 and, when it tore down mid-render, "Can't
|
||||
reach the local backend"). The previous fix only *blanked* `"[object Object]"`,
|
||||
which silently dropped the design — so an Indonesian **female** voice came out
|
||||
**male**. Now the stored instruct is sanitized down to valid tags at every
|
||||
seam (save, edit, and when a profile drives Generate or Dub), and when the
|
||||
stored value is unusable the tags are **rebuilt from the design's saved
|
||||
category picks (`vd_states`)** so the intended gender/age/pitch/accent survive.
|
||||
A migration (0007) heals existing poisoned profiles in place — no reinstall,
|
||||
no manual fix. (#550 #571 #594 #596)
|
||||
- **"Transcribe stream dropped … Likely ASR backend failed to load" now shows
|
||||
the *real* reason.** When transcription failed to load its ASR model (the
|
||||
reported case was WhisperX on Windows — typically a faster-whisper /
|
||||
CTranslate2-cuDNN mismatch, a missing model download, or the torch-2.6
|
||||
weights-only VAD regression), the UI dead-ended on a generic "stream dropped"
|
||||
message with no actionable cause. Two root causes: (1) WhisperX loads lazily
|
||||
*inside* transcription, so the load failure was buried in per-chunk errors and
|
||||
retried on every chunk; the transcribe pre-flight now eagerly loads the ASR
|
||||
model (new `ASRBackend.ensure_loaded()`), surfacing the genuine cause once, up
|
||||
front, as a structured error. (2) Pre-flight and audio-load errors closed the
|
||||
SSE stream with a bare `error` and no terminal `done`, so the browser's native
|
||||
EventSource connection-drop could race and win against the structured error —
|
||||
discarding the real cause and falling back to the generic message; every
|
||||
terminal error now emits `done`, and the frontend latches the structured cause
|
||||
so a connection drop can't overwrite it. Net: WhisperX load failures are
|
||||
diagnosable instead of a silent dead-end. Fail-before/pass-after regression
|
||||
test included. (#578)
|
||||
- **Dubbing: the PLAY button on the dubbed-video preview did nothing.** Same
|
||||
autoplay-policy trap that #510 fixed for the standalone audio player, but the
|
||||
dub editor's timeline player was missed. WaveSurfer builds its `AudioContext`
|
||||
at mount — before any user gesture — so on Windows WebView2 (and Linux
|
||||
Firefox/Chrome, Android Chrome) it stays `"suspended"`; `playPause()` then
|
||||
resolves with no sound and the preview just sits there. Every playback entry
|
||||
point in the dub timeline (the toolbar Play button and the per-segment "play
|
||||
this slot") now resumes the context via the shared `unlockAudio()` on the
|
||||
click before starting playback, and swallowed play() rejections are logged
|
||||
instead of hidden. A source-contract regression test pins the invariant so a
|
||||
future refactor can't quietly reintroduce a silent play path. macOS is
|
||||
unaffected (its context was never blocked). (#595)
|
||||
- **Voice design: the script text field couldn't be expanded.** The Script
|
||||
textarea was a `flex: 1` item inside a flex column, so flex-grow recomputed
|
||||
its height on every reflow and snapped the user's drag back — `resize:
|
||||
vertical` is silently ignored on a flex-grown item in Chromium/WebView2. The
|
||||
field now owns its own height (starts taller, and the corner grip grows it
|
||||
reliably on every platform). (#595)
|
||||
- **An interrupted model download now self-repairs instead of dead-ending.**
|
||||
When the OmniVoice TTS cache was missing weight shards (the usual aftermath of
|
||||
an interrupted first download), the next synthesize failed with a 500 and a
|
||||
"delete the model and install it again" instruction — a manual dead-end. The
|
||||
backend now detects the truncated-cache error on load, re-fetches just the
|
||||
missing files via `snapshot_download` (already-present blobs are skipped, so a
|
||||
near-complete cache repairs in seconds and a healthy cache is never touched),
|
||||
and retries the load automatically. Offline mode (`HF_HUB_OFFLINE`) is
|
||||
respected — repair never makes a network call the user opted out of — and if
|
||||
the re-fetch still can't fix it, the actionable delete-and-reinstall message
|
||||
is preserved as the fallback. (#581) The repair now also **retries** the
|
||||
re-fetch (3 attempts, resuming each time) so a single transient blip — the very
|
||||
thing that interrupts a download in the first place — doesn't bounce you back
|
||||
to a manual reinstall; tune with `OMNIVOICE_MODEL_REPAIR_RETRIES`. And if a
|
||||
resume-repair still won't load — the signature of a *corrupt* file that kept
|
||||
its size, which a resume trusts and never re-fetches — it now **force
|
||||
re-downloads** the model files once before giving up, so even a bit-rotted
|
||||
cache self-heals without a manual reinstall. (#739)
|
||||
- **Dubbing a YouTube URL no longer dies on a transient "Broken pipe."**
|
||||
Pasting a video link could fail outright with `download: Unable to download
|
||||
video: [Errno 32] Broken pipe` — a broken pipe raised while the write side of
|
||||
a pipe closes mid-stream (a killed ffmpeg merge child, a CDN reset during
|
||||
muxing). yt-dlp's own per-fragment retries don't cover that case, so a single
|
||||
transient blip aborted the whole ingest. The URL download now retries up to
|
||||
twice on broken-pipe / network-drop failures, wiping the partial download
|
||||
between attempts, and only surfaces the (already-actionable) "connection
|
||||
dropped — just retry" hint after the retries are exhausted. Unsupported links
|
||||
still fail fast with their own hint — no wasted retries. (#579, #598)
|
||||
- **`No module named 'omnivoice'` on installs whose venv lost its editable
|
||||
record.** An interrupted or offline `uv sync` (common during an in-place
|
||||
upgrade) could install all dependencies yet never lay the editable install of
|
||||
the project's own `omnivoice` package — or an antivirus quarantine could
|
||||
remove it. The venv still started uvicorn, so the bootstrap's health gate
|
||||
passed it through, and the app only failed at the first generate/dub with
|
||||
`No module named 'omnivoice'`. The bootstrap now also verifies `omnivoice` is
|
||||
importable (via a cheap `find_spec`, no torch load) and forces a repair
|
||||
`uv sync` that re-lays the editable install when it isn't; the backend also
|
||||
resolves `omnivoice` from its bundled source tree at runtime as a safety net.
|
||||
No reinstall needed — relaunch and it self-repairs. (#564)
|
||||
- **"cannot schedule new futures after shutdown" no longer breaks generate/dub
|
||||
after a slow first load.** When a model load timed out, the backend reset its
|
||||
GPU worker pool to recover — but several request handlers had captured the old
|
||||
pool object at import time and kept submitting to it, so every subsequent
|
||||
generate, dub, transcribe, or translate failed with `cannot schedule new
|
||||
futures after shutdown` (a 500, or "Can't reach the local backend" when it
|
||||
took the worker down). The GPU pool is now a single self-healing handle whose
|
||||
worker pool is rebuilt on demand, so a reset can never strand an in-flight or
|
||||
later request. No settings change; the recovery is automatic. (#589 #599)
|
||||
- **Transcription / dubbing works on Windows again.** WhisperX failed to load on
|
||||
Windows because speechbrain's guard that suppresses stray optional-integration
|
||||
imports used a POSIX-only path check, so a `k2_fsa` import error aborted the
|
||||
whole transcription. Fixed cross-platform — covers the entire class of optional
|
||||
integrations, not just k2. (#630 #611 #647)
|
||||
- **A slow transcription no longer looks like a dead backend.** Whole-file
|
||||
transcribe paths (dub QC, dictation, OpenAI-compat) ran unbounded, so a
|
||||
VRAM-starved `large-v3` could spin for minutes and hold a GPU worker — surfacing
|
||||
as "Can't reach the local backend". They're now time-bounded and return a clear,
|
||||
actionable 504 (free VRAM / pick a smaller ASR model / use CPU) instead of
|
||||
hanging. New troubleshooting section documents it. (#656)
|
||||
- **Windows preview playback fixed.** The audiobook/clone preview's streaming
|
||||
fallback fetched `localhost`, which on Windows resolves to IPv6 and missed the
|
||||
IPv4-only backend — so previews failed with "decode error" / "no supported
|
||||
sources". The preview API now targets `127.0.0.1` (matching the main client),
|
||||
and the expected decode→stream fallback is logged calmly instead of as a scary
|
||||
error. (#653 #659)
|
||||
- **A stale dub session resets cleanly instead of erroring.** Reopening the Dub
|
||||
tab after the backend restarted tried to resume a job that no longer existed and
|
||||
surfaced "Job not found" as a bug-report error. It now quietly clears the dead
|
||||
session and invites a fresh upload. (#660)
|
||||
- **A bad voice-style instruct is a clear 400, not a scary 500.** Typing free-form
|
||||
prose (or a non-English description) into the style/instruct field returned a
|
||||
500 telling you to Flush for memory you never ran out of; it now returns a clean
|
||||
400 that lists the valid style tags. The Voice Clone UI also drops unrecognized
|
||||
style text locally and generates anyway. (#664 #612)
|
||||
- **The ⊕ Insert token popover stays on screen.** On Voice Clone it could grow
|
||||
tall enough to clip off the top of the window; it's now a compact, scrollable
|
||||
box anchored above the button. (#672)
|
||||
- **First-run no longer hangs on Apple Silicon.** The MCP session-manager startup
|
||||
is now timeout-bounded so a slow/stuck mount can't wedge the whole backend boot
|
||||
on M1. (#632)
|
||||
|
||||
### CI
|
||||
|
||||
- **Feature-coverage test system.** A backend route-inventory test diffs all 213
|
||||
HTTP/WebSocket endpoints against a committed snapshot (plus a critical-endpoint
|
||||
guard and a route-count floor), and a frontend feature-coverage test asserts
|
||||
every app mode is wired to a page and every feature has its i18n namespace — so
|
||||
an endpoint or page silently disappearing now fails CI on every PR.
|
||||
- **`bun desktop` no longer kills its own dev backend.** The dev launcher runs the
|
||||
API and the Tauri app side-by-side, but the app's backend manager would "take
|
||||
ownership" of port 3900 and kill the API the moment it booted (before it was
|
||||
healthy), tearing the whole session down. The dev app now sets
|
||||
`TAURI_SKIP_BACKEND` so it attaches to the running API instead of fighting it —
|
||||
production launch is unaffected. (#745)
|
||||
|
||||
## [0.3.7] — 2026-06-20
|
||||
|
||||
A stabilization release that clears the wave of issues reported on the 0.3.6
|
||||
line — across voice design, dubbing, transcription, install, and the Linux/web
|
||||
UI — and lands two more opt-in cloning engines. The throughline is **non-English
|
||||
correctness and cross-platform playback**: cloned and designed voices now hold
|
||||
their language end-to-end, and audio plays inline in Linux/Android browsers,
|
||||
not just macOS. It also carries the v0.3.6 startup-crash fixes, so anyone still
|
||||
hitting "Can't reach the local backend" on v0.3.5/v0.3.6 only needs to update.
|
||||
|
||||
### Added
|
||||
|
||||
- **Two opt-in heavyweight TTS engines: MOSS-TTS-v1.5 (8B) and dots.tts (2B).**
|
||||
Both are zero-shot voice-cloning engines, each running in its own isolated
|
||||
subprocess venv (they pin a `transformers` version that conflicts with the
|
||||
parent's `>=5.3` — MOSS `==5.0`, dots.tts `==4.57`) via the same dedicated-venv
|
||||
pattern as IndexTTS-2, so they can't disturb the default install or its
|
||||
lockfile. Point `OMNIVOICE_MOSS_TTS_V15_DIR` / `OMNIVOICE_DOTS_TTS_DIR` at a
|
||||
local clone to enable. CUDA/CPU only — neither claims Apple-Silicon MPS, and
|
||||
dots.tts is gated off on Windows (upstream is Linux/macOS only). See
|
||||
[docs/engines/moss-tts-v15.md](docs/engines/moss-tts-v15.md) and
|
||||
[docs/engines/dots-tts.md](docs/engines/dots-tts.md). (#498)
|
||||
|
||||
### Fixed
|
||||
|
||||
- **Non-English voices drifted to English / the wrong language.** Three
|
||||
independent root causes, all in the language path: (1) a voice profile's
|
||||
stored language was never read back into generation, so a German archetype
|
||||
that *previewed* in German *generated* in English (the preview passed the
|
||||
language; the user's Generate call didn't); (2) the audiobook/longform synth
|
||||
hardcoded `language=None`, letting the engine re-autodetect per chunk so a
|
||||
non-English clone could flip language mid-render on short/ambiguous lines; and
|
||||
(3) the duration estimator weighted Unicode combining marks at zero, so
|
||||
decomposed (NFD) diacritic text — common for Vietnamese — under-allocated
|
||||
frames and came out rushed. The profile/request language is now threaded
|
||||
through both the single-shot and longform paths (request wins, profile fills
|
||||
the gap), and text is NFC-normalized before duration estimation. Each fix has
|
||||
a fail-before/pass-after regression test. (#533, #505, #502)
|
||||
- **Audio playback on Linux Firefox/Chrome and Android Chrome.** Two separate
|
||||
root causes both masquerade as "the play button doesn't work" on non-macOS
|
||||
browsers — and both are invisible when developing on macOS, which is why they
|
||||
shipped. (1) The backend served `.wav` / `.flac` with Python's default
|
||||
`audio/x-wav` / `audio/x-flac` (vendor-experimental, never IANA-registered);
|
||||
macOS CoreAudio MIME-sniffs leniently and plays anyway, but Linux FFmpeg and
|
||||
Android ExoPlayer strictly honor the declared type and prompt to download.
|
||||
Fixed by registering the canonical `audio/wav` / `audio/flac` types before
|
||||
any `StaticFiles` mount. (2) WaveSurfer's `AudioContext` is constructed at
|
||||
component-mount time — i.e. before any user gesture — so on Linux FF/Chrome
|
||||
and Android Chrome it stays `suspended`, `decodeAudioData` hangs, the
|
||||
`ready` event never fires, and the play button never enables. macOS
|
||||
Safari/Chrome auto-resume on first interaction. Fixed by patching
|
||||
`window.AudioContext` to track every instance and resuming them on the first
|
||||
`pointerdown` / `keydown` / `touchstart`, plus resuming inline on the play
|
||||
click itself. The MIME fix has a backend regression test; the unlock path
|
||||
has a Vitest unit test covering idempotency, post-unlock contexts, and
|
||||
error isolation. (#510)
|
||||
- **Voice Studio "Save design as profile" poisoned the profile with
|
||||
"[object Object]" and then 400'd every generation** ("Unsupported instruct
|
||||
items found in [object Object]"). The save passed the instruct *builder
|
||||
object* to the form instead of its string. Fixed at the source + defended with
|
||||
a coercion helper; the engine now tolerates the sentinel, and a migration
|
||||
heals already-saved profiles. (#550, #545, #542, #537, #530, #525)
|
||||
- **Profile / persona / consent endpoints 500'd with `no such column:
|
||||
consent_audio_path`** (and the same class for `kind`/`vd_states`/…) after an
|
||||
in-place upgrade. The alembic migration existed but couldn't always apply
|
||||
(stamped at a removed revision, or alembic not importable) and the failure was
|
||||
swallowed. The runtime schema now self-heals — it ADDs any missing additive
|
||||
column from the canonical schema on startup. (#552, #547)
|
||||
- **Stories: the global reading-speed slider was ignored by preview and stem
|
||||
export.** The #415 global speed only flowed through the full longform export;
|
||||
per-segment preview and stem export still resolved a hardcoded `track.speed ||
|
||||
1.0`, so audio played at 1.0× even with the global set to e.g. 0.70×. A shared
|
||||
`effectiveSpeed(track, global)` helper (per-line override → global → engine
|
||||
default) now drives all three generation paths. (#508)
|
||||
- **Generate / Settings / Clone buttons were missing / unpressable on Linux.**
|
||||
The UI-scale fix round-trips correctly on Chromium, but older WebKitGTK treats
|
||||
`zoom` as a layout no-op, leaving a ~23% black band that pushed the bottom CTAs
|
||||
off-screen. The shell now probes the engine and fills the window when `zoom`
|
||||
doesn't lay out. (#523, #524)
|
||||
- **Settings tabs with little content rendered as a stunted box in a black
|
||||
void** (reported on Appearance). The page is now a flex column with a
|
||||
min-height floor — short tabs fill the panel, tall tabs grow and scroll
|
||||
exactly as before. The Appearance panel's previously hardcoded English
|
||||
strings ("UI scale", "Color theme", "Font") were also routed through i18n,
|
||||
per the localization rule. (#507)
|
||||
- **The engine "Install" button 500'd with "No virtual environment found."**
|
||||
`uv pip install` now targets the running interpreter (`--python
|
||||
sys.executable`) instead of relying on a venv it couldn't auto-discover.
|
||||
(#529, #527)
|
||||
- **Transcription failed with "no segments" on GPUs without efficient float16.**
|
||||
Both CTranslate2 ASR backends now fall back float16 → int8 instead of crashing
|
||||
at model load; a transcribe stream can no longer close without a terminal
|
||||
error event; and an incomplete `transformers` install reports an actionable
|
||||
message instead of "Could not import module 'AutoFeatureExtractor'".
|
||||
(#551, #549, #516)
|
||||
- **Audiobook import 500'd** with `'AudiobookPlan' object has no attribute
|
||||
'chapter_count'` for every format (.txt/.md/.epub/.pdf). (#543)
|
||||
- **Windows: generated audio auto-played in a separate, un-closeable black
|
||||
window.** Renders now play in-app through the shared playback manager. (#532)
|
||||
- **Cryptic video-download errors** now carry actionable hints: an unsupported
|
||||
link shape ("paste a direct video page, not a share/feed link") vs a transient
|
||||
network drop ("just retry — the partial download was cleaned up"). (#554, #536)
|
||||
- **A relocated, copied, or restored backend venv ("No module named
|
||||
'encodings'") now self-heals** (rebuilds once) instead of failing on every
|
||||
launch.
|
||||
- **The donate goal bar showed fabricated progress** ($137.50 / $200, 23
|
||||
sponsors). It now reflects the real figures ($10 / $200, 1 sponsor) in both the
|
||||
runtime JSON and the TypeScript fallback. (#513)
|
||||
- The **"Can't reach the local backend" startup-crash wave** (pkg_resources
|
||||
#248, `scalar_fastapi` #307, exit-106 broken venv) was fixed in v0.3.6 — this
|
||||
release carries those fixes, so updating from v0.3.5/older resolves them.
|
||||
|
||||
### Changed
|
||||
|
||||
- **Version is now single-sourced from `frontend/package.json`.** Five
|
||||
hand-maintained literals drifting is exactly what shipped a 0.3.6 build that
|
||||
called itself 0.3.5. `package.json` is canonical (vite already injects it as
|
||||
`__APP_VERSION__`), `tauri.conf.json` reads its bundle version from it
|
||||
(`"version": "../package.json"`), and the remaining toolchain-required mirrors
|
||||
(Cargo.toml, pyproject.toml, the frozen-backend fallback) are CI-guarded to
|
||||
stay in lockstep. (#503)
|
||||
- **Updater: the Preview channel actually tracks `main` again.** It was stuck at
|
||||
`0.3.5-41` because its only build trigger was a manual dispatch; a nightly
|
||||
rebuild now enforces "preview = main" (no-opping on days `main` didn't move).
|
||||
Two latent hazards are closed: the `preview` release is re-asserted as a
|
||||
prerelease every run (a non-prerelease preview could hijack the Stable
|
||||
channel's "Latest"), and its manifest can no longer silently drop the
|
||||
Intel-Mac (darwin-x86_64) target. (#500)
|
||||
|
||||
### Internal
|
||||
|
||||
- **The frozen desktop backend reported `0.3.5` regardless of its real version.**
|
||||
In a synced env, `core.version.APP_VERSION` resolves from package metadata
|
||||
(correct, so CI stayed green), but the PyInstaller-frozen build has no
|
||||
`.dist-info`, hit `PackageNotFoundError`, and fell back to a hardcoded literal.
|
||||
The spec now bundles `omnivoice` metadata so the primary path works frozen too,
|
||||
and the resolution chain is metadata → pyproject → named fallback. This also
|
||||
fixes **About → Version rendering blank** in the web/Pinokio build (no Tauri,
|
||||
backend idle), which now falls back to the build-time version. (#501)
|
||||
|
||||
## [0.3.6] — 2026-06-16
|
||||
|
||||
A large release (168 commits since v0.3.5). The headline is the **Longform
|
||||
suite** — produce full audiobooks and multi-voice stories from text, EPUB, or
|
||||
PDF — alongside a real **engine-routing** layer that tells you up front when an
|
||||
engine will fall back to CPU instead of finding out mid-synth. Dubbing,
|
||||
first-run, and install reliability all get a pass too.
|
||||
|
||||
### Added
|
||||
|
||||
- **Longform: Stories + Audiobook editors.** Two new tabs turn long text into
|
||||
finished audio. **Audiobook** takes a script (or imports plain text / EPUB /
|
||||
PDF), auto-splits it into chapters, and renders a chaptered `.m4b` with
|
||||
metadata, cover art, and per-chapter preview/resume. **Stories** is a
|
||||
multi-voice editor — assign a different voice per line, preview, and export
|
||||
the whole thing through the same server-side renderer. Both share one render
|
||||
core (loudness, metadata, cover art) and one live SSE progress stream, and
|
||||
you can convert a project between Story and Audiobook in place.
|
||||
(#402, #403, #404, #408, #409, #411, #412, #413, #426, #435, #436, #447)
|
||||
- **Longform: PDF & EPUB ingest.** "Import" on the Audiobook tab accepts EPUB
|
||||
and PDF (not just plain text) and auto-chapters the result, so an existing
|
||||
ebook becomes an audiobook without manual copy-paste. (#412, #459)
|
||||
- **Longform: two-pass loudnorm mastering.** Audiobook/Story exports now run a
|
||||
measure-then-normalize loudnorm pass for accurate ACX/podcast loudness
|
||||
targets. A slow or broken measure pass degrades gracefully to single-pass
|
||||
rather than aborting the render. (#449, #455)
|
||||
- **Longform: crash-resume.** An interrupted render is resumable without
|
||||
re-submitting the original input — the compiled plan is persisted to the job
|
||||
dir and finished chapters are reused, so a crash mid-book doesn't cost you the
|
||||
whole render. (#470)
|
||||
- **Longform: pronunciation control + SSML-lite prosody.** A per-render
|
||||
pronunciation lexicon (word respelling) plus an in-app pronunciation editor
|
||||
and markup reference, and inline prosody markers — `[slow]` / `[fast]` /
|
||||
`[emphasis]` / `[spell]` — for fine-grained delivery. (#419, #421, #422)
|
||||
- **Stories: global reading-speed control.** A toolbar slider (0.5–2.0×) sets
|
||||
one speed for every line that doesn't have its own per-line override; the
|
||||
per-line slider still wins. Persisted as a UI preference. (#415, #416)
|
||||
- **Unified LongformProject store.** Audiobook metadata, scripts, and prefs
|
||||
persist in a single project store (with a `v4→v5` migration), and finished
|
||||
books/stories now show up alongside other work in **Projects**. (#417, #443,
|
||||
#444)
|
||||
- **Portable personas (`.ovsvoice`).** Export any voice as a self-contained,
|
||||
fully-local persona bundle — identity, optional reference clip, consent
|
||||
attestation, SPDX license, and a watermarked preview — and import it back into
|
||||
@@ -17,6 +701,183 @@ The bundled TTS model package (`pyproject.toml`) is versioned independently.
|
||||
be forged by hand-editing a bundle (real recording + consent text + attestation
|
||||
required). Legacy `.omnivoice` files still import. See
|
||||
[docs/persona-format.md](docs/persona-format.md). (#29)
|
||||
- **Engine routing — no more silent CPU fallback.** A host device probe and
|
||||
routing resolver now decide where each engine actually runs, and the verdict
|
||||
is surfaced before you hit Synthesize: the **Settings → Engines** picker shows
|
||||
a per-engine compatibility matrix, and **preflight** / **diagnose** report the
|
||||
active engine's GPU verdict (accelerated / caveat / CPU-fallback /
|
||||
unavailable). At synth time every TTS entry point (`/generate`,
|
||||
`/v1/audio/speech`) enforces the same routing — an engine that can't use this
|
||||
host's GPU returns an explicit error or an `X-OmniVoice-Routing` header instead
|
||||
of silently dropping to CPU or dying mid-synth. (#21)
|
||||
- **Diagnostics suite.** New self-check tooling for when something's wrong: a
|
||||
`/system/diagnose` report (and matching backend `--diagnose`), a persistent
|
||||
**error journal** surfaced in Settings, and a scrubbed **diagnostic bundle**
|
||||
(home dirs stripped to `~/`, no tokens/keys) you can attach to a bug report.
|
||||
Paired with structured GitHub **Issue Forms** (bug / install / feature) for
|
||||
cleaner reports. (#433, #456)
|
||||
- **Dubbing: multi-speaker per-speaker voice assignment.** When diarization
|
||||
detects multiple speakers, each segment is now bound to its speaker's cloned
|
||||
voice automatically instead of landing on "Default" and needing manual fixes;
|
||||
per-segment reference clips are still preferred for quality where present. Also
|
||||
adds an optional speaker-count hint for diarization. (#275, #486, #490)
|
||||
- **Dubbing: Smart Fit timing + second-pass QC.** A Smart Fit timing strategy
|
||||
(planner, fingerprints, per-segment video retime + drift absorption + fitted
|
||||
subtitles) plus a second-pass ASR QC that flags lines whose dub drifts from the
|
||||
target timing — wired into the dub editor UI. Includes a timeline segment
|
||||
editor (drag, snap-to-onset, keyboard a11y), speech-onset alignment, regional
|
||||
dialect targeting, and per-segment clone references. (#280, #347, #350, #369,
|
||||
#370, #458)
|
||||
- **Dubbing: dedicated Dub home.** A projects/history landing for dubbing with
|
||||
project rename. (#435)
|
||||
- **Voice Console workspace.** Clone and Design are consolidated into one Voice
|
||||
workspace with right-side panels, a shared waveform player, an identity recipe
|
||||
line / Active-voice card, and a free-text "describe your voice" field that maps
|
||||
natural language to design parameters. (#317, #374, #376, #378, #395, #396,
|
||||
#397)
|
||||
- **Unified first-run setup.** Nothing installs until you confirm a plan: pick an
|
||||
install mode (installed / portable), a storage location, and (on restricted
|
||||
networks) custom PyPI/HF/python-build-standalone mirrors — with a
|
||||
minimum-free-space gate before anything downloads. Followed by a guided
|
||||
studio-console wizard with platform-aware hints, resume reassurance, and
|
||||
download ETAs. (#286, #295, #297, #298)
|
||||
- **Dictation: local-LLM refinement.** Opt-in local-LLM cleanup of final
|
||||
transcripts (collapsing Whisper hallucination loops), available on both live
|
||||
dictation and the REST `/transcribe` path; plus opt-in NLMS acoustic echo
|
||||
cancellation for dictating over playback. Configure a remote LLM endpoint
|
||||
(Ollama / vLLM / LM Studio) in Settings. (#356, #357, #363, #399, #400, #457)
|
||||
- **Unlimited-length TTS + streaming.** Sentence-boundary chunking with
|
||||
crossfade removes the per-generation length cap, and a new sentence-by-sentence
|
||||
`/ws/tts` streams audio as it's produced. An inline `[pause Nms]` marker
|
||||
inserts measured silence in generated speech. (#276, #357, #358)
|
||||
- **MCP server v1.** OmniVoice mounts an MCP server on `/mcp` (with a stdio shim
|
||||
and per-agent voice binding) so it can act as a local TTS/STT provider for
|
||||
agentic pipelines. (#368)
|
||||
- **Remote-backend access.** Point the desktop UI at a remote backend URL with a
|
||||
bearer key (Tailscale-documented), and an opt-in Hugging Face token field in
|
||||
the setup flow. (#303, #364)
|
||||
- **"Fund Claude Max" support experience.** The donate page gets a real goal bar
|
||||
with a "Join N supporters" social-proof line and suggested amounts, plus Pip
|
||||
the mascot and a non-blocking "postcard" toast that appears only *after* a
|
||||
success (a finished dub, a saved clone, a longform export) — never on errors,
|
||||
setup, or first run — with escalating cooldowns and a one-click "don't ask
|
||||
again". (#494)
|
||||
|
||||
### Fixed
|
||||
|
||||
- **Transcription/dubbing failed when ffmpeg wasn't on `PATH`** (notably on
|
||||
Windows). WhisperX now decodes audio through OmniVoice's own validated ffmpeg
|
||||
binary instead of a bare `PATH` lookup, so ASR works without a system ffmpeg
|
||||
install. (#479)
|
||||
- **Translation defaulted the source language to English.** Dubbing/translation
|
||||
now guesses the source language from the text instead of assuming `en`,
|
||||
fixing wrong-direction translations. (#478)
|
||||
- **Cinematic / LLM dubbing features failed out of the box** because `openai`
|
||||
wasn't bundled. The client is now a runtime dependency, so those paths work on
|
||||
a fresh install. (#484)
|
||||
- **`pkg_resources missing` install dead-end (#248).** The auto-repair ran
|
||||
`uv pip install setuptools`, which `uv` treated as a no-op when setuptools
|
||||
*metadata* was present but its files had been removed (commonly by Windows
|
||||
Defender quarantine or a partial extract). Both repair sites now use
|
||||
`--reinstall` to force re-extraction, and the error/hint text suggests the
|
||||
working command plus an antivirus-exclusion note. (#248)
|
||||
- **A stuck backend trapped users on a buttonless splash (#474).** The bootstrap
|
||||
splash now has a per-stage stall watchdog: if a non-terminal stage sits past
|
||||
its budget (20 min for dep install, 120 s otherwise), it flips to the failed
|
||||
state with actionable hints, the live log, and Retry / Clean-&-Retry — instead
|
||||
of polling forever with no way out. (#474)
|
||||
- **Changing the model-download location in Settings had no effect (#480).** The
|
||||
desktop launcher injected a stale models dir that overrode the per-user value,
|
||||
so new downloads kept going to the old folder and "Effective location" stayed
|
||||
wrong. The per-user env file now wins, so the in-app Settings path is
|
||||
authoritative. (#480)
|
||||
- **Backend crashed on app upgrade with a stale venv (#307).** Dependencies are
|
||||
now synced on upgrade, and a structurally broken venv self-heals instead of
|
||||
exiting `106`. `scalar_fastapi` is now optional so its absence can't break
|
||||
startup. (#307, #314)
|
||||
- **`/generate` ignored the selected TTS engine (#312)** and GGUF speech-control
|
||||
parameters weren't forwarded — both now honored. (#306, #312)
|
||||
- **TTS generation failed on some GPUs.** `torch.compile` failures now fall back
|
||||
to eager execution so generation never hard-fails on unsupported GPUs, and
|
||||
cudagraph-compiled inference is pinned to one dedicated thread to avoid
|
||||
crashes. (#278, #315)
|
||||
- **Re-dub ignored transcript edits (#281).** Fingerprints are canonicalized, the
|
||||
preview cache is busted, and the mux is atomic, so editing the transcript and
|
||||
re-dubbing actually reflects your changes. Translated subtitles now burn in
|
||||
correctly and subtitle save no longer throws a JSON error. (#281, #309)
|
||||
- **macOS: app wouldn't open without using Terminal.** Builds are now ad-hoc
|
||||
signed (with signing/notarization verification), so the app launches normally.
|
||||
(#290)
|
||||
- **macOS dictation auto-paste stole focus**; it now writes the clipboard
|
||||
natively without grabbing focus, and microphone-permission handling adds OS
|
||||
usage descriptions, a WebView grant handler, and an actionable denied-state UI.
|
||||
(#287, #323)
|
||||
- **Clone-reference transcription was broken** (it used a removed transformers
|
||||
pipeline); it now routes through the ASR registry. A crash-isolated
|
||||
faster-whisper subprocess backend keeps an ASR crash from taking down the app.
|
||||
(#308, #393)
|
||||
- **Realtime status probe hit a gated route.** It now probes the auth-exempt
|
||||
`/health` instead of the gated `/model/status`, and the UI polls the backend
|
||||
over HTTP before opening the WebSocket to avoid startup `ECONNREFUSED`. (#439,
|
||||
#450)
|
||||
- **Non-executable or unreachable engine binaries showed cryptic errors** — these
|
||||
now produce actionable messages. (#437, #438, #454, #466)
|
||||
- **Design-profile save was coupled to a TTS render (#476)**, so saving a profile
|
||||
needlessly triggered synthesis; the two are now decoupled. (#476)
|
||||
- **UI scale / black bands.** The app shell now scales via `transform: scale` and
|
||||
always fills the viewport, fixing the WebKitGTK black-band issue on Linux and
|
||||
cramped/black layouts at narrow widths — a permanent fix across platforms.
|
||||
(#445, #452)
|
||||
- **Clone popover/CTA clipping and a non-resizable textarea** are fixed, the
|
||||
WaveformPlayer no longer pauses itself on play or ignores clicks, and several
|
||||
layout/history-display issues (phantom sidebar gap, title clamping, flicker)
|
||||
are cleaned up. (#379, #384, #398, #481)
|
||||
- **Windows: `desktop-prod` now runs from cmd/PowerShell** via a cross-platform
|
||||
launcher, `tqdm` is disabled on non-TTY to avoid an `OSError`, and ffmpeg
|
||||
validation guards against `WinError 193`. (#282, #305, #377)
|
||||
- **MLX import hardened** against PyInstaller dylib failures, with a proper
|
||||
platform gate so it's only loaded where it works. (#390)
|
||||
|
||||
### Changed
|
||||
|
||||
- **Restricted-network support.** A Hugging Face mirror (`HF_ENDPOINT`) setting,
|
||||
custom PyPI / HF / python-build-standalone mirrors in first-run setup, and
|
||||
region presets help installs complete behind restrictive networks. (#286, #391)
|
||||
- **Engine memory management.** Subprocess-engine sidecars now unload on demand
|
||||
and idle-reap to free VRAM. (#401, #406)
|
||||
- **Faster, more accurate model downloads** via a Xet fast path with accurate
|
||||
progress reporting, plus a model-management cleanup pass. (#424, #428)
|
||||
- **Voice profiles unified** under one model with a `kind` discriminator and
|
||||
stored design params, and consent-locked profiles (`verified_own_voice` +
|
||||
spoken-consent flow). (#354, #376)
|
||||
- **Updater** preview channel now offers the newest build across channels, and
|
||||
preview versions carry an MSI-legal numeric pre-release stamp. (#293, #326)
|
||||
- **Performance.** Voice-clone prompt embeddings are cached, and dub retime
|
||||
batches seek to their window instead of decoding from frame 0. (#387, #427)
|
||||
|
||||
### License
|
||||
|
||||
- **Relicensed from FSL-1.1-ALv2 to AGPL-3.0 (open-core).** The project is now
|
||||
under the GNU Affero General Public License v3, with a paid commercial license
|
||||
retained for proprietary/closed-source use without AGPL obligations. The
|
||||
bundled `omnivoice/` TTS model package stays Apache-2.0 upstream
|
||||
(AGPL-compatible). Manifests declare `AGPL-3.0-only`; the in-app Commercial
|
||||
License copy and README are updated, and the old "converts to Apache 2.0 after
|
||||
two years" FAQ is removed. In-app commercial-license strings are translated
|
||||
across all 20 locales. (#292)
|
||||
|
||||
### CI
|
||||
|
||||
- **macOS Intel (x86_64) build target reinstated** on `macos-15-intel`, so Intel
|
||||
Mac users get installers again. (#342)
|
||||
- **Docker Hub publishing.** Images now also publish to Docker Hub
|
||||
(`palashdeb/omnivoice-studio`), with the Docker Hub overview maintained in-repo
|
||||
and auto-synced from `main` (sync is non-fatal so it can't redden a build).
|
||||
(#375, #410, #414)
|
||||
- **Docs-drift guard.** A daily job compares the canonical feature inventory
|
||||
against README / docs / registries to catch stale docs. (#353)
|
||||
- **Security scans never cancel on `main`,** so merge trains no longer leave red
|
||||
✗ on intermediate commits. (#340)
|
||||
|
||||
## [0.3.5] — 2026-06-03
|
||||
|
||||
|
||||
@@ -190,7 +190,7 @@ Everything else (new engines, fancy features) is downstream of "the thing instal
|
||||
<!-- GSD:conventions-start source:CONVENTIONS.md -->
|
||||
## Conventions
|
||||
|
||||
**Versioning (hard rule, owner-set 2026-06-11):** main is always **latest release + 1 patch**. The moment `vX.Y.Z` is released, main's version files (`frontend/src-tauri/tauri.conf.json`, `frontend/src-tauri/Cargo.toml`, `pyproject.toml`, **and `frontend/package.json`** — keep all **four** in lockstep; `package.json` drives the runtime `__APP_VERSION__` via vite, shown in the first-run footer + every auto bug report, so a drift ships a build that misreports its own version — guarded by `tests/test_app_version.py::test_all_version_files_in_lockstep`) bump to `X.Y.(Z+1)`. Consequences:
|
||||
**Versioning (hard rule, owner-set 2026-06-11; single-source 2026-06-16):** main is always **latest release + 1 patch**. **`frontend/package.json` is the SINGLE SOURCE OF TRUTH for the app version** — vite injects `__APP_VERSION__` from it (first-run footer + every auto bug report), and `frontend/src-tauri/tauri.conf.json` reads its bundle version from it (`"version": "../package.json"`, so the MSI/dmg/updater version can't drift from the UI). Three toolchain-required **mirrors** are kept equal to it and bumped in lockstep — `frontend/src-tauri/Cargo.toml` + `pyproject.toml` (cargo/uv need a literal) and `backend/core/version.py`'s `_FALLBACK_VERSION` (the frozen-backend last resort; at runtime the backend reads its version from package metadata via `importlib.metadata`, which `backend.spec`'s `copy_metadata('omnivoice')` makes work in the frozen build too). Never hand-edit any mirror or re-hardcode a literal in `tauri.conf.json`. Guarded by `tests/test_app_version.py` (`test_all_version_files_in_lockstep` + `test_tauri_version_derives_from_package_json`). The moment `vX.Y.Z` is released, bump `package.json` (+ the mirrors) to `X.Y.(Z+1)`. Consequences:
|
||||
- Every PR and preview build identifies as the **next** version. Preview builds stamp `X.Y.(Z+1)-N` (run number), which semver-sorts **above** the last stable `X.Y.Z` — the updater ordering is natural, no comparator tricks needed.
|
||||
- Releasing = tag `vX.Y.(Z+1)` from main (version files already match), then immediately bump main to `X.Y.(Z+2)`. The post-release bump is automated by the `version-bump` job in release.yml; if it fails, do it manually in the same day.
|
||||
- Docker: `ghcr.io/debpalash/omnivoice-studio:latest` = **main** (rolling preview); `:X.Y.Z` + `:X.Y` + `:stable` = tagged releases. `:latest` is the preview channel by design — stable users pin `:stable` or a version tag.
|
||||
@@ -198,6 +198,8 @@ Everything else (new engines, fancy features) is downstream of "the thing instal
|
||||
|
||||
**Docs-sync (hard rule, owner-set 2026-06-11):** any change that alters something these docs describe — README.md, CONTRIBUTING.md, SECURITY.md, SUPPORT.md, LICENSE, or `docs/**` (install flows, Docker tag semantics, platform support, versioning/release behavior, review process, supported versions) — must update those docs **in the same PR** as the change. If a doc impact is discovered after merge, the docs fix is the immediate next commit, not backlog. Stale docs are treated as bugs.
|
||||
|
||||
**Release notes / changelog (hard rule, owner-set 2026-06-16):** every tagged release gets a **high-quality, user-facing `## [X.Y.Z] — DATE` section in `CHANGELOG.md`** before (or in the same hour as) the tag — never the "Auto-generated release for vX.Y.Z…" fallback. `release.yml` extracts that section verbatim as the GitHub Release body (the `Extract CHANGELOG section for tag` step), so a missing/empty section ships a bare release. Quality bar = the existing house style: a one-paragraph headline, then `### Added` / `### Fixed` / `### Changed` / `### License` / `### CI` subsections; each entry is a **bold one-line lead** (what the user gets), 1–3 lines of plain-English why, and the `(#NNN)` issue/PR ref — grouped by theme, written for users, **not** raw commit dumps. This applies to **preview builds too**: preview release notes summarize what's new on `main` since the last stable, in the same style. Workflow: as features merge, keep `## [Unreleased]` current; at release time rename it to the version + date. If a release was already cut with the fallback body, the next action is to backfill `CHANGELOG.md` **and** `gh release edit <tag>` the live body — not backlog.
|
||||
|
||||
**Localization (hard rule):** No hardcoded non-English (CJK) **user-facing text** anywhere in the codebase except the translation layer (`frontend/src/i18n/`). All UI strings go through i18n (`t('...')` keys in `locales/*.json`); native language names live in `i18n/index.ts` (`LANGUAGES`). Functional CJK is allowed and tracked via the allowlist in `tests/test_no_hardcoded_cjk.py` — text-processing regexes, model/engine vocabulary & identifiers (e.g. CosyVoice speaker IDs), localized error matching, demo/eval data, and test fixtures. CI fails on any hardcoded CJK outside the allowlist; to add legitimate functional CJK, extend `_ALLOWED_FILES` there with a justification.
|
||||
|
||||
**Fix quality (hard rule, owner-set 2026-06-16):** Fix issues *properly* and future-maintenance-proof — don't stop at the symptom. Root-cause fully, fix the whole **class** of the bug (not just the one reported instance), add a fail-before/pass-after regression test, and harden against recurrence (e.g. if a lockfile drift only fails in Docker, also make CI catch it). Go the extra mile where it durably pays off. Be token-efficient about it — extra **effort**, not extra **verbosity**: no padding, no redundant re-checks, the smallest correct change that is also recurrence-proof. Don't be shy to spend the effort a proper fix needs; do be shy about wasting tokens.
|
||||
|
||||
+25
-1
@@ -159,7 +159,7 @@ class MyEngineBackend(TTSBackend):
|
||||
|
||||
- **Components**: Functional components with hooks
|
||||
- **State**: Zustand stores in `src/stores/`, organized by slice
|
||||
- **CSS**: Vanilla CSS in component-level files — no Tailwind
|
||||
- **CSS**: **Utilities-first + shadcn/ui, one stylesheet.** UI is built on the shadcn/ui primitives in `src/components/ui/` (wrapped by the `src/ui/` barrel, themed to the OmniVoice palette), composed with Tailwind v4 utility classes. **All styling now lives in a single file — `src/index.css`**: the `@theme` / `[data-theme]` token foundation plus the irreducible set utilities can't express (`@keyframes`, glassmorphism/`backdrop-filter`, pseudo-elements, `:has()`, unlayered cascade overrides, and styling hooks on library-generated DOM like virtualized rows / WaveSurfer). The per-component `.css` files were eliminated in the CSS→Tailwind/shadcn migration — **do not create new ones.** Reach for shadcn primitives + utilities; if a rule is genuinely irreducible, add it to `src/index.css` with a provenance comment. (The only other `.css` is the test-only visual harness. See `docs/shadcn-migration.md`.)
|
||||
- **Naming**: `PascalCase` for components, `camelCase` for hooks and utils
|
||||
|
||||
### Rust (Tauri)
|
||||
@@ -169,6 +169,30 @@ class MyEngineBackend(TTSBackend):
|
||||
|
||||
---
|
||||
|
||||
## Frontend file structure & size limits
|
||||
|
||||
Frontend code stays modular so an edit loads one small file, not a 1900-line
|
||||
one. The rules:
|
||||
|
||||
- **Size caps:** **soft 300 lines**, **hard 500 lines** per `.jsx` file.
|
||||
Anything over 500 lines must be split. (The cap does **not** apply to
|
||||
`src/index.css` — it is the single, intentional styling foundation and the
|
||||
only app stylesheet; see the CSS rule above.)
|
||||
- **Pages are thin orchestrators.** A file in `frontend/src/pages/` is just
|
||||
layout + routing + state wiring that composes feature components — no inline
|
||||
sub-component over ~50 lines.
|
||||
- **One component per file.** Co-locate `Foo.jsx` + `Foo.test.jsx` together in a
|
||||
per-page feature folder under `frontend/src/components/` (e.g.
|
||||
`components/settings/`, `components/dub/`). Styling is **not** co-located —
|
||||
it's utilities + shadcn, with any irreducible rules in `src/index.css`.
|
||||
- **Shared bits go in a `primitives/` folder** inside the feature folder
|
||||
(`components/settings/primitives/` is the existing example).
|
||||
- **Enforced by ESLint `max-lines`** (`max: 500`) — **warn-only for now** so it
|
||||
never breaks CI, with the goal of upgrading to `error` once the backlog of
|
||||
oversized files clears.
|
||||
|
||||
---
|
||||
|
||||
## Commit Messages
|
||||
|
||||
Write clear, concise messages. The PR title becomes the squash-merge commit.
|
||||
|
||||
@@ -4,34 +4,26 @@
|
||||
<h3>The open-source ElevenLabs alternative.</h3>
|
||||
<p>Real-time dictation, zero-shot voice cloning, and cinematic video dubbing — all on your desktop.<br/>Open-source, no API keys, fully local. <b>646 languages.</b></p>
|
||||
|
||||
<p>
|
||||
<a href="https://github.com/debpalash/OmniVoice-Studio/stargazers"><img src="https://img.shields.io/github/stars/debpalash/OmniVoice-Studio?style=flat-square&color=f59e0b" alt="Stars" /></a>
|
||||
<a href="https://github.com/debpalash/OmniVoice-Studio/releases/latest"><img src="https://img.shields.io/github/v/release/debpalash/OmniVoice-Studio?style=flat-square&color=10b981" alt="Release" /></a>
|
||||
<a href="LICENSE"><img src="https://img.shields.io/badge/license-AGPL--3.0-blue?style=flat-square" alt="License" /></a>
|
||||
<a href="https://github.com/debpalash/OmniVoice-Studio/issues"><img src="https://img.shields.io/github/issues/debpalash/OmniVoice-Studio?style=flat-square&color=ef4444" alt="Issues" /></a>
|
||||
<a href="https://discord.gg/bzQavDfVV9"><img src="https://img.shields.io/badge/Discord-Join_Community-5865F2?style=flat-square&logo=discord&logoColor=white" alt="Discord" /></a>
|
||||
</p>
|
||||
|
||||
<p>
|
||||
<a href="#quickstart">Quickstart</a> ·
|
||||
<a href="#features">Features</a> ·
|
||||
<a href="#why-omnivoice-studio">Why OmniVoice Studio?</a> ·
|
||||
<a href="#why-ovs">Why OVS</a> ·
|
||||
<a href="#tts-engines">TTS Engines</a> ·
|
||||
<a href="#asr-engines">ASR Engines</a> ·
|
||||
<a href="#sponsor--donate">Donate</a> ·
|
||||
<a href="#contributing">Contributing</a> ·
|
||||
<a href="https://discord.gg/bzQavDfVV9">Discord</a> ·
|
||||
<a href="README_CN.md"><strong>简体中文</strong></a>
|
||||
</p>
|
||||
|
||||
<p>
|
||||
<a href="https://github.com/debpalash/OmniVoice-Studio/releases/latest"><img src="https://img.shields.io/badge/macOS-DMG_(Apple_Silicon)-000?style=for-the-badge&logo=apple&logoColor=white" alt="Download macOS DMG" /></a>
|
||||
<!-- Pre-built macOS bundle is Apple Silicon. Intel Macs: build from source (docs/install/macos.md); a pre-built Intel target is tracked in #279. -->
|
||||
<a href="https://github.com/debpalash/OmniVoice-Studio/releases/latest"><img src="https://img.shields.io/badge/Windows-MSI_(x64)-0078D4?style=for-the-badge&logo=windows&logoColor=white" alt="Download Windows MSI" /></a>
|
||||
<a href="https://github.com/debpalash/OmniVoice-Studio/releases/latest"><img src="https://img.shields.io/badge/Linux-AppImage_(x64)-FCC624?style=for-the-badge&logo=linux&logoColor=black" alt="Download Linux AppImage" /></a>
|
||||
<a href="https://github.com/debpalash/OmniVoice-Studio/releases/latest"><img src="https://img.shields.io/badge/Debian-.deb-A81D33?style=for-the-badge&logo=debian&logoColor=white" alt="Download Debian .deb" /></a>
|
||||
</p>
|
||||
<p>
|
||||
<sub><b>macOS:</b> first launch needs a one-time approval — right-click → <b>Open</b> (or System Settings → Privacy & Security → <b>"Open Anyway"</b> on macOS 15). No Terminal needed. <a href="docs/install/macos.md#gatekeeper-quarantine">Why?</a></sub>
|
||||
<a href="https://github.com/debpalash/OmniVoice-Studio/stargazers"><img src="https://img.shields.io/github/stars/debpalash/OmniVoice-Studio?style=flat-square&color=f59e0b" alt="Stars" /></a>
|
||||
<a href="https://github.com/debpalash/OmniVoice-Studio/releases/latest"><img src="https://img.shields.io/github/v/release/debpalash/OmniVoice-Studio?style=flat-square&color=10b981" alt="Release" /></a>
|
||||
<a href="LICENSE"><img src="https://img.shields.io/badge/license-AGPL--3.0-blue?style=flat-square" alt="License" /></a>
|
||||
<a href="https://github.com/debpalash/OmniVoice-Studio/issues"><img src="https://img.shields.io/github/issues/debpalash/OmniVoice-Studio?style=flat-square&color=ef4444" alt="Issues" /></a>
|
||||
<a href="https://discord.gg/bzQavDfVV9"><img src="https://img.shields.io/badge/Discord-Join_Community-5865F2?style=flat-square&logo=discord&logoColor=white" alt="Discord" /></a>
|
||||
<a href="https://ko-fi.com/debpalash"><img src="https://img.shields.io/badge/Ko--fi-Support_Us-FF5E5B?style=flat-square&logo=ko-fi&logoColor=white" alt="Ko-fi" /></a>
|
||||
<a href="https://paypal.me/palashCoder"><img src="https://img.shields.io/badge/PayPal-Donate-00457C?style=flat-square&logo=paypal&logoColor=white" alt="PayPal" /></a>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
@@ -58,20 +50,28 @@
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
<td align="center" width="33%">
|
||||
<td align="center" width="25%">
|
||||
<h3>🎙️ Voice Cloning</h3>
|
||||
<p>3-second clip → mirror any voice.<br/><b>646 languages</b>, zero-shot.</p>
|
||||
</td>
|
||||
<td align="center" width="33%">
|
||||
<td align="center" width="25%">
|
||||
<h3>🎨 Voice Design</h3>
|
||||
<p>Gender, age, accent, pitch, speed,<br/>emotion, dialect — <b>dial it in</b>.</p>
|
||||
</td>
|
||||
<td align="center" width="33%">
|
||||
<td align="center" width="25%">
|
||||
<h3>🎬 Video Dubbing</h3>
|
||||
<p>YouTube URL or file → transcribe →<br/>translate → re-voice → <b>MP4</b>.</p>
|
||||
</td>
|
||||
<td align="center" width="25%">
|
||||
<h3>📖 Audiobook Editor</h3>
|
||||
<p>Import text, EPUB, or PDF. Auto-chapter,<br/>loudnorm, metadata. Export <b>.m4b</b>.</p>
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center" valign="top">
|
||||
<h3>🎭 Stories</h3>
|
||||
<p>Multi-voice editor. Assign voices<br/>per-line, preview, <b>export full cast</b>.</p>
|
||||
</td>
|
||||
<td align="center" valign="top">
|
||||
<h3>⌨️ Dictation Widget</h3>
|
||||
<p><code>⌘+⇧+Space</code> from <b>any app</b>.<br/>Transcribes, auto-pastes, disappears.</p>
|
||||
@@ -98,6 +98,10 @@
|
||||
<h3>🛡️ AI Watermark</h3>
|
||||
<p>AudioSeal (Meta). <b>Invisible</b>,<br/>survives compression.</p>
|
||||
</td>
|
||||
<td align="center" valign="top">
|
||||
<h3>🔬 Diagnostics</h3>
|
||||
<p>Self-check, error journal,<br/>scrubbed <b>diagnostic bundle</b>.</p>
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center" valign="top">
|
||||
@@ -110,7 +114,29 @@
|
||||
</td>
|
||||
<td align="center" valign="top">
|
||||
<h3>🧩 Extensible</h3>
|
||||
<p>Subclass <code>TTSBackend</code>,<br/>add any engine in <b>~50 lines</b>.</p>
|
||||
<p>Subclass <code>TTSbackend</code>,<br/>add any engine in <b>~50 lines</b>.</p>
|
||||
</td>
|
||||
<td align="center" valign="top">
|
||||
<h3>🧭 Engine Routing</h3>
|
||||
<p>Preflight GPU check per engine.<br/><b>No silent CPU fallback</b>.</p>
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center" valign="top">
|
||||
<h3>🎒 Portable Personas</h3>
|
||||
<p>Export voices as <code>.ovsvoice</code><br/>bundles — identity + <b>watermark</b>.</p>
|
||||
</td>
|
||||
<td align="center" valign="top">
|
||||
<h3>♾️ Unlimited TTS</h3>
|
||||
<p>Sentence-chunked generation.<br/><b>No length cap</b>. Streaming via WS.</p>
|
||||
</td>
|
||||
<td align="center" valign="top">
|
||||
<h3>🌐 Remote Backend</h3>
|
||||
<p>Point UI at a remote server.<br/>Tailscale-friendly. <b>Bearer auth</b>.</p>
|
||||
</td>
|
||||
<td align="center" valign="top">
|
||||
<h3>🧠 Dictation + LLM</h3>
|
||||
<p>Local LLM cleanup of transcripts.<br/>Optional echo <b>cancellation</b>.</p>
|
||||
</td>
|
||||
</tr>
|
||||
</table>
|
||||
@@ -119,6 +145,15 @@
|
||||
|
||||
## Quickstart
|
||||
|
||||
<div align="center">
|
||||
<a href="https://github.com/debpalash/OmniVoice-Studio/releases/latest"><img src="https://img.shields.io/badge/macOS-DMG_(Apple_Silicon)-000?style=for-the-badge&logo=apple&logoColor=white" alt="Download macOS DMG" /></a>
|
||||
<a href="https://github.com/debpalash/OmniVoice-Studio/releases/latest"><img src="https://img.shields.io/badge/Windows-MSI_(x64)-0078D4?style=for-the-badge&logo=windows&logoColor=white" alt="Download Windows MSI" /></a>
|
||||
<a href="https://github.com/debpalash/OmniVoice-Studio/releases/latest"><img src="https://img.shields.io/badge/Linux-AppImage_(x64)-FCC624?style=for-the-badge&logo=linux&logoColor=black" alt="Download Linux AppImage" /></a>
|
||||
<a href="https://github.com/debpalash/OmniVoice-Studio/releases/latest"><img src="https://img.shields.io/badge/Debian-.deb-A81D33?style=for-the-badge&logo=debian&logoColor=white" alt="Download Debian .deb" /></a>
|
||||
<br/>
|
||||
<sub><b>macOS:</b> first launch needs a one-time approval — right-click → <b>Open</b> (or System Settings → Privacy & Security → <b>"Open Anyway"</b> on macOS 15). No Terminal needed. <a href="docs/install/macos.md#gatekeeper-quarantine">Why?</a></sub>
|
||||
</div>
|
||||
|
||||
Per-OS install guides — pick yours and follow it end-to-end:
|
||||
|
||||
- **macOS** — [docs/install/macos.md](docs/install/macos.md)
|
||||
@@ -191,7 +226,7 @@ options, see [docs/downloading-models.md](docs/downloading-models.md).
|
||||
|
||||
---
|
||||
|
||||
## Why OmniVoice Studio?
|
||||
## Why OVS?
|
||||
|
||||
ElevenLabs charges **$5–$330/mo** and processes your audio on their servers. OmniVoice Studio runs **on your hardware, with no usage limits.**
|
||||
|
||||
@@ -200,12 +235,17 @@ ElevenLabs charges **$5–$330/mo** and processes your audio on their servers. O
|
||||
| **Pricing** | $5–$330/mo, per-character billing | Free & open-source (AGPL-3.0) · [Commercial license](#license) for proprietary use |
|
||||
| **Voice Cloning** | ✅ 3s clip | ✅ 3s clip, zero-shot |
|
||||
| **Voice Design** | ✅ Gender, age | ✅ Gender, age, accent, pitch, style, dialect |
|
||||
| **Audiobook / Stories** | ❌ | ✅ Full audiobook editor + multi-voice stories (EPUB/PDF import, .m4b export) |
|
||||
| **Languages** | 32 | **646** |
|
||||
| **Video Dubbing** | ✅ Cloud-only | ✅ Fully local |
|
||||
| **Data Privacy** | Audio sent to cloud | **Nothing leaves your machine** |
|
||||
| **API Keys** | Required | Not needed |
|
||||
| **GPU Support** | N/A (cloud) | CUDA · Apple Silicon · ROCm · CPU |
|
||||
| **Desktop App** | ❌ | ✅ macOS · Windows · Linux |
|
||||
| **TTS Engines** | 1 | **11** (OmniVoice, CosyVoice 3, GPT-SoVITS, VoxCPM2, MOSS-TTS-Nano, KittenTTS, MLX-Audio, Sherpa-ONNX, IndexTTS 2, OmniVoice GGUF, Supertonic 3) |
|
||||
| **ASR Engines** | 1 | **9** (WhisperX, Faster-Whisper, MLX Whisper, PyTorch Whisper, Parakeet, Moonshine, FunASR, isolated Faster-Whisper, sherpa-onnx live dictation) |
|
||||
| **MCP Server** | ❌ | ✅ Use from Claude, Cursor, any MCP client |
|
||||
| **Self-check** | ❌ | ✅ Diagnostics suite, error journal, scrubbed debug bundles |
|
||||
| **Customizable** | ❌ Closed | ✅ Fork it, extend it, ship it |
|
||||
|
||||
OmniVoice Studio gives you professional-grade AI tools without the subscription or the cloud.
|
||||
@@ -241,12 +281,21 @@ OmniVoice ships a multi-engine TTS backend. The default engine (OmniVoice) is al
|
||||
|--------|:---------:|:-----:|:--------:|:-----:|:---------:|:-------:|:-------:|
|
||||
| **OmniVoice** (default) | 600+ | ✅ | ✅ | ✅ CUDA/CPU | ✅ MPS | ✅ CUDA/CPU | Built-in |
|
||||
| **CosyVoice 3** | 9 + 18 dialects | ✅ | ✅ | ✅ CUDA/CPU | ✅ MPS | ✅ CUDA/CPU | Apache-2.0 |
|
||||
| **MLX-Audio** (Kokoro, Qwen3-TTS, CSM, Dia, …) | Multi | Varies | Varies | ❌ | ✅ Native | ❌ | Varies |
|
||||
| **GPT-SoVITS** | 5 | ✅ | — | ✅ CUDA/CPU | — | ✅ CUDA/CPU | MIT |
|
||||
| **VoxCPM2** | 30 | ✅ | ✅ | ✅ CUDA/CPU | ✅ MPS | ✅ CUDA/CPU | Apache-2.0 |
|
||||
| **MOSS-TTS-Nano** | 20 | ✅ | ❌ | ✅ CUDA/CPU | ✅ CPU | ✅ CUDA/CPU | Apache-2.0 |
|
||||
| **KittenTTS** | English | ❌ | ❌ | ✅ CPU | ✅ CPU | ✅ CPU | MIT |
|
||||
| **MOSS-TTS-Nano** | 20 | ✅ | — | ✅ CUDA/CPU | ✅ CPU | ✅ CUDA/CPU | Apache-2.0 |
|
||||
| **KittenTTS** | English | — | — | ✅ CPU | ✅ CPU | ✅ CPU | MIT |
|
||||
| **MLX-Audio** (Kokoro, Qwen3-TTS, CSM, Dia, …) | Multi | Varies | Varies | ❌ | ✅ Native | ❌ | Varies |
|
||||
| **Sherpa-ONNX** | 20+ | — | — | ✅ CUDA/CPU | ✅ CPU | ✅ CUDA/CPU | Apache-2.0 |
|
||||
| **IndexTTS 2** ⚡ | Multi | ✅ | — | ✅ CUDA | — | ✅ CUDA | Apache-2.0 |
|
||||
| **OmniVoice GGUF** ⚡ | 600+ | ✅ | ✅ | ✅ CPU | ✅ CPU | ✅ CPU | Built-in |
|
||||
| **Supertonic 3** ⚡ | 31 | — | — | ✅ CPU | ✅ CPU | ✅ CPU | OpenRAIL-M |
|
||||
| **MOSS-TTS-v1.5** ⚡ (8B) | 31 | ✅ | — | ✅ CUDA/CPU | ✅ CPU | ✅ CUDA/CPU | Apache-2.0 |
|
||||
| **dots.tts** ⚡ (2B) | 24 | ✅ | — | ✅ CUDA/CPU | ✅ CPU | ❌ | Apache-2.0 |
|
||||
|
||||
> **CUDA** = GPU-accelerated · **MPS** = Apple Silicon Metal · **CPU** = runs everywhere, slower for large models · KittenTTS and MOSS-TTS-Nano run realtime on CPU · MLX-Audio is Apple Silicon only.
|
||||
> **CUDA** = GPU-accelerated · **MPS** = Apple Silicon Metal · **CPU** = runs everywhere, slower for large models · KittenTTS and MOSS-TTS-Nano run realtime on CPU · MLX-Audio is Apple Silicon only · ⚡ = lazy-registered (installed on first use)
|
||||
>
|
||||
> **MOSS-TTS-v1.5** (8B, ~16 GB weights) and **dots.tts** (2B, ~9 GB weights) are heavyweight opt-in engines that run in their own isolated venv from a local clone — see [MOSS-TTS-v1.5](docs/engines/moss-tts-v15.md) and [dots.tts](docs/engines/dots-tts.md). Neither claims Apple-Silicon **MPS** (upstream is CUDA/CPU only; on a Mac they run on CPU). dots.tts upstream is Linux/macOS only — no Windows path.
|
||||
|
||||
### ASR Engines
|
||||
|
||||
@@ -256,31 +305,36 @@ OmniVoice ships a multi-engine ASR (speech-to-text) backend that powers dictatio
|
||||
|--------|-------------------------|:---------:|----------|
|
||||
| **WhisperX** (default) | `whisperx` | ~100 | Dubbing & subtitles — word-level timing via wav2vec2 forced alignment |
|
||||
| **Faster-Whisper** | `faster-whisper` | ~100 | Fast transcription on Linux / macOS / Windows (CTranslate2) |
|
||||
| **Faster-Whisper (isolated)** | `faster-whisper-isolated` | ~100 | Same as Faster-Whisper but crash-isolated in a subprocess — an ASR crash won't take down the app |
|
||||
| **MLX Whisper** | `mlx-whisper` | ~100 | Native Apple Silicon speed (Apple MLX / Metal) |
|
||||
| **PyTorch Whisper** | `pytorch-whisper` | ~100 | CUDA / CPU fallback via 🤗 Transformers |
|
||||
| **PyTorch Whisper** | `pytorch-whisper` | ~100 | CUDA / CPU fallback via 🤗 Transformers (no cuDNN 8 needed) |
|
||||
| **Parakeet TDT** | `nemo-parakeet` | English + 25 EU | SOTA English accuracy, auto language detection (NVIDIA NeMo, GPU only) |
|
||||
| **Moonshine** | `moonshine` | English | Edge / low-latency, ONNX |
|
||||
| **FunASR** | `funasr` | 50+ | All-in-one multilingual — built-in VAD + inline speaker diarization (SenseVoice) |
|
||||
| **sherpa-onnx** (live dictation) | `sherpa-onnx-asr` | 25 EU + 90+ | Live, faster-than-real-time dictation — small streaming/offline ONNX models (Parakeet TDT v3/v2, streaming Zipformer & Paraformer, Whisper Tiny), CPU, identical on macOS / Windows / Linux. Picked per-model in **Settings → Voice**. |
|
||||
|
||||
> Whisper-family engines cover ~100 languages; **FunASR / SenseVoice** adds an all-in-one multilingual path with built-in voice-activity detection and inline speaker diarization. Every engine runs on-device — no API keys, no cloud.
|
||||
> Whisper-family engines cover ~100 languages; **FunASR / SenseVoice** adds an all-in-one multilingual path with built-in voice-activity detection and inline speaker diarization. **sherpa-onnx** powers the live dictation model picker — you talk and text appears as you speak. Every engine runs on-device — no API keys, no cloud.
|
||||
|
||||
> **GPU without efficient float16?** On older NVIDIA GPUs (Maxwell/Pascal, GTX 16xx) or after a CTranslate2/cuDNN mismatch, the CTranslate2 ASR engines (WhisperX, Faster-Whisper) can't run `float16` and OmniVoice automatically retries on `int8` — no config needed. If transcription still fails, pin the compute type with the `ASR_COMPUTE_TYPE` env var (escape hatch): `ASR_COMPUTE_TYPE=int8` (or `float32` for CPU). Set it to `int8` and restart the backend.
|
||||
|
||||
---
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
┌─────────────────────────────────────────────────┐
|
||||
│ Frontend (React) │
|
||||
│ DubTab · VoicePreview · BatchQueue · Gallery │
|
||||
├─────────────────────────────────────────────────┤
|
||||
│ Backend (FastAPI) │
|
||||
│ 97 API endpoints · SSE streaming · SQLite │
|
||||
├──────────┬──────────┬──────────┬────────────────┤
|
||||
│ WhisperX │ Demucs │OmniVoice │ Pyannote │
|
||||
│ ASR │ Source │ TTS │ Diarization │
|
||||
│ │ Sep. │ │ │
|
||||
└──────────┴──────────┴──────────┴────────────────┘
|
||||
CUDA / MPS / ROCm / CPU (auto-detected)
|
||||
┌─────────────────────────────────────────────────────────────┐
|
||||
│ Frontend (React) │
|
||||
│ DubTab · VoiceConsole · Stories · Audiobook · Gallery │
|
||||
│ Dictation · BatchQueue · Diagnostics · MCP Client │
|
||||
├─────────────────────────────────────────────────────────────┤
|
||||
│ Backend (FastAPI) │
|
||||
│ 100+ API endpoints · SSE+WSS streaming · SQLite │
|
||||
├──────────┬──────────┬──────────┬──────────┬────────────────┤
|
||||
│ WhisperX │ Demucs │OmniVoice │ Pyannote │ Engine Routing │
|
||||
│ (+7 ASR │ Source │ (+10 │ Diariz- │ ↳ GPU preflight │
|
||||
│ engines) │ Sep. │ TTS) │ ation │ ↳ No silent CPU │
|
||||
└──────────┴──────────┴──────────┴──────────┴────────────────┘
|
||||
CUDA / MPS / ROCm / CPU (auto-detected + routed)
|
||||
```
|
||||
|
||||
---
|
||||
@@ -291,27 +345,55 @@ OmniVoice ships a multi-engine ASR (speech-to-text) backend that powers dictatio
|
||||
|
||||
| Category | Features |
|
||||
|----------|----------|
|
||||
| **Dubbing** | Full pipeline (transcribe→translate→synthesize→mux), scene-aware splitting, lip-sync scoring, streaming TTS |
|
||||
| **Voice** | Zero-shot cloning, voice design, A/B comparison, voice preview widget, gallery with favorites/tags |
|
||||
| **Audio** | Demucs vocal isolation, per-segment gain, selective track export, stem/SRT/VTT/MP3 export |
|
||||
| **Longform** | Audiobook editor (text/EPUB/PDF → chaptered .m4b), Stories multi-voice editor, two-pass loudnorm mastering, crash-resume for interrupted renders, pronunciation control + SSML-lite prosody |
|
||||
| **Dubbing** | Full pipeline (transcribe→translate→synthesize→mux), scene-aware splitting, lip-sync scoring, streaming TTS, per-speaker voice assignment, Smart Fit timing + second-pass QC, dedicated Dub home |
|
||||
| **Voice** | Zero-shot cloning, voice design, A/B comparison, voice preview widget, gallery with favorites/tags, portable persona bundles (`.ovsvoice`), voice console workspace |
|
||||
| **Audio** | Demucs vocal isolation, per-segment gain, selective track export, stem/SRT/VTT/MP3 export, unlimited-length TTS via sentence-chunked generation |
|
||||
| **Multi-Lang** | Multi-language batch picker, batch dubbing queue with sequential GPU execution |
|
||||
| **Diarization** | Pyannote ML diarization, auto speaker clone extraction, per-speaker voice assignment |
|
||||
| **Infra** | Docker deployment, CUDA/MPS/ROCm auto-detect, cuDNN 8 compat, VRAM-aware model offloading |
|
||||
| **ASR** | 9 engines (WhisperX, Faster-Whisper, isolated Faster-Whisper, MLX Whisper, PyTorch Whisper, Parakeet TDT, Moonshine, FunASR/SenseVoice, sherpa-onnx live dictation), crash-isolated subprocess backend |
|
||||
| **TTS** | 11 engines (OmniVoice, CosyVoice 3, GPT-SoVITS, VoxCPM2, MOSS-TTS-Nano, KittenTTS, MLX-Audio, Sherpa-ONNX, + lazy: IndexTTS 2, OmniVoice GGUF, Supertonic 3), engine routing with GPU preflight |
|
||||
| **Infra** | Docker deployment, CUDA/MPS/ROCm auto-detect, cuDNN 8 compat, VRAM-aware model offloading, engine routing (no silent CPU fallback), diagnostics suite & error journal, restricted-network mirror support |
|
||||
| **AI Provenance** | AudioSeal invisible watermarking (SynthID-like), video logo overlay, watermark detection API |
|
||||
| **UX** | Undo/redo, keyboard shortcuts, drag-and-drop, session persistence, glassmorphism design system |
|
||||
| **UX** | Undo/redo, keyboard shortcuts, drag-and-drop, session persistence, glassmorphism design system, UI scale fix for Linux/WebKitGTK |
|
||||
| **Real-time Events** | WebSocket event bus — instant sidebar refresh on data mutations, exponential backoff reconnect |
|
||||
| **State Management** | Zustand store migration — `uiSlice`, `pillSlice`, `dubSlice`, `generateSlice`, `prefsSlice`, `glossarySlice` |
|
||||
| **Desktop** | Cross-platform Tauri installers (macOS DMG, Windows MSI, Linux deb/AppImage), auto-update infrastructure |
|
||||
| **Windows Hardening** | Cross-platform log paths, Triton workaround, HF symlink bypass, 300s health check timeout |
|
||||
| **Dictation** | Global system-wide hotkey (`⌘+⇧+Space`), frameless floating widget, streaming ASR via WebSocket, auto-paste |
|
||||
| **Desktop** | Cross-platform Tauri installers (macOS DMG/Intel, Windows MSI, Linux deb/AppImage), auto-update infrastructure, single-instance enforcement, close-to-tray, macOS Gatekeeper fix |
|
||||
| **Dictation** | Global system-wide hotkey (`⌘+⇧+Space`), frameless floating widget, streaming ASR via WebSocket, auto-paste, customizable hotkey, local-LLM transcript refinement |
|
||||
| **Batch Pipeline** | Full batch TTS: extract → transcribe → translate → generate → mix → export, with live progress tracking |
|
||||
| **MCP Server** | OmniVoice as a local TTS/STT provider for Claude, Cursor, and any MCP client |
|
||||
| **Remote Backend** | Point the desktop UI at a remote backend URL with bearer auth (Tailscale-documented) |
|
||||
| **Reliability** | Stall watchdog on bootstrap splash, per-engine GPU compatibility matrix, actionable errors for non-executable engine binaries, setuptools auto-repair |
|
||||
|
||||
### 🔜 Up Next
|
||||
|
||||
- 🎬 **Lip-sync v2** — visual speech timing with wav2lip
|
||||
- 📖 **Audiobook Editor** — chapter-aware long-form narration
|
||||
- 🌐 **Hosted Demo** — try OmniVoice without installing anything
|
||||
- 🔌 **Plugin Marketplace** — community-contributed TTS engines and effects
|
||||
- 🎵 **Real-time Voice Changer** — live microphone transformation during calls
|
||||
|
||||
---
|
||||
|
||||
## Sponsor / Donate
|
||||
|
||||
OmniVoice Studio is built by one developer using Claude Code and AI agents — and the agent bills are real. Over the last three months I've spent thousands of dollars on Claude subscriptions to keep the features shipping, the bugs fixed, and your issues answered. If OmniVoice has created value for you, helping cover those bills means I can keep developing full-time.
|
||||
|
||||
<div align="center">
|
||||
|
||||
**This month's agent bill fund**
|
||||
|
||||
<img src="https://img.shields.io/badge/raised_%2410_of_%24200-5%25-EAB308?style=for-the-badge" alt="$10 / $200 raised" />
|
||||
|
||||
<br/><br/>
|
||||
|
||||
<a href="https://ko-fi.com/debpalash"><img src="https://img.shields.io/badge/Ko--fi-Support_❤️-FF5E5B?style=for-the-badge&logo=ko-fi&logoColor=white" alt="Ko-fi" /></a>
|
||||
|
||||
<a href="https://paypal.me/palashCoder"><img src="https://img.shields.io/badge/PayPal-Donate-00457C?style=for-the-badge&logo=paypal&logoColor=white" alt="PayPal" /></a>
|
||||
|
||||
<br/>
|
||||
<sub>Every dollar goes directly to agent bills — keeping OmniVoice development continuous.</sub>
|
||||
|
||||
</div>
|
||||
|
||||
---
|
||||
|
||||
@@ -380,7 +462,7 @@ Yes. MPS acceleration is auto-detected. MLX-optimized Whisper models are availab
|
||||
<details>
|
||||
<summary><b>Can I add my own TTS engine?</b></summary>
|
||||
<br/>
|
||||
Yes. OmniVoice uses a <b>built-in backend registry</b>. To add an engine in ~50 lines, subclass <code>TTSBackend</code> in <code>backend/services/tts_backend.py</code> and add it to the <code>_REGISTRY</code> dictionary at the bottom. Six engines are built in: OmniVoice, CosyVoice, MLX-Audio (14+ sub-engines), VoxCPM2, MOSS-TTS-Nano, and KittenTTS. See the <a href="#tts-engines">TTS Engines</a> section for details.
|
||||
Yes. OmniVoice uses a <b>built-in backend registry</b>. To add an engine in ~50 lines, subclass <code>TTSBackend</code> in <code>backend/services/tts_backend.py</code> and add it to the <code>_REGISTRY</code> dictionary. Eleven engines are built in: OmniVoice, CosyVoice 3, GPT-SoVITS, MLX-Audio (14+ sub-engines), VoxCPM2, MOSS-TTS-Nano, KittenTTS, Sherpa-ONNX, plus lazy-registered IndexTTS 2, OmniVoice GGUF, and Supertonic 3. See the <a href="#tts-engines">TTS Engines</a> section for details.
|
||||
</details>
|
||||
|
||||
---
|
||||
@@ -410,6 +492,9 @@ OmniVoice Studio is built on the shoulders of exceptional open-source work:
|
||||
| [**CTranslate2**](https://github.com/OpenNMT/CTranslate2) | Optimized Transformer inference on CPU and GPU |
|
||||
| [**AudioSeal (Meta)**](https://github.com/facebookresearch/audioseal) | Invisible neural audio watermarking for AI provenance |
|
||||
| [**Tauri**](https://tauri.app) | Native desktop app framework |
|
||||
| [**Supertone / Supertonic 3**](https://huggingface.co/Supertone/supertonic-3) | ONNX TTS engine — 31 languages, CPU-efficient |
|
||||
| [**Sherpa-ONNX**](https://github.com/k2-fsa/sherpa-onnx) | WASM-ready universal TTS/ASR runtime |
|
||||
| [**GPT-SoVITS**](https://github.com/RVC-Boss/GPT-SoVITS) | Zero-shot TTS engine — 5 languages, RTF 0.014 |
|
||||
|
||||
---
|
||||
|
||||
@@ -419,7 +504,8 @@ OmniVoice Studio is built on the shoulders of exceptional open-source work:
|
||||
|
||||
If you read this far, you're our kind of person.<br/>
|
||||
**[⭐ Star this repo](https://github.com/debpalash/OmniVoice-Studio)** so others can find it too.<br/>
|
||||
**[💬 Join the Discord](https://discord.gg/bzQavDfVV9)** to share what you build.
|
||||
**[💬 Join the Discord](https://discord.gg/bzQavDfVV9)** to share what you build.<br/>
|
||||
**[❤️ Support development](https://ko-fi.com/debpalash)** — fund the AI agent bills that keep OmniVoice shipping.
|
||||
|
||||
<br/>
|
||||
|
||||
|
||||
@@ -372,9 +372,13 @@ OmniVoice 配备多引擎 TTS 后端。默认引擎(OmniVoice)始终可用
|
||||
| **MLX-Audio**(Kokoro, Qwen3-TTS, CSM, Dia 等) | 多语言 | 因引擎而异 | 因引擎而异 | ❌ | ✅ 原生 | ❌ | 因引擎而异 |
|
||||
| **VoxCPM2** | 30 | ✅ | ✅ | ✅ CUDA/CPU | ✅ MPS | ✅ CUDA/CPU | Apache-2.0 |
|
||||
| **MOSS-TTS-Nano** | 20 | ✅ | ❌ | ✅ CUDA/CPU | ✅ CPU | ✅ CUDA/CPU | Apache-2.0 |
|
||||
| **MOSS-TTS-v1.5**(8B,可选装) | 31 | ✅ | ❌ | ✅ CUDA/CPU | ✅ CPU | ✅ CUDA/CPU | Apache-2.0 |
|
||||
| **dots.tts**(2B,可选装) | 24 | ✅ | ❌ | ✅ CUDA/CPU | ✅ CPU | ❌ | Apache-2.0 |
|
||||
| **KittenTTS** | 英语 | ❌ | ❌ | ✅ CPU | ✅ CPU | ✅ CPU | MIT |
|
||||
|
||||
> **CUDA** = GPU 加速 · **MPS** = Apple Silicon Metal · **CPU** = 随处可运行,大模型较慢 · KittenTTS 和 MOSS-TTS-Nano 可在 CPU 上实时运行 · MLX-Audio 仅限 Apple Silicon。
|
||||
>
|
||||
> **MOSS-TTS-v1.5**(8B,约 16 GB 权重)和 **dots.tts**(2B,约 9 GB 权重)是重量级可选引擎,从本地克隆在独立 venv 中运行——参见 [MOSS-TTS-v1.5](docs/engines/moss-tts-v15.md) 和 [dots.tts](docs/engines/dots-tts.md)。两者均不支持 Apple Silicon **MPS**(上游仅支持 CUDA/CPU;在 Mac 上以 CPU 运行)。dots.tts 上游仅支持 Linux/macOS——无 Windows 路径。
|
||||
|
||||
---
|
||||
|
||||
|
||||
+7
-1
@@ -13,12 +13,18 @@
|
||||
# Run: uv run pyinstaller backend.spec --noconfirm --clean
|
||||
import platform
|
||||
import sys
|
||||
from PyInstaller.utils.hooks import collect_data_files, collect_all, collect_submodules
|
||||
from PyInstaller.utils.hooks import collect_data_files, collect_all, collect_submodules, copy_metadata
|
||||
|
||||
IS_MAC_ARM = sys.platform == "darwin" and platform.machine() == "arm64"
|
||||
|
||||
datas = []
|
||||
binaries = []
|
||||
|
||||
# Bundle the omnivoice package's .dist-info so importlib.metadata.version()
|
||||
# resolves inside the frozen build. Without it the backend can't read its own
|
||||
# version and falls back to the literal in backend/core/version.py — which is
|
||||
# how a 0.3.6 desktop build shipped reporting "0.3.5" in About + bug reports.
|
||||
datas += copy_metadata('omnivoice')
|
||||
hiddenimports = [
|
||||
# Web stack
|
||||
'uvicorn', 'uvicorn.logging', 'uvicorn.loops', 'uvicorn.loops.auto',
|
||||
|
||||
@@ -18,7 +18,6 @@ Design notes
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import hashlib
|
||||
import logging
|
||||
import os
|
||||
@@ -137,7 +136,7 @@ async def _render_archetype_wav(a: dict, out_path: Path) -> None:
|
||||
from api.routers.generation import ( # noqa: WPS433 — intentional lazy import
|
||||
get_model,
|
||||
_run_inference,
|
||||
_gpu_pool,
|
||||
run_on_gpu_pool_guarded,
|
||||
_safe_torchaudio_save,
|
||||
)
|
||||
|
||||
@@ -147,8 +146,6 @@ async def _render_archetype_wav(a: dict, out_path: Path) -> None:
|
||||
language = None
|
||||
text = (a.get("sample_script") or "").strip() or _FALLBACK_SCRIPT
|
||||
|
||||
loop = asyncio.get_running_loop()
|
||||
|
||||
def _infer(seed: int):
|
||||
return _run_inference(
|
||||
model, # _model
|
||||
@@ -171,14 +168,18 @@ async def _render_archetype_wav(a: dict, out_path: Path) -> None:
|
||||
"broadcast", # effect_preset
|
||||
)
|
||||
|
||||
audio_tensor = await loop.run_in_executor(_gpu_pool, _infer, _PREVIEW_SEED)
|
||||
# Bounded + pool-reset on hang so a wedged preview render can't starve the
|
||||
# GPU pool and brick the backend (#730 class).
|
||||
audio_tensor = await run_on_gpu_pool_guarded(
|
||||
lambda: _infer(_PREVIEW_SEED), what="Archetype preview generate")
|
||||
if _is_unusable_audio(audio_tensor):
|
||||
# Blank OR a degenerate tonal buzz — retry once on a different seed to
|
||||
# step off the bad diffusion trajectory. Static message only: the
|
||||
# archetype id is request-derived (CodeQL log-injection); the seed is a
|
||||
# module constant, safe to log.
|
||||
logger.warning("Archetype rendered unusable at seed %d — retrying once", _PREVIEW_SEED)
|
||||
audio_tensor = await loop.run_in_executor(_gpu_pool, _infer, _PREVIEW_SEED + 1)
|
||||
audio_tensor = await run_on_gpu_pool_guarded(
|
||||
lambda: _infer(_PREVIEW_SEED + 1), what="Archetype preview generate")
|
||||
if _is_unusable_audio(audio_tensor):
|
||||
raise RuntimeError("the voice engine returned no audible audio for this archetype")
|
||||
|
||||
|
||||
@@ -164,6 +164,7 @@ async def audiobook_cover(cover: UploadFile = File(...)) -> dict:
|
||||
class AudiobookRequest(BaseModel):
|
||||
text: str
|
||||
default_voice: str | None = None # voice profile id; None = engine default
|
||||
language: str | None = None # None/"Auto" → profile language, else autodetect (#505)
|
||||
bitrate: str = "128k"
|
||||
format: str = "m4b" # "m4b" | "mp3"
|
||||
loudness: str | None = None # None/"off" | "acx" | "podcast" (opt-in)
|
||||
@@ -215,7 +216,35 @@ def _resolve_voice(profile_id: str | None) -> dict:
|
||||
return out
|
||||
|
||||
|
||||
def _build_synth(default_voice: str | None) -> dict:
|
||||
def _resolve_default_language(language: str | None, default_voice: str | None) -> str | None:
|
||||
"""Pick the language to thread into the longform synth callable.
|
||||
|
||||
Priority (mirrors the single-shot /generate path, #533): an explicit
|
||||
non-Auto request ``language`` wins; otherwise the selected profile's stored
|
||||
language drives it; otherwise ``None`` (genuine Auto — the engine
|
||||
autodetects, exactly as before). Hardcoding ``None`` here (#505 B2) let the
|
||||
engine re-autodetect per chunk, so a non-English clone flipped to the wrong
|
||||
language on short/ambiguous chapters.
|
||||
"""
|
||||
if language and language != "Auto":
|
||||
return language
|
||||
if default_voice:
|
||||
from core.db import db_conn
|
||||
with db_conn() as conn:
|
||||
row = conn.execute(
|
||||
"SELECT language FROM voice_profiles WHERE id=?", (default_voice,)
|
||||
).fetchone()
|
||||
if row:
|
||||
try:
|
||||
prof_lang = row["language"]
|
||||
except (KeyError, IndexError):
|
||||
prof_lang = None
|
||||
if prof_lang and prof_lang != "Auto":
|
||||
return prof_lang
|
||||
return None
|
||||
|
||||
|
||||
def _build_synth(default_voice: str | None, language: str | None = None) -> dict:
|
||||
"""Describe how to synthesize for the active TTS engine.
|
||||
|
||||
Returns a dict with ``mode``, ``resolve`` (voice-id → resolved refs, cached
|
||||
@@ -223,6 +252,11 @@ def _build_synth(default_voice: str | None) -> dict:
|
||||
``get_model``; other engines carry a ready ``synth`` + ``sample_rate``.
|
||||
:func:`_prepare_synth` turns this into a uniform ``(synth, sr, resolve,
|
||||
engine_id)`` once the (async) model is in hand.
|
||||
|
||||
``language`` (already resolved by :func:`_resolve_default_language`) is
|
||||
threaded into every chunk's ``generate`` so a non-English clone stays in its
|
||||
language instead of re-autodetecting per chunk (#505 B2). ``None`` keeps the
|
||||
engine's autodetect behavior unchanged.
|
||||
"""
|
||||
from services.tts_backend import OmniVoiceBackend, active_backend_id, get_backend_class
|
||||
|
||||
@@ -239,14 +273,14 @@ def _build_synth(default_voice: str | None) -> dict:
|
||||
if cls is OmniVoiceBackend:
|
||||
from services.model_manager import get_model
|
||||
return {"mode": "omnivoice", "resolve": resolve,
|
||||
"engine_id": engine_id, "get_model": get_model}
|
||||
"engine_id": engine_id, "get_model": get_model, "language": language}
|
||||
|
||||
backend = cls()
|
||||
|
||||
def synth(text, voice_id, speed=None):
|
||||
v = resolve(voice_id)
|
||||
return backend.generate(
|
||||
text, language=None, ref_audio=v["ref_audio"],
|
||||
text, language=language, ref_audio=v["ref_audio"],
|
||||
ref_text=v["ref_text"], instruct=v["instruct"], duration=None,
|
||||
speed=float(speed) if speed else 1.0,
|
||||
)
|
||||
@@ -254,20 +288,22 @@ def _build_synth(default_voice: str | None) -> dict:
|
||||
"synth": synth, "sample_rate": backend.sample_rate}
|
||||
|
||||
|
||||
async def _prepare_synth(default_voice: str | None):
|
||||
async def _prepare_synth(default_voice: str | None, language: str | None = None):
|
||||
"""Resolve :func:`_build_synth` into ``(synth, sample_rate, resolve,
|
||||
engine_id)`` — awaiting the OmniVoice model load when needed. Shared by the
|
||||
full job and the per-chapter preview."""
|
||||
info = _build_synth(default_voice)
|
||||
full job and the per-chapter preview. ``language`` is threaded into every
|
||||
chunk so a non-English clone holds its language (#505 B2)."""
|
||||
info = _build_synth(default_voice, language=language)
|
||||
resolve, engine_id = info["resolve"], info["engine_id"]
|
||||
if info["mode"] == "omnivoice":
|
||||
lang = info["language"]
|
||||
model = await info["get_model"]()
|
||||
sr = getattr(model, "sampling_rate", 24000)
|
||||
|
||||
def synth(text, voice_id, speed=None):
|
||||
v = resolve(voice_id)
|
||||
return model.generate(
|
||||
text=text, language=None, ref_audio=v["ref_audio"],
|
||||
text=text, language=lang, ref_audio=v["ref_audio"],
|
||||
ref_text=v["ref_text"], instruct=v["instruct"], duration=None,
|
||||
speed=float(speed) if speed else 1.0,
|
||||
)[0]
|
||||
@@ -323,6 +359,7 @@ class AudiobookPreviewRequest(BaseModel):
|
||||
text: str
|
||||
chapter_index: int = 0
|
||||
default_voice: str | None = None
|
||||
language: str | None = None # None/"Auto" → profile language, else autodetect
|
||||
lexicon: dict | None = None
|
||||
|
||||
|
||||
@@ -346,7 +383,10 @@ async def audiobook_preview(req: AudiobookPreviewRequest) -> dict:
|
||||
chapter = plan.chapters[req.chapter_index]
|
||||
cache_dir = os.path.join(OUTPUTS_DIR, "longform_cache") # shared with _render_longform_sse
|
||||
os.makedirs(cache_dir, exist_ok=True)
|
||||
synth, sr, resolve, engine_id = await _prepare_synth(req.default_voice)
|
||||
synth, sr, resolve, engine_id = await _prepare_synth(
|
||||
req.default_voice,
|
||||
language=_resolve_default_language(req.language, req.default_voice),
|
||||
)
|
||||
loop = asyncio.get_running_loop()
|
||||
wav_path, dur, was_cached = await loop.run_in_executor(
|
||||
_gpu_pool, _render_chapter_cached, chapter, synth, sr, engine_id, resolve, cache_dir,
|
||||
@@ -364,6 +404,7 @@ async def _render_longform_sse(
|
||||
plan,
|
||||
*,
|
||||
default_voice: str | None,
|
||||
language: str | None = None,
|
||||
fmt: str = "m4b",
|
||||
bitrate: str = "128k",
|
||||
loudness: str | None = None,
|
||||
@@ -412,7 +453,8 @@ async def _render_longform_sse(
|
||||
for c in plan.chapters
|
||||
],
|
||||
params={
|
||||
"default_voice": default_voice, "fmt": fmt, "bitrate": bitrate,
|
||||
"default_voice": default_voice, "language": language,
|
||||
"fmt": fmt, "bitrate": bitrate,
|
||||
"loudness": loudness, "cover_path": cover_path,
|
||||
"metadata": metadata, "lexicon": lexicon,
|
||||
},
|
||||
@@ -453,7 +495,9 @@ async def _render_longform_sse(
|
||||
loop = asyncio.get_running_loop()
|
||||
|
||||
try:
|
||||
synth, sr, resolve, engine_id = await _prepare_synth(default_voice)
|
||||
synth, sr, resolve, engine_id = await _prepare_synth(
|
||||
default_voice, language=_resolve_default_language(language, default_voice)
|
||||
)
|
||||
|
||||
total = len(plan.chapters)
|
||||
chapter_files: list[str] = []
|
||||
@@ -557,7 +601,8 @@ async def audiobook_synthesize(req: AudiobookRequest):
|
||||
plan = parse_audiobook_script(req.text, default_voice=req.default_voice)
|
||||
return StreamingResponse(
|
||||
_render_longform_sse(
|
||||
plan, default_voice=req.default_voice, fmt=req.format, bitrate=req.bitrate,
|
||||
plan, default_voice=req.default_voice, language=req.language,
|
||||
fmt=req.format, bitrate=req.bitrate,
|
||||
loudness=req.loudness, cover_path=req.cover_path, metadata=req.metadata,
|
||||
lexicon=req.lexicon, job_type="audiobook",
|
||||
),
|
||||
@@ -582,6 +627,7 @@ class LongformChapter(BaseModel):
|
||||
class LongformRenderRequest(BaseModel):
|
||||
chapters: list[LongformChapter] = []
|
||||
default_voice: str | None = None
|
||||
language: str | None = None # None/"Auto" → profile language, else autodetect (#505)
|
||||
bitrate: str = "128k"
|
||||
format: str = "m4b"
|
||||
loudness: str | None = None
|
||||
@@ -612,7 +658,8 @@ async def longform_render(req: LongformRenderRequest):
|
||||
plan = AudiobookPlan(chapters=chapters)
|
||||
return StreamingResponse(
|
||||
_render_longform_sse(
|
||||
plan, default_voice=req.default_voice, fmt=req.format, bitrate=req.bitrate,
|
||||
plan, default_voice=req.default_voice, language=req.language,
|
||||
fmt=req.format, bitrate=req.bitrate,
|
||||
loudness=req.loudness, cover_path=req.cover_path, metadata=req.metadata,
|
||||
lexicon=req.lexicon, job_type="story",
|
||||
),
|
||||
@@ -702,7 +749,7 @@ async def resume_longform(job_id: str):
|
||||
# never names a work dir / output file (defence-in-depth path-injection).
|
||||
return StreamingResponse(
|
||||
_render_longform_sse(
|
||||
plan, default_voice=p.get("default_voice"),
|
||||
plan, default_voice=p.get("default_voice"), language=p.get("language"),
|
||||
fmt=p.get("fmt", "m4b"), bitrate=p.get("bitrate", "128k"),
|
||||
loudness=p.get("loudness"), cover_path=p.get("cover_path"),
|
||||
metadata=p.get("metadata"), lexicon=p.get("lexicon"),
|
||||
|
||||
@@ -142,7 +142,7 @@ async def _run_batch_pipeline(job_id: str, job: dict):
|
||||
_set_progress(job, "transcribe", 0)
|
||||
|
||||
from services.asr_backend import get_active_asr_backend
|
||||
from services.model_manager import _gpu_pool, _cpu_pool
|
||||
from services.model_manager import _gpu_pool, _cpu_pool, run_on_gpu_pool_guarded
|
||||
from services.segmentation import (
|
||||
segment_transcript, assign_speakers_heuristic,
|
||||
)
|
||||
@@ -162,7 +162,12 @@ async def _run_batch_pipeline(job_id: str, job: dict):
|
||||
pass
|
||||
return segments, detected_lang
|
||||
|
||||
segments, source_lang = await loop.run_in_executor(_gpu_pool, _transcribe)
|
||||
# Bound the batch transcribe (#730) so a wedged whisperx/CTranslate2 call
|
||||
# can't hold its GPU-pool worker forever and starve the rest of the backend
|
||||
# ("can't reach backend"); run_transcribe_guarded also resets the pool on
|
||||
# timeout to restore capacity.
|
||||
from services.asr_backend import run_transcribe_guarded
|
||||
segments, source_lang = await run_transcribe_guarded(_gpu_pool, _transcribe, what="Batch")
|
||||
source_lang = (source_lang or "en").split("_")[0][:2].lower()
|
||||
job["segments"] = segments
|
||||
job["source_lang"] = source_lang
|
||||
@@ -311,7 +316,9 @@ async def _run_batch_pipeline(job_id: str, job: dict):
|
||||
return torch.zeros(1, int(dur * sr))
|
||||
|
||||
try:
|
||||
audio_tensor = await loop.run_in_executor(_gpu_pool, _gen)
|
||||
# Bounded + pool-reset on hang so a wedged batch segment can't
|
||||
# starve the GPU pool and brick the backend (#730 class).
|
||||
audio_tensor = await run_on_gpu_pool_guarded(_gen, what="Batch generate")
|
||||
|
||||
# Fit to slot
|
||||
target_samples_seg = int(seg_duration * sr)
|
||||
|
||||
@@ -18,7 +18,7 @@ import os
|
||||
import tempfile
|
||||
import time
|
||||
|
||||
from fastapi import APIRouter, File, Form, UploadFile
|
||||
from fastapi import APIRouter, File, Form, HTTPException, UploadFile
|
||||
from typing import Optional
|
||||
|
||||
router = APIRouter()
|
||||
@@ -96,9 +96,17 @@ async def transcribe_audio(
|
||||
return result, backend.id
|
||||
|
||||
from services.model_manager import _gpu_pool
|
||||
loop = asyncio.get_running_loop()
|
||||
from services.asr_backend import ASRTimeoutError, run_transcribe_guarded
|
||||
t0 = time.perf_counter()
|
||||
result, engine_id = await loop.run_in_executor(_gpu_pool, _run)
|
||||
try:
|
||||
result, engine_id = await run_transcribe_guarded(
|
||||
_gpu_pool, _run, what="Dictation",
|
||||
)
|
||||
except ASRTimeoutError as e:
|
||||
# Backend is alive — ASR couldn't finish. 504 with guidance, not a
|
||||
# silent hang the UI reads as "can't reach the local backend".
|
||||
logger.warning("Capture transcription timed out: %s", e)
|
||||
raise HTTPException(status_code=504, detail=str(e))
|
||||
elapsed = round(time.perf_counter() - t0, 2)
|
||||
|
||||
# Normalize result shape
|
||||
|
||||
@@ -101,6 +101,30 @@ def _pcm16_to_wav(pcm: bytes, sample_rate: int) -> str | None:
|
||||
return None
|
||||
|
||||
|
||||
def _select_sherpa_spec(websocket: WebSocket):
|
||||
"""Resolve the sherpa dictation model for this WS session, or None.
|
||||
|
||||
A ``?model=<id>`` query param wins (the frontend can pin a model per
|
||||
session); otherwise the persisted ``dictation.model_id`` pref is used (only
|
||||
when dictation is enabled). Returns the :class:`SherpaModelSpec` or None
|
||||
(None → the legacy Whisper/WebM path runs unchanged).
|
||||
"""
|
||||
try:
|
||||
from services import sherpa_dictation as sd
|
||||
except Exception:
|
||||
return None
|
||||
requested = websocket.query_params.get("model")
|
||||
if requested:
|
||||
return sd.get_spec(requested) # explicit selection (may be None if bad)
|
||||
# Fall back to the persisted dictation pref.
|
||||
try:
|
||||
from services.asr_backend import dictation_model_id
|
||||
mid = dictation_model_id()
|
||||
except Exception:
|
||||
mid = None
|
||||
return sd.get_spec(mid) if mid else None
|
||||
|
||||
|
||||
@router.websocket("/ws/transcribe")
|
||||
async def ws_transcribe(websocket: WebSocket):
|
||||
"""Stream audio in, get partial + final transcription out."""
|
||||
@@ -119,6 +143,24 @@ async def ws_transcribe(websocket: WebSocket):
|
||||
|
||||
await websocket.accept()
|
||||
|
||||
# Live-dictation engine selection. When a sherpa-onnx model is selected
|
||||
# (via ?model= or the dictation.model_id pref) AND sherpa is installed,
|
||||
# run the dedicated low-latency handler. Otherwise fall through to the
|
||||
# legacy Whisper/WebM path, byte-for-byte unchanged.
|
||||
spec = _select_sherpa_spec(websocket)
|
||||
if spec is not None:
|
||||
from services.asr_backend import SherpaDictationBackend
|
||||
ok, _reason = SherpaDictationBackend.is_available()
|
||||
if ok:
|
||||
if spec.streaming:
|
||||
await _run_sherpa_streaming(websocket, spec)
|
||||
else:
|
||||
await _run_sherpa_offline(websocket, spec)
|
||||
return
|
||||
# sherpa not installed → fall through to the legacy path so the user
|
||||
# still gets dictation (just not live partials).
|
||||
logger.info("sherpa dictation selected but unavailable — legacy path")
|
||||
|
||||
# Opt-in dictate-over-playback AEC (parity Action 8b). Default OFF →
|
||||
# identical legacy behaviour. When on, frames are 1-byte-tagged raw PCM
|
||||
# and the cleaned mic stream is muxed via stdlib wave (not ffmpeg).
|
||||
@@ -292,6 +334,321 @@ async def ws_transcribe(websocket: WebSocket):
|
||||
pass
|
||||
|
||||
|
||||
# ── sherpa-onnx live dictation handlers ─────────────────────────────────────
|
||||
#
|
||||
# Both handlers read raw int16 mono PCM frames (reusing the AEC framing: an
|
||||
# opt-in 1-byte type prefix when ?aec=1, else bare PCM) at ?sr= (default 16000).
|
||||
# This is the low-latency transport — no WebM/ffmpeg in the hot path.
|
||||
|
||||
# How often the offline-kind handler re-decodes the growing buffer for a live
|
||||
# partial (streaming-kind decodes every frame, no cadence needed).
|
||||
SHERPA_OFFLINE_PARTIAL_S = float(os.environ.get("OMNIVOICE_SHERPA_OFFLINE_PARTIAL", "0.8"))
|
||||
|
||||
|
||||
def _pcm16_to_f32(pcm: bytes):
|
||||
"""int16 little-endian mono PCM bytes → float32 numpy in [-1, 1]."""
|
||||
import numpy as np
|
||||
if not pcm:
|
||||
return np.zeros(0, dtype=np.float32)
|
||||
# Guard against an odd trailing byte from a split frame.
|
||||
if len(pcm) % 2:
|
||||
pcm = pcm[:-1]
|
||||
return np.frombuffer(pcm, dtype=np.int16).astype(np.float32) / 32768.0
|
||||
|
||||
|
||||
async def _sherpa_session(websocket: WebSocket):
|
||||
"""Shared WS receive setup for the sherpa handlers.
|
||||
|
||||
Returns ``(get_frame, state)`` where ``get_frame`` is an async callable
|
||||
that yields the next near-end (mic) PCM bytes, ``b""`` for a keepalive/ref
|
||||
frame, or ``None`` on EOF/disconnect. ``state`` carries sample rate, AEC,
|
||||
and the disconnect flag for the caller's finaliser.
|
||||
"""
|
||||
pcm_sr = 16000
|
||||
try:
|
||||
pcm_sr = int(websocket.query_params.get("sr", "16000"))
|
||||
except (TypeError, ValueError):
|
||||
pcm_sr = 16000
|
||||
aec = None
|
||||
if websocket.query_params.get("aec") in ("1", "true", "on"):
|
||||
try:
|
||||
from services.aec import NlmsEchoCanceller
|
||||
aec = NlmsEchoCanceller(sample_rate=pcm_sr)
|
||||
except Exception as e:
|
||||
logger.warning("AEC requested but disabled (sherpa): %s", e)
|
||||
aec = None
|
||||
return pcm_sr, aec
|
||||
|
||||
|
||||
async def _recv_pcm_frame(websocket: WebSocket, aec):
|
||||
"""Receive one frame; return (kind, pcm_bytes).
|
||||
|
||||
kind ∈ {"near","eof","skip"}. Demuxes AEC-tagged frames when ``aec`` is on
|
||||
and feeds the playback reference into the canceller. A text "EOF" or an
|
||||
empty/closed socket yields kind "eof".
|
||||
"""
|
||||
msg = await websocket.receive()
|
||||
mtype = msg.get("type")
|
||||
if mtype == "websocket.disconnect":
|
||||
return "eof", b""
|
||||
if mtype != "websocket.receive":
|
||||
return "skip", b""
|
||||
data = msg.get("bytes")
|
||||
if data is not None:
|
||||
if len(data) == 0:
|
||||
return "eof", b""
|
||||
if aec is not None:
|
||||
kind, payload = _demux_aec_frame(data)
|
||||
if kind == "far":
|
||||
aec.push_far_end(payload)
|
||||
return "skip", b""
|
||||
if not payload:
|
||||
return "skip", b""
|
||||
return "near", aec.process_near_end(payload)
|
||||
return "near", data
|
||||
if msg.get("text") == "EOF":
|
||||
return "eof", b""
|
||||
return "skip", b""
|
||||
|
||||
|
||||
async def _run_sherpa_streaming(websocket: WebSocket, spec):
|
||||
"""True streaming: feed the OnlineRecognizer frame-by-frame, emit `partial`
|
||||
every time the decoded text grows, and `final` on sherpa's endpoint (silence)
|
||||
detection and on EOF. <300ms perceived latency on CPU for the tiny models.
|
||||
"""
|
||||
import numpy as np
|
||||
from services.asr_backend import SherpaDictationBackend
|
||||
|
||||
pcm_sr, aec = await _sherpa_session(websocket)
|
||||
logger.info("sherpa streaming dictation: model=%s sr=%d aec=%s",
|
||||
spec.id, pcm_sr, bool(aec))
|
||||
|
||||
backend = SherpaDictationBackend(model_id=spec.id)
|
||||
# Build the recognizer off the event loop (download-on-first-use + ONNX
|
||||
# session init can take a moment); keep the socket responsive.
|
||||
try:
|
||||
await asyncio.to_thread(backend.ensure_loaded)
|
||||
except Exception as e:
|
||||
logger.error("sherpa streaming load failed: %s", e)
|
||||
try:
|
||||
await websocket.send_json({"type": "error", "detail": str(e)})
|
||||
await websocket.close()
|
||||
except Exception:
|
||||
pass
|
||||
return
|
||||
rec = backend._rec
|
||||
stream = rec.create_stream()
|
||||
|
||||
last_partial = ""
|
||||
committed: list[str] = [] # finalized utterances this session
|
||||
client_disconnected = False
|
||||
|
||||
async def _send(payload) -> bool:
|
||||
nonlocal client_disconnected
|
||||
if client_disconnected:
|
||||
return False
|
||||
try:
|
||||
await websocket.send_json(payload)
|
||||
return True
|
||||
except Exception:
|
||||
client_disconnected = True
|
||||
return False
|
||||
|
||||
def _decode_after_feed(pcm: bytes):
|
||||
"""Blocking: feed one PCM frame, decode, return (text, is_endpoint).
|
||||
Runs in a thread so the ONNX work never blocks the event loop."""
|
||||
samples = _pcm16_to_f32(pcm)
|
||||
if len(samples):
|
||||
stream.accept_waveform(pcm_sr, samples)
|
||||
while rec.is_ready(stream):
|
||||
rec.decode_stream(stream)
|
||||
endpoint = rec.is_endpoint(stream)
|
||||
text = (rec.get_result(stream) or "").strip()
|
||||
return text, endpoint
|
||||
|
||||
def _flush_final():
|
||||
"""Blocking: pad + drain the stream for the trailing utterance."""
|
||||
tail = np.zeros(int(0.5 * pcm_sr), dtype=np.float32)
|
||||
stream.accept_waveform(pcm_sr, tail)
|
||||
stream.input_finished()
|
||||
while rec.is_ready(stream):
|
||||
rec.decode_stream(stream)
|
||||
return (rec.get_result(stream) or "").strip()
|
||||
|
||||
try:
|
||||
while True:
|
||||
kind, pcm = await _recv_pcm_frame(websocket, aec)
|
||||
if kind == "eof":
|
||||
break
|
||||
if kind == "skip":
|
||||
continue
|
||||
text, endpoint = await asyncio.to_thread(_decode_after_feed, pcm)
|
||||
if endpoint:
|
||||
# Commit this utterance; reset for the next one.
|
||||
if text:
|
||||
committed.append(text)
|
||||
await _send({"type": "final", "text": text,
|
||||
"segments": [{"start": 0.0, "end": None, "text": text}],
|
||||
"language": "auto", "engine": backend.id})
|
||||
rec.reset(stream)
|
||||
last_partial = ""
|
||||
elif text and text != last_partial:
|
||||
last_partial = text
|
||||
await _send({"type": "partial", "text": text})
|
||||
except WebSocketDisconnect:
|
||||
client_disconnected = True
|
||||
except Exception as e:
|
||||
logger.warning("sherpa streaming loop ended: %s", e)
|
||||
client_disconnected = True
|
||||
|
||||
# Drain the trailing (un-endpointed) utterance on EOF.
|
||||
try:
|
||||
tail_text = await asyncio.to_thread(_flush_final)
|
||||
except Exception as e:
|
||||
logger.debug("sherpa streaming flush failed: %s", e)
|
||||
tail_text = ""
|
||||
if tail_text and tail_text != (committed[-1] if committed else None):
|
||||
committed.append(tail_text)
|
||||
|
||||
full = " ".join(t for t in committed if t).strip()
|
||||
segments = [{"start": 0.0, "end": None, "text": t} for t in committed if t]
|
||||
if not client_disconnected:
|
||||
if full:
|
||||
try:
|
||||
from services.refinement import maybe_refine
|
||||
refined = await asyncio.to_thread(maybe_refine, full)
|
||||
except Exception:
|
||||
refined = None
|
||||
payload = {"type": "final", "text": full, "segments": segments,
|
||||
"language": "auto", "engine": backend.id}
|
||||
if refined and refined != full:
|
||||
payload["refined_text"] = refined
|
||||
await _send(payload)
|
||||
else:
|
||||
await _send({"type": "final", "text": "", "segments": [],
|
||||
"language": "auto", "engine": backend.id})
|
||||
try:
|
||||
await websocket.close()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
async def _run_sherpa_offline(websocket: WebSocket, spec):
|
||||
"""Offline-kind sherpa model with live partials: buffer raw PCM and
|
||||
re-decode the growing buffer every ~800ms so the user still sees text
|
||||
appear while speaking; finalize on EOF/silence."""
|
||||
from services.asr_backend import SherpaDictationBackend
|
||||
|
||||
pcm_sr, aec = await _sherpa_session(websocket)
|
||||
logger.info("sherpa offline dictation: model=%s sr=%d aec=%s",
|
||||
spec.id, pcm_sr, bool(aec))
|
||||
|
||||
backend = SherpaDictationBackend(model_id=spec.id)
|
||||
try:
|
||||
await asyncio.to_thread(backend.ensure_loaded)
|
||||
except Exception as e:
|
||||
logger.error("sherpa offline load failed: %s", e)
|
||||
try:
|
||||
await websocket.send_json({"type": "error", "detail": str(e)})
|
||||
await websocket.close()
|
||||
except Exception:
|
||||
pass
|
||||
return
|
||||
|
||||
buf = bytearray()
|
||||
last_partial = ""
|
||||
running = True
|
||||
client_disconnected = False
|
||||
last_audio = time.monotonic()
|
||||
|
||||
async def _send(payload) -> bool:
|
||||
nonlocal client_disconnected
|
||||
if client_disconnected:
|
||||
return False
|
||||
try:
|
||||
await websocket.send_json(payload)
|
||||
return True
|
||||
except Exception:
|
||||
client_disconnected = True
|
||||
return False
|
||||
|
||||
def _decode_buffer() -> str:
|
||||
samples = _pcm16_to_f32(bytes(buf))
|
||||
if not len(samples):
|
||||
return ""
|
||||
return backend._decode_offline(samples, pcm_sr)
|
||||
|
||||
async def receive():
|
||||
nonlocal running, client_disconnected, last_audio
|
||||
try:
|
||||
while running:
|
||||
kind, pcm = await _recv_pcm_frame(websocket, aec)
|
||||
if kind == "eof":
|
||||
running = False
|
||||
break
|
||||
if kind == "skip":
|
||||
continue
|
||||
buf.extend(pcm)
|
||||
last_audio = time.monotonic()
|
||||
except WebSocketDisconnect:
|
||||
client_disconnected = True
|
||||
running = False
|
||||
except Exception as e:
|
||||
logger.debug("sherpa offline receive ended: %s", e)
|
||||
running = False
|
||||
|
||||
async def partials():
|
||||
nonlocal last_partial, running
|
||||
while running:
|
||||
await asyncio.sleep(SHERPA_OFFLINE_PARTIAL_S)
|
||||
if not running or len(buf) < 2000:
|
||||
continue
|
||||
try:
|
||||
text = await asyncio.to_thread(_decode_buffer)
|
||||
except Exception as e:
|
||||
logger.debug("sherpa offline partial failed: %s", e)
|
||||
continue
|
||||
if text and text != last_partial:
|
||||
last_partial = text
|
||||
await _send({"type": "partial", "text": text})
|
||||
|
||||
recv_task = asyncio.create_task(receive())
|
||||
part_task = asyncio.create_task(partials())
|
||||
await asyncio.wait([recv_task, part_task], return_when=asyncio.FIRST_COMPLETED)
|
||||
running = False
|
||||
for t in (recv_task, part_task):
|
||||
if not t.done():
|
||||
t.cancel()
|
||||
try:
|
||||
await t
|
||||
except (asyncio.CancelledError, Exception):
|
||||
pass
|
||||
|
||||
try:
|
||||
full = await asyncio.to_thread(_decode_buffer)
|
||||
except Exception as e:
|
||||
logger.error("sherpa offline final failed: %s", e)
|
||||
full = ""
|
||||
full = (full or "").strip()
|
||||
segments = [{"start": 0.0, "end": None, "text": full}] if full else []
|
||||
if not client_disconnected:
|
||||
payload = {"type": "final", "text": full, "segments": segments,
|
||||
"language": "auto", "engine": backend.id}
|
||||
if full:
|
||||
try:
|
||||
from services.refinement import maybe_refine
|
||||
refined = await asyncio.to_thread(maybe_refine, full)
|
||||
if refined and refined != full:
|
||||
payload["refined_text"] = refined
|
||||
except Exception:
|
||||
pass
|
||||
await _send(payload)
|
||||
try:
|
||||
await websocket.close()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
async def _transcribe_buffer(chunks: list[bytes], *, pcm_sr: int | None = None) -> str:
|
||||
"""Quick partial transcription of the current audio buffer."""
|
||||
|
||||
@@ -301,15 +658,17 @@ async def _transcribe_buffer(chunks: list[bytes], *, pcm_sr: int | None = None)
|
||||
|
||||
try:
|
||||
from services.model_manager import _gpu_pool
|
||||
from services.asr_backend import get_capture_asr_backend
|
||||
from services.asr_backend import get_capture_asr_backend, run_transcribe_guarded
|
||||
|
||||
def _run():
|
||||
backend = get_capture_asr_backend()
|
||||
result = backend.transcribe(tmp, word_timestamps=False)
|
||||
return result.get("text", "")
|
||||
|
||||
loop = asyncio.get_running_loop()
|
||||
text = await loop.run_in_executor(_gpu_pool, _run)
|
||||
# Bound dictation transcribes (#730): a wedged whisperx/CTranslate2 call
|
||||
# must not hold its GPU-pool worker forever and starve TTS / other ASR
|
||||
# into a "can't reach backend"; on timeout the pool is reset to recover.
|
||||
text = await run_transcribe_guarded(_gpu_pool, _run, what="Dictation")
|
||||
return text.strip()
|
||||
finally:
|
||||
try:
|
||||
@@ -327,7 +686,7 @@ async def _transcribe_buffer_full(chunks: list[bytes], *, pcm_sr: int | None = N
|
||||
|
||||
try:
|
||||
from services.model_manager import _gpu_pool
|
||||
from services.asr_backend import get_capture_asr_backend
|
||||
from services.asr_backend import get_capture_asr_backend, run_transcribe_guarded
|
||||
|
||||
def _run():
|
||||
backend = get_capture_asr_backend()
|
||||
@@ -362,8 +721,9 @@ async def _transcribe_buffer_full(chunks: list[bytes], *, pcm_sr: int | None = N
|
||||
"engine": backend.id,
|
||||
}
|
||||
|
||||
loop = asyncio.get_running_loop()
|
||||
return await loop.run_in_executor(_gpu_pool, _run)
|
||||
# Bounded + pool-resetting on timeout (#730), same rationale as the
|
||||
# partial path above.
|
||||
return await run_transcribe_guarded(_gpu_pool, _run, what="Dictation")
|
||||
finally:
|
||||
try:
|
||||
os.unlink(tmp)
|
||||
|
||||
@@ -0,0 +1,127 @@
|
||||
"""
|
||||
Dictation router — sherpa-onnx live-dictation engine.
|
||||
|
||||
Exposes the seven sherpa-onnx dictation models and the dictation prefs the
|
||||
frontend dictation UI binds to.
|
||||
|
||||
GET /dictation/models → the 7 models + install state (frontend model list)
|
||||
GET /dictation/prefs → { enabled, mode, model_id }
|
||||
POST /dictation/prefs → persist any subset of those prefs
|
||||
|
||||
Install state reuses the same HF-cache check the model store uses, so a model
|
||||
shown "installed" here is the same snapshot the backend will load.
|
||||
|
||||
Prefs are stored in the shared ``prefs.json`` store under the ``dictation.*``
|
||||
namespace (``dictation.enabled``, ``dictation.mode``, ``dictation.model_id``),
|
||||
mirroring how the ASR/TTS engine picks persist.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
from fastapi import APIRouter, Depends, HTTPException
|
||||
from pydantic import BaseModel
|
||||
from typing import Optional
|
||||
|
||||
from api.dependencies import require_loopback
|
||||
from core import prefs
|
||||
from services import sherpa_dictation as sd
|
||||
|
||||
router = APIRouter()
|
||||
logger = logging.getLogger("omnivoice.dictation")
|
||||
|
||||
# Pref keys (the binding contract — the frontend writes exactly these).
|
||||
PREF_ENABLED = "dictation.enabled"
|
||||
PREF_MODE = "dictation.mode"
|
||||
PREF_MODEL_ID = "dictation.model_id"
|
||||
|
||||
_DEFAULT_ENABLED = True
|
||||
_DEFAULT_MODE = "toggle"
|
||||
_VALID_MODES = ("toggle", "hold")
|
||||
|
||||
|
||||
def _read_prefs() -> dict:
|
||||
mid = prefs.get(PREF_MODEL_ID, sd.DEFAULT_MODEL_ID)
|
||||
if not sd.is_sherpa_model(mid):
|
||||
mid = sd.DEFAULT_MODEL_ID
|
||||
mode = prefs.get(PREF_MODE, _DEFAULT_MODE)
|
||||
if mode not in _VALID_MODES:
|
||||
mode = _DEFAULT_MODE
|
||||
return {
|
||||
"enabled": bool(prefs.get(PREF_ENABLED, _DEFAULT_ENABLED)),
|
||||
"mode": mode,
|
||||
"model_id": mid,
|
||||
}
|
||||
|
||||
|
||||
@router.get("/dictation/models", dependencies=[Depends(require_loopback)])
|
||||
def list_dictation_models():
|
||||
"""The seven sherpa-onnx dictation models + install state.
|
||||
|
||||
Each entry: id, repo_id, label, tag ("offline"|"streaming"), recommended,
|
||||
size_gb, languages, kind, and install state (installed/installing). The
|
||||
``installed`` flag is computed from the same HF cache the model store reads,
|
||||
so it matches the model-store row state.
|
||||
"""
|
||||
available, reason = sd.sherpa_available()
|
||||
out = []
|
||||
for spec in sd.list_specs():
|
||||
out.append({
|
||||
"id": spec.id,
|
||||
"repo_id": spec.repo_id,
|
||||
"label": spec.label,
|
||||
"tag": spec.tag,
|
||||
"recommended": spec.recommended,
|
||||
"size_gb": spec.size_gb,
|
||||
"languages": spec.languages,
|
||||
"kind": spec.kind,
|
||||
"installed": sd.is_installed(spec),
|
||||
})
|
||||
return {
|
||||
"models": out,
|
||||
"engine_available": available,
|
||||
"engine_reason": None if available else reason,
|
||||
"default_model_id": sd.DEFAULT_MODEL_ID,
|
||||
}
|
||||
|
||||
|
||||
@router.get("/dictation/prefs", dependencies=[Depends(require_loopback)])
|
||||
def get_dictation_prefs():
|
||||
return _read_prefs()
|
||||
|
||||
|
||||
class DictationPrefsUpdate(BaseModel):
|
||||
enabled: Optional[bool] = None
|
||||
mode: Optional[str] = None
|
||||
model_id: Optional[str] = None
|
||||
|
||||
|
||||
@router.post("/dictation/prefs", dependencies=[Depends(require_loopback)])
|
||||
def set_dictation_prefs(req: DictationPrefsUpdate):
|
||||
"""Persist any subset of the dictation prefs. Validates ``mode`` and
|
||||
``model_id`` so a bad value can't wedge the capture engine."""
|
||||
if req.mode is not None:
|
||||
if req.mode not in _VALID_MODES:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"mode must be one of {_VALID_MODES}",
|
||||
)
|
||||
prefs.set_(PREF_MODE, req.mode)
|
||||
if req.model_id is not None:
|
||||
if not sd.is_sherpa_model(req.model_id):
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"unknown dictation model_id {req.model_id!r}",
|
||||
)
|
||||
# Normalise to the canonical dictation id (accept repo_id too).
|
||||
prefs.set_(PREF_MODEL_ID, sd.get_spec(req.model_id).id)
|
||||
if req.enabled is not None:
|
||||
prefs.set_(PREF_ENABLED, bool(req.enabled))
|
||||
# Rebuild the cached capture singleton so the change takes effect at once.
|
||||
try:
|
||||
from services import asr_backend
|
||||
asr_backend._capture_backend = None
|
||||
asr_backend._capture_backend_key = None
|
||||
except Exception:
|
||||
pass
|
||||
return _read_prefs()
|
||||
@@ -23,6 +23,9 @@ from services.segmentation import (
|
||||
assign_speakers_from_diarization,
|
||||
assign_speakers_from_turns,
|
||||
assign_speakers_heuristic,
|
||||
resplit_segments_by_diarization,
|
||||
resplit_segments_by_turns,
|
||||
_words_from_whisper,
|
||||
clean_up_segments,
|
||||
)
|
||||
from services.onset_align import snap_segment_starts
|
||||
@@ -31,6 +34,25 @@ from services import dub_pipeline
|
||||
router = APIRouter()
|
||||
logger = logging.getLogger("omnivoice.api")
|
||||
|
||||
|
||||
def _reset_pool_on_wedge(pool) -> None:
|
||||
"""Abandon a GPU pool whose worker is wedged on a timed-out transcribe (#730).
|
||||
|
||||
Python can't kill the stuck thread, but dropping the poisoned pool means the
|
||||
next submit (the next chunk, or a concurrent TTS generate) gets a fresh
|
||||
worker instead of queueing behind the wedged one — the same recovery the
|
||||
whole-file paths get inside ``run_transcribe_guarded``. Best-effort and a
|
||||
no-op for a pool without ``reset`` (a plain executor), so it never raises on
|
||||
the failure path it's trying to recover from.
|
||||
"""
|
||||
_reset = getattr(pool, "reset", None)
|
||||
if callable(_reset):
|
||||
try:
|
||||
_reset()
|
||||
except Exception:
|
||||
logger.exception("GPU pool reset after transcribe timeout failed")
|
||||
|
||||
|
||||
# ── Legacy-name aliases to services/dub_pipeline.py ────────────────────────
|
||||
# Phase 2.4 moved the business logic into a service. Other routers
|
||||
# (dub_generate, dub_translate, dub_export) + internal call sites below still
|
||||
@@ -425,10 +447,22 @@ async def dub_transcribe_stream(
|
||||
try:
|
||||
# The PyTorch-Whisper backend lazily builds its own pipeline
|
||||
# when no preloaded `_asr_pipe` is present (issue #255), so it
|
||||
# no longer needs OMNIVOICE_PRELOAD_TTS_ASR=1 — don't reject it
|
||||
# here; any load failure surfaces per-chunk with a real cause.
|
||||
# no longer needs OMNIVOICE_PRELOAD_TTS_ASR=1.
|
||||
_asr_backend = get_active_asr_backend(asr_pipe=getattr(_model, "_asr_pipe", None))
|
||||
# Eagerly load the model HERE so a real load failure (e.g.
|
||||
# WhisperX: missing weights, CTranslate2/cuDNN mismatch, the
|
||||
# torch-2.6 weights-only VAD regression) surfaces once, with
|
||||
# its actual cause, as a clean preflight `error` event —
|
||||
# instead of being buried in N cryptic per-chunk failures
|
||||
# and retried on every chunk (#578). Run in a thread so the
|
||||
# (blocking) load doesn't stall the event loop.
|
||||
_ensure_loaded = getattr(_asr_backend, "ensure_loaded", None)
|
||||
if callable(_ensure_loaded):
|
||||
await asyncio.get_running_loop().run_in_executor(
|
||||
_gpu_pool, _ensure_loaded
|
||||
)
|
||||
except Exception as e:
|
||||
logger.exception("transcribe preflight: ASR load failed (job=%s)", job_id)
|
||||
from core.failure import build_failure
|
||||
f = build_failure(e, stage="transcribe-preflight", include_diagnostic=False)
|
||||
preflight_error = "ASR backend initialization failed: " + f["reason"] + (
|
||||
@@ -436,9 +470,16 @@ async def dub_transcribe_stream(
|
||||
)
|
||||
scene_cuts = job.get("scene_cuts") or []
|
||||
|
||||
async def gen():
|
||||
async def _gen_body():
|
||||
if preflight_error:
|
||||
yield _sse_event("error", {"detail": preflight_error})
|
||||
# Always follow a terminal `error` with `done` so the stream closes
|
||||
# via a named event, not a raw connection drop. A bare error+close
|
||||
# races the browser's native EventSource error (which carries no
|
||||
# `data`); if that native error wins, the client falls back to the
|
||||
# misleading generic "stream dropped … ASR backend failed" message
|
||||
# and the real cause (in `detail`) is lost (#578).
|
||||
yield _sse_event("error", {"detail": preflight_error, "retryable": True})
|
||||
yield _sse_event("done", {})
|
||||
return
|
||||
import math
|
||||
import tempfile
|
||||
@@ -453,7 +494,9 @@ async def dub_transcribe_stream(
|
||||
try:
|
||||
audio_np, sr = await loop.run_in_executor(_cpu_pool, _load)
|
||||
except Exception as e:
|
||||
yield _sse_event("error", {"detail": f"audio load failed: {e}"})
|
||||
# Terminal error → always emit `done` (see preflight note, #578).
|
||||
yield _sse_event("error", {"detail": f"audio load failed: {e}", "retryable": True})
|
||||
yield _sse_event("done", {})
|
||||
return
|
||||
|
||||
total = float(len(audio_np)) / float(sr) if sr else 0.0
|
||||
@@ -470,6 +513,9 @@ async def dub_transcribe_stream(
|
||||
logger.warning("offload_tts_for_asr failed (continuing): %s", e)
|
||||
|
||||
all_segments: list[dict] = []
|
||||
# Words (global-timeline) retained so diarization can re-split a segment
|
||||
# that spans two speakers' turns at the word boundary (#486).
|
||||
all_words: list = []
|
||||
detected_lang = None
|
||||
next_seg_id = 0
|
||||
chunk_errors: list[str] = []
|
||||
@@ -541,6 +587,11 @@ async def dub_transcribe_stream(
|
||||
"Transcribe chunk %d/%d timed out after %.0fs (job=%s)",
|
||||
i + 1, chunks_n, TRANSCRIBE_CHUNK_TIMEOUT_S, job_id,
|
||||
)
|
||||
# #730: the wedged chunk thread keeps holding its GPU-pool worker.
|
||||
# Abandon the poisoned pool so the next chunk (and any TTS work)
|
||||
# gets a fresh worker instead of queueing behind the stuck one —
|
||||
# same recovery the whole-file paths get via run_transcribe_guarded.
|
||||
_reset_pool_on_wedge(_gpu_pool)
|
||||
part = {
|
||||
"chunks": [], "language": None,
|
||||
"error": f"Chunk {i+1} timed out after {TRANSCRIBE_CHUNK_TIMEOUT_S:.0f}s — "
|
||||
@@ -553,6 +604,12 @@ async def dub_transcribe_stream(
|
||||
detected_lang = part["language"]
|
||||
asr_speaker_turns.extend(part.get("speaker_turns") or [])
|
||||
chunk_segs = segment_transcript(part, duration=t1, scene_cuts=scene_cuts)
|
||||
# Same word source segment_transcript used (already global-timeline),
|
||||
# kept for the post-diarization speaker re-split (#486).
|
||||
try:
|
||||
all_words.extend(_words_from_whisper(part))
|
||||
except Exception:
|
||||
pass
|
||||
# #280: Whisper often stretches a segment's start back over
|
||||
# leading music/silence (classic case: speech begins at 0:03,
|
||||
# transcript says 0.0 → the dub plays 3 s early). Snap starts
|
||||
@@ -631,7 +688,10 @@ async def dub_transcribe_stream(
|
||||
# use its speaker turns directly and skip pyannote entirely (#182).
|
||||
if asr_speaker_turns:
|
||||
logger.info("Using inline ASR diarization (%d turns); skipping pyannote.", len(asr_speaker_turns))
|
||||
return assign_speakers_from_turns(all_segments, asr_speaker_turns), None
|
||||
assigned = assign_speakers_from_turns(all_segments, asr_speaker_turns)
|
||||
# #486: split any segment that spans two speakers' turns at the
|
||||
# word boundary (single-speaker segments pass through unchanged).
|
||||
return resplit_segments_by_turns(assigned, all_words, asr_speaker_turns), None
|
||||
|
||||
from services.model_manager import (
|
||||
DIARIZATION_ERR_LICENSE,
|
||||
@@ -708,7 +768,10 @@ async def dub_transcribe_stream(
|
||||
diar = diar_pipe(asr_audio_target, num_speakers=num_speakers)
|
||||
else:
|
||||
diar = diar_pipe(asr_audio_target)
|
||||
return assign_speakers_from_diarization(all_segments, diar), None
|
||||
assigned = assign_speakers_from_diarization(all_segments, diar)
|
||||
# #486: split any segment that spans two speakers' turns at the
|
||||
# word boundary (single-speaker segments pass through unchanged).
|
||||
return resplit_segments_by_diarization(assigned, all_words, diar), None
|
||||
except Exception as e:
|
||||
logger.error(f"Diarization failed: {e}")
|
||||
# Mid-run failure — classify against the same sentinels so a
|
||||
@@ -856,6 +919,25 @@ async def dub_transcribe_stream(
|
||||
})
|
||||
yield _sse_event("done", {})
|
||||
|
||||
async def gen():
|
||||
# Terminal-event guard (#516): the SSE stream must NEVER close without a
|
||||
# terminal event. Any unanticipated exception in the body (e.g. an ASR
|
||||
# load that escapes the per-chunk handler) previously dropped the
|
||||
# connection, which the frontend can only report as "stream dropped,
|
||||
# likely ASR failed" — hiding the real cause. Emit a structured `error`
|
||||
# (with the actionable hint from build_failure) then `done`, so the user
|
||||
# sees the real failure + a Retry instead of a silent disconnect.
|
||||
try:
|
||||
async for ev in _gen_body():
|
||||
yield ev
|
||||
except Exception as e: # noqa: BLE001 — last-resort stream finalizer
|
||||
logger.exception("transcribe stream crashed (job=%s)", job_id)
|
||||
from core.failure import build_failure
|
||||
f = build_failure(e, stage="transcribe", include_diagnostic=False)
|
||||
detail = f["reason"] + (f" — {f['hint']}" if f.get("hint") else "")
|
||||
yield _sse_event("error", {"detail": detail, "retryable": True})
|
||||
yield _sse_event("done", {})
|
||||
|
||||
return StreamingResponse(
|
||||
gen(),
|
||||
media_type="text/event-stream",
|
||||
@@ -959,7 +1041,12 @@ async def dub_transcribe(job_id: str):
|
||||
try:
|
||||
loop = asyncio.get_running_loop()
|
||||
try:
|
||||
segments_result = await loop.run_in_executor(_gpu_pool, _transcribe)
|
||||
# Bound the whole-file transcribe (#730): a wedged whisperx/CTranslate2
|
||||
# call would otherwise hold its GPU-pool worker forever and starve
|
||||
# every other request into a "can't reach backend". run_transcribe_guarded
|
||||
# also resets the pool on timeout so capacity is restored.
|
||||
from services.asr_backend import run_transcribe_guarded
|
||||
segments_result = await run_transcribe_guarded(_gpu_pool, _transcribe, what="Dub")
|
||||
except asyncio.CancelledError:
|
||||
job["aborted"] = True
|
||||
raise
|
||||
|
||||
@@ -1187,8 +1187,14 @@ async def dub_qc_pass(job_id: str, lang: str = Query(None), drift_threshold: flo
|
||||
|
||||
try:
|
||||
from services.model_manager import _get_gpu_pool
|
||||
loop = asyncio.get_running_loop()
|
||||
recognized, engine_id = await loop.run_in_executor(_get_gpu_pool(), _recognize)
|
||||
from services.asr_backend import ASRTimeoutError, run_transcribe_guarded
|
||||
recognized, engine_id = await run_transcribe_guarded(
|
||||
_get_gpu_pool(), _recognize, what="QC",
|
||||
)
|
||||
except ASRTimeoutError as e:
|
||||
# Backend is alive; ASR just couldn't finish in time. 504, not 500/connection.
|
||||
logger.warning("dub QC ASR pass timed out for %s: %s", job_id, e)
|
||||
raise HTTPException(status_code=504, detail=str(e))
|
||||
except Exception as e:
|
||||
logger.exception("dub QC ASR pass failed for %s", job_id)
|
||||
raise HTTPException(status_code=500, detail=f"QC transcription failed: {e}")
|
||||
|
||||
+415
-186
@@ -11,7 +11,7 @@ from core.db import db_conn
|
||||
from core.config import DUB_DIR, VOICES_DIR, dub_seg_path
|
||||
from core.tasks import task_manager
|
||||
from schemas.requests import DubRequest
|
||||
from services.model_manager import get_model, _gpu_pool
|
||||
from services.model_manager import get_model, _gpu_pool, run_on_gpu_pool_guarded
|
||||
from services.audio_dsp import apply_mastering, normalize_audio, apply_effects_chain, get_effect_chain
|
||||
from services.audio_io import atomic_save_wav, _safe_torchaudio_save
|
||||
from services.ffmpeg_utils import (
|
||||
@@ -28,6 +28,7 @@ from services.incremental import segment_fingerprint, fit_fingerprint
|
||||
from services.fit_planner import FitParams, plan_fit
|
||||
from services.watermark import embed_watermark
|
||||
from api.routers.dub_core import _get_job, _save_job
|
||||
from omnivoice.utils.voice_design import heal_design_instruct
|
||||
|
||||
logger = logging.getLogger("omnivoice.dub")
|
||||
|
||||
@@ -106,6 +107,123 @@ async def dub_generate(job_id: str, req: DubRequest):
|
||||
all_segment_wavs = []
|
||||
sync_scores = []
|
||||
|
||||
# Throttle the device cache flush. empty_cache() is a synchronous
|
||||
# device stall, so calling it every segment (as the old code did)
|
||||
# serialised the GPU loop; the batched-I/O design it replaced kept
|
||||
# it off the hot path on purpose. Flush every ~16 releases instead —
|
||||
# frequent enough to bound VRAM, rare enough to stay invisible.
|
||||
_RELEASE_FLUSH_EVERY = 16
|
||||
_release_count = {"n": 0}
|
||||
|
||||
def _release_audio_tensors(*objs) -> None:
|
||||
"""Best-effort VRAM cleanup after a segment is safely on disk.
|
||||
|
||||
Tensors are freed by the callers' own ``del`` once they fall out
|
||||
of scope; this only throttles the device cache flush. ``*objs`` is
|
||||
kept for call-site compatibility but intentionally unused — a local
|
||||
``del`` here would only unbind the parameter, never the caller's
|
||||
reference.
|
||||
"""
|
||||
_release_count["n"] += 1
|
||||
if _release_count["n"] % _RELEASE_FLUSH_EVERY != 0:
|
||||
return
|
||||
try:
|
||||
if torch.cuda.is_available():
|
||||
torch.cuda.empty_cache()
|
||||
elif hasattr(torch.backends, "mps") and torch.backends.mps.is_available():
|
||||
torch.mps.empty_cache()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# mix_<id> scratch WAVs written for silence/cached-fail/error slots are
|
||||
# pure assembly inputs (no preview/regen contract), so they're deleted
|
||||
# once the final track is written.
|
||||
_mix_temp_paths: list[str] = []
|
||||
|
||||
def _store_mix_wav(start: float, end: float, wav: torch.Tensor, sr: int, seg_key: str):
|
||||
"""Write one segment to disk and keep only its path in the mix manifest.
|
||||
|
||||
A zero/negative-length buffer is never written (``atomic_save_wav``
|
||||
raises on empty audio); instead a harmless zero-length in-memory
|
||||
entry is returned, which the assembly tolerates via its ``e > s``
|
||||
guard.
|
||||
"""
|
||||
if wav.shape[-1] <= 0:
|
||||
return (start, end, torch.zeros(1, 0), sr)
|
||||
path = dub_seg_path(job_id, seg_key)
|
||||
os.makedirs(os.path.dirname(path), exist_ok=True)
|
||||
atomic_save_wav(path, wav.detach().cpu(), sr)
|
||||
if seg_key.startswith("mix_"):
|
||||
_mix_temp_paths.append(path)
|
||||
_release_audio_tensors(wav)
|
||||
return (start, end, path, sr)
|
||||
|
||||
def _entry_num_samples(entry) -> int:
|
||||
# Zero/negative-duration slots are kept as in-memory tensors (never
|
||||
# written to disk); report their length directly.
|
||||
if isinstance(entry[2], torch.Tensor):
|
||||
return int(entry[2].shape[-1])
|
||||
try:
|
||||
info = torchaudio.info(entry[2])
|
||||
return int(info.num_frames)
|
||||
except Exception:
|
||||
wav, _sr = torchaudio.load(entry[2])
|
||||
n = int(wav.shape[-1])
|
||||
_release_audio_tensors(wav)
|
||||
return n
|
||||
|
||||
def _load_entry_wav(entry, target_sr: int) -> torch.Tensor:
|
||||
if isinstance(entry[2], torch.Tensor):
|
||||
return entry[2]
|
||||
wav, loaded_sr = torchaudio.load(entry[2])
|
||||
if loaded_sr != target_sr:
|
||||
import torchaudio.functional as AF
|
||||
wav = AF.resample(wav, loaded_sr, target_sr)
|
||||
return wav
|
||||
|
||||
def _write_memmap_wav_atomic(target_path: str, samples, sample_rate: int) -> None:
|
||||
"""Write a mono float32 memmap to int16 WAV without loading it all.
|
||||
|
||||
Intentionally does NOT watermark: the final track is assembled from
|
||||
per-segment WAVs that were already watermarked once at synthesis
|
||||
time (see the seg-write path below), exactly as ``main`` does.
|
||||
Re-marking here would double-mark every segment in the final mix.
|
||||
"""
|
||||
import tempfile
|
||||
import wave
|
||||
import numpy as np
|
||||
|
||||
target_dir = os.path.dirname(target_path) or "."
|
||||
target_base = os.path.basename(target_path)
|
||||
fd, tmp_path = tempfile.mkstemp(
|
||||
prefix=f".{target_base}.",
|
||||
suffix=".wav",
|
||||
dir=target_dir,
|
||||
)
|
||||
os.close(fd)
|
||||
chunk_samples = max(sample_rate * 30, 1)
|
||||
try:
|
||||
with wave.open(tmp_path, "wb") as wf:
|
||||
wf.setnchannels(1)
|
||||
wf.setsampwidth(2)
|
||||
wf.setframerate(sample_rate)
|
||||
total_len = int(samples.shape[0])
|
||||
for off in range(0, total_len, chunk_samples):
|
||||
chunk = np.array(samples[off: off + chunk_samples], dtype=np.float32, copy=True)
|
||||
if chunk.size == 0:
|
||||
continue
|
||||
np.nan_to_num(chunk, copy=False, nan=0.0, posinf=1.0, neginf=-1.0)
|
||||
chunk = np.clip(chunk, -1.0, 1.0)
|
||||
pcm = (chunk * 32767.0).astype("<i2", copy=False)
|
||||
wf.writeframes(pcm.tobytes())
|
||||
os.replace(tmp_path, target_path)
|
||||
except BaseException:
|
||||
try:
|
||||
os.unlink(tmp_path)
|
||||
except OSError:
|
||||
pass
|
||||
raise
|
||||
|
||||
# Phase 4.1 — partial regen. If `regen_only` is set, we only run TTS
|
||||
# on segments whose id is in that set; the others reuse their existing
|
||||
# `seg_i.wav` on disk and slot into the final mix unchanged.
|
||||
@@ -125,9 +243,9 @@ async def dub_generate(job_id: str, req: DubRequest):
|
||||
# reorder; index-keyed readers (preview/export) resolve via this manifest.
|
||||
job["seg_order"] = [seg_ids[k] if k < len(seg_ids) else f"seg_{k}" for k in range(len(req.segments))]
|
||||
|
||||
# Deferred disk writes: collect (index, tensor, sr, seg_id, fingerprint,
|
||||
# num_step) tuples during the hot loop and batch-flush after all TTS
|
||||
# completes. Eliminates ~200ms/seg of synchronous I/O from the GPU path.
|
||||
# Per-segment metadata to persist after the hot loop. Audio itself is
|
||||
# written immediately and only file paths are kept, so long videos don't
|
||||
# retain every generated tensor in RAM until final assembly.
|
||||
_pending_seg_writes: list[tuple] = []
|
||||
|
||||
# Phase 4.1 bench instrumentation: measure where incremental time goes.
|
||||
@@ -149,8 +267,16 @@ async def dub_generate(job_id: str, req: DubRequest):
|
||||
seg_duration = seg.end - seg.start
|
||||
if seg_duration <= 0.05 or not seg.text.strip():
|
||||
sr = _model.sampling_rate
|
||||
silence = torch.zeros(1, int(seg_duration * sr))
|
||||
all_segment_wavs.append((seg.start, seg.end, silence, sr))
|
||||
# max(0, …): a zero/negative-duration slot must not feed a
|
||||
# negative length to torch.zeros (raises) — _store_mix_wav
|
||||
# turns the empty buffer into a harmless in-memory entry.
|
||||
silence = torch.zeros(1, max(0, int(seg_duration * sr)))
|
||||
all_segment_wavs.append(_store_mix_wav(seg.start, seg.end, silence, sr, f"mix_{seg_id}"))
|
||||
try:
|
||||
del silence
|
||||
except Exception:
|
||||
pass
|
||||
_release_audio_tensors()
|
||||
sync_scores.append(1.0)
|
||||
continue
|
||||
|
||||
@@ -181,7 +307,12 @@ async def dub_generate(job_id: str, req: DubRequest):
|
||||
cached_wav = torch.nn.functional.pad(cached_wav, (0, target_samples - current_samples))
|
||||
elif current_samples > target_samples:
|
||||
cached_wav = cached_wav[..., :target_samples]
|
||||
all_segment_wavs.append((seg.start, seg.end, cached_wav, _model.sampling_rate))
|
||||
all_segment_wavs.append(_store_mix_wav(seg.start, seg.end, cached_wav, _model.sampling_rate, f"mix_{seg_id}"))
|
||||
try:
|
||||
del cached_wav
|
||||
except Exception:
|
||||
pass
|
||||
_release_audio_tensors()
|
||||
sync_scores.append(getattr(seg, 'sync_ratio', None) or 1.0)
|
||||
_t_cache += time.perf_counter() - _t_cache_0
|
||||
continue
|
||||
@@ -190,8 +321,13 @@ async def dub_generate(job_id: str, req: DubRequest):
|
||||
# is broken — cleaner than aborting the whole mix.
|
||||
yield f"data: {json.dumps({'type': 'warning', 'segment': i, 'message': f'cached seg lost, padding silence: {str(e)[:120]}'})}\n\n"
|
||||
sr = _model.sampling_rate
|
||||
silence = torch.zeros(1, int(seg_duration * sr))
|
||||
all_segment_wavs.append((seg.start, seg.end, silence, sr))
|
||||
silence = torch.zeros(1, max(0, int(seg_duration * sr)))
|
||||
all_segment_wavs.append(_store_mix_wav(seg.start, seg.end, silence, sr, f"mix_{seg_id}"))
|
||||
try:
|
||||
del silence
|
||||
except Exception:
|
||||
pass
|
||||
_release_audio_tensors()
|
||||
sync_scores.append(1.0)
|
||||
continue
|
||||
|
||||
@@ -258,7 +394,11 @@ async def dub_generate(job_id: str, req: DubRequest):
|
||||
used_seed = row["seed"]
|
||||
|
||||
if not instruct_str:
|
||||
instruct_str = row["instruct"]
|
||||
try:
|
||||
_vd = row["vd_states"]
|
||||
except (KeyError, IndexError):
|
||||
_vd = None
|
||||
instruct_str = heal_design_instruct(row["instruct"], _vd)
|
||||
|
||||
if used_seed is not None:
|
||||
torch.manual_seed(used_seed)
|
||||
@@ -401,10 +541,14 @@ async def dub_generate(job_id: str, req: DubRequest):
|
||||
# where dur_s is the slot hint.
|
||||
_dur_for_tts = seg_duration if strategy == "strict_slot" else None
|
||||
|
||||
audio_tensor = await loop.run_in_executor(
|
||||
_gpu_pool, _gen,
|
||||
seg.text, seg_lang, seg_instruct, _dur_for_tts,
|
||||
_num_step, req.guidance_scale, seg_speed, seg_profile, seg_effect_preset,
|
||||
# Bounded + pool-reset on hang so a wedged dub segment can't
|
||||
# starve the GPU pool and brick the backend (#730 class).
|
||||
audio_tensor = await run_on_gpu_pool_guarded(
|
||||
lambda: _gen(
|
||||
seg.text, seg_lang, seg_instruct, _dur_for_tts,
|
||||
_num_step, req.guidance_scale, seg_speed, seg_profile, seg_effect_preset,
|
||||
),
|
||||
what="Dub generate",
|
||||
)
|
||||
_t_tts += time.perf_counter() - _t_tts_0
|
||||
|
||||
@@ -452,7 +596,7 @@ async def dub_generate(job_id: str, req: DubRequest):
|
||||
except Exception as e:
|
||||
logger.debug("seg fingerprint skipped for %s: %s", seg_id, e)
|
||||
|
||||
_pending_seg_writes.append((i, audio_tensor, _model.sampling_rate, seg_id, _seg_fp, _num_step))
|
||||
_pending_seg_writes.append((i, _model.sampling_rate, seg_id, _seg_fp, _num_step))
|
||||
|
||||
# RVC needs the WAV on disk, so write it immediately only
|
||||
# when RVC is active (uncommon path).
|
||||
@@ -474,32 +618,54 @@ async def dub_generate(job_id: str, req: DubRequest):
|
||||
except Exception as e:
|
||||
yield f"data: {json.dumps({'type': 'warning', 'segment': i, 'message': f'RVC skipped: {str(e)[:120]}'})}\n\n"
|
||||
|
||||
all_segment_wavs.append((seg.start, seg.end, audio_tensor, _model.sampling_rate))
|
||||
# Watermark this FRESH TTS output exactly once, right before it
|
||||
# is persisted. The same seg_<id>.wav is BOTH the downloadable
|
||||
# per-segment file AND the assembly input for the final track,
|
||||
# so marking it here (and nowhere else) gives the downloadable
|
||||
# WAV its mark back and the final mix inherits it — no double-
|
||||
# mark. Cached-reuse audio is already marked; silence/zero slots
|
||||
# carry no speech to mark, so neither is re-watermarked.
|
||||
audio_tensor = embed_watermark(audio_tensor, _model.sampling_rate)
|
||||
|
||||
seg_wav_path = dub_seg_path(job_id, seg_id)
|
||||
try:
|
||||
# Keep the existing per-segment WAV contract for previews
|
||||
# and partial regeneration, but do not keep the tensor in RAM.
|
||||
atomic_save_wav(seg_wav_path, audio_tensor, _model.sampling_rate)
|
||||
except Exception as e:
|
||||
logger.warning("seg write failed for %s: %s", seg_id, e)
|
||||
# If the durable segment write fails, still preserve a mix
|
||||
# copy so this generation can finish.
|
||||
all_segment_wavs.append(_store_mix_wav(seg.start, seg.end, audio_tensor, _model.sampling_rate, f"mix_{seg_id}"))
|
||||
try:
|
||||
del audio_tensor
|
||||
except Exception:
|
||||
pass
|
||||
_release_audio_tensors()
|
||||
else:
|
||||
all_segment_wavs.append((seg.start, seg.end, seg_wav_path, _model.sampling_rate))
|
||||
try:
|
||||
del audio_tensor
|
||||
except Exception:
|
||||
pass
|
||||
_release_audio_tensors()
|
||||
except Exception as e:
|
||||
yield f"data: {json.dumps({'type': 'error', 'segment': i, 'error': str(e)})}\n\n"
|
||||
sr = _model.sampling_rate
|
||||
all_segment_wavs.append((seg.start, seg.end, torch.zeros(1, int(seg_duration * sr)), sr))
|
||||
all_segment_wavs.append(_store_mix_wav(seg.start, seg.end, torch.zeros(1, max(0, int(seg_duration * sr))), sr, f"mix_{seg_id}"))
|
||||
sync_scores.append(1.0)
|
||||
|
||||
_t_loop_end = time.perf_counter()
|
||||
|
||||
yield f"data: {json.dumps({'type': 'assembling'})}\n\n"
|
||||
|
||||
# ── Batch disk-write phase ────────────────────────────────────
|
||||
# Flush all per-segment WAVs and fingerprints in one burst now
|
||||
# that the GPU-hot loop is done. This keeps I/O off the critical
|
||||
# path and cuts ~200ms × N_segments of latency.
|
||||
# ── Batch metadata phase ──────────────────────────────────────
|
||||
# Per-segment WAVs were written during the loop to keep RAM bounded.
|
||||
# Flush only lightweight fingerprints/quality metadata here.
|
||||
_t_diskw_0 = time.perf_counter()
|
||||
hashes = job.setdefault("seg_hashes", {})
|
||||
quality_map = job.setdefault("seg_num_step", {})
|
||||
for (_si, _wav, _sr, _sid, _fp, _nstep) in _pending_seg_writes:
|
||||
seg_wav_path = dub_seg_path(job_id, _sid)
|
||||
try:
|
||||
# Apply invisible watermark before writing to disk
|
||||
_wav = embed_watermark(_wav, _sr)
|
||||
atomic_save_wav(seg_wav_path, _wav, _sr)
|
||||
except Exception as e:
|
||||
logger.warning("deferred seg write failed for %s: %s", _sid, e)
|
||||
for (_si, _sr, _sid, _fp, _nstep) in _pending_seg_writes:
|
||||
if _fp is not None:
|
||||
hashes[_sid] = _fp
|
||||
quality_map[_sid] = _nstep
|
||||
@@ -527,8 +693,8 @@ async def dub_generate(job_id: str, req: DubRequest):
|
||||
|
||||
if strategy == "stretch_video":
|
||||
cursor = 0.0
|
||||
for i, (orig_start, orig_end, wav, _) in enumerate(all_segment_wavs):
|
||||
wl_i = wav.shape[-1]
|
||||
for i, (orig_start, orig_end, wav_path, _) in enumerate(all_segment_wavs):
|
||||
wl_i = _entry_num_samples((orig_start, orig_end, wav_path, sr))
|
||||
natural_dur = (wl_i / sr) if wl_i > 0 else max(0.0, orig_end - orig_start)
|
||||
if i == 0:
|
||||
# Preserve the pre-roll (silence before the first seg).
|
||||
@@ -578,9 +744,9 @@ async def dub_generate(job_id: str, req: DubRequest):
|
||||
"start": s,
|
||||
"end": e,
|
||||
}
|
||||
for i, (s, e, _w, _) in enumerate(all_segment_wavs)
|
||||
for i, (s, e, _path, _) in enumerate(all_segment_wavs)
|
||||
],
|
||||
[w.shape[-1] / sr for (_s, _e, w, _) in all_segment_wavs],
|
||||
[_entry_num_samples(entry) / sr for entry in all_segment_wavs],
|
||||
orig_total_dur,
|
||||
fit_params,
|
||||
)
|
||||
@@ -593,183 +759,245 @@ async def dub_generate(job_id: str, req: DubRequest):
|
||||
# not from the plan — so subtitles land exactly on the audio.
|
||||
fitted_cues: list[dict] = []
|
||||
|
||||
full_audio = torch.zeros(1, total_samples)
|
||||
lang_code = req.language_code or "und"
|
||||
track_path = os.path.join(DUB_DIR, job_id, f"dubbed_{lang_code}.wav")
|
||||
os.makedirs(os.path.dirname(track_path), exist_ok=True)
|
||||
|
||||
for i, (start, end, wav, _) in enumerate(all_segment_wavs):
|
||||
seg_ref = req.segments[i] if i < len(req.segments) else None
|
||||
seg_gain = getattr(seg_ref, "gain", None) if seg_ref is not None else None
|
||||
seg_gain = seg_gain if seg_gain is not None else 1.0
|
||||
seg_gain = max(0.0, min(2.0, seg_gain))
|
||||
adjusted = wav * seg_gain
|
||||
wl = adjusted.shape[-1]
|
||||
natural_dur = wl / sr if wl > 0 else 0.0
|
||||
orig_dur = max(0.0, end - start)
|
||||
import gc
|
||||
import tempfile
|
||||
import numpy as np
|
||||
|
||||
if strategy == "stretch_video":
|
||||
# Mode B: audio at natural rate, placed on the stretched
|
||||
# timeline. No trim, no atempo. dub_export handles the video.
|
||||
new_start, _new_end = new_layout[i]
|
||||
place_at = new_start
|
||||
fit_status.append({
|
||||
"status": "video_stretched",
|
||||
"stretch_ratio": round(natural_dur / max(orig_dur, 1e-3), 3),
|
||||
})
|
||||
mix_samples = max(total_samples, 1)
|
||||
fd, mix_path = tempfile.mkstemp(
|
||||
prefix=f".{os.path.basename(track_path)}.mix.",
|
||||
suffix=".f32",
|
||||
dir=os.path.dirname(track_path),
|
||||
)
|
||||
os.close(fd)
|
||||
try:
|
||||
with open(mix_path, "r+b") as mix_file:
|
||||
mix_file.truncate(mix_samples * 4)
|
||||
mix_audio = np.memmap(mix_path, dtype=np.float32, mode="r+", shape=(mix_samples,))
|
||||
|
||||
elif strategy == "smart_fit":
|
||||
# Smart Fit: apply the planner's audio_rate via the same
|
||||
# pitch-preserving atempo pipe strict_slot uses, place the
|
||||
# result at the planned new_start, and hard-trim whatever
|
||||
# the caps couldn't absorb. The video side (video_ratio per
|
||||
# chunk) is persisted below for the export pipeline.
|
||||
sf = fit_plan.segments[i]
|
||||
place_at = sf.new_start
|
||||
if sf.audio_rate > 1.0 + 1e-6 and wl > 0:
|
||||
target = max(1, int(round(wl / sf.audio_rate)))
|
||||
try:
|
||||
adjusted = await _pitch_preserving_stretch(
|
||||
adjusted, target, sr,
|
||||
)
|
||||
except Exception as e:
|
||||
logger.warning(
|
||||
"atempo stretch failed for seg %d (%.2f×), "
|
||||
"falling back to linear interp: %s",
|
||||
i, sf.audio_rate, e,
|
||||
)
|
||||
adjusted = torch.nn.functional.interpolate(
|
||||
adjusted.unsqueeze(0),
|
||||
size=target,
|
||||
mode='linear',
|
||||
align_corners=False,
|
||||
).squeeze(0)
|
||||
wl = adjusted.shape[-1]
|
||||
# Residual overflow → hard-trim to the segment's new video
|
||||
# slot (fade below keeps the cut pop-free).
|
||||
new_slot_samples = int(max(0.0, sf.new_end - sf.new_start) * sr)
|
||||
if new_slot_samples > 0 and wl > new_slot_samples:
|
||||
adjusted = adjusted[..., :new_slot_samples]
|
||||
wl = adjusted.shape[-1]
|
||||
# Truthful per-segment verdict for the UI badge.
|
||||
entry = {"status": sf.status}
|
||||
if sf.audio_rate > 1.0 + 1e-6:
|
||||
entry["audio_rate"] = round(sf.audio_rate, 3)
|
||||
if sf.video_ratio > 1.0 + 1e-6:
|
||||
entry["video_ratio"] = round(sf.video_ratio, 3)
|
||||
if sf.overflow_s > 0:
|
||||
entry["overflow_s"] = round(sf.overflow_s, 3)
|
||||
fit_status.append(entry)
|
||||
# Cue times from the ACTUAL stretched sample positions.
|
||||
fitted_cues.append({
|
||||
"id": sf.seg_id,
|
||||
"start": round(place_at, 4),
|
||||
"end": round(place_at + wl / sr, 4),
|
||||
})
|
||||
for i, (start, end, wav_path, _) in enumerate(all_segment_wavs):
|
||||
seg_ref = req.segments[i] if i < len(req.segments) else None
|
||||
seg_gain = getattr(seg_ref, "gain", None) if seg_ref is not None else None
|
||||
seg_gain = seg_gain if seg_gain is not None else 1.0
|
||||
seg_gain = max(0.0, min(2.0, seg_gain))
|
||||
wav = _load_entry_wav((start, end, wav_path, sr), sr)
|
||||
adjusted = wav * seg_gain
|
||||
if adjusted.ndim == 2 and adjusted.shape[0] > 1:
|
||||
adjusted = adjusted.mean(dim=0, keepdim=True)
|
||||
wl = adjusted.shape[-1]
|
||||
natural_dur = wl / sr if wl > 0 else 0.0
|
||||
orig_dur = max(0.0, end - start)
|
||||
|
||||
elif strategy == "concise":
|
||||
# Mode A: never compress. Allow the audio to extend into the
|
||||
# silent gap before the next seg (existing heuristic) plus
|
||||
# any extra `overflow_budget_s`. Beyond that, hard-trim with
|
||||
# a short fade so we never overlap the next speaker.
|
||||
place_at = start
|
||||
effective_end = end
|
||||
if i + 1 < len(all_segment_wavs):
|
||||
next_start = all_segment_wavs[i + 1][0]
|
||||
gap = next_start - end
|
||||
if gap > GAP_OVERFLOW_BUFFER_S:
|
||||
effective_end = end + min(
|
||||
gap - GAP_OVERFLOW_BUFFER_S, GAP_OVERFLOW_MAX_S,
|
||||
)
|
||||
effective_end += overflow_budget_s
|
||||
slot_samples_eff = int(max(0.0, (effective_end - start)) * sr)
|
||||
if slot_samples_eff > 0 and wl > slot_samples_eff:
|
||||
overflow_s = (wl - slot_samples_eff) / sr
|
||||
adjusted = adjusted[..., :slot_samples_eff]
|
||||
wl = adjusted.shape[-1]
|
||||
if strategy == "stretch_video":
|
||||
# Mode B: audio at natural rate, placed on the stretched
|
||||
# timeline. No trim, no atempo. dub_export handles the video.
|
||||
new_start, _new_end = new_layout[i]
|
||||
place_at = new_start
|
||||
fit_status.append({
|
||||
"status": "overflows",
|
||||
"overflow_s": round(overflow_s, 3),
|
||||
"status": "video_stretched",
|
||||
"stretch_ratio": round(natural_dur / max(orig_dur, 1e-3), 3),
|
||||
})
|
||||
else:
|
||||
fit_status.append({"status": "fits"})
|
||||
|
||||
else:
|
||||
# strict_slot (legacy): preserve the previous atempo / trim /
|
||||
# off semantics so existing callers and back-compat tests
|
||||
# keep passing.
|
||||
place_at = start
|
||||
effective_end = end
|
||||
if i + 1 < len(all_segment_wavs):
|
||||
next_start = all_segment_wavs[i + 1][0]
|
||||
gap = next_start - end
|
||||
if gap > GAP_OVERFLOW_BUFFER_S:
|
||||
effective_end = end + min(
|
||||
gap - GAP_OVERFLOW_BUFFER_S, GAP_OVERFLOW_MAX_S,
|
||||
)
|
||||
slot_samples = int(max(0.0, (effective_end - start)) * sr)
|
||||
if slot_fit != "off" and slot_samples > 0 and wl > slot_samples:
|
||||
if slot_fit == "time_stretch":
|
||||
ratio = wl / slot_samples
|
||||
capped_ratio = min(ratio, MAX_STRETCH_RATIO)
|
||||
capped_target = int(wl / capped_ratio)
|
||||
elif strategy == "smart_fit":
|
||||
# Smart Fit: apply the planner's audio_rate via the same
|
||||
# pitch-preserving atempo pipe strict_slot uses, place the
|
||||
# result at the planned new_start, and hard-trim whatever
|
||||
# the caps couldn't absorb. The video side (video_ratio per
|
||||
# chunk) is persisted below for the export pipeline.
|
||||
sf = fit_plan.segments[i]
|
||||
place_at = sf.new_start
|
||||
if sf.audio_rate > 1.0 + 1e-6 and wl > 0:
|
||||
target = max(1, int(round(wl / sf.audio_rate)))
|
||||
try:
|
||||
adjusted = await _pitch_preserving_stretch(
|
||||
adjusted, capped_target, sr,
|
||||
adjusted, target, sr,
|
||||
)
|
||||
if adjusted.shape[-1] > slot_samples:
|
||||
adjusted = adjusted[..., :slot_samples]
|
||||
if ratio > MAX_STRETCH_RATIO:
|
||||
logger.info(
|
||||
"seg %d compression %.2f× exceeded cap; "
|
||||
"stretched to %.2f×, tail trimmed",
|
||||
i, ratio, capped_ratio,
|
||||
)
|
||||
except Exception as e:
|
||||
logger.warning(
|
||||
"atempo stretch failed for seg %d (%.2f×), "
|
||||
"falling back to linear interp: %s",
|
||||
i, ratio, e,
|
||||
i, sf.audio_rate, e,
|
||||
)
|
||||
adjusted = torch.nn.functional.interpolate(
|
||||
adjusted.unsqueeze(0),
|
||||
size=slot_samples,
|
||||
size=target,
|
||||
mode='linear',
|
||||
align_corners=False,
|
||||
).squeeze(0)
|
||||
else: # "trim"
|
||||
adjusted = adjusted[..., :slot_samples]
|
||||
wl = adjusted.shape[-1]
|
||||
# Residual overflow → hard-trim to the segment's new video
|
||||
# slot (fade below keeps the cut pop-free).
|
||||
new_slot_samples = int(max(0.0, sf.new_end - sf.new_start) * sr)
|
||||
if new_slot_samples > 0 and wl > new_slot_samples:
|
||||
adjusted = adjusted[..., :new_slot_samples]
|
||||
wl = adjusted.shape[-1]
|
||||
# Truthful per-segment verdict for the UI badge.
|
||||
entry = {"status": sf.status}
|
||||
if sf.audio_rate > 1.0 + 1e-6:
|
||||
entry["audio_rate"] = round(sf.audio_rate, 3)
|
||||
if sf.video_ratio > 1.0 + 1e-6:
|
||||
entry["video_ratio"] = round(sf.video_ratio, 3)
|
||||
if sf.overflow_s > 0:
|
||||
entry["overflow_s"] = round(sf.overflow_s, 3)
|
||||
fit_status.append(entry)
|
||||
# Cue times from the ACTUAL stretched sample positions.
|
||||
fitted_cues.append({
|
||||
"id": sf.seg_id,
|
||||
"start": round(place_at, 4),
|
||||
"end": round(place_at + wl / sr, 4),
|
||||
})
|
||||
|
||||
elif strategy == "concise":
|
||||
# Mode A: never compress. Allow the audio to extend into the
|
||||
# silent gap before the next seg (existing heuristic) plus
|
||||
# any extra `overflow_budget_s`. Beyond that, hard-trim with
|
||||
# a short fade so we never overlap the next speaker.
|
||||
place_at = start
|
||||
effective_end = end
|
||||
if i + 1 < len(all_segment_wavs):
|
||||
next_start = all_segment_wavs[i + 1][0]
|
||||
gap = next_start - end
|
||||
if gap > GAP_OVERFLOW_BUFFER_S:
|
||||
effective_end = end + min(
|
||||
gap - GAP_OVERFLOW_BUFFER_S, GAP_OVERFLOW_MAX_S,
|
||||
)
|
||||
effective_end += overflow_budget_s
|
||||
slot_samples_eff = int(max(0.0, (effective_end - start)) * sr)
|
||||
if slot_samples_eff > 0 and wl > slot_samples_eff:
|
||||
overflow_s = (wl - slot_samples_eff) / sr
|
||||
adjusted = adjusted[..., :slot_samples_eff]
|
||||
wl = adjusted.shape[-1]
|
||||
fit_status.append({
|
||||
"status": "overflows",
|
||||
"overflow_s": round(overflow_s, 3),
|
||||
})
|
||||
else:
|
||||
fit_status.append({"status": "fits"})
|
||||
|
||||
else:
|
||||
# strict_slot (legacy): preserve the previous atempo / trim /
|
||||
# off semantics so existing callers and back-compat tests
|
||||
# keep passing.
|
||||
place_at = start
|
||||
effective_end = end
|
||||
if i + 1 < len(all_segment_wavs):
|
||||
next_start = all_segment_wavs[i + 1][0]
|
||||
gap = next_start - end
|
||||
if gap > GAP_OVERFLOW_BUFFER_S:
|
||||
effective_end = end + min(
|
||||
gap - GAP_OVERFLOW_BUFFER_S, GAP_OVERFLOW_MAX_S,
|
||||
)
|
||||
slot_samples = int(max(0.0, (effective_end - start)) * sr)
|
||||
if slot_fit != "off" and slot_samples > 0 and wl > slot_samples:
|
||||
if slot_fit == "time_stretch":
|
||||
ratio = wl / slot_samples
|
||||
capped_ratio = min(ratio, MAX_STRETCH_RATIO)
|
||||
capped_target = int(wl / capped_ratio)
|
||||
try:
|
||||
adjusted = await _pitch_preserving_stretch(
|
||||
adjusted, capped_target, sr,
|
||||
)
|
||||
if adjusted.shape[-1] > slot_samples:
|
||||
adjusted = adjusted[..., :slot_samples]
|
||||
if ratio > MAX_STRETCH_RATIO:
|
||||
logger.info(
|
||||
"seg %d compression %.2f× exceeded cap; "
|
||||
"stretched to %.2f×, tail trimmed",
|
||||
i, ratio, capped_ratio,
|
||||
)
|
||||
except Exception as e:
|
||||
logger.warning(
|
||||
"atempo stretch failed for seg %d (%.2f×), "
|
||||
"falling back to linear interp: %s",
|
||||
i, ratio, e,
|
||||
)
|
||||
adjusted = torch.nn.functional.interpolate(
|
||||
adjusted.unsqueeze(0),
|
||||
size=slot_samples,
|
||||
mode='linear',
|
||||
align_corners=False,
|
||||
).squeeze(0)
|
||||
else: # "trim"
|
||||
adjusted = adjusted[..., :slot_samples]
|
||||
wl = adjusted.shape[-1]
|
||||
fit_status.append({
|
||||
"status": "fits",
|
||||
"compression_applied": (slot_fit == "time_stretch"
|
||||
and wl != int(natural_dur * sr)),
|
||||
})
|
||||
|
||||
# Common: short fades to avoid pops, then mix into disk-backed audio.
|
||||
fade_ms = 15
|
||||
fade_samples = int((fade_ms / 1000.0) * sr)
|
||||
if wl > fade_samples * 2:
|
||||
ramp_up = torch.linspace(0, 1, fade_samples, device=adjusted.device)
|
||||
ramp_down = torch.linspace(1, 0, fade_samples, device=adjusted.device)
|
||||
adjusted[0, :fade_samples] *= ramp_up
|
||||
adjusted[0, -fade_samples:] *= ramp_down
|
||||
|
||||
s = int(place_at * sr)
|
||||
if s < 0:
|
||||
adjusted = adjusted[..., -s:]
|
||||
wl = adjusted.shape[-1]
|
||||
fit_status.append({
|
||||
"status": "fits",
|
||||
"compression_applied": (slot_fit == "time_stretch"
|
||||
and wl != int(natural_dur * sr)),
|
||||
})
|
||||
s = 0
|
||||
e = min(s + wl, total_samples)
|
||||
if s < total_samples and e > s:
|
||||
mix_len = e - s
|
||||
seg_np = (
|
||||
adjusted[:, :mix_len]
|
||||
.detach()
|
||||
.cpu()
|
||||
.to(torch.float32)
|
||||
.clamp(-1.0, 1.0)
|
||||
.squeeze(0)
|
||||
.numpy()
|
||||
)
|
||||
mix_audio[s:e] += seg_np
|
||||
try:
|
||||
del wav, adjusted
|
||||
except Exception:
|
||||
pass
|
||||
_release_audio_tensors()
|
||||
|
||||
# Common: short fades to avoid pops, then mix into full_audio.
|
||||
fade_ms = 15
|
||||
fade_samples = int((fade_ms / 1000.0) * sr)
|
||||
if wl > fade_samples * 2:
|
||||
ramp_up = torch.linspace(0, 1, fade_samples, device=adjusted.device)
|
||||
ramp_down = torch.linspace(1, 0, fade_samples, device=adjusted.device)
|
||||
adjusted[0, :fade_samples] *= ramp_up
|
||||
adjusted[0, -fade_samples:] *= ramp_down
|
||||
|
||||
s = int(place_at * sr)
|
||||
e = min(s + wl, total_samples)
|
||||
if s < total_samples:
|
||||
full_audio[:, s:e] += adjusted[:, :e - s]
|
||||
|
||||
lang_code = req.language_code or "und"
|
||||
track_path = os.path.join(DUB_DIR, job_id, f"dubbed_{lang_code}.wav")
|
||||
_t_save_0 = time.perf_counter()
|
||||
# Apply invisible watermark to the final assembled track
|
||||
full_audio = embed_watermark(full_audio, sr)
|
||||
atomic_save_wav(track_path, full_audio, sr)
|
||||
_t_save = time.perf_counter() - _t_save_0
|
||||
_t_mix = _t_save_0 - _t_loop_end
|
||||
_t_save_0 = time.perf_counter()
|
||||
mix_audio.flush()
|
||||
_write_memmap_wav_atomic(track_path, mix_audio[:mix_samples], sr)
|
||||
_t_save = time.perf_counter() - _t_save_0
|
||||
_t_mix = _t_save_0 - _t_loop_end
|
||||
finally:
|
||||
try:
|
||||
mix_audio.flush()
|
||||
mix_mmap = getattr(mix_audio, "_mmap", None)
|
||||
if mix_mmap is not None:
|
||||
mix_mmap.close()
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
del mix_audio
|
||||
except Exception:
|
||||
pass
|
||||
gc.collect()
|
||||
try:
|
||||
os.unlink(mix_path)
|
||||
except OSError:
|
||||
pass
|
||||
# The final track is written; the mix_<id> scratch WAVs (silence /
|
||||
# cached-fail / error slots) have served their only purpose as
|
||||
# assembly inputs and would otherwise leak into the job dir.
|
||||
for _mp in _mix_temp_paths:
|
||||
try:
|
||||
os.unlink(_mp)
|
||||
except OSError:
|
||||
pass
|
||||
# Per-track metadata. For stretch_video, the dub wav is at the new
|
||||
# (longer) timeline, so we record its actual duration here too — the
|
||||
# mux step needs this to know whether to use the original video as-is
|
||||
# or stretch it per the plan.
|
||||
track_dur = full_audio.shape[-1] / sr if full_audio.shape[-1] > 0 else 0.0
|
||||
track_dur = total_samples / sr if total_samples > 0 else 0.0
|
||||
job["dubbed_tracks"][lang_code] = {
|
||||
"path": track_path,
|
||||
"language": req.language,
|
||||
@@ -928,8 +1156,9 @@ async def preview_segment(job_id: str, req: SegmentPreviewRequest):
|
||||
)
|
||||
return normalize_audio(mastered, target_dBFS=-2.0)
|
||||
|
||||
loop = asyncio.get_running_loop()
|
||||
audio_tensor = await loop.run_in_executor(_gpu_pool, _gen)
|
||||
# Bounded + pool-reset on hang so a wedged preview generate can't starve the
|
||||
# GPU pool and brick the backend (#730 class).
|
||||
audio_tensor = await run_on_gpu_pool_guarded(_gen, what="Dub preview generate")
|
||||
|
||||
sr = getattr(_model, "sampling_rate", 24000)
|
||||
buf = io.BytesIO()
|
||||
|
||||
@@ -413,11 +413,16 @@ async def dub_translate(req: TranslateRequest):
|
||||
try:
|
||||
import argostranslate # noqa: F401
|
||||
except ImportError:
|
||||
# Single-source the install command from the engine registry so
|
||||
# this 400 and the proactive Install button in the Engine
|
||||
# selector can never drift (see translation_engines.install_command).
|
||||
from services.translation_engines import install_command
|
||||
cmd = install_command("argos") or "uv pip install argostranslate"
|
||||
friendly = (
|
||||
f"The '{provider}' translation engine needs the optional "
|
||||
f"`argostranslate` Python package, which isn't installed in "
|
||||
f"this backend. Install it with `uv pip install argostranslate` "
|
||||
f"(or `pip install argostranslate`) and restart the server, or "
|
||||
f"this backend. Install it with `{cmd}` "
|
||||
f"and restart the server, or "
|
||||
f"switch the Engine dropdown to another provider."
|
||||
)
|
||||
return JSONResponse(status_code=400, content={"error": friendly})
|
||||
@@ -470,11 +475,16 @@ async def dub_translate(req: TranslateRequest):
|
||||
try:
|
||||
import deep_translator # noqa: F401
|
||||
except ImportError:
|
||||
# Same single-source install command as the Engine selector's Install
|
||||
# button (translation_engines.install_command) — google/deepl/
|
||||
# microsoft/mymemory all share the deep_translator package.
|
||||
from services.translation_engines import install_command
|
||||
cmd = install_command(provider) or "uv pip install deep_translator"
|
||||
friendly = (
|
||||
f"The '{provider}' translation engine needs the optional "
|
||||
f"`deep_translator` Python package, which isn't installed in "
|
||||
f"this backend. Install it with `uv pip install deep_translator` "
|
||||
f"(or `pip install deep_translator`) and restart the server, or "
|
||||
f"this backend. Install it with `{cmd}` "
|
||||
f"and restart the server, or "
|
||||
f"switch the Engine dropdown to Argos (local, bundled), NLLB "
|
||||
f"(local, heavier), or OpenAI (LLM)."
|
||||
)
|
||||
@@ -573,11 +583,14 @@ async def _maybe_cinematic(translated, req, src_lang, loop):
|
||||
base = {"translated": translated, "target_lang": req.target_lang, "source_lang": src_lang,
|
||||
"quality_used": "fast", **_dialect_flags(req, applied=False)}
|
||||
|
||||
if quality != "cinematic":
|
||||
# Autofit is Cinematic + a strict "never exceed the slot" fit pass, so both
|
||||
# qualities take the LLM refine path below. Fast (and anything else) returns
|
||||
# the plain translation unchanged.
|
||||
if quality not in ("cinematic", "autofit"):
|
||||
return base
|
||||
|
||||
if not cinematic_available():
|
||||
logger.warning("cinematic requested but no LLM configured — returning Fast result.")
|
||||
logger.warning("%s requested but no LLM configured — returning Fast result.", quality)
|
||||
base["cinematic_skipped"] = "no-llm-configured"
|
||||
return base
|
||||
|
||||
@@ -660,6 +673,7 @@ async def _maybe_cinematic(translated, req, src_lang, loop):
|
||||
slot_seconds=float(slot),
|
||||
target_lang=req.target_lang,
|
||||
source_text=source_by_id.get(seg_id),
|
||||
strict=(quality == "autofit"),
|
||||
)
|
||||
if fit.get("text"):
|
||||
out["text"] = fit["text"]
|
||||
@@ -675,6 +689,6 @@ async def _maybe_cinematic(translated, req, src_lang, loop):
|
||||
"translated": merged,
|
||||
"target_lang": req.target_lang,
|
||||
"source_lang": src_lang,
|
||||
"quality_used": "cinematic",
|
||||
"quality_used": quality,
|
||||
**_dialect_flags(req, applied=bool(dialect_hint)),
|
||||
}
|
||||
|
||||
@@ -2,6 +2,7 @@ import os
|
||||
import io
|
||||
import uuid
|
||||
import time
|
||||
import random
|
||||
import asyncio
|
||||
import tempfile
|
||||
import contextlib
|
||||
@@ -11,16 +12,36 @@ from typing import Optional
|
||||
from fastapi import APIRouter, File, Form, UploadFile, HTTPException
|
||||
from fastapi.responses import StreamingResponse
|
||||
|
||||
from core.db import db_conn
|
||||
import sqlite3
|
||||
from core.db import db_conn, ensure_schema
|
||||
from core.config import OUTPUTS_DIR, VOICES_DIR
|
||||
from services.model_manager import get_model, _gpu_pool
|
||||
import functools
|
||||
from services.model_manager import (
|
||||
get_model, _gpu_pool, run_on_gpu_pool_guarded, GpuJobTimeoutError,
|
||||
)
|
||||
from services.audio_io import _safe_torchaudio_save
|
||||
from core import event_bus
|
||||
from omnivoice.utils.voice_design import heal_design_instruct
|
||||
|
||||
router = APIRouter()
|
||||
logger = logging.getLogger("omnivoice.generate")
|
||||
|
||||
|
||||
def _profile_instruct(row):
|
||||
"""Validator-safe instruct for a stored profile row.
|
||||
|
||||
Sanitizes the persisted instruct (dropping the ``"[object Object]"``
|
||||
sentinel / freeform prose that older builds saved) and, for a design row,
|
||||
rebuilds the tags from ``vd_states`` when the stored value is unusable — so
|
||||
a poisoned/legacy profile never 400-s generation (#550 #571 #594 #596).
|
||||
"""
|
||||
try:
|
||||
vd = row["vd_states"]
|
||||
except (KeyError, IndexError):
|
||||
vd = None
|
||||
return heal_design_instruct(row["instruct"], vd)
|
||||
|
||||
|
||||
def _render_with_pauses(gen_span, segments, sample_rate):
|
||||
"""Synthesize ``[(text, pause_ms), ...]`` spans and stitch silence between
|
||||
them (issue #276).
|
||||
@@ -60,6 +81,23 @@ def _render_with_pauses(gen_span, segments, sample_rate):
|
||||
return torch.cat(parts, dim=-1)
|
||||
|
||||
|
||||
def _sanitize_audio(audio_out):
|
||||
"""Replace non-finite samples (NaN / ±inf) with silence so a model glitch
|
||||
can't produce an unreadable WAV (#629). Returns the input unchanged when it's
|
||||
already finite or isn't a tensor. Never raises."""
|
||||
try:
|
||||
import torch
|
||||
if torch.is_tensor(audio_out) and not bool(torch.isfinite(audio_out).all()):
|
||||
logger.warning(
|
||||
"Generated audio contained non-finite samples (NaN/inf) — "
|
||||
"sanitizing to silence to keep the WAV decodable (#629)."
|
||||
)
|
||||
return torch.nan_to_num(audio_out, nan=0.0, posinf=0.0, neginf=0.0)
|
||||
except Exception:
|
||||
pass
|
||||
return audio_out
|
||||
|
||||
|
||||
def _apply_effect_chain(audio_out, sample_rate, effect_preset, *, skip_mastering=False):
|
||||
"""Shared post-DSP for /generate: preset validation → mastering →
|
||||
effect chain → loudness normalization.
|
||||
@@ -75,6 +113,14 @@ def _apply_effect_chain(audio_out, sample_rate, effect_preset, *, skip_mastering
|
||||
apply_effects_chain, get_effect_chain,
|
||||
)
|
||||
|
||||
# #629: a numerical glitch in the model (observed on MPS) can leave NaN/±inf
|
||||
# samples, which write an unreadable WAV that then fails decoding with an
|
||||
# opaque "ffmpeg returned error code: 183 / Invalid data" — surfaced to the
|
||||
# user as a misleading "ran out of memory". Replace non-finite samples with
|
||||
# silence here, before any DSP/encode touches the audio, so the output is
|
||||
# always a valid WAV. Covers the raw path too (it returns just below).
|
||||
audio_out = _sanitize_audio(audio_out)
|
||||
|
||||
preset = effect_preset or "broadcast"
|
||||
if preset not in EFFECT_PRESETS:
|
||||
raise ValueError(
|
||||
@@ -127,6 +173,66 @@ def _oom_friendly_reraise(e):
|
||||
f"or run `chmod +x` on the engine binary named in the error. "
|
||||
f"Underlying error: {e}"
|
||||
) from e
|
||||
# #629: a decode/ffmpeg failure on the rendered audio is NOT out of memory —
|
||||
# it's unreadable audio (usually a transient numerical glitch). Say so rather
|
||||
# than sending the user down the OOM path.
|
||||
if "ffmpeg returned error" in es or "Decoding failed" in es or "Invalid data found" in es:
|
||||
raise RuntimeError(
|
||||
f"The engine produced unreadable audio (a decode step failed) — this is "
|
||||
f"usually a transient glitch. Use the Flush button to reload the model, "
|
||||
f"then regenerate. Underlying error: {e}"
|
||||
) from e
|
||||
# #664: a bad voice-design instruct (free-form prose, mixed EN/ZH, or
|
||||
# conflicting tags) raises "Unsupported instruct items …" / "Cannot mix …
|
||||
# in a single instruct" / "Conflicting instruct items …" from omnivoice's
|
||||
# _resolve_instruct. That's a USER-INPUT validation error, not an OOM. Match
|
||||
# on the message signature (NOT the type — a lower layer can wrap the original
|
||||
# ValueError, which is why the route's `except ValueError` guard misses it)
|
||||
# and re-raise as a clean ValueError so the route returns a 400 with the
|
||||
# instruct guidance, instead of a 500 telling the user to Flush for memory
|
||||
# they never ran out of. (Complements the client-side guard in #658/#612.)
|
||||
_low = es.lower()
|
||||
if ("unsupported instruct items" in _low
|
||||
or "conflicting instruct items" in _low
|
||||
or "in a single instruct" in _low):
|
||||
raise ValueError(es) from e
|
||||
# #705: a corrupt or wrong-architecture native component (a .dll / .pyd / .exe
|
||||
# — torch, ffmpeg, or a bundled engine binary) fails to load/spawn on Windows
|
||||
# with "[WinError 193] %1 is not a valid Win32 application". That is NOT OOM,
|
||||
# and Flush won't help — reinstalling/repairing the component is the real fix.
|
||||
if "[winerror 193]" in _low or "is not a valid win32 application" in _low:
|
||||
raise RuntimeError(
|
||||
f"A native component (a DLL / .pyd / .exe — e.g. torch, ffmpeg, or an "
|
||||
f"engine binary) is corrupt or built for the wrong architecture "
|
||||
f"([WinError 193]). Reinstall or repair that component — the Flush "
|
||||
f"button won't help here. Underlying error: {e}"
|
||||
) from e
|
||||
# #715: a "[Errno 32] Broken pipe" (BrokenPipeError) surfacing from
|
||||
# generation is NOT out of memory — it means the backend's stdout/stderr
|
||||
# pipe to the desktop shell that launched it closed mid-render (an orphaned
|
||||
# backend whose parent shell exited or relaunched). main.py wraps
|
||||
# sys.stdout/stderr to swallow EPIPE, but a C-level write inside the native
|
||||
# engine/torch can still raise one past that guard. Flush won't help —
|
||||
# relaunching the app re-parents the backend to a live shell.
|
||||
# #756: the GPU's compute capability isn't in this PyTorch build's arch list,
|
||||
# so CUDA can't launch kernels ("no kernel image is available for execution").
|
||||
# NOT OOM. get_best_device() now falls back to CPU up front, but classify the
|
||||
# raw error too in case CUDA was forced (OMNIVOICE_FORCE_CUDA) or a sub-path
|
||||
# still ran on the GPU — point at the real fix, not the Flush button.
|
||||
if "no kernel image is available" in _low:
|
||||
raise RuntimeError(
|
||||
f"Your GPU isn't supported by the installed PyTorch build (CUDA can't "
|
||||
f"launch kernels for its compute capability). Switch the compute device "
|
||||
f"to CPU in Settings, or install a matching PyTorch (e.g. a cu128 build "
|
||||
f"for newer GPUs). The Flush button won't help. Underlying error: {e}"
|
||||
) from e
|
||||
if isinstance(e, BrokenPipeError) or "broken pipe" in _low or "errno 32" in _low:
|
||||
raise RuntimeError(
|
||||
f"The backend lost its output pipe mid-generation — the desktop app "
|
||||
f"that launched it closed or relaunched ([Errno 32] Broken pipe). "
|
||||
f"Restart the app and try again; the Flush button won't help here. "
|
||||
f"Underlying error: {e}"
|
||||
) from e
|
||||
raise RuntimeError(
|
||||
f"TTS engine stopped mid-generation. This usually means it ran out of memory. "
|
||||
f"Try the Flush button to reload the model, then regenerate. Underlying error: {e}"
|
||||
@@ -318,7 +424,21 @@ async def generate_speech(
|
||||
# boundaries and crossfaded. 0 disables chunking (whole text to engine).
|
||||
max_chunk_chars: int = Form(800, ge=0),
|
||||
crossfade_ms: int = Form(50, ge=0, le=1000),
|
||||
# Expressive-TTS Spec 01: apply the user pronunciation dictionary + inline
|
||||
# [[…]] overrides to the text before synthesis. Default ON; the global
|
||||
# OMNIVOICE_PRONUNCIATION pref can disable it for power users. Omitting it
|
||||
# with an empty dictionary is byte-identical to legacy behavior.
|
||||
pronounce: bool = Form(True),
|
||||
):
|
||||
# #502: NFC-normalize the input text so decomposed (NFD) diacritics — common
|
||||
# in pasted Vietnamese and other Latin-with-marks text — are composed to the
|
||||
# single codepoints the tokenizer/model expect, instead of base-letter +
|
||||
# combining-mark sequences that render as distorted/garbled speech. NFC is a
|
||||
# no-op for already-composed text; mirrors the duration estimator
|
||||
# (utils/duration.py) so the estimate and the synthesis see the same text.
|
||||
import unicodedata
|
||||
text = unicodedata.normalize("NFC", text)
|
||||
|
||||
# ── Engine resolution (issue #312) ──────────────────────────────────────
|
||||
# The request runs on the engine selected in Settings (POST /engines/select,
|
||||
# env var OMNIVOICE_TTS_BACKEND wins), or an explicit per-request `engine`
|
||||
@@ -400,7 +520,7 @@ async def generate_speech(
|
||||
if not ref_text:
|
||||
ref_text = row["ref_text"]
|
||||
if not instruct:
|
||||
instruct = row["instruct"]
|
||||
instruct = _profile_instruct(row)
|
||||
if used_seed is None and row["seed"] is not None:
|
||||
used_seed = row["seed"]
|
||||
elif profile_kind == "design":
|
||||
@@ -410,14 +530,14 @@ async def generate_speech(
|
||||
if ref_audio_path and not ref_text and row["ref_text"]:
|
||||
ref_text = row["ref_text"]
|
||||
if not instruct:
|
||||
instruct = row["instruct"]
|
||||
instruct = _profile_instruct(row)
|
||||
if used_seed is None and row["seed"] is not None:
|
||||
used_seed = row["seed"]
|
||||
elif row["instruct"] and not row["is_locked"] and not row["ref_audio_path"]:
|
||||
# Legacy design-shaped row (pre-0004 archetype materialization
|
||||
# failure path): instruct-only conditioning.
|
||||
if not instruct:
|
||||
instruct = row["instruct"]
|
||||
instruct = _profile_instruct(row)
|
||||
if used_seed is None and row["seed"] is not None:
|
||||
used_seed = row["seed"]
|
||||
else:
|
||||
@@ -430,6 +550,20 @@ async def generate_speech(
|
||||
used_seed = row["seed"]
|
||||
if language == "Auto":
|
||||
language = None
|
||||
# #533: a profile's stored language must drive generation when the
|
||||
# request didn't pin one. Without this the German (etc.) archetype
|
||||
# generates with language=None and the model drifts to English —
|
||||
# even though the archetype PREVIEW renders correctly (archetypes.py
|
||||
# passes the language). An EXPLICIT non-Auto request language still
|
||||
# wins; we only fill the gap. `row` is a sqlite3.Row, so guard the
|
||||
# column lookup for pre-language DBs mid-upgrade.
|
||||
if language is None:
|
||||
try:
|
||||
prof_lang = row["language"]
|
||||
except (KeyError, IndexError):
|
||||
prof_lang = None
|
||||
if prof_lang and prof_lang != "Auto":
|
||||
language = prof_lang
|
||||
elif ref_audio is not None:
|
||||
try:
|
||||
with tempfile.NamedTemporaryFile(delete=False, suffix=".wav") as f:
|
||||
@@ -446,32 +580,88 @@ async def generate_speech(
|
||||
# fallback behaves exactly as before.
|
||||
if ref_audio_path and not ref_text:
|
||||
from services.asr_backend import transcribe_reference
|
||||
ref_text = await asyncio.get_running_loop().run_in_executor(
|
||||
_gpu_pool, transcribe_reference, ref_audio_path
|
||||
)
|
||||
# Same #730 hang risk as any whisperx transcribe — bound + reset the pool
|
||||
# so a wedged reference transcribe can't brick the backend. This path is
|
||||
# best-effort (transcribe_reference returns None on failure → the model's
|
||||
# built-in ASR fallback), so a timeout degrades to None rather than
|
||||
# failing the whole generate.
|
||||
try:
|
||||
ref_text = await run_on_gpu_pool_guarded(
|
||||
functools.partial(transcribe_reference, ref_audio_path),
|
||||
what="Reference transcribe",
|
||||
)
|
||||
except GpuJobTimeoutError as e:
|
||||
logger.warning("reference transcribe hung (%s); using model ASR fallback", e)
|
||||
ref_text = None
|
||||
|
||||
# #526: materialize a concrete seed when none was supplied (and no profile
|
||||
# pinned one) so the take is reproducible and we can hand it back via the
|
||||
# X-Seed header for the "keep this seed" control. An explicit request seed
|
||||
# or a profile's stored seed still wins — used_seed is only filled when it
|
||||
# is still None here, never overwritten.
|
||||
if used_seed is None:
|
||||
used_seed = random.randint(0, 2**31 - 1)
|
||||
|
||||
# Expressive-TTS Spec 01: apply the user pronunciation dictionary + inline
|
||||
# [[…]] one-off overrides to the text, here — AFTER `language` is fully
|
||||
# resolved (a profile may fill it above) so per-language entries match the
|
||||
# real render language, and BEFORE the text reaches either inference path
|
||||
# (native OmniVoice or a pluggable backend) and the chunk splitter. This is
|
||||
# the single point user text → normalized text → model, so the transform
|
||||
# covers generate for every engine. Pure text substitution → identical on
|
||||
# mac/Win/Linux. A disabled pref or empty dictionary is a pass-through, so
|
||||
# plain text stays byte-identical (#G5 backward-compat).
|
||||
from core import prefs as _prefs
|
||||
_pron_env = os.environ.get("OMNIVOICE_PRONUNCIATION")
|
||||
if _pron_env is not None:
|
||||
# Env wins (power-user override); "0"/"false"/"no"/"off" disable it.
|
||||
_pron_enabled = _pron_env.strip().lower() not in ("0", "false", "no", "off", "")
|
||||
else:
|
||||
_pron_enabled = bool(_prefs.get("pronunciation_enabled", True))
|
||||
if pronounce and _pron_enabled:
|
||||
from services.pronunciation import apply_pronunciation, load_entries_from_db
|
||||
try:
|
||||
_pron_rows = load_entries_from_db()
|
||||
except Exception: # noqa: BLE001 — table missing / DB locked → no-op
|
||||
_pron_rows = []
|
||||
text = apply_pronunciation(text, _pron_rows, language)
|
||||
else:
|
||||
# Even with the dictionary off, inline [[…]] overrides are an explicit,
|
||||
# in-text authoring choice → always honored (and never left as literal
|
||||
# double-bracket text the model would mispronounce).
|
||||
from services.pronunciation import apply_inline_overrides
|
||||
text = apply_inline_overrides(text)
|
||||
|
||||
start_time = time.time()
|
||||
try:
|
||||
loop = asyncio.get_running_loop()
|
||||
if _backend is not None:
|
||||
audio_tensor = await loop.run_in_executor(
|
||||
_gpu_pool, _run_backend_inference,
|
||||
_backend, text, language, ref_audio_path, ref_text, instruct,
|
||||
duration, num_step, guidance_scale, speed, denoise,
|
||||
postprocess_output, used_seed, effect_preset,
|
||||
max_chunk_chars, crossfade_ms,
|
||||
# Bounded + pool-reset on hang so a wedged generate can't starve the
|
||||
# GPU pool and brick the backend ("can't reach backend", #730 class).
|
||||
audio_tensor = await run_on_gpu_pool_guarded(
|
||||
functools.partial(
|
||||
_run_backend_inference,
|
||||
_backend, text, language, ref_audio_path, ref_text, instruct,
|
||||
duration, num_step, guidance_scale, speed, denoise,
|
||||
postprocess_output, used_seed, effect_preset,
|
||||
max_chunk_chars, crossfade_ms,
|
||||
),
|
||||
what="TTS generate",
|
||||
)
|
||||
# Read after generation: engines with lazy model loading report
|
||||
# their real rate only once weights are up.
|
||||
sample_rate = _backend.sample_rate
|
||||
else:
|
||||
audio_tensor = await loop.run_in_executor(
|
||||
_gpu_pool, _run_inference,
|
||||
_model, text, language, ref_audio_path, ref_text, instruct, duration,
|
||||
num_step, guidance_scale, speed, t_shift, denoise,
|
||||
postprocess_output, layer_penalty_factor, position_temperature,
|
||||
class_temperature, used_seed, effect_preset,
|
||||
max_chunk_chars, crossfade_ms,
|
||||
audio_tensor = await run_on_gpu_pool_guarded(
|
||||
functools.partial(
|
||||
_run_inference,
|
||||
_model, text, language, ref_audio_path, ref_text, instruct, duration,
|
||||
num_step, guidance_scale, speed, t_shift, denoise,
|
||||
postprocess_output, layer_penalty_factor, position_temperature,
|
||||
class_temperature, used_seed, effect_preset,
|
||||
max_chunk_chars, crossfade_ms,
|
||||
),
|
||||
what="TTS generate",
|
||||
)
|
||||
sample_rate = _model.sampling_rate
|
||||
# Invisible AudioSeal provenance watermark on the final audio. Embedding
|
||||
@@ -493,13 +683,29 @@ async def generate_speech(
|
||||
|
||||
audio_dur = round(audio_tensor.shape[-1] / sample_rate, 2)
|
||||
|
||||
with db_conn() as conn:
|
||||
conn.execute(
|
||||
"INSERT INTO generation_history (id, text, mode, language, instruct, profile_id, audio_path, duration_seconds, generation_time, seed, created_at) VALUES (?,?,?,?,?,?,?,?,?,?,?)",
|
||||
(audio_id, text[:200], history_mode or ("clone" if ref_audio_path else "design"),
|
||||
language or "Auto", instruct or "", resolved_profile_id,
|
||||
audio_filename, audio_dur, gen_time, used_seed, time.time())
|
||||
)
|
||||
# #710: the clip is already generated and saved above. A history-write
|
||||
# failure — e.g. "no such table: generation_history" on a DB that missed
|
||||
# schema init — must NOT 500 the user's generation. Self-heal the schema
|
||||
# once and retry; if it still fails, log and return the audio anyway.
|
||||
def _write_history():
|
||||
with db_conn() as conn:
|
||||
conn.execute(
|
||||
"INSERT INTO generation_history (id, text, mode, language, instruct, profile_id, audio_path, duration_seconds, generation_time, seed, created_at) VALUES (?,?,?,?,?,?,?,?,?,?,?)",
|
||||
(audio_id, text[:200], history_mode or ("clone" if ref_audio_path else "design"),
|
||||
language or "Auto", instruct or "", resolved_profile_id,
|
||||
audio_filename, audio_dur, gen_time, used_seed, time.time())
|
||||
)
|
||||
try:
|
||||
_write_history()
|
||||
except sqlite3.OperationalError as e:
|
||||
logger.warning("generation history write failed (%s); healing schema + retrying", e)
|
||||
try:
|
||||
ensure_schema()
|
||||
_write_history()
|
||||
except Exception as e2:
|
||||
logger.warning("history write still failed after schema heal; returning audio anyway: %s", e2)
|
||||
except Exception as e:
|
||||
logger.warning("generation history write failed; returning audio anyway: %s", e)
|
||||
event_bus.emit("generation_history", {"action": "created", "id": audio_id})
|
||||
|
||||
buffer = io.BytesIO()
|
||||
@@ -535,6 +741,12 @@ async def generate_speech(
|
||||
)
|
||||
except HTTPException:
|
||||
raise
|
||||
except GpuJobTimeoutError as e:
|
||||
# A wedged GPU generate — the pool was already reset to restore capacity
|
||||
# (#730 class). Report the actionable timeout instead of the misleading
|
||||
# "can't reach backend" the frontend shows when the pool starves.
|
||||
logger.error("Generate timed out: %s", e)
|
||||
raise HTTPException(status_code=503, detail=str(e)) from e
|
||||
except ValueError as e:
|
||||
logger.error("Validation failed: %s", e)
|
||||
raise HTTPException(status_code=400, detail=str(e)) from e
|
||||
|
||||
@@ -22,7 +22,6 @@ from __future__ import annotations
|
||||
import io
|
||||
import logging
|
||||
import os
|
||||
import asyncio
|
||||
import tempfile
|
||||
from typing import Literal, Optional
|
||||
|
||||
@@ -30,7 +29,7 @@ from fastapi import APIRouter, File, Form, HTTPException, UploadFile
|
||||
from fastapi.responses import StreamingResponse
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from services.model_manager import _gpu_pool
|
||||
from services.model_manager import _gpu_pool, run_on_gpu_pool_guarded
|
||||
|
||||
logger = logging.getLogger("omnivoice.openai_compat")
|
||||
|
||||
@@ -313,8 +312,10 @@ async def create_speech(req: SpeechRequest):
|
||||
kw["voice"] = voice
|
||||
|
||||
try:
|
||||
loop = asyncio.get_running_loop()
|
||||
wav, sr = await loop.run_in_executor(_gpu_pool, _run_tts, backend, req.input, kw)
|
||||
# Bounded + pool-reset on hang so a wedged TTS request can't starve the
|
||||
# GPU pool and brick the backend (#730 class).
|
||||
wav, sr = await run_on_gpu_pool_guarded(
|
||||
lambda: _run_tts(backend, req.input, kw), what="OpenAI TTS generate")
|
||||
except Exception as e:
|
||||
logger.exception("OpenAI TTS failed: %s", e)
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
@@ -384,12 +385,15 @@ async def create_transcription(
|
||||
try:
|
||||
backend = get_active_asr_backend()
|
||||
|
||||
# Run transcription in the thread pool to avoid blocking the event loop
|
||||
loop = asyncio.get_running_loop()
|
||||
# Run transcription in the thread pool to avoid blocking the event loop,
|
||||
# bounded so a stuck/starved ASR returns a 504 with guidance instead of
|
||||
# hanging the request forever (see run_transcribe_guarded).
|
||||
from services.asr_backend import run_transcribe_guarded
|
||||
word_ts = response_format == "verbose_json"
|
||||
result = await loop.run_in_executor(
|
||||
result = await run_transcribe_guarded(
|
||||
_gpu_pool,
|
||||
lambda: backend.transcribe(tmp_path, word_timestamps=word_ts),
|
||||
what="OpenAI",
|
||||
)
|
||||
|
||||
# Extract the full text from segments
|
||||
@@ -456,6 +460,12 @@ async def create_transcription(
|
||||
# Default: json
|
||||
return TranscriptionResponse(text=full_text)
|
||||
|
||||
except HTTPException:
|
||||
raise
|
||||
except TimeoutError as e:
|
||||
# ASRTimeoutError (subclass): backend alive, ASR too heavy for compute.
|
||||
logger.warning("OpenAI transcription timed out: %s", e)
|
||||
raise HTTPException(status_code=504, detail=str(e))
|
||||
except Exception as e:
|
||||
logger.exception("OpenAI transcription failed: %s", e)
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
|
||||
@@ -60,6 +60,11 @@ async def export_persona(
|
||||
profile = dict(row)
|
||||
|
||||
tag_list = [t.strip() for t in tags.split(",") if t.strip()]
|
||||
# #693: if OMNIVOICE_MODEL is set, record the *resolved* checkpoint in the
|
||||
# exported bundle so a leaked engine id (e.g. "omnivoice") can't be baked in;
|
||||
# keep "" when unset (the bundle's "engine unspecified" marker).
|
||||
from services.model_manager import resolve_omnivoice_checkpoint
|
||||
engine_id = resolve_omnivoice_checkpoint() if os.environ.get("OMNIVOICE_MODEL", "").strip() else ""
|
||||
try:
|
||||
loop = asyncio.get_running_loop()
|
||||
content = await loop.run_in_executor(
|
||||
@@ -70,7 +75,7 @@ async def export_persona(
|
||||
license_spdx=license_spdx,
|
||||
tags=tag_list,
|
||||
include_reference=include_reference,
|
||||
engine_id=os.environ.get("OMNIVOICE_MODEL", ""),
|
||||
engine_id=engine_id,
|
||||
omnivoice_version=APP_VERSION,
|
||||
),
|
||||
)
|
||||
|
||||
@@ -12,6 +12,7 @@ from core.db import db_conn
|
||||
from core.config import VOICES_DIR, OUTPUTS_DIR
|
||||
from core import event_bus
|
||||
from core.personalities import get_personalities
|
||||
from omnivoice.utils.voice_design import heal_design_instruct, sanitize_instruct
|
||||
|
||||
router = APIRouter()
|
||||
|
||||
@@ -76,6 +77,13 @@ async def create_profile(
|
||||
# instruct — that's still a valid, saveable voice: synthesis falls back
|
||||
# to neutral instruct-only conditioning (see generation.py design path).
|
||||
# Don't gate save on a non-empty instruct.
|
||||
#
|
||||
# Defence-in-depth against the "[object Object]" / freeform-prose poison
|
||||
# (#550 #571 #594 #596): never persist an instruct the engine validator
|
||||
# would reject. Sanitize the submitted instruct and, if it's unusable,
|
||||
# rebuild the tags from vd_states — so the row is always generation-safe
|
||||
# regardless of which frontend build saved it.
|
||||
instruct = heal_design_instruct(instruct, parsed)
|
||||
|
||||
profile_id = str(uuid.uuid4())[:8]
|
||||
|
||||
@@ -167,6 +175,10 @@ def update_profile(profile_id: str, patch: ProfileUpdate):
|
||||
continue
|
||||
if col == "name" and not val.strip():
|
||||
raise HTTPException(status_code=400, detail="A voice profile needs a name.")
|
||||
if col == "instruct":
|
||||
# Never let an edit persist a validator-rejecting instruct (prose /
|
||||
# "[object Object]"); keep only whitelist tags (#550 #571 #594 #596).
|
||||
val = sanitize_instruct(val)
|
||||
fields.append(f"{col} = ?")
|
||||
params.append(val.strip() if col in ("name", "language") else val)
|
||||
if not fields:
|
||||
|
||||
@@ -0,0 +1,306 @@
|
||||
"""
|
||||
Pronunciation dictionary router — Expressive-TTS Spec 01 Phase 1.
|
||||
|
||||
CRUD for the DB-backed, per-language pronunciation dictionary the
|
||||
``PronunciationPanel`` (Settings → Pronunciation) edits, plus a model-free
|
||||
``/pronunciation/test`` dry-run. Entries are applied as pure text substitution
|
||||
before synthesis (see ``services/pronunciation.apply_pronunciation`` and the
|
||||
generate path), so a saved entry actually changes the audio on every engine.
|
||||
|
||||
Endpoints (loopback-only, like the dictation router):
|
||||
GET /pronunciation → list every entry
|
||||
POST /pronunciation → create one entry
|
||||
PUT /pronunciation/{entry_id} → update an entry (partial)
|
||||
DELETE /pronunciation/{entry_id} → remove an entry
|
||||
POST /pronunciation/test → dry-run substitution (no model)
|
||||
GET /pronunciation/export → all entries as JSON (round-trips import)
|
||||
POST /pronunciation/import → bulk add entries from JSON
|
||||
|
||||
Scope: ``language='*'`` is global (applies to every request); a 2-letter code
|
||||
(``'en'``, ``'de'``) applies only when the request language matches.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
import time
|
||||
import uuid
|
||||
from typing import List, Optional
|
||||
|
||||
from fastapi import APIRouter, Depends, HTTPException
|
||||
from pydantic import BaseModel
|
||||
|
||||
from api.dependencies import require_loopback
|
||||
from core.db import db_conn
|
||||
from services.pronunciation import apply_pronunciation, entries_for_language
|
||||
|
||||
logger = logging.getLogger("omnivoice.pronunciation")
|
||||
router = APIRouter()
|
||||
|
||||
_VALID_TYPES = ("respelling", "ipa", "cmu")
|
||||
_ALL_LANG = "*"
|
||||
|
||||
# IPA: the input is validated as a non-empty string of Unicode letters / IPA
|
||||
# extension codepoints + the usual suprasegmental marks; we reject ASCII control
|
||||
# and the bracket/pipe chars that would collide with the inline grammar. This is
|
||||
# a charset gate (catches obvious garbage early), not a full IPA grammar.
|
||||
_IPA_BAD = re.compile(r"[\[\]\|\x00-\x1f]")
|
||||
# CMU / ARPABET: space-separated phoneme tokens (letters + an optional 0-2 stress
|
||||
# digit), e.g. "N AH0 V AE1 D AH0". Reject anything else.
|
||||
_CMU_TOKEN = re.compile(r"^[A-Za-z]{1,3}[0-2]?$")
|
||||
|
||||
|
||||
def _validate_type_replacement(etype: str, replacement: str) -> None:
|
||||
"""Raise 400 on a phoneme replacement that's obviously malformed.
|
||||
|
||||
Respelling rows accept any text. IPA rows must be a non-empty string free of
|
||||
bracket/pipe/control chars. CMU rows must be space-separated ARPABET tokens.
|
||||
Validating on save (not at synth) means a model never sees garbage phonemes
|
||||
(Spec 01 §R3 — never pass unvalidated phoneme strings to a model).
|
||||
"""
|
||||
if etype == "respelling":
|
||||
return
|
||||
rep = (replacement or "").strip()
|
||||
if not rep:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"A {etype.upper()} entry needs a phoneme string in 'replacement'.",
|
||||
)
|
||||
if etype == "ipa":
|
||||
if _IPA_BAD.search(rep):
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail="That IPA string contains brackets, a pipe, or control characters. "
|
||||
"Use plain IPA symbols, e.g. ˈnɛvʌdə.",
|
||||
)
|
||||
elif etype == "cmu":
|
||||
tokens = rep.split()
|
||||
if not tokens or any(not _CMU_TOKEN.match(tok) for tok in tokens):
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail="That doesn't look like CMU/ARPABET. Use space-separated tokens with "
|
||||
"optional stress digits, e.g. N AH0 V AE1 D AH0.",
|
||||
)
|
||||
|
||||
|
||||
def _norm_language(language: Optional[str]) -> str:
|
||||
"""Normalize a scope to '*' (global) or a lowercase 2-letter code."""
|
||||
if not language:
|
||||
return _ALL_LANG
|
||||
s = str(language).strip()
|
||||
if not s or s == _ALL_LANG or s.lower() == "auto":
|
||||
return _ALL_LANG
|
||||
return s.lower()[:2]
|
||||
|
||||
|
||||
def _row_to_dict(r) -> dict:
|
||||
d = dict(r)
|
||||
d["enabled"] = bool(d.get("enabled"))
|
||||
# ``scope`` is the UI-facing alias for ``language`` ('*' shows as Global).
|
||||
d["scope"] = d.get("language") or _ALL_LANG
|
||||
return d
|
||||
|
||||
|
||||
# ── Schemas ──────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
class PronEntry(BaseModel):
|
||||
term: str
|
||||
replacement: str = ""
|
||||
type: str = "respelling"
|
||||
language: str = _ALL_LANG
|
||||
enabled: bool = True
|
||||
|
||||
|
||||
class PronEntryUpdate(BaseModel):
|
||||
term: Optional[str] = None
|
||||
replacement: Optional[str] = None
|
||||
type: Optional[str] = None
|
||||
language: Optional[str] = None
|
||||
enabled: Optional[bool] = None
|
||||
|
||||
|
||||
class PronTestRequest(BaseModel):
|
||||
text: str
|
||||
language: Optional[str] = None
|
||||
|
||||
|
||||
class PronImportRequest(BaseModel):
|
||||
entries: List[PronEntry]
|
||||
replace: bool = False # True → clear existing rows first
|
||||
|
||||
|
||||
# ── CRUD ─────────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
@router.get("/pronunciation", dependencies=[Depends(require_loopback)])
|
||||
def list_entries():
|
||||
with db_conn() as conn:
|
||||
rows = conn.execute(
|
||||
"SELECT id, term, replacement, type, language, enabled, created_at "
|
||||
"FROM pronunciation_entries ORDER BY created_at ASC, id ASC"
|
||||
).fetchall()
|
||||
return [_row_to_dict(r) for r in rows]
|
||||
|
||||
|
||||
@router.post("/pronunciation", dependencies=[Depends(require_loopback)])
|
||||
def create_entry(entry: PronEntry):
|
||||
term = entry.term.strip()
|
||||
if not term:
|
||||
raise HTTPException(status_code=400, detail="A pronunciation entry needs a term.")
|
||||
etype = (entry.type or "respelling").strip().lower()
|
||||
if etype not in _VALID_TYPES:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"Unknown entry type {entry.type!r}. Use one of: {', '.join(_VALID_TYPES)}.",
|
||||
)
|
||||
_validate_type_replacement(etype, entry.replacement)
|
||||
eid = str(uuid.uuid4())[:12]
|
||||
now = time.time()
|
||||
lang = _norm_language(entry.language)
|
||||
with db_conn() as conn:
|
||||
conn.execute(
|
||||
"INSERT INTO pronunciation_entries (id, term, replacement, type, language, enabled, created_at) "
|
||||
"VALUES (?, ?, ?, ?, ?, ?, ?)",
|
||||
(eid, term, entry.replacement, etype, lang, 1 if entry.enabled else 0, now),
|
||||
)
|
||||
row = conn.execute(
|
||||
"SELECT id, term, replacement, type, language, enabled, created_at "
|
||||
"FROM pronunciation_entries WHERE id = ?", (eid,)
|
||||
).fetchone()
|
||||
return _row_to_dict(row)
|
||||
|
||||
|
||||
@router.put("/pronunciation/{entry_id}", dependencies=[Depends(require_loopback)])
|
||||
def update_entry(entry_id: str, patch: PronEntryUpdate):
|
||||
with db_conn() as conn:
|
||||
existing = conn.execute(
|
||||
"SELECT id, term, replacement, type, language, enabled, created_at "
|
||||
"FROM pronunciation_entries WHERE id = ?", (entry_id,)
|
||||
).fetchone()
|
||||
if existing is None:
|
||||
raise HTTPException(status_code=404, detail="No such pronunciation entry.")
|
||||
|
||||
# Resolve the post-update type + replacement so phoneme validation runs
|
||||
# against the final state (e.g. switching type without changing text).
|
||||
new_type = (patch.type.strip().lower() if patch.type is not None else existing["type"]) or "respelling"
|
||||
if new_type not in _VALID_TYPES:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"Unknown entry type {patch.type!r}. Use one of: {', '.join(_VALID_TYPES)}.",
|
||||
)
|
||||
new_replacement = patch.replacement if patch.replacement is not None else existing["replacement"]
|
||||
_validate_type_replacement(new_type, new_replacement)
|
||||
|
||||
fields, params = [], []
|
||||
if patch.term is not None:
|
||||
term = patch.term.strip()
|
||||
if not term:
|
||||
raise HTTPException(status_code=400, detail="A pronunciation entry needs a term.")
|
||||
fields.append("term = ?"); params.append(term)
|
||||
if patch.replacement is not None:
|
||||
fields.append("replacement = ?"); params.append(patch.replacement)
|
||||
if patch.type is not None:
|
||||
fields.append("type = ?"); params.append(new_type)
|
||||
if patch.language is not None:
|
||||
fields.append("language = ?"); params.append(_norm_language(patch.language))
|
||||
if patch.enabled is not None:
|
||||
fields.append("enabled = ?"); params.append(1 if patch.enabled else 0)
|
||||
if not fields:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail="PUT body was empty. Include at least one field to change, or DELETE the entry.",
|
||||
)
|
||||
params.append(entry_id)
|
||||
# nosec B608 - `fields` are fixed literal assignments ("term = ?", …) from
|
||||
# the allowlist above; every user value is a bound `?` parameter, never
|
||||
# interpolated. The f-string only joins constant column fragments.
|
||||
conn.execute(
|
||||
f"UPDATE pronunciation_entries SET {', '.join(fields)} WHERE id = ?", # nosec B608
|
||||
params,
|
||||
)
|
||||
row = conn.execute(
|
||||
"SELECT id, term, replacement, type, language, enabled, created_at "
|
||||
"FROM pronunciation_entries WHERE id = ?", (entry_id,)
|
||||
).fetchone()
|
||||
return _row_to_dict(row)
|
||||
|
||||
|
||||
@router.delete("/pronunciation/{entry_id}", dependencies=[Depends(require_loopback)])
|
||||
def delete_entry(entry_id: str):
|
||||
with db_conn() as conn:
|
||||
cur = conn.execute("DELETE FROM pronunciation_entries WHERE id = ?", (entry_id,))
|
||||
return {"deleted": cur.rowcount > 0}
|
||||
|
||||
|
||||
# ── Dry-run + import/export ───────────────────────────────────────────────────
|
||||
|
||||
|
||||
@router.post("/pronunciation/test", dependencies=[Depends(require_loopback)])
|
||||
def test_substitution(req: PronTestRequest):
|
||||
"""Show the post-substitution text for ``req.text`` — no model call.
|
||||
|
||||
Applies the same dictionary + inline ``[[…]]`` resolution the synth path
|
||||
runs, so the user sees exactly what the engine will be handed.
|
||||
"""
|
||||
with db_conn() as conn:
|
||||
rows = conn.execute(
|
||||
"SELECT id, term, replacement, type, language, enabled, created_at "
|
||||
"FROM pronunciation_entries"
|
||||
).fetchall()
|
||||
substituted = apply_pronunciation(req.text, rows, req.language)
|
||||
applied = entries_for_language(rows, req.language)
|
||||
return {
|
||||
"input": req.text,
|
||||
"substituted": substituted,
|
||||
"changed": substituted != req.text,
|
||||
"applied_terms": sorted(applied.keys(), key=len, reverse=True),
|
||||
}
|
||||
|
||||
|
||||
@router.get("/pronunciation/export", dependencies=[Depends(require_loopback)])
|
||||
def export_entries():
|
||||
"""Every entry as a JSON-serializable list (round-trips ``/import``)."""
|
||||
with db_conn() as conn:
|
||||
rows = conn.execute(
|
||||
"SELECT term, replacement, type, language, enabled "
|
||||
"FROM pronunciation_entries ORDER BY created_at ASC, id ASC"
|
||||
).fetchall()
|
||||
return {"entries": [
|
||||
{"term": r["term"], "replacement": r["replacement"], "type": r["type"],
|
||||
"language": r["language"], "enabled": bool(r["enabled"])}
|
||||
for r in rows
|
||||
]}
|
||||
|
||||
|
||||
@router.post("/pronunciation/import", dependencies=[Depends(require_loopback)])
|
||||
def import_entries(req: PronImportRequest):
|
||||
"""Bulk-add entries. ``replace=true`` clears the table first.
|
||||
|
||||
Each entry is validated like ``POST /pronunciation``; one bad row fails the
|
||||
whole import (400) so the table is never left half-applied.
|
||||
"""
|
||||
now = time.time()
|
||||
cleaned = []
|
||||
for e in req.entries:
|
||||
term = e.term.strip()
|
||||
if not term:
|
||||
continue # silently skip blank terms — they're a no-op anyway
|
||||
etype = (e.type or "respelling").strip().lower()
|
||||
if etype not in _VALID_TYPES:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"Entry {term!r}: unknown type {e.type!r}.",
|
||||
)
|
||||
_validate_type_replacement(etype, e.replacement)
|
||||
cleaned.append((str(uuid.uuid4())[:12], term, e.replacement, etype,
|
||||
_norm_language(e.language), 1 if e.enabled else 0, now))
|
||||
with db_conn() as conn:
|
||||
if req.replace:
|
||||
conn.execute("DELETE FROM pronunciation_entries")
|
||||
conn.executemany(
|
||||
"INSERT INTO pronunciation_entries (id, term, replacement, type, language, enabled, created_at) "
|
||||
"VALUES (?, ?, ?, ?, ?, ?, ?)",
|
||||
cleaned,
|
||||
)
|
||||
return {"imported": len(cleaned), "replaced": req.replace}
|
||||
@@ -234,6 +234,97 @@ def set_llm_endpoint(body: _LLMEndpointBody):
|
||||
return _llm_endpoint_state()
|
||||
|
||||
|
||||
# ── Multi-provider LLM registry (Settings → LLM Providers) ────────────────
|
||||
# Keys persist ENCRYPTED via settings_store.set_secret (never .env, never
|
||||
# returned). base_url/model/account overrides are non-secret. Loopback-gated
|
||||
# by the router dep, so LAN peers can't read masks or write keys.
|
||||
|
||||
class _LLMProviderBody(BaseModel):
|
||||
api_key: str | None = Field(None, description="API key; '' clears it, None leaves unchanged")
|
||||
base_url: str | None = None
|
||||
model: str | None = None
|
||||
account_id: str | None = Field(None, description="Cloudflare account id")
|
||||
make_active: bool = False
|
||||
|
||||
|
||||
class _LLMActiveBody(BaseModel):
|
||||
provider: str = Field(..., description="provider id to activate")
|
||||
|
||||
|
||||
@router.get("/llm-providers")
|
||||
def list_llm_providers():
|
||||
"""All providers with resolved base_url/model + whether a key is configured.
|
||||
|
||||
Never returns key material — only `has_key`/`key_from_env` booleans.
|
||||
"""
|
||||
from services import llm_providers
|
||||
return {
|
||||
"active": llm_providers.active_provider_id(),
|
||||
"providers": [llm_providers.describe(p) for p in llm_providers.all_providers()],
|
||||
}
|
||||
|
||||
|
||||
@router.put("/llm-providers/{provider_id}")
|
||||
def save_llm_provider(provider_id: str, body: _LLMProviderBody):
|
||||
"""Save a provider's key (encrypted) + optional base_url/model/account.
|
||||
|
||||
A None field is left unchanged; an empty api_key clears the stored key.
|
||||
"""
|
||||
from services import llm_providers
|
||||
if llm_providers.get_provider(provider_id) is None:
|
||||
raise HTTPException(status_code=404, detail=f"unknown provider {provider_id!r}")
|
||||
if body.api_key is not None:
|
||||
llm_providers.save_key(provider_id, body.api_key.strip())
|
||||
llm_providers.save_overrides(
|
||||
provider_id, base_url=body.base_url, model=body.model,
|
||||
account_id=body.account_id,
|
||||
)
|
||||
if body.make_active:
|
||||
llm_providers.set_active_provider(provider_id)
|
||||
return list_llm_providers()
|
||||
|
||||
|
||||
@router.post("/llm-providers/active")
|
||||
def set_active_llm_provider(body: _LLMActiveBody):
|
||||
from services import llm_providers
|
||||
if llm_providers.get_provider(body.provider) is None:
|
||||
raise HTTPException(status_code=404, detail=f"unknown provider {body.provider!r}")
|
||||
llm_providers.set_active_provider(body.provider)
|
||||
return list_llm_providers()
|
||||
|
||||
|
||||
@router.post("/llm-providers/{provider_id}/test")
|
||||
def test_llm_provider(provider_id: str):
|
||||
"""One cheap round-trip against a provider to prove the key/URL work.
|
||||
|
||||
Temporarily activates the provider for the probe by resolving its config
|
||||
directly (does not change the persisted active selection).
|
||||
"""
|
||||
from services import llm_providers
|
||||
p = llm_providers.get_provider(provider_id)
|
||||
if p is None:
|
||||
raise HTTPException(status_code=404, detail=f"unknown provider {provider_id!r}")
|
||||
base_url = llm_providers.resolve_base_url(p)
|
||||
api_key = llm_providers.resolve_api_key(p)
|
||||
if not base_url:
|
||||
return {"ok": False, "detail": "No Base URL set for this provider."}
|
||||
if not api_key:
|
||||
return {"ok": False, "detail": "No API key configured for this provider."}
|
||||
try:
|
||||
from openai import OpenAI
|
||||
client = OpenAI(api_key=api_key, base_url=base_url)
|
||||
res = client.chat.completions.create(
|
||||
model=llm_providers.resolve_model(p),
|
||||
messages=[{"role": "user", "content": "Reply with the single word: ok"}],
|
||||
timeout=20,
|
||||
)
|
||||
reply = (res.choices[0].message.content or "").strip()
|
||||
return {"ok": True, "model": llm_providers.resolve_model(p), "reply": reply[:80]}
|
||||
except Exception as e: # noqa: BLE001 — surface a clean, scrubbed error to the UI
|
||||
from core.scrub import scrub_text
|
||||
return {"ok": False, "detail": scrub_text(f"{type(e).__name__}: {e}")}
|
||||
|
||||
|
||||
# ── License acceptance (Phase 3 Plan 03-01 / TTS-05) ──────────────────────
|
||||
# Frontend ``SupertonicLicenseDialog`` flips the engine-license bit via this
|
||||
# endpoint. The handler is loopback-gated (router-level dep) and the
|
||||
|
||||
@@ -21,7 +21,17 @@ from pydantic import BaseModel
|
||||
from core import prefs
|
||||
from utils import hf_progress
|
||||
from utils import download_aggregator
|
||||
from .models import KNOWN_MODELS, invalidate_cache
|
||||
# Weight-floor scan (MM2-07 / #352) lives in ``models.py`` — the lowest module in
|
||||
# the setup import graph — so install-time validation here, the first-run
|
||||
# install-state detector (#622), and load-time repair share one set of floors and
|
||||
# can't drift apart. ``_MIN_WEIGHT_BYTES``/``_WEIGHT_FLOORS`` re-exported for tests.
|
||||
from .models import ( # noqa: F401
|
||||
KNOWN_MODELS,
|
||||
invalidate_cache,
|
||||
snapshot_has_weights,
|
||||
_MIN_WEIGHT_BYTES,
|
||||
_WEIGHT_FLOORS,
|
||||
)
|
||||
|
||||
logger = logging.getLogger("omnivoice.setup.download")
|
||||
router = APIRouter()
|
||||
@@ -120,11 +130,15 @@ def compute_plan(plan_files) -> dict:
|
||||
|
||||
|
||||
def _segmented_enabled() -> bool:
|
||||
"""Opt-in IDM-style accelerator (FDL-09), default OFF. Most useful when Xet
|
||||
is inactive (the app's default): the legacy-LFS path is single-stream, so
|
||||
this restores parallel speed AND gives real live byte progress."""
|
||||
"""IDM-style multi-connection accelerator (FDL-09), default **ON**. The app
|
||||
forces the legacy-LFS path (HF_HUB_DISABLE_XET=1) for clear progress, but that
|
||||
path is single-stream and slow — this restores parallel byte-range speed AND
|
||||
real live progress, and falls back to snapshot_download on any error so it
|
||||
can never compromise a correct install. Default-on so first-run downloads are
|
||||
fast out of the box (pairs with an HF token for higher rate limits); set
|
||||
OMNIVOICE_SEGMENTED_DOWNLOAD=0 to force the single-stream path."""
|
||||
return _truthy(prefs.resolve(
|
||||
"segmented_downloader", env="OMNIVOICE_SEGMENTED_DOWNLOAD", default=False,
|
||||
"segmented_downloader", env="OMNIVOICE_SEGMENTED_DOWNLOAD", default=True,
|
||||
))
|
||||
|
||||
|
||||
@@ -223,51 +237,26 @@ def _safe_put(queue: asyncio.Queue, event) -> None:
|
||||
# model.safetensors" (#352). 5 MB clears every weight format we ship
|
||||
# (safetensors/bin shards, onnx, pt, gguf) without false-positiving on
|
||||
# config-only aux repos.
|
||||
_MIN_WEIGHT_BYTES = 5 * 1024 * 1024
|
||||
|
||||
# Per-role weight-file floors (MM2-07). A valid model has at least one
|
||||
# recognized weight file at or above its extension's floor. ONNX graphs are
|
||||
# legitimately small (a complete model can be well under 5 MB), so a single
|
||||
# 5 MB rule false-positives on them as "truncated" (#352 over-trigger); give
|
||||
# .onnx a lower floor while still rejecting a 0/KB partial. Tensor formats keep
|
||||
# the original 5 MB floor.
|
||||
_WEIGHT_FLOORS = {
|
||||
".safetensors": _MIN_WEIGHT_BYTES,
|
||||
".bin": _MIN_WEIGHT_BYTES,
|
||||
".ckpt": _MIN_WEIGHT_BYTES,
|
||||
".pt": _MIN_WEIGHT_BYTES,
|
||||
".pth": _MIN_WEIGHT_BYTES,
|
||||
".gguf": _MIN_WEIGHT_BYTES,
|
||||
".onnx": 64 * 1024, # a real ONNX graph is ≥ tens of KB; a truncated one is bytes
|
||||
}
|
||||
|
||||
|
||||
def _validate_snapshot_has_weights(repo_id: str, snapshot_path: str) -> None:
|
||||
"""Raise OSError when a finished snapshot has no plausible weight file —
|
||||
surfaces the truncated-download class (#352) at install time, where the
|
||||
retry loop and the UI's re-download path can deal with it, instead of at
|
||||
first synthesis with an opaque transformers error.
|
||||
|
||||
A snapshot is valid if it contains a recognized weight file meeting its
|
||||
per-extension floor (MM2-07) OR any file ≥ the global 5 MB floor (the
|
||||
original lenient catch — kept so this is never stricter than before)."""
|
||||
Delegates the weight check to ``models.snapshot_has_weights`` (single source of
|
||||
the floors); only the install-time error message lives here."""
|
||||
if snapshot_has_weights(snapshot_path):
|
||||
return
|
||||
biggest = 0
|
||||
try:
|
||||
biggest = 0
|
||||
for root, _dirs, files in os.walk(snapshot_path, followlinks=True):
|
||||
for f in files:
|
||||
try:
|
||||
size = os.path.getsize(os.path.join(root, f))
|
||||
biggest = max(biggest, os.path.getsize(os.path.join(root, f)))
|
||||
except OSError:
|
||||
continue
|
||||
biggest = max(biggest, size)
|
||||
ext = os.path.splitext(f)[1].lower()
|
||||
floor = _WEIGHT_FLOORS.get(ext)
|
||||
if floor is not None and size >= floor:
|
||||
return # a recognized weight file of plausible size
|
||||
if size >= _MIN_WEIGHT_BYTES:
|
||||
return # original lenient catch (non-standard weight names)
|
||||
except OSError:
|
||||
return # can't inspect — don't block the install on the checker itself
|
||||
pass
|
||||
raise OSError(
|
||||
f"{repo_id}: download finished but no model weights were found in the "
|
||||
"snapshot (largest file "
|
||||
@@ -449,10 +438,11 @@ async def install_model(req: InstallModelRequest):
|
||||
raise _InstallCancelled()
|
||||
_attempt += 1
|
||||
try:
|
||||
# Opt-in segmented accelerator (FDL-09): parallel byte-range
|
||||
# fetch with real live progress, for the legacy-LFS path.
|
||||
# Any failure falls through to snapshot_download — the
|
||||
# accelerator can never compromise a correct install.
|
||||
# Segmented accelerator (FDL-09, default ON): parallel
|
||||
# byte-range fetch with real live progress, for the
|
||||
# legacy-LFS path. Any failure falls through to
|
||||
# snapshot_download — the accelerator can never compromise a
|
||||
# correct install.
|
||||
_snapshot_path = None
|
||||
if _attempt == 1 and _segmented_enabled() and not _xet_active():
|
||||
try:
|
||||
|
||||
@@ -146,6 +146,94 @@ def _hub_cache_roots() -> list[str]:
|
||||
return roots
|
||||
|
||||
|
||||
# ── Weight-presence (truncated-cache) detection ─────────────────────────────
|
||||
# A cache that downloaded config/tokenizer files but not the weight shard still
|
||||
# occupies bytes on disk, so a size-only "installed" check (#352/#581/#606) reads
|
||||
# it as installed and the first-run wizard hides the re-download button, stranding
|
||||
# the user (#622). These helpers tell a *complete* snapshot from a truncated one by
|
||||
# checking for a plausible weight file — the same class `download.py` guards at
|
||||
# install time and `model_manager.py` repairs at load time. Shared here (the lowest
|
||||
# module in the setup import graph; `download.py` imports from this module) so the
|
||||
# floors live in exactly one place and can't drift between the three call sites.
|
||||
|
||||
_MIN_WEIGHT_BYTES = 5 * 1024 * 1024 # tensor formats: a real shard is ≥ a few MB
|
||||
|
||||
# Per-extension floors. ONNX graphs are legitimately small (a complete model can be
|
||||
# well under 5 MB), so they get a lower floor that still rejects a bytes-only partial.
|
||||
_WEIGHT_FLOORS = {
|
||||
".safetensors": _MIN_WEIGHT_BYTES,
|
||||
".bin": _MIN_WEIGHT_BYTES,
|
||||
".ckpt": _MIN_WEIGHT_BYTES,
|
||||
".pt": _MIN_WEIGHT_BYTES,
|
||||
".pth": _MIN_WEIGHT_BYTES,
|
||||
".gguf": _MIN_WEIGHT_BYTES,
|
||||
".onnx": 64 * 1024,
|
||||
}
|
||||
|
||||
|
||||
def snapshot_has_weights(snapshot_path: str) -> bool:
|
||||
"""True when a finished snapshot dir holds a plausible weight file.
|
||||
|
||||
A snapshot is complete if it contains a recognized weight file meeting its
|
||||
per-extension floor OR any file ≥ the global 5 MB floor (the lenient catch for
|
||||
non-standard weight names). Returns True when the path can't be inspected — an
|
||||
un-walkable dir must never be reported as truncated, only a confirmed weight-less
|
||||
one. `getsize` follows symlinks, so HF's snapshot→blob links resolve correctly;
|
||||
a broken link (missing blob) raises OSError and is skipped, i.e. counts as absent.
|
||||
"""
|
||||
try:
|
||||
for root, _dirs, files in os.walk(snapshot_path, followlinks=True):
|
||||
for f in files:
|
||||
try:
|
||||
size = os.path.getsize(os.path.join(root, f))
|
||||
except OSError:
|
||||
continue
|
||||
ext = os.path.splitext(f)[1].lower()
|
||||
floor = _WEIGHT_FLOORS.get(ext)
|
||||
if floor is not None and size >= floor:
|
||||
return True
|
||||
if size >= _MIN_WEIGHT_BYTES:
|
||||
return True
|
||||
except OSError:
|
||||
return True # can't inspect — don't mislabel as truncated
|
||||
return False
|
||||
|
||||
|
||||
def _snapshot_dirs(repo_id: str) -> list[str]:
|
||||
"""Existing snapshot revision dirs for a repo across the candidate cache roots."""
|
||||
name = _repo_dir_name(repo_id)
|
||||
dirs: list[str] = []
|
||||
for root in _hub_cache_roots():
|
||||
snaps = os.path.join(root, name, "snapshots")
|
||||
try:
|
||||
for rev in os.listdir(snaps):
|
||||
rev_dir = os.path.join(snaps, rev)
|
||||
if os.path.isdir(rev_dir):
|
||||
dirs.append(rev_dir)
|
||||
except OSError:
|
||||
continue
|
||||
return dirs
|
||||
|
||||
|
||||
def cache_is_complete(model: dict) -> bool:
|
||||
"""True when this model's on-disk cache is usable (not a truncated download).
|
||||
|
||||
Config-only repos (``config_only: true`` in models.yaml — e.g. pyannote's
|
||||
diarisation pipeline, whose real weights live in referenced sub-repos) carry no
|
||||
weight file of their own, so the weight check would false-positive them as
|
||||
incomplete (#622 caveat). They're exempt: cache presence alone means complete.
|
||||
A weight-bearing repo is complete only if at least one of its snapshots has
|
||||
weights; if no snapshot dir is found on disk we can't prove truncation, so we
|
||||
don't downgrade (the size-based caller already decided it's cached).
|
||||
"""
|
||||
if model.get("config_only"):
|
||||
return True
|
||||
dirs = _snapshot_dirs(model["repo_id"])
|
||||
if not dirs:
|
||||
return True
|
||||
return any(snapshot_has_weights(d) for d in dirs)
|
||||
|
||||
|
||||
def _is_cached_on_disk(repo_id: str) -> bool:
|
||||
"""Direct-filesystem fallback for is_cached when scan_cache_dir is unavailable.
|
||||
|
||||
@@ -286,9 +374,15 @@ def list_models():
|
||||
out = []
|
||||
for m in KNOWN_MODELS:
|
||||
cached = cached_by_repo.get(m["repo_id"])
|
||||
on_disk = cached is not None and cached["size_on_disk"] > 0
|
||||
# A size-positive cache can still be a truncated download (config landed,
|
||||
# weight shard didn't). Treat that as not-installed + incomplete so the
|
||||
# wizard re-offers the download instead of stranding the user (#622).
|
||||
incomplete = on_disk and not cache_is_complete(m)
|
||||
out.append({
|
||||
**m,
|
||||
"installed": cached is not None and cached["size_on_disk"] > 0,
|
||||
"installed": on_disk and not incomplete,
|
||||
"incomplete": incomplete,
|
||||
"size_on_disk_bytes": cached["size_on_disk"] if cached else 0,
|
||||
"nb_files": cached["nb_files"] if cached else 0,
|
||||
"supported": _model_supported(m),
|
||||
@@ -383,6 +477,9 @@ def recommendations():
|
||||
entries = []
|
||||
for rid in recommended_ids:
|
||||
meta = known_by_id.get(rid, {})
|
||||
# Mirror /models: a truncated cache (weights missing) is not installed, so
|
||||
# the wizard counts it toward the remaining download instead of "all set".
|
||||
installed = rid in cached_ids and cache_is_complete(meta or {"repo_id": rid})
|
||||
entries.append({
|
||||
"repo_id": rid,
|
||||
"label": meta.get("label", rid),
|
||||
@@ -390,7 +487,7 @@ def recommendations():
|
||||
"size_gb": meta.get("size_gb", 0),
|
||||
"required": bool(meta.get("required", False)),
|
||||
"note": meta.get("note"),
|
||||
"installed": rid in cached_ids,
|
||||
"installed": installed,
|
||||
})
|
||||
|
||||
to_download_gb = sum(e["size_gb"] for e in entries if not e["installed"])
|
||||
|
||||
@@ -18,7 +18,7 @@ import shutil
|
||||
|
||||
from core.config import OUTPUTS_DIR, DATA_DIR, CRASH_LOG_PATH, LOG_PATH, IDLE_TIMEOUT_SECONDS
|
||||
from core.version import APP_VERSION
|
||||
from services.model_manager import get_model_status, get_best_device
|
||||
from services.model_manager import get_model_status, get_best_device, resolve_omnivoice_checkpoint
|
||||
from services.ffmpeg_utils import find_ffmpeg, run_ffmpeg
|
||||
|
||||
# Router-level loopback gate. Every route mounted on `router` (GET + POST,
|
||||
@@ -208,7 +208,7 @@ def system_info():
|
||||
"outputs_dir": OUTPUTS_DIR,
|
||||
"crash_log_path": CRASH_LOG_PATH,
|
||||
"idle_timeout_seconds": IDLE_TIMEOUT_SECONDS,
|
||||
"model_checkpoint": os.environ.get("OMNIVOICE_MODEL", "k2-fsa/OmniVoice"),
|
||||
"model_checkpoint": resolve_omnivoice_checkpoint(), # #693: show the effective checkpoint, not a leaked raw value
|
||||
"asr_model": os.environ.get("ASR_MODEL", "Systran/faster-whisper-large-v3"),
|
||||
"translate_provider": os.environ.get("TRANSLATE_PROVIDER", "google"),
|
||||
"has_hf_token": _has_hf_token(),
|
||||
|
||||
@@ -182,8 +182,8 @@ async def ws_tts(websocket: WebSocket):
|
||||
sentences = [text]
|
||||
|
||||
# Run generation in the GPU pool
|
||||
from services.model_manager import _gpu_pool
|
||||
loop = asyncio.get_running_loop()
|
||||
import functools
|
||||
from services.model_manager import run_on_gpu_pool_guarded
|
||||
|
||||
def _generate(sentence_text):
|
||||
from services.audio_dsp import apply_mastering, normalize_audio
|
||||
@@ -204,8 +204,13 @@ async def ws_tts(websocket: WebSocket):
|
||||
started = False
|
||||
|
||||
for sentence in sentences:
|
||||
wav_tensor, sr = await loop.run_in_executor(
|
||||
_gpu_pool, _generate, sentence
|
||||
# Bounded + pool-reset on hang so a wedged generate can't
|
||||
# starve the GPU pool and brick the backend (#730 class). On
|
||||
# timeout GpuJobTimeoutError propagates to the handler below,
|
||||
# which sends an actionable error frame.
|
||||
wav_tensor, sr = await run_on_gpu_pool_guarded(
|
||||
functools.partial(_generate, sentence),
|
||||
what="TTS generate",
|
||||
)
|
||||
|
||||
if not started:
|
||||
|
||||
Binary file not shown.
@@ -14,6 +14,10 @@
|
||||
# required (optional) — true if the app needs this model to function
|
||||
# platforms (optional) — restrict to specific OS+arch tags (e.g. darwin-arm64, cuda)
|
||||
# note (optional) — shown in the UI as a tooltip/footnote
|
||||
# config_only (optional) — true for pipeline repos that ship no weight file of
|
||||
# their own (weights live in referenced sub-repos). Such
|
||||
# a cache is legitimately tiny, so the truncated-download
|
||||
# (weights-missing) detector must NOT flag it incomplete.
|
||||
# ─────────────────────────────────────────────────────────────────────────
|
||||
|
||||
models:
|
||||
@@ -116,12 +120,83 @@ models:
|
||||
size_gb: 0.05
|
||||
note: "Smallest/fastest Moonshine, sub-200ms latency. Lower accuracy than base. Requires moonshine-onnx."
|
||||
|
||||
# ── sherpa-onnx live dictation (ONNX, CPU, streaming + offline) ────────
|
||||
# Live faster-than-real-time dictation via the k2-fsa/sherpa-onnx runtime.
|
||||
# `engine: sherpa-onnx`, `dictation_id` (backend model id), and `tag`
|
||||
# (offline | streaming) are extra fields the model-store list passes through
|
||||
# so the dictation UI can filter/group these (role=ASR, engine=sherpa-onnx).
|
||||
# Requires `uv add sherpa-onnx` (CPU wheels, all platforms).
|
||||
|
||||
- repo_id: "csukuangfj/sherpa-onnx-nemo-parakeet-tdt-0.6b-v3-int8"
|
||||
label: "Parakeet TDT v3 (sherpa-onnx — dictation, 25 EU langs)"
|
||||
role: ASR
|
||||
size_gb: 0.18
|
||||
engine: sherpa-onnx
|
||||
dictation_id: sherpa-parakeet-tdt-v3
|
||||
tag: offline
|
||||
note: "Recommended live-dictation default. CPU, int8 ONNX. Requires sherpa-onnx."
|
||||
|
||||
- repo_id: "csukuangfj/sherpa-onnx-nemo-parakeet-tdt-0.6b-v2-int8"
|
||||
label: "Parakeet TDT v2 (sherpa-onnx — dictation, English)"
|
||||
role: ASR
|
||||
size_gb: 0.17
|
||||
engine: sherpa-onnx
|
||||
dictation_id: sherpa-parakeet-tdt-v2
|
||||
tag: offline
|
||||
note: "English live dictation. CPU, int8 ONNX. Requires sherpa-onnx."
|
||||
|
||||
- repo_id: "csukuangfj/sherpa-onnx-streaming-zipformer-bilingual-zh-en-2023-02-20"
|
||||
label: "Zipformer Bilingual (sherpa-onnx — streaming, zh+en)"
|
||||
role: ASR
|
||||
size_gb: 0.13
|
||||
engine: sherpa-onnx
|
||||
dictation_id: sherpa-zipformer-bilingual-zh-en
|
||||
tag: streaming
|
||||
note: "True streaming partials as you speak (zh+en). CPU. Requires sherpa-onnx."
|
||||
|
||||
- repo_id: "csukuangfj/sherpa-onnx-streaming-paraformer-bilingual-zh-en"
|
||||
label: "Paraformer Bilingual (sherpa-onnx — streaming, zh+en)"
|
||||
role: ASR
|
||||
size_gb: 0.115
|
||||
engine: sherpa-onnx
|
||||
dictation_id: sherpa-paraformer-bilingual-zh-en
|
||||
tag: streaming
|
||||
note: "True streaming partials (zh+en). CPU. Requires sherpa-onnx."
|
||||
|
||||
- repo_id: "csukuangfj/sherpa-onnx-streaming-zipformer-en-20M-2023-02-17"
|
||||
label: "Zipformer Streaming EN 20M (sherpa-onnx — streaming, English)"
|
||||
role: ASR
|
||||
size_gb: 0.128
|
||||
engine: sherpa-onnx
|
||||
dictation_id: sherpa-zipformer-en-20m
|
||||
tag: streaming
|
||||
note: "Tiny English streaming model, very low latency. CPU. Requires sherpa-onnx."
|
||||
|
||||
- repo_id: "csukuangfj/sherpa-onnx-streaming-zipformer-zh-14M-2023-02-23"
|
||||
label: "Zipformer Streaming ZH 14M (sherpa-onnx — streaming, Chinese)"
|
||||
role: ASR
|
||||
size_gb: 0.074
|
||||
engine: sherpa-onnx
|
||||
dictation_id: sherpa-zipformer-zh-14m
|
||||
tag: streaming
|
||||
note: "Tiny Chinese streaming model, very low latency. CPU. Requires sherpa-onnx."
|
||||
|
||||
- repo_id: "csukuangfj/sherpa-onnx-whisper-tiny"
|
||||
label: "Whisper Tiny (sherpa-onnx — dictation, 90+ langs)"
|
||||
role: ASR
|
||||
size_gb: 0.116
|
||||
engine: sherpa-onnx
|
||||
dictation_id: sherpa-whisper-tiny
|
||||
tag: offline
|
||||
note: "Multilingual offline dictation (auto-detect). CPU, int8 ONNX. Requires sherpa-onnx."
|
||||
|
||||
# ── Diarisation ───────────────────────────────────────────────────────
|
||||
|
||||
- repo_id: "pyannote/speaker-diarization-3.1"
|
||||
label: "pyannote speaker diarisation (multi-speaker videos)"
|
||||
role: Diarisation
|
||||
size_gb: 0.8
|
||||
config_only: true # pipeline repo; real weights live in referenced sub-repos
|
||||
note: "Needs an HF_TOKEN with license accepted."
|
||||
|
||||
# ── Optional TTS ──────────────────────────────────────────────────────
|
||||
|
||||
+105
-6
@@ -157,6 +157,22 @@ _BASE_SCHEMA = """
|
||||
last_seen_at REAL,
|
||||
created_at REAL
|
||||
);
|
||||
|
||||
-- Expressive-TTS Spec 01 Phase 1: user pronunciation dictionary. A
|
||||
-- per-language word→respelling map applied as pure text substitution
|
||||
-- before synthesis (Settings → Pronunciation). Fresh installs create it
|
||||
-- here; existing DBs get it via alembic 0008_pronunciation_dictionary.
|
||||
-- Both paths converge on this identical schema (dual-path discipline).
|
||||
CREATE TABLE IF NOT EXISTS pronunciation_entries (
|
||||
id TEXT PRIMARY KEY,
|
||||
term TEXT NOT NULL,
|
||||
replacement TEXT NOT NULL DEFAULT '',
|
||||
type TEXT NOT NULL DEFAULT 'respelling',
|
||||
language TEXT NOT NULL DEFAULT '*',
|
||||
enabled INTEGER NOT NULL DEFAULT 1,
|
||||
created_at REAL
|
||||
);
|
||||
CREATE INDEX IF NOT EXISTS idx_pron_lang ON pronunciation_entries(language);
|
||||
"""
|
||||
|
||||
# Only tables/columns this module is allowed to ALTER. Prevents SQL injection via
|
||||
@@ -206,6 +222,72 @@ def _migrate(conn, current: int) -> int:
|
||||
return current
|
||||
|
||||
|
||||
def _reconcile_additive_columns(conn) -> None:
|
||||
"""Make the live schema converge to ``_BASE_SCHEMA`` by ADDing any column the
|
||||
canonical schema declares but an existing table is missing — the belt for
|
||||
when alembic can't run on an upgraded DB.
|
||||
|
||||
``CREATE TABLE IF NOT EXISTS`` (init_db) never adds columns to a table that
|
||||
already exists, the legacy ``_migrate`` only knows pre-0.3 columns, and
|
||||
``_run_alembic_upgrade`` swallows failures. So a DB whose ``alembic_version``
|
||||
is stamped at a removed revision (e.g. after running a preview build), or
|
||||
where alembic isn't importable in the bundled interpreter, would otherwise
|
||||
lose every alembic-era additive column forever — the ``no such column:
|
||||
consent_audio_path`` 500 (#552/#547), and the same class for
|
||||
``kind``/``vd_states``/``is_demo``/.... Additive only: never drops or retypes
|
||||
a column, so it is safe and backward-compatible with existing user data. The
|
||||
canonical names/types/defaults come solely from ``_BASE_SCHEMA`` (developer
|
||||
controlled), so the ALTER is injection-safe.
|
||||
"""
|
||||
canon = sqlite3.connect(":memory:")
|
||||
try:
|
||||
canon.executescript(_BASE_SCHEMA)
|
||||
_tables_sql = "SELECT name FROM sqlite_master WHERE type='table' AND name NOT LIKE 'sqlite_%'"
|
||||
live_tables = {r[0] for r in conn.execute(_tables_sql)}
|
||||
for table in (r[0] for r in canon.execute(_tables_sql)):
|
||||
if table not in live_tables:
|
||||
continue # whole table missing → init_db's CREATE already made it
|
||||
have = {r[1] for r in conn.execute(f"PRAGMA table_info({table})")}
|
||||
# (cid, name, type, notnull, dflt_value, pk)
|
||||
for _cid, name, ctype, notnull, dflt, _pk in canon.execute(f"PRAGMA table_info({table})"):
|
||||
if name in have or not _IDENT_RE.match(name):
|
||||
continue
|
||||
ddl = f'ALTER TABLE "{table}" ADD COLUMN "{name}" {ctype or "TEXT"}'
|
||||
if dflt is not None:
|
||||
ddl += f" DEFAULT {dflt}"
|
||||
elif notnull:
|
||||
ddl += " DEFAULT ''" # SQLite requires a default to ADD a NOT NULL column
|
||||
try:
|
||||
conn.execute(ddl)
|
||||
logger.info("schema reconcile: added missing column %s.%s", table, name)
|
||||
except sqlite3.OperationalError as exc:
|
||||
if "duplicate column" not in str(exc).lower():
|
||||
logger.warning("schema reconcile ALTER %s.%s failed: %s", table, name, exc)
|
||||
conn.commit()
|
||||
finally:
|
||||
canon.close()
|
||||
|
||||
|
||||
def ensure_schema() -> None:
|
||||
"""Idempotently ensure the base tables + additive columns exist.
|
||||
|
||||
A runtime self-heal for a DB that somehow missed init — e.g. a write hitting
|
||||
``no such table: generation_history`` (#710) because ``init_db()``'s
|
||||
``executescript`` never took on that DB. Safe to call anytime: it's just
|
||||
``CREATE ... IF NOT EXISTS`` plus the additive-only column reconcile, so it
|
||||
never drops or retypes anything and is backward-compatible with user data.
|
||||
Cheaper than ``init_db()`` (skips the legacy ``_migrate`` + alembic), so a
|
||||
write path can call it on a schema error and retry without a 500.
|
||||
"""
|
||||
conn = get_db()
|
||||
try:
|
||||
conn.executescript(_BASE_SCHEMA)
|
||||
_reconcile_additive_columns(conn)
|
||||
conn.commit()
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
def init_db():
|
||||
conn = get_db()
|
||||
try:
|
||||
@@ -214,6 +296,11 @@ def init_db():
|
||||
new_version = _migrate(conn, version)
|
||||
if new_version != version:
|
||||
conn.execute(f"PRAGMA user_version = {new_version}")
|
||||
# Converge any alembic-era additive columns that CREATE TABLE IF NOT
|
||||
# EXISTS + the legacy _migrate don't add to a pre-existing table
|
||||
# (consent_audio_path, kind, ...). Runs regardless of whether alembic
|
||||
# below succeeds, so an unrunnable alembic can't leave a 500-ing schema.
|
||||
_reconcile_additive_columns(conn)
|
||||
conn.commit()
|
||||
finally:
|
||||
conn.close()
|
||||
@@ -227,10 +314,12 @@ def init_db():
|
||||
|
||||
def _run_alembic_upgrade() -> None:
|
||||
"""Best-effort `alembic upgrade head` on startup. Non-fatal: if alembic
|
||||
isn't reachable (e.g. user running a stripped-down install or migrations
|
||||
were already applied out-of-band), log a warning and move on. The
|
||||
_BASE_SCHEMA CREATE TABLE IF NOT EXISTS above guarantees the runtime
|
||||
schema is correct regardless."""
|
||||
isn't reachable (e.g. a stripped-down install) or its version is stamped at
|
||||
a revision no longer in versions/ (e.g. after running a preview build), log
|
||||
a warning and move on. The schema is still kept correct by
|
||||
_reconcile_additive_columns (run in init_db above and again here on failure)
|
||||
— CREATE TABLE IF NOT EXISTS alone does NOT add columns to a pre-existing
|
||||
table, so the reconcile is what actually guarantees additive columns land."""
|
||||
try:
|
||||
import os
|
||||
from alembic import command
|
||||
@@ -248,6 +337,16 @@ def _run_alembic_upgrade() -> None:
|
||||
cfg.set_main_option("sqlalchemy.url", f"sqlite:///{DB_PATH}")
|
||||
command.upgrade(cfg, "head")
|
||||
except Exception as exc:
|
||||
# Don't block startup on a migration tooling problem. The runtime
|
||||
# schema is already correct via _BASE_SCHEMA.
|
||||
# Don't block startup on a migration tooling problem. Converge the schema
|
||||
# directly so a swallowed failure (alembic not importable, or
|
||||
# alembic_version stamped at a removed revision) still lands the additive
|
||||
# columns instead of 500-ing on `no such column` (#552/#547).
|
||||
logger.warning("alembic upgrade head skipped: %s", exc)
|
||||
try:
|
||||
conn = get_db()
|
||||
try:
|
||||
_reconcile_additive_columns(conn)
|
||||
finally:
|
||||
conn.close()
|
||||
except Exception as exc2: # noqa: BLE001
|
||||
logger.warning("schema reconcile after alembic failure also failed: %s", exc2)
|
||||
|
||||
@@ -37,6 +37,11 @@ _HINTS: dict[str, str] = {
|
||||
"APPIMAGE_WEBKIT_WHITESCREEN": "Launch with WEBKIT_DISABLE_DMABUF_RENDERER=1 set.",
|
||||
"HF_AUTH_FAILED": "Set a valid HF_TOKEN in Settings → Hugging Face and retry.",
|
||||
"PYANNOTE_LICENSE_REQUIRED": "Accept the pyannote model licenses on Hugging Face, then retry.",
|
||||
"COMPUTE_TYPE_UNSUPPORTED": "Your GPU doesn't support float16 — OmniVoice retried on int8. If transcription still fails, set OMNIVOICE/ASR_COMPUTE_TYPE=int8 or use CPU.",
|
||||
"TRANSFORMERS_IMPORT": "Your transformers install is incomplete. Reinstall it (`uv pip install --reinstall transformers`) or switch ASR to faster-whisper (Settings → Models).",
|
||||
"UNSUPPORTED_VIDEO_URL": "This link isn't a directly downloadable video. Paste a direct video page (e.g. a youtube.com/watch?v=… or douyin.com/video/<id> link), not a share/profile/feed link — or download the file and drop it in directly.",
|
||||
"VIDEO_DOWNLOAD_NETWORK": "The connection to the video server dropped mid-download (often a transient CDN/network blip or a regional rate-limit). Just retry — OmniVoice already cleaned up the partial download. If it keeps failing, check your network/VPN.",
|
||||
"BROKEN_VENV": "The Python backend environment was moved or damaged. OmniVoice rebuilds it automatically on the next launch; if it keeps failing, use Clean & Retry on the setup screen.",
|
||||
}
|
||||
|
||||
|
||||
@@ -55,10 +60,58 @@ def classify(reason: str) -> str:
|
||||
return "APPIMAGE_WEBKIT_WHITESCREEN"
|
||||
if "pyannote" in low or ("gated" in low and "model" in low) or "accept the" in low:
|
||||
return "PYANNOTE_LICENSE_REQUIRED"
|
||||
# ASR robustness (#551 / #549): name the class so the no-segments toast is
|
||||
# actionable. Place before the generic returns so a compute-type/transformers
|
||||
# failure gets its hint rather than falling through to "".
|
||||
if "compute type" in low or "efficient float16" in low:
|
||||
return "COMPUTE_TYPE_UNSUPPORTED"
|
||||
if (
|
||||
"could not import module" in low
|
||||
or "autofeatureextractor" in low
|
||||
# A corrupted/incomplete transformers install: a model load lazily
|
||||
# resolves a module file that's MISSING from site-packages (an
|
||||
# interrupted `uv sync`, antivirus removal, or a partial update), e.g.
|
||||
# `[Errno 2] No such file or directory:
|
||||
# '.../site-packages/transformers/models/qwen3/modeling_qwen3.py'`.
|
||||
# That's a FileNotFoundError, not an ImportError, so the matches above
|
||||
# miss it and the user got a useless "try restarting". Substring-match
|
||||
# the package + the missing-file signal (separately, so it works on both
|
||||
# POSIX `/` and Windows `\` paths).
|
||||
or (
|
||||
("no such file" in low or "errno 2" in low)
|
||||
and "transformers" in low
|
||||
and "site-packages" in low
|
||||
)
|
||||
):
|
||||
return "TRANSFORMERS_IMPORT"
|
||||
if ("huggingface" in low or "hf_token" in low or "401" in low or "unauthorized" in low) and (
|
||||
"token" in low or "auth" in low or "401" in low or "unauthorized" in low
|
||||
):
|
||||
return "HF_AUTH_FAILED"
|
||||
# Video download (#554/#536): a non-downloadable URL shape vs a transient
|
||||
# network drop — both previously surfaced as a bare yt-dlp string with no
|
||||
# next step. UNSUPPORTED first (more specific) so "Unable to download video:
|
||||
# Broken pipe" still classifies as a network blip.
|
||||
if "unsupported url" in low or "no video formats" in low or "is not a valid url" in low:
|
||||
return "UNSUPPORTED_VIDEO_URL"
|
||||
if (
|
||||
"broken pipe" in low
|
||||
or "connection reset" in low
|
||||
or "unable to download video" in low
|
||||
or "remote end closed" in low
|
||||
or "timed out" in low
|
||||
):
|
||||
return "VIDEO_DOWNLOAD_NETWORK"
|
||||
# A relocated/corrupted venv whose interpreter can't bootstrap its stdlib —
|
||||
# the Rust self-heal rebuilds it; this names the class for the toast.
|
||||
if "no module named 'encodings'" in low:
|
||||
return "BROKEN_VENV"
|
||||
# #564: the interpreter starts fine but the backend can't import its OWN
|
||||
# `omnivoice` package (a venv missing the editable install). Same self-heal
|
||||
# class — Clean & Retry / the bootstrap repair rebuilds it. The trailing
|
||||
# quote keeps a legitimately-named `omnivoice_*` helper from matching.
|
||||
if "no module named 'omnivoice'" in low:
|
||||
return "BROKEN_VENV"
|
||||
return ""
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,77 @@
|
||||
"""Resolve the project's own ``omnivoice`` package from source when the venv's
|
||||
editable install is missing (#564).
|
||||
|
||||
``omnivoice`` is normally an editable install in the backend venv. An interrupted
|
||||
or offline ``uv sync`` can install dependencies yet never lay the editable record
|
||||
(``_editable_impl_omnivoice.pth``), or an antivirus quarantine can remove it —
|
||||
leaving a venv that starts uvicorn but cannot ``import omnivoice``, so it boots
|
||||
fine and only fails at the first model call (``No module named 'omnivoice'``).
|
||||
|
||||
The desktop layout always copies ``omnivoice/`` next to ``backend/``, so we fall
|
||||
back to importing it from there. The bootstrap now also gates on omnivoice being
|
||||
importable (re-syncing to re-lay the editable install), but this keeps the
|
||||
backend resilient even when that repair hasn't run yet.
|
||||
"""
|
||||
import os
|
||||
import sys
|
||||
|
||||
|
||||
def find_omnivoice_source_root(candidates):
|
||||
"""Return the first candidate dir holding ``omnivoice/__init__.py``, else None."""
|
||||
for root in candidates:
|
||||
if root and os.path.isfile(os.path.join(root, "omnivoice", "__init__.py")):
|
||||
return root
|
||||
return None
|
||||
|
||||
|
||||
def _candidate_roots(backend_dir):
|
||||
"""Source roots to probe, most-specific first.
|
||||
|
||||
``OMNIVOICE_PROJECT_ROOT`` lets the launcher point at the staged project dir
|
||||
explicitly; otherwise the desktop layout puts ``omnivoice/`` beside
|
||||
``backend/`` (parent of ``backend_dir``).
|
||||
"""
|
||||
roots = []
|
||||
env = os.environ.get("OMNIVOICE_PROJECT_ROOT")
|
||||
if env:
|
||||
roots.append(env)
|
||||
roots.append(os.path.dirname(os.path.abspath(backend_dir)))
|
||||
return roots
|
||||
|
||||
|
||||
def _already_importable():
|
||||
import importlib.util
|
||||
try:
|
||||
return importlib.util.find_spec("omnivoice") is not None
|
||||
except (ImportError, ValueError):
|
||||
# A half-laid spec (e.g. a stale .pth pointing at a deleted dir) raises
|
||||
# rather than returning None — treat it as "not importable" so we fall
|
||||
# back to the on-disk source.
|
||||
return False
|
||||
|
||||
|
||||
def ensure_omnivoice_importable(backend_dir, logger=None):
|
||||
"""Make ``import omnivoice`` work, falling back to the sibling source tree.
|
||||
|
||||
No-op when the editable/site-packages install already resolves it. Otherwise
|
||||
appends the first source root containing ``omnivoice/`` to ``sys.path``
|
||||
(appended, never inserted, so a real install keeps precedence). Returns the
|
||||
root that was added, or ``None`` if none was needed or found.
|
||||
"""
|
||||
if _already_importable():
|
||||
return None
|
||||
root = find_omnivoice_source_root(_candidate_roots(backend_dir))
|
||||
if root and root not in sys.path:
|
||||
sys.path.append(root)
|
||||
if logger:
|
||||
logger.warning(
|
||||
"omnivoice not importable from the venv (missing/broken editable "
|
||||
"install) — resolving it from source at %s (#564)", root,
|
||||
)
|
||||
elif logger and root is None:
|
||||
logger.error(
|
||||
"omnivoice is not importable and no source tree was found next to "
|
||||
"%s — the install is incomplete; relaunch to let the bootstrap "
|
||||
"repair the venv (#564)", backend_dir,
|
||||
)
|
||||
return root
|
||||
@@ -61,9 +61,15 @@ def seed_sample_project():
|
||||
if count > 0:
|
||||
return # Not first run — skip
|
||||
|
||||
# Check if demo audio exists
|
||||
# The demo clip is committed at backend/assets/samples/demo_voice.wav and
|
||||
# bundled with the app (#621). If it's somehow absent (e.g. a partial
|
||||
# checkout), skip the seed gracefully rather than seeding a profile that
|
||||
# points at a missing file — run scripts/build_demos.sh to regenerate it.
|
||||
if not os.path.isfile(_DEMO_AUDIO):
|
||||
logger.warning("Demo audio not found at %s — skipping onboarding seed", _DEMO_AUDIO)
|
||||
logger.warning(
|
||||
"Demo audio not found at %s — skipping onboarding seed "
|
||||
"(regenerate with scripts/build_demos.sh)", _DEMO_AUDIO,
|
||||
)
|
||||
return
|
||||
|
||||
# Copy demo audio to voices directory
|
||||
|
||||
+34
-7
@@ -36,18 +36,39 @@ _TOKEN_PATTERNS = (
|
||||
re.compile(r"github_pat_[A-Za-z0-9_]{20,}"), # GitHub fine-grained PAT
|
||||
re.compile(r"gh[pousr]_[A-Za-z0-9]{30,}"), # GitHub classic tokens
|
||||
re.compile(r"sk-[A-Za-z0-9_\-]{20,}"), # OpenAI-style API keys
|
||||
# A backend error can carry a secret from *any* provider (the LLM-providers
|
||||
# feature ships a dozen), so match the common credential shapes too, not
|
||||
# just the four vendors above — a leaked key in a public issue is real harm.
|
||||
re.compile(r"eyJ[A-Za-z0-9_\-]{8,}\.[A-Za-z0-9_\-]{8,}\.[A-Za-z0-9_\-]{6,}"), # JWT (Bearer)
|
||||
re.compile(r"AIza[0-9A-Za-z_\-]{35}"), # Google API key
|
||||
re.compile(r"xox[baprs]-[A-Za-z0-9\-]{10,}"), # Slack token
|
||||
re.compile(r"AKIA[0-9A-Z]{16}"), # AWS access key id
|
||||
re.compile(r"(?i)bearer\s+[A-Za-z0-9._\-]{16,}"), # opaque bearer tokens
|
||||
)
|
||||
|
||||
# Secrets carried in a URL query string (`?token=…`, `&api_key=…`). Redact the
|
||||
# VALUE while keeping the param name + separator so the URL stays legible. Bare
|
||||
# `key=` is intentionally excluded — too common in non-secret text; shaped keys
|
||||
# are already caught above and named env vars by the sweep below.
|
||||
_URL_SECRET_RE = re.compile(
|
||||
r"((?:access[_-]?token|api[_-]?key|apikey|auth[_-]?token|token|secret|password|passwd|pwd)=)"
|
||||
r"([^&\s\"'#]{6,})",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
# Home-directory shapes for all three supported platforms. Matched
|
||||
# pattern-wise (not just this machine's $HOME) so paths quoted from a
|
||||
# user's pasted log on another OS get cleaned too.
|
||||
# IGNORECASE because Windows is case-insensitive and tools routinely emit the
|
||||
# lowercase `c:\users\<name>` form, which the CLAUDE.md redaction spec still
|
||||
# requires to become `~`. `Users`/`users`, `Home`/`home` all match.
|
||||
_HOME_PATTERNS = (
|
||||
# Windows-with-forward-slashes must run BEFORE the bare macOS shape, or
|
||||
# `/Users/<name>` inside `C:/Users/<name>` gets eaten first, leaving `C:~`.
|
||||
re.compile(r"[A-Za-z]:/Users/[^/\s\"']+"), # Windows, forward slashes (file URLs, normalized traces)
|
||||
re.compile(r"/Users/[^/\s\"']+"), # macOS
|
||||
re.compile(r"/home/[^/\s\"']+"), # Linux
|
||||
re.compile(r"[A-Za-z]:\\Users\\[^\\\s\"']+"), # Windows, backslashes
|
||||
re.compile(r"[A-Za-z]:/Users/[^/\s\"']+", re.IGNORECASE), # Windows, forward slashes
|
||||
re.compile(r"/Users/[^/\s\"']+", re.IGNORECASE), # macOS
|
||||
re.compile(r"/home/[^/\s\"']+", re.IGNORECASE), # Linux
|
||||
re.compile(r"[A-Za-z]:\\Users\\[^\\\s\"']+", re.IGNORECASE), # Windows, backslashes
|
||||
)
|
||||
|
||||
# Values shorter than this are too entropy-poor to be real secrets and too
|
||||
@@ -85,19 +106,25 @@ def scrub_text(text: str | None) -> str:
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# 2. Credential-shaped substrings.
|
||||
# 2. Credential-shaped substrings + URL query secrets.
|
||||
for pat in _TOKEN_PATTERNS:
|
||||
try:
|
||||
s = pat.sub(REDACTED, s)
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
s = _URL_SECRET_RE.sub(lambda m: m.group(1) + REDACTED, s)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# 3. This process's real home dir (covers symlinked/nonstandard homes
|
||||
# the generic patterns miss), then the per-OS shapes.
|
||||
# the generic patterns miss), then the per-OS shapes. Boundary-aware so
|
||||
# a home of `/Users/john` doesn't rewrite `/Users/johnny` to `~ny`
|
||||
# (leaking the fragment + mangling the path).
|
||||
try:
|
||||
home = os.path.expanduser("~")
|
||||
if home and home not in ("/", "~"):
|
||||
s = s.replace(home, "~")
|
||||
s = re.sub(re.escape(home) + r"(?=[/\\\s\"']|$)", "~", s)
|
||||
except Exception:
|
||||
pass
|
||||
for pat in _HOME_PATTERNS:
|
||||
|
||||
+33
-4
@@ -2,14 +2,43 @@
|
||||
|
||||
Read from the installed package metadata (driven by ``pyproject.toml``) so the
|
||||
FastAPI/API version and exported-bundle metadata never drift to a stale literal
|
||||
again (the prior "0.4.0" / "0.2.7" bug). Falls back to a literal only when
|
||||
running from a raw source checkout that was never ``uv sync``'d.
|
||||
— the prior "0.4.0" / "0.2.7" bug, and the v0.3.6 desktop build that reported
|
||||
"0.3.5" because the *frozen* backend couldn't read its own metadata.
|
||||
|
||||
Resolution order:
|
||||
1. installed package metadata — correct in any ``uv sync``'d env and, thanks
|
||||
to ``copy_metadata('omnivoice')`` in ``backend.spec``, in the frozen build;
|
||||
2. ``pyproject.toml`` walked up from this file — correct for a raw source
|
||||
checkout that was never installed;
|
||||
3. ``_FALLBACK_VERSION`` — a last resort, kept in lockstep with the four
|
||||
version files by ``tests/test_app_version.py`` so it can never silently
|
||||
drift again.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from importlib.metadata import PackageNotFoundError, version
|
||||
from pathlib import Path
|
||||
|
||||
# Last-resort literal. Guarded by
|
||||
# tests/test_app_version.py::test_all_version_files_in_lockstep and bumped by
|
||||
# release.yml's version-bump job, so it stays equal to
|
||||
# pyproject/tauri.conf/Cargo/package.json.
|
||||
_FALLBACK_VERSION = "0.3.8"
|
||||
|
||||
|
||||
def _fallback_version() -> str:
|
||||
"""Version for contexts where package metadata is unavailable."""
|
||||
for parent in Path(__file__).resolve().parents:
|
||||
pyproject = parent / "pyproject.toml"
|
||||
if pyproject.is_file():
|
||||
match = re.search(r'(?m)^version\s*=\s*"([^"]+)"', pyproject.read_text())
|
||||
if match:
|
||||
return match.group(1)
|
||||
return _FALLBACK_VERSION
|
||||
|
||||
|
||||
try:
|
||||
APP_VERSION = version("omnivoice")
|
||||
except PackageNotFoundError: # non-installed source checkout
|
||||
APP_VERSION = "0.3.5"
|
||||
except PackageNotFoundError: # frozen build w/o metadata, or non-installed checkout
|
||||
APP_VERSION = _fallback_version()
|
||||
|
||||
@@ -50,6 +50,22 @@ def _recv(stream):
|
||||
return json.loads(bytes(body).decode("utf-8"))
|
||||
|
||||
|
||||
# NOTE: keep this compute_type fallback in lockstep with
|
||||
# services/asr_backend.py:_compute_type_candidates / _is_compute_type_error.
|
||||
# This sidecar runs in a child proc with a clean import path, so we duplicate a
|
||||
# tiny copy rather than cross-importing the heavy services package (#551).
|
||||
def _ct_candidates(device):
|
||||
override = os.environ.get("ASR_COMPUTE_TYPE")
|
||||
if override:
|
||||
return [override]
|
||||
return ["float16", "int8_float16", "int8"] if device == "cuda" else ["int8", "float32"]
|
||||
|
||||
|
||||
def _is_ct_error(msg):
|
||||
low = msg.lower()
|
||||
return "compute type" in low or "efficient float16" in low
|
||||
|
||||
|
||||
def _get_model():
|
||||
global _model
|
||||
if _model is None:
|
||||
@@ -60,8 +76,20 @@ def _get_model():
|
||||
device = "cuda" if torch.cuda.is_available() else "cpu"
|
||||
except Exception:
|
||||
device = "cpu"
|
||||
compute = "float16" if device == "cuda" else "int8"
|
||||
_model = WhisperModel(name, device=device, compute_type=compute)
|
||||
# Degrade fp16 → int8 rather than crash on GPUs without efficient fp16
|
||||
# (older Maxwell/Pascal, GTX 16xx, CTranslate2/cuDNN mismatch) (#551).
|
||||
last_err = None
|
||||
for compute in _ct_candidates(device):
|
||||
try:
|
||||
_model = WhisperModel(name, device=device, compute_type=compute)
|
||||
break
|
||||
except (ValueError, RuntimeError) as e:
|
||||
last_err = e
|
||||
if _is_ct_error(str(e)):
|
||||
continue
|
||||
raise
|
||||
else:
|
||||
raise last_err
|
||||
return _model
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,187 @@
|
||||
"""dots.tts sidecar package (issue #498).
|
||||
|
||||
dots.tts is rednote-hilab's 2B fully-continuous autoregressive TTS — widely
|
||||
cited as among the strongest open zero-shot voice-cloning models. 24
|
||||
languages, 48 kHz output, Apache-2.0 (code + checkpoints).
|
||||
|
||||
It runs in its own subprocess **and its own venv**, isolated from the
|
||||
OmniVoice parent, for the same ``transformers`` reason as IndexTTS and
|
||||
MOSS-TTS-v1.5: dots.tts pins ``transformers==4.57.0`` (verified against
|
||||
``constraints/recommended.txt``), while OmniVoice pins
|
||||
``transformers>=5.3.0``. The two cannot share one interpreter.
|
||||
|
||||
Cross-platform honesty (the strict default-parity rule): dots.tts's
|
||||
upstream package declares **Linux + macOS** classifiers only — **no
|
||||
Windows** — and its device code is **CUDA-or-CPU with no MPS branch**
|
||||
(verified in ``runtime.py``). So:
|
||||
|
||||
* It is **opt-in** (engine-picker selection + a user-provided clone),
|
||||
never a default — so it never becomes a broken default on any platform.
|
||||
* ``is_available()`` returns ``False`` with a clear reason on **Windows**
|
||||
rather than offering an engine that can't run there. Windows users are
|
||||
pointed at WSL2 / a Linux or macOS host.
|
||||
* ``gpu_compat = ("cuda", "cpu")`` — no MPS claim. On Apple Silicon the
|
||||
upstream package runs on CPU (slow but correct); the faster MLX path is
|
||||
a community port we deliberately don't auto-wire here.
|
||||
|
||||
Three public entry points: ``DotsTTSBackend`` (this module), ``main.py``
|
||||
(sidecar, runs under dots.tts's ``transformers==4.57`` venv — never imported
|
||||
by the parent), and ``bootstrap.py`` (venv probe + lazy bootstrap).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import sys
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from services.subprocess_backend import SubprocessBackend
|
||||
|
||||
if TYPE_CHECKING:
|
||||
import torch # noqa: F401
|
||||
|
||||
logger = logging.getLogger("omnivoice.dots_tts")
|
||||
|
||||
|
||||
class DotsTTSBackend(SubprocessBackend):
|
||||
"""dots.tts (rednote-hilab) — 2B, 24 langs, zero-shot clone, CUDA/CPU.
|
||||
|
||||
Runs in a long-lived sidecar over length-prefixed JSON-over-stdio in a
|
||||
dedicated venv (``transformers==4.57.0``). First synthesize cold-loads
|
||||
the ~9 GB checkpoint (bf16 on CUDA); subsequent calls reuse the process.
|
||||
|
||||
Installation (OmniVoice prefers a user's existing ``${DIR}/.venv``)::
|
||||
|
||||
git clone https://github.com/rednote-hilab/dots.tts.git
|
||||
cd dots.tts
|
||||
uv venv && uv pip install -e . -c constraints/recommended.txt
|
||||
|
||||
Set ``OMNIVOICE_DOTS_TTS_DIR`` to the clone root. OmniVoice creates
|
||||
``backend/engines/dots_tts/.venv`` lazily on first launch if no venv
|
||||
exists yet; the user's existing ``${DIR}/.venv`` is preferred if present.
|
||||
|
||||
Best cloning quality uses the ``dots.tts-soar`` checkpoint (the default)
|
||||
and BOTH a reference clip and its exact transcript (continuation
|
||||
cloning). License: Apache-2.0.
|
||||
"""
|
||||
|
||||
id = "dots-tts"
|
||||
display_name = (
|
||||
"dots.tts (2B, 24 langs, zero-shot clone, CUDA/CPU, 48 kHz, Apache-2.0)"
|
||||
)
|
||||
supports_voice_design = False # requires ref audio for timbre cloning
|
||||
# dots.tts emits 48 kHz (verified via checkpoint vocoder.sample_rate).
|
||||
_DEFAULT_SAMPLE_RATE = 48000
|
||||
# CUDA + CPU only; no MPS branch in upstream runtime.py.
|
||||
gpu_compat = ("cuda", "cpu")
|
||||
|
||||
# ── availability ───────────────────────────────────────────────────────
|
||||
|
||||
@classmethod
|
||||
def is_available(cls) -> tuple[bool, str]:
|
||||
# Cross-platform parity: dots.tts upstream is Linux/macOS-only (no
|
||||
# Windows classifier, no Windows install path). Refuse cleanly on
|
||||
# Windows instead of advertising an engine that can't run.
|
||||
if sys.platform == "win32":
|
||||
return False, (
|
||||
"dots.tts is not supported on Windows — upstream targets "
|
||||
"Linux and macOS only. Run OmniVoice under WSL2, or use a "
|
||||
"Linux/macOS host. See docs/engines/dots-tts.md."
|
||||
)
|
||||
|
||||
# Do NOT import dots_tts here: it pins transformers==4.57, which can't
|
||||
# coexist with the parent's transformers>=5.3 in one interpreter —
|
||||
# the reason for the subprocess isolation. Verify the venv on disk
|
||||
# only; a real health-check is gated on the user's "Test engine"
|
||||
# action in Settings.
|
||||
from engines.dots_tts.bootstrap import (
|
||||
DOTS_TTS_SIDECAR_SCRIPT,
|
||||
is_dots_tts_installed,
|
||||
)
|
||||
if not is_dots_tts_installed():
|
||||
return False, (
|
||||
"dots.tts venv not found. Set OMNIVOICE_DOTS_TTS_DIR to your "
|
||||
"dots.tts clone (the directory containing pyproject.toml) and "
|
||||
"restart OmniVoice. CUDA or CPU only (no MPS). See "
|
||||
"docs/engines/dots-tts.md for the full install walk-through."
|
||||
)
|
||||
if not DOTS_TTS_SIDECAR_SCRIPT.exists():
|
||||
return False, (
|
||||
"dots.tts sidecar script missing at "
|
||||
f"{DOTS_TTS_SIDECAR_SCRIPT} — reinstall OmniVoice."
|
||||
)
|
||||
return True, "ok (CUDA when present, else CPU)"
|
||||
|
||||
@classmethod
|
||||
def venv_python(cls):
|
||||
from engines.dots_tts.bootstrap import resolve_dots_tts_venv
|
||||
return resolve_dots_tts_venv()
|
||||
|
||||
@classmethod
|
||||
def sidecar_script(cls):
|
||||
from engines.dots_tts.bootstrap import DOTS_TTS_SIDECAR_SCRIPT
|
||||
return DOTS_TTS_SIDECAR_SCRIPT
|
||||
|
||||
# ── TTSBackend protocol ────────────────────────────────────────────────
|
||||
|
||||
@property
|
||||
def sample_rate(self) -> int:
|
||||
return self._DEFAULT_SAMPLE_RATE
|
||||
|
||||
@property
|
||||
def supported_languages(self) -> list[str]:
|
||||
# 24 languages with auto-detect; expose "multi" on the protocol
|
||||
# surface and translate the caller's language at synthesize time.
|
||||
return ["multi"]
|
||||
|
||||
# ── generate (parent-side arbitration) ─────────────────────────────────
|
||||
|
||||
def generate(self, text: str, **kw) -> "torch.Tensor":
|
||||
"""Synthesize one utterance through the dots.tts sidecar.
|
||||
|
||||
kwargs honored:
|
||||
* ``ref_audio`` — reference clip path → ``prompt_audio_path``
|
||||
(zero-shot cloning). Optional.
|
||||
* ``ref_text`` — the reference transcript → ``prompt_text``.
|
||||
Best cloning fidelity ("continuation"). Upstream
|
||||
REQUIRES ``prompt_audio_path`` when ``prompt_text``
|
||||
is set, so we drop a stray ref_text with no
|
||||
ref_audio rather than let the sidecar raise.
|
||||
* ``language`` — ISO code / name / None (auto-detect).
|
||||
* ``num_step`` — flow-matching steps → ``num_steps`` (default 10;
|
||||
use 4 for the ``dots.tts-mf`` checkpoint).
|
||||
* ``guidance_scale`` — CFG (default 1.2; >2 amplifies energy).
|
||||
|
||||
Returns a tensor of shape (1, n_samples) at :attr:`sample_rate`.
|
||||
"""
|
||||
forwarded: dict = {}
|
||||
|
||||
ref_audio = kw.get("ref_audio")
|
||||
if ref_audio:
|
||||
forwarded["ref_audio"] = ref_audio
|
||||
ref_text = kw.get("ref_text")
|
||||
if ref_text:
|
||||
# continuation cloning — only valid alongside ref_audio.
|
||||
forwarded["ref_text"] = ref_text
|
||||
elif kw.get("ref_text"):
|
||||
logger.info(
|
||||
"dots-tts: ref_text supplied without ref_audio; ignoring "
|
||||
"(upstream requires prompt_audio_path when prompt_text is set)."
|
||||
)
|
||||
|
||||
language = kw.get("language")
|
||||
if language:
|
||||
forwarded["language"] = str(language)
|
||||
|
||||
# OmniVoice's generic num_step default is 16; dots.tts's own default
|
||||
# is 10. Honor an explicit value, else use the dots-appropriate 10.
|
||||
num_step = kw.get("num_step")
|
||||
forwarded["num_steps"] = int(num_step) if num_step is not None else 10
|
||||
|
||||
# dots.tts's own CFG default is 1.2 (the generic 2.0 over-energises).
|
||||
guidance = kw.get("guidance_scale")
|
||||
forwarded["guidance_scale"] = float(guidance) if guidance is not None else 1.2
|
||||
|
||||
return super().generate(text, **forwarded)
|
||||
|
||||
|
||||
__all__ = ["DotsTTSBackend"]
|
||||
@@ -0,0 +1,226 @@
|
||||
"""dots.tts venv probe + lazy bootstrap (issue #498).
|
||||
|
||||
Resolves which Python interpreter runs the dots.tts sidecar. Mirrors
|
||||
``engines.indextts.bootstrap`` / ``engines.moss_tts_v15.bootstrap`` because
|
||||
dots.tts has the same shape of problem: a hard ``transformers==4.57.0`` pin
|
||||
that conflicts with the parent's ``transformers>=5.3`` — so it runs in its
|
||||
own venv.
|
||||
|
||||
Probe order (priority — existing power-user installs win, zero migration):
|
||||
|
||||
1. ``${OMNIVOICE_DOTS_TTS_DIR}/.venv/`` — the user's clone-level venv.
|
||||
2. ``backend/engines/dots_tts/.venv/`` — this package's own venv.
|
||||
3. Bootstrap: ``uv venv`` then ``uv pip install -e <clone> -c
|
||||
<clone>/constraints/recommended.txt`` (the upstream-pinned stack:
|
||||
torch==2.8.0, transformers==4.57.0, …).
|
||||
|
||||
Caching: memoised after first success. Tests reset via :func:`invalidate`.
|
||||
|
||||
Security: same posture as IndexTTS — bootstrap never touches HF_TOKEN; the
|
||||
sidecar's stderr is redacted by the parent's ``HFTokenRedactor``; the
|
||||
editable install comes from a user-controlled clone they already trust.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
logger = logging.getLogger("omnivoice.dots_tts.bootstrap")
|
||||
|
||||
#: Absolute path to the sidecar entrypoint.
|
||||
DOTS_TTS_SIDECAR_SCRIPT: Path = Path(__file__).parent / "main.py"
|
||||
|
||||
#: This package's owned venv (Probe 2).
|
||||
_ENGINES_VENV_DIR: Path = Path(__file__).parent / ".venv"
|
||||
|
||||
#: Env var pointing at the user's dots.tts clone root.
|
||||
_CLONE_DIR_ENV: str = "OMNIVOICE_DOTS_TTS_DIR"
|
||||
|
||||
#: Per-process resolution cache. Cleared by :func:`invalidate` for tests.
|
||||
_resolved_python: Optional[Path] = None
|
||||
|
||||
_IMPORT_PROBE_TIMEOUT_S = 15
|
||||
_UV_VENV_TIMEOUT_S = 120
|
||||
_UV_PIP_INSTALL_TIMEOUT_S = 1800
|
||||
|
||||
|
||||
# ── public API ────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def invalidate() -> None:
|
||||
"""Clear the resolved-python cache. Tests call this between scenarios."""
|
||||
global _resolved_python
|
||||
_resolved_python = None
|
||||
|
||||
|
||||
def is_dots_tts_installed() -> bool:
|
||||
"""Cheap file-existence check for a usable dots.tts venv. Does NOT spawn
|
||||
the venv Python — that's saved for :func:`resolve_dots_tts_venv`."""
|
||||
for cand in _probe_paths():
|
||||
if cand.is_file():
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def resolve_dots_tts_venv() -> Path:
|
||||
"""Resolve the sidecar's Python interpreter (probe order in the module
|
||||
docstring). Memoised. Raises :exc:`RuntimeError` if none can be located
|
||||
and the bootstrap path is unavailable."""
|
||||
global _resolved_python
|
||||
if _resolved_python is not None:
|
||||
return _resolved_python
|
||||
|
||||
clone_dir = os.environ.get(_CLONE_DIR_ENV)
|
||||
|
||||
# Probe 1 — user's clone-level venv.
|
||||
if clone_dir:
|
||||
cand = _venv_python_path(Path(clone_dir) / ".venv")
|
||||
if cand.is_file() and _venv_can_import_dots(cand):
|
||||
logger.info(
|
||||
"dots.tts venv resolved from %s: %s", _CLONE_DIR_ENV, cand,
|
||||
)
|
||||
_resolved_python = cand
|
||||
return cand
|
||||
|
||||
# Probe 2 — this package's own venv.
|
||||
cand = _venv_python_path(_ENGINES_VENV_DIR)
|
||||
if cand.is_file() and _venv_can_import_dots(cand):
|
||||
logger.info("dots.tts venv resolved from engines path: %s", cand)
|
||||
_resolved_python = cand
|
||||
return cand
|
||||
|
||||
# Probe 3 — bootstrap.
|
||||
if not clone_dir:
|
||||
raise RuntimeError(
|
||||
"dots.tts is not installed. Set the "
|
||||
f"{_CLONE_DIR_ENV} environment variable to your dots.tts clone "
|
||||
"(the directory that contains pyproject.toml and constraints/), "
|
||||
"then restart OmniVoice. See docs/engines/dots-tts.md for the "
|
||||
"full install walk-through."
|
||||
)
|
||||
|
||||
cand = _bootstrap_engines_venv(Path(clone_dir))
|
||||
_resolved_python = cand
|
||||
return cand
|
||||
|
||||
|
||||
# ── internals ─────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def _venv_python_path(venv_dir: Path) -> Path:
|
||||
if sys.platform == "win32":
|
||||
return venv_dir / "Scripts" / "python.exe"
|
||||
return venv_dir / "bin" / "python"
|
||||
|
||||
|
||||
def _probe_paths() -> list[Path]:
|
||||
out: list[Path] = []
|
||||
clone_dir = os.environ.get(_CLONE_DIR_ENV)
|
||||
if clone_dir:
|
||||
out.append(_venv_python_path(Path(clone_dir) / ".venv"))
|
||||
out.append(_venv_python_path(_ENGINES_VENV_DIR))
|
||||
return out
|
||||
|
||||
|
||||
def _venv_can_import_dots(python_path: Path) -> bool:
|
||||
"""Spawn the candidate python and verify ``import dots_tts.runtime`` works.
|
||||
Bounded by ``_IMPORT_PROBE_TIMEOUT_S``. False on any failure."""
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
[str(python_path), "-c", "import dots_tts.runtime"],
|
||||
capture_output=True,
|
||||
timeout=_IMPORT_PROBE_TIMEOUT_S,
|
||||
)
|
||||
except (subprocess.TimeoutExpired, OSError) as exc:
|
||||
logger.debug("dots.tts import probe failed for %s: %s", python_path, exc)
|
||||
return False
|
||||
if proc.returncode != 0:
|
||||
logger.debug(
|
||||
"dots.tts import probe non-zero for %s: %s",
|
||||
python_path,
|
||||
proc.stderr.decode("utf-8", errors="replace")[:200],
|
||||
)
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def _locate_uv() -> Optional[str]:
|
||||
bundled = os.environ.get("OMNIVOICE_BUNDLED_UV")
|
||||
if bundled and Path(bundled).is_file():
|
||||
return bundled
|
||||
return shutil.which("uv")
|
||||
|
||||
|
||||
def _bootstrap_engines_venv(clone_dir: Path) -> Path:
|
||||
"""Create engines/dots_tts/.venv and editable-install the user's clone
|
||||
with the upstream constraints file."""
|
||||
uv = _locate_uv()
|
||||
if not uv:
|
||||
raise RuntimeError(
|
||||
"uv is required to bootstrap the dots.tts venv but was not found "
|
||||
"on PATH (and OMNIVOICE_BUNDLED_UV was not set). Install uv from "
|
||||
"https://docs.astral.sh/uv/ and re-launch OmniVoice, or set "
|
||||
"OMNIVOICE_BUNDLED_UV to the absolute path of a uv binary."
|
||||
)
|
||||
|
||||
logger.info(
|
||||
"Bootstrapping dots.tts venv at %s from %s (this can take several "
|
||||
"minutes on first launch)", _ENGINES_VENV_DIR, clone_dir,
|
||||
)
|
||||
|
||||
try:
|
||||
subprocess.run(
|
||||
[uv, "venv", str(_ENGINES_VENV_DIR)],
|
||||
check=True, timeout=_UV_VENV_TIMEOUT_S, capture_output=True,
|
||||
)
|
||||
except subprocess.CalledProcessError as exc:
|
||||
raise RuntimeError(
|
||||
f"uv venv failed for dots.tts bootstrap at {_ENGINES_VENV_DIR}: "
|
||||
f"{exc.stderr.decode('utf-8', errors='replace') if exc.stderr else exc}"
|
||||
) from exc
|
||||
|
||||
python_path = _venv_python_path(_ENGINES_VENV_DIR)
|
||||
install_cmd = [
|
||||
uv, "pip", "install",
|
||||
"--python", str(python_path),
|
||||
"-e", str(clone_dir),
|
||||
]
|
||||
# Apply the upstream pin set when it ships with the clone.
|
||||
constraints = clone_dir / "constraints" / "recommended.txt"
|
||||
if constraints.is_file():
|
||||
install_cmd += ["-c", str(constraints)]
|
||||
try:
|
||||
subprocess.run(
|
||||
install_cmd, check=True,
|
||||
timeout=_UV_PIP_INSTALL_TIMEOUT_S, capture_output=True,
|
||||
)
|
||||
except subprocess.CalledProcessError as exc:
|
||||
raise RuntimeError(
|
||||
"uv pip install -e failed during dots.tts bootstrap "
|
||||
f"({clone_dir}): "
|
||||
f"{exc.stderr.decode('utf-8', errors='replace') if exc.stderr else exc}. "
|
||||
"See docs/engines/dots-tts.md."
|
||||
) from exc
|
||||
|
||||
if not _venv_can_import_dots(python_path):
|
||||
raise RuntimeError(
|
||||
"dots.tts bootstrap completed but `import dots_tts.runtime` still "
|
||||
f"fails from {python_path}. Verify that {clone_dir} is a valid "
|
||||
"dots.tts clone. See docs/engines/dots-tts.md."
|
||||
)
|
||||
|
||||
logger.info("dots.tts venv bootstrap successful: %s", python_path)
|
||||
return python_path
|
||||
|
||||
|
||||
__all__ = [
|
||||
"DOTS_TTS_SIDECAR_SCRIPT",
|
||||
"invalidate",
|
||||
"is_dots_tts_installed",
|
||||
"resolve_dots_tts_venv",
|
||||
]
|
||||
@@ -0,0 +1,255 @@
|
||||
"""dots.tts sidecar entry point (issue #498).
|
||||
|
||||
Runs inside ``engines/dots_tts/.venv`` (or the user's existing
|
||||
``${OMNIVOICE_DOTS_TTS_DIR}/.venv``) with ``transformers==4.57.0``, isolated
|
||||
from the OmniVoice parent (``transformers>=5.3``). Same isolation rationale
|
||||
as the IndexTTS / MOSS-TTS-v1.5 sidecars.
|
||||
|
||||
Stdlib-only at import time; ``dots_tts`` + torch are imported lazily on the
|
||||
first synthesize op so the ``ready`` frame fits inside the parent's 30 s
|
||||
spawn handshake even on a cold filesystem.
|
||||
|
||||
Wire protocol — length-prefixed JSON over stdin/stdout, byte-identical to
|
||||
``backend/services/subprocess_backend.py``::
|
||||
|
||||
[ 4-byte big-endian uint32 length ][ N bytes UTF-8 JSON ]
|
||||
|
||||
Op flow:
|
||||
1. Sidecar -> parent: {"op": "ready", "engine": "dots-tts",
|
||||
"sample_rate": 48000}
|
||||
2. parent -> sidecar: {"op": "ping"} -> {"op": "pong", "vram_mb": N}
|
||||
3. parent -> sidecar: {"op": "synthesize", "text": "...",
|
||||
"ref_audio": "/path/ref.wav",
|
||||
"ref_text": "transcript", "language": "EN",
|
||||
"num_steps": 10, "guidance_scale": 1.2}
|
||||
-> {"op": "progress", ...} (cold load) then
|
||||
-> {"op": "audio", "audio_pcm_b64": "...", "sample_rate": 48000,
|
||||
"n_samples": N}
|
||||
4. parent -> sidecar: {"op": "shutdown"} -> exit 0
|
||||
|
||||
Restrictions: NO imports from OmniVoice parent code (different venv). NO
|
||||
logging of ``os.environ`` contents. Single-frame DoS cap matches the
|
||||
parent's ``MAX_FRAME_BYTES``.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import json
|
||||
import os
|
||||
import struct
|
||||
import sys
|
||||
import traceback
|
||||
|
||||
|
||||
# Mirrors backend/services/subprocess_backend.py::MAX_FRAME_BYTES.
|
||||
MAX_FRAME_BYTES = 64 * 1024 * 1024
|
||||
|
||||
#: dots.tts emits 48 kHz (checkpoint vocoder.sample_rate). Advertised in the
|
||||
#: ready frame; the real value is re-read from each generate() result.
|
||||
DOTS_SAMPLE_RATE = 48000
|
||||
|
||||
#: Default checkpoint. ``-soar`` is the best-cloning variant; ``-mf`` is the
|
||||
#: fastest (use num_steps=4). Overridable for air-gapped / mirror installs.
|
||||
_DEFAULT_REPO = "rednote-hilab/dots.tts-soar"
|
||||
|
||||
|
||||
# ── wire protocol ─────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def _send(stream, obj: dict) -> None:
|
||||
body = json.dumps(obj, separators=(",", ":")).encode("utf-8")
|
||||
stream.write(struct.pack("!I", len(body)))
|
||||
stream.write(body)
|
||||
stream.flush()
|
||||
|
||||
|
||||
def _recv(stream):
|
||||
header = stream.read(4)
|
||||
if len(header) < 4:
|
||||
return None # EOF
|
||||
(n,) = struct.unpack("!I", header)
|
||||
if n > MAX_FRAME_BYTES:
|
||||
raise IOError(f"frame too large: {n}")
|
||||
body = bytearray()
|
||||
while len(body) < n:
|
||||
chunk = stream.read(n - len(body))
|
||||
if not chunk:
|
||||
raise IOError("short read")
|
||||
body.extend(chunk)
|
||||
return json.loads(bytes(body).decode("utf-8"))
|
||||
|
||||
|
||||
def _measure_vram_mb() -> float:
|
||||
"""This sidecar's own GPU memory in MB (MM2-08). 0 on CPU. Never raises."""
|
||||
try:
|
||||
import torch
|
||||
if torch.cuda.is_available():
|
||||
return round(torch.cuda.memory_allocated() / (1024 ** 2), 1)
|
||||
except Exception:
|
||||
pass
|
||||
return 0.0
|
||||
|
||||
|
||||
# ── model loading (lazy, on first synthesize) ─────────────────────────────
|
||||
|
||||
|
||||
# Module-level singleton — (runtime,). Device is auto-selected inside the
|
||||
# dots.tts runtime (cuda-or-cpu, no MPS); we don't pass a device.
|
||||
_runtime = None
|
||||
|
||||
|
||||
def _load_runtime(stdout):
|
||||
"""Cold-construct the dots.tts runtime.
|
||||
|
||||
``DotsTtsRuntime.from_pretrained`` auto-selects cuda-or-cpu internally
|
||||
(no MPS path). precision is bf16 on CUDA; on CPU we fall back to fp32
|
||||
(bf16 CPU kernels are spotty). Both overridable via env.
|
||||
"""
|
||||
global _runtime
|
||||
if _runtime is not None:
|
||||
return _runtime
|
||||
|
||||
_send(stdout, {"op": "progress", "stage": "loading_model", "percent": 0})
|
||||
|
||||
import torch
|
||||
from dots_tts.runtime import DotsTtsRuntime # type: ignore[import-not-found]
|
||||
|
||||
repo = os.environ.get("OMNIVOICE_DOTS_TTS_MODEL", _DEFAULT_REPO)
|
||||
default_precision = "bfloat16" if torch.cuda.is_available() else "float32"
|
||||
precision = os.environ.get("OMNIVOICE_DOTS_TTS_PRECISION", default_precision)
|
||||
optimize = os.environ.get("OMNIVOICE_DOTS_TTS_OPTIMIZE", "0") == "1"
|
||||
|
||||
_send(stdout, {"op": "progress", "stage": "loading_model", "percent": 50})
|
||||
|
||||
_runtime = DotsTtsRuntime.from_pretrained(
|
||||
repo,
|
||||
precision=precision,
|
||||
optimize=optimize,
|
||||
)
|
||||
|
||||
_send(stdout, {"op": "progress", "stage": "loading_model", "percent": 100})
|
||||
return _runtime
|
||||
|
||||
|
||||
def _tensor_to_pcm_b64(audio, sample_rate: int) -> tuple[str, int, int]:
|
||||
"""Convert a torch waveform tensor (1, N) in [-1, 1] to base64 int16 PCM."""
|
||||
import numpy as np
|
||||
|
||||
arr = audio.detach().to("cpu").float().numpy()
|
||||
arr = np.asarray(arr, dtype=np.float32).squeeze()
|
||||
if arr.ndim > 1:
|
||||
arr = arr.mean(axis=0) # defensive downmix to mono
|
||||
arr = np.clip(arr, -1.0, 1.0)
|
||||
pcm = (arr * 32767.0).astype(np.int16).tobytes()
|
||||
return base64.b64encode(pcm).decode("ascii"), int(sample_rate), int(arr.shape[0])
|
||||
|
||||
|
||||
def _normalize_language(raw):
|
||||
"""Map OmniVoice's language value to what dots.tts accepts, or None.
|
||||
|
||||
dots.tts accepts None/"auto_detect", ISO codes upper-cased ("EN"/"ZH"),
|
||||
or names ("english"). A 2-letter ISO code is upper-cased; anything else
|
||||
is passed through; empty / "auto" → None (auto-detect)."""
|
||||
if not raw or not isinstance(raw, str):
|
||||
return None
|
||||
s = raw.strip()
|
||||
if not s or s.lower() == "auto":
|
||||
return None
|
||||
if len(s) == 2 and s.isalpha():
|
||||
return s.upper()
|
||||
return s
|
||||
|
||||
|
||||
def _handle_synthesize(msg: dict, stdout) -> None:
|
||||
"""Dispatch one synthesize request. Emits the audio frame or raises."""
|
||||
text = msg.get("text")
|
||||
if not text or not isinstance(text, str):
|
||||
raise ValueError("synthesize: missing or non-string 'text'")
|
||||
|
||||
runtime = _load_runtime(stdout)
|
||||
|
||||
gen_kwargs: dict = {
|
||||
"text": text,
|
||||
"num_steps": int(msg.get("num_steps", 10)),
|
||||
"guidance_scale": float(msg.get("guidance_scale", 1.2)),
|
||||
}
|
||||
|
||||
ref_audio = msg.get("ref_audio")
|
||||
if ref_audio:
|
||||
gen_kwargs["prompt_audio_path"] = ref_audio
|
||||
ref_text = msg.get("ref_text")
|
||||
if ref_text:
|
||||
# continuation cloning — upstream requires prompt_audio_path when
|
||||
# prompt_text is set (the parent already enforces this).
|
||||
gen_kwargs["prompt_text"] = ref_text
|
||||
|
||||
language = _normalize_language(msg.get("language"))
|
||||
if language:
|
||||
gen_kwargs["language"] = language
|
||||
|
||||
result = runtime.generate(**gen_kwargs)
|
||||
audio = result["audio"]
|
||||
sample_rate = int(result.get("sample_rate", DOTS_SAMPLE_RATE))
|
||||
|
||||
pcm_b64, sr, n_samples = _tensor_to_pcm_b64(audio, sample_rate)
|
||||
_send(stdout, {
|
||||
"op": "audio",
|
||||
"audio_pcm_b64": pcm_b64,
|
||||
"sample_rate": sr,
|
||||
"n_samples": n_samples,
|
||||
})
|
||||
|
||||
|
||||
# ── main loop ─────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def main() -> int:
|
||||
stdin = sys.stdin.buffer
|
||||
stdout = sys.stdout.buffer
|
||||
|
||||
# Ready handshake fires BEFORE any heavy import.
|
||||
_send(stdout, {
|
||||
"op": "ready",
|
||||
"engine": "dots-tts",
|
||||
"sample_rate": DOTS_SAMPLE_RATE,
|
||||
})
|
||||
|
||||
while True:
|
||||
try:
|
||||
msg = _recv(stdin)
|
||||
except Exception as exc:
|
||||
_send(stdout, {
|
||||
"op": "error",
|
||||
"stage": "recv",
|
||||
"message": f"{type(exc).__name__}: {exc}",
|
||||
"traceback": traceback.format_exc(),
|
||||
})
|
||||
return 1
|
||||
if msg is None:
|
||||
return 0
|
||||
|
||||
op = msg.get("op") if isinstance(msg, dict) else None
|
||||
try:
|
||||
if op == "ping":
|
||||
_send(stdout, {"op": "pong", "vram_mb": _measure_vram_mb()})
|
||||
elif op == "synthesize":
|
||||
_handle_synthesize(msg, stdout)
|
||||
elif op == "shutdown":
|
||||
return 0
|
||||
else:
|
||||
_send(stdout, {
|
||||
"op": "error",
|
||||
"stage": "dispatch",
|
||||
"message": f"unknown op: {op!r}",
|
||||
})
|
||||
except Exception as exc:
|
||||
_send(stdout, {
|
||||
"op": "error",
|
||||
"stage": op or "unknown",
|
||||
"message": f"{type(exc).__name__}: {exc}",
|
||||
"traceback": traceback.format_exc(),
|
||||
})
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,194 @@
|
||||
"""MOSS-TTS-v1.5 sidecar package (issue #498).
|
||||
|
||||
MOSS-TTS-v1.5 is OpenMOSS's 8B flagship TTS — a Qwen3-8B language backbone
|
||||
plus a 1.6B audio codec, 31 languages, zero-shot voice cloning, token-level
|
||||
duration control and inline ``[pause Ns]`` markers. Apache-2.0.
|
||||
|
||||
It runs in its own subprocess **and its own venv**, isolated from the
|
||||
OmniVoice parent process, for the *same* reason IndexTTS does: a hard
|
||||
``transformers`` version conflict. MOSS-TTS-v1.5's ``torch-runtime`` extra
|
||||
pins ``transformers==5.0.0`` (verified against the upstream
|
||||
``pyproject.toml``), while OmniVoice pins ``transformers>=5.3.0``. The two
|
||||
cannot share one interpreter — so MOSS lives behind ``SubprocessBackend``
|
||||
with a dedicated venv, exactly like ``engines.indextts``.
|
||||
|
||||
Three public entry points live in this package:
|
||||
|
||||
* ``MossTTSV15Backend`` (this module) — the SubprocessBackend subclass
|
||||
that ``services.tts_backend._LAZY_REGISTRY`` resolves on first access.
|
||||
Defined HERE (not in ``services.tts_backend``) to break the import
|
||||
cycle: ``services.subprocess_backend`` imports ``TTSBackend`` from
|
||||
``services.tts_backend``, so the backend class must live downstream of
|
||||
that module finishing its import. Same indirection as IndexTTS /
|
||||
Supertonic-3.
|
||||
* ``main.py`` — the sidecar entrypoint (runs under MOSS's venv with
|
||||
``transformers==5.0.0``; never imported by the parent).
|
||||
* ``bootstrap.py`` — the venv-probe + lazy-bootstrap helper.
|
||||
|
||||
Do NOT import ``main.py`` from the parent process — it runs under a
|
||||
different venv (``transformers==5.0.0``) and importing it in-process would
|
||||
re-introduce the exact conflict this isolation exists to avoid.
|
||||
|
||||
Hardware honesty (cross-platform rule): MOSS-TTS-v1.5's upstream documents
|
||||
only CUDA and CPU. There is **no documented or tested MPS path** — the
|
||||
custom ``trust_remote_code`` modelling code and the separate audio
|
||||
tokenizer are unverified on Apple Silicon. We therefore advertise
|
||||
``gpu_compat = ("cuda", "cpu")`` and the sidecar selects ``cuda`` when
|
||||
present else ``cpu`` — it never silently routes to MPS where it might
|
||||
crash. On Apple Silicon the engine honestly resolves to CPU (slow but
|
||||
correct), and the engine is opt-in regardless, so it never becomes a
|
||||
broken default on any platform.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from services.subprocess_backend import SubprocessBackend
|
||||
|
||||
if TYPE_CHECKING:
|
||||
import torch # noqa: F401
|
||||
|
||||
logger = logging.getLogger("omnivoice.moss_tts_v15")
|
||||
|
||||
#: 1 second of audio ≈ 12.5 codec tokens (MOSS-TTS-v1.5 model card). Used to
|
||||
#: translate OmniVoice's ``duration`` (seconds) into the model's ``tokens``
|
||||
#: duration-control argument.
|
||||
TOKENS_PER_SECOND: float = 12.5
|
||||
|
||||
|
||||
class MossTTSV15Backend(SubprocessBackend):
|
||||
"""MOSS-TTS-v1.5 (OpenMOSS) — 8B, 31 langs, zero-shot clone, CUDA/CPU.
|
||||
|
||||
Runs in a long-lived sidecar over length-prefixed JSON-over-stdio in a
|
||||
dedicated venv (``transformers==5.0.0``). The first synthesize cold-loads
|
||||
~16 GB of bf16 weights (CUDA) / fp32 (CPU); subsequent calls reuse the
|
||||
process and the in-memory model.
|
||||
|
||||
Installation (transparent to power users who already cloned MOSS-TTS —
|
||||
OmniVoice prefers their existing ``${DIR}/.venv``)::
|
||||
|
||||
git clone https://github.com/OpenMOSS/MOSS-TTS.git
|
||||
cd MOSS-TTS
|
||||
# CUDA host:
|
||||
uv venv && uv pip install -e ".[torch-runtime]"
|
||||
# non-CUDA host (CPU): install plain torch/transformers instead of +cu128
|
||||
|
||||
Set ``OMNIVOICE_MOSS_TTS_V15_DIR`` to the clone root. OmniVoice creates
|
||||
``backend/engines/moss_tts_v15/.venv`` lazily on first launch if no venv
|
||||
exists yet (CUDA hosts only — the upstream ``torch-runtime`` extra is
|
||||
``+cu128``); the user's existing ``${DIR}/.venv`` is preferred if
|
||||
present, so no re-install is needed.
|
||||
|
||||
License: Apache-2.0 (code + weights) — no acceptance gate needed.
|
||||
"""
|
||||
|
||||
id = "moss-tts-v15"
|
||||
display_name = (
|
||||
"MOSS-TTS-v1.5 (8B, 31 langs, zero-shot clone, CUDA/CPU, Apache-2.0)"
|
||||
)
|
||||
supports_voice_design = False # requires ref audio for timbre cloning
|
||||
_DEFAULT_SAMPLE_RATE = 24000
|
||||
# Honest hardware surface: upstream documents CUDA + CPU only. MPS is
|
||||
# undocumented / untested, so we do NOT claim it (cross-platform rule).
|
||||
gpu_compat = ("cuda", "cpu")
|
||||
|
||||
# ── availability ───────────────────────────────────────────────────────
|
||||
|
||||
@classmethod
|
||||
def is_available(cls) -> tuple[bool, str]:
|
||||
# IMPORTANT: do NOT attempt to import MOSS / its transformers==5.0.0
|
||||
# here. The parent pins transformers>=5.3 — co-importing the two in
|
||||
# one interpreter is exactly the conflict this subprocess isolation
|
||||
# exists to avoid. We only verify the venv exists on disk; a real
|
||||
# health-check (spawn + ping) is gated on the user's "Test engine"
|
||||
# action in Settings, same as IndexTTS.
|
||||
from engines.moss_tts_v15.bootstrap import (
|
||||
MOSS_TTS_V15_SIDECAR_SCRIPT,
|
||||
is_moss_tts_v15_installed,
|
||||
)
|
||||
if not is_moss_tts_v15_installed():
|
||||
return False, (
|
||||
"MOSS-TTS-v1.5 venv not found. Set OMNIVOICE_MOSS_TTS_V15_DIR "
|
||||
"to your MOSS-TTS clone (the directory containing pyproject.toml) "
|
||||
"and restart OmniVoice. CUDA or CPU only (no MPS). See "
|
||||
"docs/engines/moss-tts-v15.md for the full install walk-through."
|
||||
)
|
||||
if not MOSS_TTS_V15_SIDECAR_SCRIPT.exists():
|
||||
return False, (
|
||||
"MOSS-TTS-v1.5 sidecar script missing at "
|
||||
f"{MOSS_TTS_V15_SIDECAR_SCRIPT} — reinstall OmniVoice."
|
||||
)
|
||||
return True, "ok (CUDA when present, else CPU)"
|
||||
|
||||
@classmethod
|
||||
def venv_python(cls):
|
||||
from engines.moss_tts_v15.bootstrap import resolve_moss_tts_v15_venv
|
||||
return resolve_moss_tts_v15_venv()
|
||||
|
||||
@classmethod
|
||||
def sidecar_script(cls):
|
||||
from engines.moss_tts_v15.bootstrap import MOSS_TTS_V15_SIDECAR_SCRIPT
|
||||
return MOSS_TTS_V15_SIDECAR_SCRIPT
|
||||
|
||||
# ── TTSBackend protocol ────────────────────────────────────────────────
|
||||
|
||||
@property
|
||||
def sample_rate(self) -> int:
|
||||
return self._DEFAULT_SAMPLE_RATE
|
||||
|
||||
@property
|
||||
def supported_languages(self) -> list[str]:
|
||||
# 31 languages with multilingual handling; expose "multi" on the
|
||||
# protocol surface (same as OmniVoice / CosyVoice / Supertonic-3) and
|
||||
# translate the caller's language at synthesize time.
|
||||
return ["multi"]
|
||||
|
||||
# ── generate (parent-side arbitration) ─────────────────────────────────
|
||||
|
||||
def generate(self, text: str, **kw) -> "torch.Tensor":
|
||||
"""Synthesize one utterance through the MOSS-TTS-v1.5 sidecar.
|
||||
|
||||
kwargs honored:
|
||||
* ``ref_audio`` — path to a reference clip. When present, MOSS
|
||||
runs zero-shot voice cloning (``reference=``).
|
||||
Optional: without it the model uses its own
|
||||
default voice.
|
||||
* ``ref_text`` — accepted but unused in clone mode (MOSS's
|
||||
zero-shot path needs only the audio); kept in
|
||||
the signature so the common call-site doesn't
|
||||
need engine-specific knowledge.
|
||||
* ``language`` — ISO code or name; mapped to a MOSS language name
|
||||
in the sidecar, omitted (auto-detect) if unknown.
|
||||
* ``duration`` — target seconds → ``tokens`` (1 s ≈ 12.5 tokens).
|
||||
* ``max_new_tokens`` — generation cap (default 4096).
|
||||
|
||||
Returns a tensor of shape (1, n_samples) at :attr:`sample_rate`.
|
||||
"""
|
||||
forwarded: dict = {}
|
||||
|
||||
ref_audio = kw.get("ref_audio")
|
||||
if ref_audio:
|
||||
forwarded["ref_audio"] = ref_audio
|
||||
ref_text = kw.get("ref_text")
|
||||
if ref_text:
|
||||
forwarded["ref_text"] = ref_text
|
||||
|
||||
language = kw.get("language")
|
||||
if language:
|
||||
forwarded["language"] = str(language)
|
||||
|
||||
duration = kw.get("duration")
|
||||
if duration is not None:
|
||||
target_tokens = int(float(duration) * TOKENS_PER_SECOND)
|
||||
if target_tokens > 0:
|
||||
forwarded["tokens"] = target_tokens
|
||||
|
||||
max_new_tokens = kw.get("max_new_tokens")
|
||||
if max_new_tokens is not None:
|
||||
forwarded["max_new_tokens"] = int(max_new_tokens)
|
||||
|
||||
return super().generate(text, **forwarded)
|
||||
|
||||
|
||||
__all__ = ["MossTTSV15Backend", "TOKENS_PER_SECOND"]
|
||||
@@ -0,0 +1,272 @@
|
||||
"""MOSS-TTS-v1.5 venv probe + lazy bootstrap (issue #498).
|
||||
|
||||
The parent process needs to know *which Python interpreter* to spawn the
|
||||
MOSS-TTS-v1.5 sidecar under. This module owns that resolution. It mirrors
|
||||
``engines.indextts.bootstrap`` because MOSS has the same shape of problem:
|
||||
a hard ``transformers`` pin (``==5.0.0``) that conflicts with the parent's
|
||||
``transformers>=5.3`` — so MOSS runs in its own venv.
|
||||
|
||||
Probe order (priority — existing power-user installs win, zero migration):
|
||||
|
||||
1. ``${OMNIVOICE_MOSS_TTS_V15_DIR}/.venv/`` — the user's clone-level
|
||||
venv. Highest priority: a user who already cloned MOSS-TTS and ran
|
||||
``uv pip install -e ".[torch-runtime]"`` (per upstream docs) gets
|
||||
reused verbatim, no re-download of the ~16 GB model.
|
||||
2. ``backend/engines/moss_tts_v15/.venv/`` — this package's own venv,
|
||||
created by step 3 if needed.
|
||||
3. Bootstrap: ``uv venv`` then ``uv pip install -e
|
||||
"${DIR}[torch-runtime]"``. Requires ``OMNIVOICE_MOSS_TTS_V15_DIR``.
|
||||
The upstream ``torch-runtime`` extra is CUDA (``+cu128``), so the
|
||||
auto-bootstrap targets CUDA hosts; non-CUDA (CPU/Mac) users set up
|
||||
their own venv per docs/engines/moss-tts-v15.md (Probe 1).
|
||||
|
||||
Caching: resolution is memoised after the first successful call. Tests
|
||||
reset via :func:`invalidate`.
|
||||
|
||||
Security: bootstrap never touches HF_TOKEN; the sidecar's stderr is drained
|
||||
by SubprocessBackend through the parent root logger where Phase 1's
|
||||
``HFTokenRedactor`` strips token bytes. ``uv pip install -e`` installs from
|
||||
a user-controlled clone the user already trusts (same posture as IndexTTS).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
logger = logging.getLogger("omnivoice.moss_tts_v15.bootstrap")
|
||||
|
||||
#: Absolute path to the sidecar entrypoint. ``MossTTSV15Backend.sidecar_script``
|
||||
#: returns this; SubprocessBackend spawns it with the resolved venv python.
|
||||
MOSS_TTS_V15_SIDECAR_SCRIPT: Path = Path(__file__).parent / "main.py"
|
||||
|
||||
#: This package's owned venv (Probe 2). The MOSS-TTS clone, when bootstrapped,
|
||||
#: is installed into this venv via ``uv pip install -e``.
|
||||
_ENGINES_VENV_DIR: Path = Path(__file__).parent / ".venv"
|
||||
|
||||
#: Env var pointing at the user's MOSS-TTS clone root.
|
||||
_CLONE_DIR_ENV: str = "OMNIVOICE_MOSS_TTS_V15_DIR"
|
||||
|
||||
#: Per-process resolution cache. Cleared by :func:`invalidate` for tests.
|
||||
_resolved_python: Optional[Path] = None
|
||||
|
||||
# Timeouts — bounded so a wedged venv never hangs the parent. The bootstrap
|
||||
# install can take many minutes on a cold cache (MOSS pulls a CUDA torch
|
||||
# build + transformers + an audio codec stack).
|
||||
_IMPORT_PROBE_TIMEOUT_S = 15
|
||||
_UV_VENV_TIMEOUT_S = 120
|
||||
_UV_PIP_INSTALL_TIMEOUT_S = 1800
|
||||
|
||||
|
||||
# ── public API ────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def invalidate() -> None:
|
||||
"""Clear the resolved-python cache. Tests call this between scenarios."""
|
||||
global _resolved_python
|
||||
_resolved_python = None
|
||||
|
||||
|
||||
def is_moss_tts_v15_installed() -> bool:
|
||||
"""Cheap file-existence check for a usable MOSS-TTS-v1.5 venv.
|
||||
|
||||
Returns True if either Probe 1 or Probe 2 has a Python executable on
|
||||
disk. Does NOT spawn the venv Python — that's saved for
|
||||
:func:`resolve_moss_tts_v15_venv`, which is only invoked on the first
|
||||
generate() / health_check(). This fires on every Settings render via
|
||||
``MossTTSV15Backend.is_available()``, so it stays cheap.
|
||||
"""
|
||||
for cand in _probe_paths():
|
||||
if cand.is_file():
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def resolve_moss_tts_v15_venv() -> Path:
|
||||
"""Resolve the path to the Python interpreter that runs the sidecar.
|
||||
|
||||
Probe order described in the module docstring. Memoised. Raises
|
||||
:exc:`RuntimeError` if no working venv can be located AND the bootstrap
|
||||
path is unavailable.
|
||||
"""
|
||||
global _resolved_python
|
||||
if _resolved_python is not None:
|
||||
return _resolved_python
|
||||
|
||||
clone_dir = os.environ.get(_CLONE_DIR_ENV)
|
||||
|
||||
# Probe 1 — user's clone-level venv (highest priority for back-compat).
|
||||
if clone_dir:
|
||||
cand = _venv_python_path(Path(clone_dir) / ".venv")
|
||||
if cand.is_file() and _venv_can_import_moss(cand):
|
||||
logger.info(
|
||||
"MOSS-TTS-v1.5 venv resolved from %s: %s", _CLONE_DIR_ENV, cand,
|
||||
)
|
||||
_resolved_python = cand
|
||||
return cand
|
||||
|
||||
# Probe 2 — this package's own venv.
|
||||
cand = _venv_python_path(_ENGINES_VENV_DIR)
|
||||
if cand.is_file() and _venv_can_import_moss(cand):
|
||||
logger.info("MOSS-TTS-v1.5 venv resolved from engines path: %s", cand)
|
||||
_resolved_python = cand
|
||||
return cand
|
||||
|
||||
# Probe 3 — bootstrap.
|
||||
if not clone_dir:
|
||||
raise RuntimeError(
|
||||
"MOSS-TTS-v1.5 is not installed. Set the "
|
||||
f"{_CLONE_DIR_ENV} environment variable to your MOSS-TTS clone "
|
||||
"(the directory that contains pyproject.toml), then restart "
|
||||
"OmniVoice. See docs/engines/moss-tts-v15.md for the full "
|
||||
"install walk-through."
|
||||
)
|
||||
|
||||
cand = _bootstrap_engines_venv(Path(clone_dir))
|
||||
_resolved_python = cand
|
||||
return cand
|
||||
|
||||
|
||||
# ── internals ─────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def _venv_python_path(venv_dir: Path) -> Path:
|
||||
"""Return the python executable path inside a venv directory.
|
||||
|
||||
Handles the Unix (``bin/python``) vs Windows (``Scripts/python.exe``)
|
||||
layout. No filesystem access — caller checks .is_file().
|
||||
"""
|
||||
if sys.platform == "win32":
|
||||
return venv_dir / "Scripts" / "python.exe"
|
||||
return venv_dir / "bin" / "python"
|
||||
|
||||
|
||||
def _probe_paths() -> list[Path]:
|
||||
"""Ordered list of candidate venv-python paths (no .is_file() check)."""
|
||||
out: list[Path] = []
|
||||
clone_dir = os.environ.get(_CLONE_DIR_ENV)
|
||||
if clone_dir:
|
||||
out.append(_venv_python_path(Path(clone_dir) / ".venv"))
|
||||
out.append(_venv_python_path(_ENGINES_VENV_DIR))
|
||||
return out
|
||||
|
||||
|
||||
def _venv_can_import_moss(python_path: Path) -> bool:
|
||||
"""Spawn the candidate python and verify the MOSS stack imports.
|
||||
|
||||
MOSS-TTS-v1.5 loads via ``transformers`` + ``trust_remote_code`` (no
|
||||
fixed top-level package to import), so the readiness signal is that the
|
||||
venv has a working ``transformers`` + ``torch`` — which only the
|
||||
``[torch-runtime]`` install provides. Bounded by
|
||||
``_IMPORT_PROBE_TIMEOUT_S`` so a wedged venv never hangs the parent.
|
||||
Returns False on any failure (non-zero exit, timeout, OSError).
|
||||
"""
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
[str(python_path), "-c", "import transformers, torch"],
|
||||
capture_output=True,
|
||||
timeout=_IMPORT_PROBE_TIMEOUT_S,
|
||||
)
|
||||
except (subprocess.TimeoutExpired, OSError) as exc:
|
||||
logger.debug("moss-tts-v15 import probe failed for %s: %s", python_path, exc)
|
||||
return False
|
||||
if proc.returncode != 0:
|
||||
logger.debug(
|
||||
"moss-tts-v15 import probe non-zero for %s: %s",
|
||||
python_path,
|
||||
proc.stderr.decode("utf-8", errors="replace")[:200],
|
||||
)
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def _locate_uv() -> Optional[str]:
|
||||
"""Find the uv binary — bundled first (Tauri-set env var), else PATH."""
|
||||
bundled = os.environ.get("OMNIVOICE_BUNDLED_UV")
|
||||
if bundled and Path(bundled).is_file():
|
||||
return bundled
|
||||
sys_uv = shutil.which("uv")
|
||||
if sys_uv:
|
||||
return sys_uv
|
||||
return None
|
||||
|
||||
|
||||
def _bootstrap_engines_venv(clone_dir: Path) -> Path:
|
||||
"""Create engines/moss_tts_v15/.venv and install the user's clone into it.
|
||||
|
||||
Runs ``uv venv <engines_venv>`` then ``uv pip install --python
|
||||
<engines_venv>/bin/python -e "<clone>[torch-runtime]"``. Verifies the
|
||||
result by re-probing the import — a successful uv invocation that still
|
||||
can't import the stack indicates a deeper environment problem (e.g. the
|
||||
``+cu128`` torch-runtime extra can't resolve on a non-CUDA host) and we
|
||||
raise with whatever stderr we captured plus a docs pointer.
|
||||
"""
|
||||
uv = _locate_uv()
|
||||
if not uv:
|
||||
raise RuntimeError(
|
||||
"uv is required to bootstrap the MOSS-TTS-v1.5 venv but was not "
|
||||
"found on PATH (and OMNIVOICE_BUNDLED_UV was not set). Install uv "
|
||||
"from https://docs.astral.sh/uv/ and re-launch OmniVoice, or set "
|
||||
"OMNIVOICE_BUNDLED_UV to the absolute path of a uv binary."
|
||||
)
|
||||
|
||||
logger.info(
|
||||
"Bootstrapping MOSS-TTS-v1.5 venv at %s from %s (this can take "
|
||||
"several minutes on first launch)", _ENGINES_VENV_DIR, clone_dir,
|
||||
)
|
||||
|
||||
try:
|
||||
subprocess.run(
|
||||
[uv, "venv", str(_ENGINES_VENV_DIR)],
|
||||
check=True,
|
||||
timeout=_UV_VENV_TIMEOUT_S,
|
||||
capture_output=True,
|
||||
)
|
||||
except subprocess.CalledProcessError as exc:
|
||||
raise RuntimeError(
|
||||
f"uv venv failed for MOSS-TTS-v1.5 bootstrap at {_ENGINES_VENV_DIR}: "
|
||||
f"{exc.stderr.decode('utf-8', errors='replace') if exc.stderr else exc}"
|
||||
) from exc
|
||||
|
||||
python_path = _venv_python_path(_ENGINES_VENV_DIR)
|
||||
try:
|
||||
subprocess.run(
|
||||
[
|
||||
uv, "pip", "install",
|
||||
"--python", str(python_path),
|
||||
"-e", f"{clone_dir}[torch-runtime]",
|
||||
],
|
||||
check=True,
|
||||
timeout=_UV_PIP_INSTALL_TIMEOUT_S,
|
||||
capture_output=True,
|
||||
)
|
||||
except subprocess.CalledProcessError as exc:
|
||||
raise RuntimeError(
|
||||
"uv pip install -e failed during MOSS-TTS-v1.5 bootstrap "
|
||||
f"({clone_dir}). On a non-CUDA host the upstream '[torch-runtime]' "
|
||||
"extra (cu128) cannot resolve — set up the venv manually per "
|
||||
"docs/engines/moss-tts-v15.md. Error: "
|
||||
f"{exc.stderr.decode('utf-8', errors='replace') if exc.stderr else exc}"
|
||||
) from exc
|
||||
|
||||
if not _venv_can_import_moss(python_path):
|
||||
raise RuntimeError(
|
||||
"MOSS-TTS-v1.5 bootstrap completed but the transformers/torch "
|
||||
f"import still fails from {python_path}. Verify that {clone_dir} "
|
||||
"is a valid MOSS-TTS clone. See docs/engines/moss-tts-v15.md."
|
||||
)
|
||||
|
||||
logger.info("MOSS-TTS-v1.5 venv bootstrap successful: %s", python_path)
|
||||
return python_path
|
||||
|
||||
|
||||
__all__ = [
|
||||
"MOSS_TTS_V15_SIDECAR_SCRIPT",
|
||||
"invalidate",
|
||||
"is_moss_tts_v15_installed",
|
||||
"resolve_moss_tts_v15_venv",
|
||||
]
|
||||
@@ -0,0 +1,303 @@
|
||||
"""MOSS-TTS-v1.5 sidecar entry point (issue #498).
|
||||
|
||||
Runs inside ``engines/moss_tts_v15/.venv`` (or the user's existing
|
||||
``${OMNIVOICE_MOSS_TTS_V15_DIR}/.venv``) with ``transformers==5.0.0``,
|
||||
isolated from the OmniVoice parent process which pins ``transformers>=5.3``.
|
||||
Same isolation rationale as the IndexTTS sidecar.
|
||||
|
||||
Stdlib-only at import time. The model + transformers + torch are imported
|
||||
lazily on the first synthesize op so the sidecar emits its ``ready`` frame
|
||||
inside the parent's 30 s spawn handshake even on a cold filesystem (an 8B
|
||||
model takes well over 30 s to cold-load).
|
||||
|
||||
Wire protocol — length-prefixed JSON over stdin/stdout, byte-identical to
|
||||
``backend/services/subprocess_backend.py``::
|
||||
|
||||
[ 4-byte big-endian uint32 length ][ N bytes UTF-8 JSON ]
|
||||
|
||||
Op flow:
|
||||
1. Sidecar -> parent: {"op": "ready", "engine": "moss-tts-v15",
|
||||
"sample_rate": 24000}
|
||||
2. parent -> sidecar: {"op": "ping"} -> {"op": "pong", "vram_mb": N}
|
||||
3. parent -> sidecar: {"op": "synthesize", "text": "...",
|
||||
"ref_audio": "/path/spk.wav", "language": "fr",
|
||||
"tokens": 325, "max_new_tokens": 4096}
|
||||
-> {"op": "progress", ...} (cold load only) then
|
||||
-> {"op": "audio", "audio_pcm_b64": "...", "sample_rate": 24000,
|
||||
"n_samples": N}
|
||||
4. parent -> sidecar: {"op": "shutdown"} -> exit 0
|
||||
|
||||
Restrictions: NO imports from OmniVoice parent code (different venv). NO
|
||||
logging of ``os.environ`` contents. Single-frame DoS cap matches the
|
||||
parent's ``MAX_FRAME_BYTES``.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import json
|
||||
import os
|
||||
import struct
|
||||
import sys
|
||||
import traceback
|
||||
|
||||
|
||||
# Mirrors backend/services/subprocess_backend.py::MAX_FRAME_BYTES.
|
||||
MAX_FRAME_BYTES = 64 * 1024 * 1024
|
||||
|
||||
#: Native sample rate MOSS-TTS-v1.5 emits. Advertised in the ready frame so
|
||||
#: the parent doesn't have to import MOSS just to learn the rate. Confirmed
|
||||
#: via ``processor.model_config.sampling_rate`` (the real value is read from
|
||||
#: the loaded model at synthesize time; this is the handshake default).
|
||||
MOSS_SAMPLE_RATE = 24000
|
||||
|
||||
#: HF repo id for the weights, overridable for air-gapped / mirror installs.
|
||||
_DEFAULT_REPO = "OpenMOSS-Team/MOSS-TTS-v1.5"
|
||||
|
||||
#: ISO-639-1 → MOSS language name. MOSS's ``build_user_message`` takes a
|
||||
#: language *name* ("French"), not a code. Unknown codes are omitted so the
|
||||
#: model auto-detects. Covers the high-traffic subset of MOSS's 31 langs.
|
||||
_ISO_TO_NAME = {
|
||||
"en": "English", "zh": "Chinese", "ja": "Japanese", "ko": "Korean",
|
||||
"fr": "French", "de": "German", "es": "Spanish", "it": "Italian",
|
||||
"pt": "Portuguese", "ru": "Russian", "ar": "Arabic", "hi": "Hindi",
|
||||
"nl": "Dutch", "pl": "Polish", "tr": "Turkish", "vi": "Vietnamese",
|
||||
"th": "Thai", "id": "Indonesian", "cs": "Czech", "el": "Greek",
|
||||
"he": "Hebrew", "fa": "Persian", "uk": "Ukrainian", "sv": "Swedish",
|
||||
}
|
||||
|
||||
|
||||
# ── wire protocol ─────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def _send(stream, obj: dict) -> None:
|
||||
body = json.dumps(obj, separators=(",", ":")).encode("utf-8")
|
||||
stream.write(struct.pack("!I", len(body)))
|
||||
stream.write(body)
|
||||
stream.flush()
|
||||
|
||||
|
||||
def _recv(stream):
|
||||
header = stream.read(4)
|
||||
if len(header) < 4:
|
||||
return None # EOF
|
||||
(n,) = struct.unpack("!I", header)
|
||||
if n > MAX_FRAME_BYTES:
|
||||
raise IOError(f"frame too large: {n}")
|
||||
body = bytearray()
|
||||
while len(body) < n:
|
||||
chunk = stream.read(n - len(body))
|
||||
if not chunk:
|
||||
raise IOError("short read")
|
||||
body.extend(chunk)
|
||||
return json.loads(bytes(body).decode("utf-8"))
|
||||
|
||||
|
||||
def _measure_vram_mb() -> float:
|
||||
"""This sidecar's own GPU memory in MB (MM2-08). The parent can't see a
|
||||
child's VRAM, so we self-report it in the pong. 0 on CPU. Never raises."""
|
||||
try:
|
||||
import torch
|
||||
if torch.cuda.is_available():
|
||||
return round(torch.cuda.memory_allocated() / (1024 ** 2), 1)
|
||||
except Exception:
|
||||
pass
|
||||
return 0.0
|
||||
|
||||
|
||||
# ── model loading (lazy, on first synthesize) ─────────────────────────────
|
||||
|
||||
|
||||
# Module-level singleton — populated on the first synthesize op and reused.
|
||||
# Holds (processor, model, device, sample_rate).
|
||||
_state = None
|
||||
|
||||
|
||||
def _load_model(stdout):
|
||||
"""Cold-construct the MOSS-TTS-v1.5 processor + model.
|
||||
|
||||
Device selection is CUDA-or-CPU only — MOSS's upstream documents no MPS
|
||||
path and the custom ``trust_remote_code`` modelling code is untested on
|
||||
Apple Silicon, so we never route to MPS where it might crash. dtype is
|
||||
bf16 on CUDA, fp32 on CPU (bf16 CPU ops are spotty). Emits progress
|
||||
frames so the parent can surface the multi-GB cold-load latency.
|
||||
"""
|
||||
global _state
|
||||
if _state is not None:
|
||||
return _state
|
||||
|
||||
_send(stdout, {"op": "progress", "stage": "loading_model", "percent": 0})
|
||||
|
||||
import torch
|
||||
from transformers import AutoModel, AutoProcessor
|
||||
|
||||
repo = os.environ.get("OMNIVOICE_MOSS_TTS_V15_MODEL", _DEFAULT_REPO)
|
||||
device = "cuda" if torch.cuda.is_available() else "cpu"
|
||||
dtype = torch.bfloat16 if device == "cuda" else torch.float32
|
||||
# "sdpa" works on CUDA + CPU and needs no extra dep. flash_attention_2
|
||||
# (Ampere+ CUDA, optional flash-attn) is opt-in via env.
|
||||
attn = os.environ.get("OMNIVOICE_MOSS_TTS_V15_ATTN", "sdpa")
|
||||
|
||||
processor = AutoProcessor.from_pretrained(repo, trust_remote_code=True)
|
||||
# The audio tokenizer is a separate sub-module that must be moved to the
|
||||
# device independently (easy to miss — see upstream README).
|
||||
processor.audio_tokenizer = processor.audio_tokenizer.to(device)
|
||||
|
||||
_send(stdout, {"op": "progress", "stage": "loading_model", "percent": 50})
|
||||
|
||||
model = AutoModel.from_pretrained(
|
||||
repo,
|
||||
trust_remote_code=True,
|
||||
attn_implementation=attn,
|
||||
torch_dtype=dtype,
|
||||
).to(device)
|
||||
model.eval()
|
||||
|
||||
sample_rate = int(getattr(processor.model_config, "sampling_rate", MOSS_SAMPLE_RATE))
|
||||
_state = (processor, model, device, sample_rate)
|
||||
|
||||
_send(stdout, {"op": "progress", "stage": "loading_model", "percent": 100})
|
||||
return _state
|
||||
|
||||
|
||||
def _tensor_to_pcm_b64(audio, sample_rate: int) -> tuple[str, int, int]:
|
||||
"""Convert a torch waveform tensor to base64 int16 PCM.
|
||||
|
||||
MOSS returns a float tensor in [-1, 1] (1-D or (1, N)); we squeeze to
|
||||
mono, clip, scale to int16, and base64 so the wire frame stays JSON-safe.
|
||||
"""
|
||||
import numpy as np
|
||||
|
||||
arr = audio.detach().to("cpu").float().numpy()
|
||||
arr = np.asarray(arr, dtype=np.float32).squeeze()
|
||||
if arr.ndim > 1:
|
||||
arr = arr.mean(axis=0) # defensive downmix to mono
|
||||
arr = np.clip(arr, -1.0, 1.0)
|
||||
pcm = (arr * 32767.0).astype(np.int16).tobytes()
|
||||
return base64.b64encode(pcm).decode("ascii"), int(sample_rate), int(arr.shape[0])
|
||||
|
||||
|
||||
def _resolve_language(raw):
|
||||
"""Map OmniVoice's language value to a MOSS language name, or None.
|
||||
|
||||
Accepts an ISO-639-1 code or a full name. Unknown / empty / "auto"
|
||||
values return None so MOSS auto-detects."""
|
||||
if not raw or not isinstance(raw, str):
|
||||
return None
|
||||
s = raw.strip()
|
||||
if not s or s.lower() == "auto":
|
||||
return None
|
||||
if s.lower() in _ISO_TO_NAME:
|
||||
return _ISO_TO_NAME[s.lower()]
|
||||
# Already a language name (or an unknown code) — pass it through; MOSS
|
||||
# ignores a language it doesn't recognise.
|
||||
return s
|
||||
|
||||
|
||||
def _handle_synthesize(msg: dict, stdout) -> None:
|
||||
"""Dispatch one synthesize request. Emits the audio frame or raises."""
|
||||
import torch
|
||||
|
||||
text = msg.get("text")
|
||||
if not text or not isinstance(text, str):
|
||||
raise ValueError("synthesize: missing or non-string 'text'")
|
||||
|
||||
processor, model, device, sample_rate = _load_model(stdout)
|
||||
|
||||
user_kwargs: dict = {"text": text}
|
||||
|
||||
ref_audio = msg.get("ref_audio")
|
||||
if ref_audio:
|
||||
# Zero-shot voice cloning: the reference audio alone is enough in
|
||||
# MOSS's clone mode (ref_text is not consumed here). The processor's
|
||||
# audio tokenizer encodes the reference into the prompt.
|
||||
user_kwargs["reference"] = [ref_audio]
|
||||
|
||||
language = _resolve_language(msg.get("language"))
|
||||
if language:
|
||||
user_kwargs["language"] = language
|
||||
|
||||
tokens = msg.get("tokens")
|
||||
if tokens is not None:
|
||||
user_kwargs["tokens"] = int(tokens)
|
||||
|
||||
max_new_tokens = int(msg.get("max_new_tokens", 4096))
|
||||
|
||||
conversations = [[processor.build_user_message(**user_kwargs)]]
|
||||
|
||||
with torch.no_grad():
|
||||
batch = processor(conversations, mode="generation")
|
||||
outputs = model.generate(
|
||||
input_ids=batch["input_ids"].to(device),
|
||||
attention_mask=batch["attention_mask"].to(device),
|
||||
max_new_tokens=max_new_tokens,
|
||||
)
|
||||
|
||||
decoded = processor.decode(outputs)
|
||||
audio = decoded[0].audio_codes_list[0]
|
||||
|
||||
pcm_b64, sr, n_samples = _tensor_to_pcm_b64(audio, sample_rate)
|
||||
_send(stdout, {
|
||||
"op": "audio",
|
||||
"audio_pcm_b64": pcm_b64,
|
||||
"sample_rate": sr,
|
||||
"n_samples": n_samples,
|
||||
})
|
||||
|
||||
|
||||
# ── main loop ─────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def main() -> int:
|
||||
stdin = sys.stdin.buffer
|
||||
stdout = sys.stdout.buffer
|
||||
|
||||
# Ready handshake fires BEFORE any heavy import — nothing above this line
|
||||
# touches transformers/torch, so we make the 30 s spawn window even cold.
|
||||
_send(stdout, {
|
||||
"op": "ready",
|
||||
"engine": "moss-tts-v15",
|
||||
"sample_rate": MOSS_SAMPLE_RATE,
|
||||
})
|
||||
|
||||
while True:
|
||||
try:
|
||||
msg = _recv(stdin)
|
||||
except Exception as exc:
|
||||
_send(stdout, {
|
||||
"op": "error",
|
||||
"stage": "recv",
|
||||
"message": f"{type(exc).__name__}: {exc}",
|
||||
"traceback": traceback.format_exc(),
|
||||
})
|
||||
return 1
|
||||
if msg is None:
|
||||
return 0
|
||||
|
||||
op = msg.get("op") if isinstance(msg, dict) else None
|
||||
try:
|
||||
if op == "ping":
|
||||
_send(stdout, {"op": "pong", "vram_mb": _measure_vram_mb()})
|
||||
elif op == "synthesize":
|
||||
_handle_synthesize(msg, stdout)
|
||||
elif op == "shutdown":
|
||||
return 0
|
||||
else:
|
||||
_send(stdout, {
|
||||
"op": "error",
|
||||
"stage": "dispatch",
|
||||
"message": f"unknown op: {op!r}",
|
||||
})
|
||||
except Exception as exc:
|
||||
# Per-op failure is recoverable — emit the error frame and stay
|
||||
# alive so the parent can retry without paying the respawn +
|
||||
# multi-GB model-load cost again.
|
||||
_send(stdout, {
|
||||
"op": "error",
|
||||
"stage": op or "unknown",
|
||||
"message": f"{type(exc).__name__}: {exc}",
|
||||
"traceback": traceback.format_exc(),
|
||||
})
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
+139
-15
@@ -9,6 +9,16 @@ _backend_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
if _backend_dir not in sys.path:
|
||||
sys.path.insert(0, _backend_dir)
|
||||
|
||||
# #564: also make the project's OWN `omnivoice` package importable from source
|
||||
# when the venv's editable install is missing/broken (interrupted/offline
|
||||
# `uv sync`, antivirus-quarantined `_editable_impl_omnivoice.pth`, …). Without
|
||||
# this the backend boots fine and only fails at the first model call with
|
||||
# `No module named 'omnivoice'`. The bootstrap now gates on omnivoice being
|
||||
# importable too (re-syncing to re-lay the editable install); this is the
|
||||
# runtime safety net. See core/omnivoice_path.py for the full rationale.
|
||||
from core.omnivoice_path import ensure_omnivoice_importable
|
||||
ensure_omnivoice_importable(_backend_dir)
|
||||
|
||||
# Triton is unavailable on Windows — disable torch.compile / dynamo / inductor
|
||||
# to prevent TritonMissing errors at inference time. Must be set before torch
|
||||
# is imported (it is lazily imported in services/model_manager.py). Uses
|
||||
@@ -327,6 +337,7 @@ from api.routers import (
|
||||
events,
|
||||
capture,
|
||||
capture_ws,
|
||||
dictation,
|
||||
openai_compat,
|
||||
tts_stream,
|
||||
marketplace,
|
||||
@@ -334,6 +345,7 @@ from api.routers import (
|
||||
sonitranslate,
|
||||
audiobook,
|
||||
longform_jobs,
|
||||
pronunciation, # Expressive-TTS Spec 01: user pronunciation dictionary
|
||||
settings as settings_router, # Phase 1 AUTH-03: HF token save/clear/state
|
||||
)
|
||||
from utils import hf_progress
|
||||
@@ -375,8 +387,90 @@ def _env_flag(name: str, default: bool = False) -> bool:
|
||||
return value.strip().lower() in {"1", "true", "yes", "on"}
|
||||
|
||||
|
||||
def _mcp_start_timeout_s() -> float:
|
||||
"""Seconds to wait for the MCP session manager to start before giving up
|
||||
and serving without it (#632). Overridable via OMNIVOICE_MCP_START_TIMEOUT_S."""
|
||||
raw = os.environ.get("OMNIVOICE_MCP_START_TIMEOUT_S", "")
|
||||
try:
|
||||
v = float(raw)
|
||||
if v > 0:
|
||||
return v
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
return 30.0
|
||||
|
||||
|
||||
async def _serve_mcp(session_manager, ready: "asyncio.Event", stop: "asyncio.Event") -> None:
|
||||
"""Own the MCP session manager's full enter→exit lifecycle in ONE task.
|
||||
|
||||
FastMCP's ``run()`` opens an anyio task group, and anyio requires the cancel
|
||||
scope to be exited in the *same task* that entered it. So we must NOT enter
|
||||
it via ``wait_for`` (which runs the enter in a throwaway sub-task) or on the
|
||||
lifespan task and exit it elsewhere — either raises "Attempted to exit cancel
|
||||
scope in a different task". This coroutine enters and exits the context
|
||||
itself: it signals ``ready`` once mounted, then idles until ``stop``.
|
||||
"""
|
||||
try:
|
||||
async with session_manager.run():
|
||||
ready.set()
|
||||
await stop.wait()
|
||||
except Exception as e:
|
||||
logger.warning("MCP session manager stopped: %s", e)
|
||||
finally:
|
||||
ready.set() # never leave startup blocked on the readiness wait
|
||||
|
||||
|
||||
async def _start_mcp_session_manager(session_manager, *, timeout: float):
|
||||
"""Start MCP off the startup critical path; wait up to ``timeout`` for it to
|
||||
signal ready. Returns ``(task, stop_event, mounted)``.
|
||||
|
||||
The MCP layer is best-effort and must never wedge backend startup. On some
|
||||
platforms (observed: Apple-Silicon M1, #632) ``run()`` can *hang* on its
|
||||
anyio task group; the old code awaited the enter before serving, so the hang
|
||||
meant "Application startup complete" never fired and the whole backend was
|
||||
unreachable with no error. Now the enter lives in its own task and we only
|
||||
*optionally* wait on a ready signal — a hang becomes a logged warning + a
|
||||
backend that serves normally without MCP.
|
||||
"""
|
||||
stop = asyncio.Event()
|
||||
if session_manager is None:
|
||||
return None, stop, False
|
||||
ready = asyncio.Event()
|
||||
task = asyncio.create_task(_serve_mcp(session_manager, ready, stop))
|
||||
try:
|
||||
await asyncio.wait_for(ready.wait(), timeout=timeout)
|
||||
mounted = not task.done() # ready is also set on failure → not mounted
|
||||
except asyncio.TimeoutError:
|
||||
logger.warning(
|
||||
"MCP session manager did not signal ready within %.0fs (#632); "
|
||||
"serving without waiting. Set OMNIVOICE_MCP_START_TIMEOUT_S to adjust.",
|
||||
timeout,
|
||||
)
|
||||
mounted = False
|
||||
return task, stop, mounted
|
||||
|
||||
|
||||
@asynccontextmanager
|
||||
async def lifespan(app: FastAPI):
|
||||
# Startup watchdog (#632): a silent hang during startup (e.g. a model-load /
|
||||
# MCP deadlock on some platforms) means "Application startup complete" never
|
||||
# logs and the app sits forever with no error. If startup hasn't finished
|
||||
# within the window, dump every thread's stack to stderr (→ backend_err.log)
|
||||
# so the hang point is captured instead of invisible. Cancelled the instant
|
||||
# startup completes, so a normal (even slow-download) boot never trips it.
|
||||
# Tune with OMNIVOICE_STARTUP_WATCHDOG_S (seconds; 0 disables). Best-effort —
|
||||
# never let the diagnostic itself break startup.
|
||||
_watchdog_armed = False
|
||||
try:
|
||||
import faulthandler
|
||||
_wd = float(os.environ.get("OMNIVOICE_STARTUP_WATCHDOG_S", "300"))
|
||||
if _wd > 0 and hasattr(faulthandler, "dump_traceback_later"):
|
||||
faulthandler.dump_traceback_later(_wd, repeat=False, exit=False)
|
||||
_watchdog_armed = True
|
||||
logger.info("Startup watchdog armed: thread dump if startup exceeds %.0fs (#632).", _wd)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
init_db()
|
||||
# Network sharing is loopback-only by default; the PIN middleware stays
|
||||
# inert until enable() sets a PIN. Seed the (disabled) state so the
|
||||
@@ -459,23 +553,37 @@ async def lifespan(app: FastAPI):
|
||||
logger.info("Capture ASR preload disabled; dictation ASR will load on first use.")
|
||||
|
||||
# ── MCP session manager (Wave 2.2) ────────────────────────────────────
|
||||
# FastMCP's Streamable-HTTP transport needs its session manager running
|
||||
# for the lifetime of the app. It's created lazily by streamable_http_app()
|
||||
# (called in mount_mcp below), so we stack its `run()` context into ours
|
||||
# via AsyncExitStack rather than replacing this lifespan. Best-effort: a
|
||||
# missing/broken MCP layer must never stop the rest of the backend.
|
||||
from contextlib import AsyncExitStack
|
||||
async with AsyncExitStack() as _mcp_stack:
|
||||
_sm = getattr(app.state, "mcp_session_manager", None)
|
||||
if _sm is not None:
|
||||
try:
|
||||
await _mcp_stack.enter_async_context(_sm.run())
|
||||
logger.info("MCP server mounted at /mcp")
|
||||
except Exception as e:
|
||||
logger.warning("MCP session manager failed to start: %s", e)
|
||||
yield
|
||||
# FastMCP's Streamable-HTTP transport needs its session manager running for
|
||||
# the lifetime of the app. Run it in its OWN task that owns the full
|
||||
# enter→exit lifecycle (anyio task-affinity, see _serve_mcp) and only wait,
|
||||
# with a timeout, for it to signal ready — so a hang on its anyio group
|
||||
# (observed on M1, #632) can never wedge "Application startup complete".
|
||||
_sm = getattr(app.state, "mcp_session_manager", None)
|
||||
mcp_task, mcp_stop, mcp_mounted = await _start_mcp_session_manager(
|
||||
_sm, timeout=_mcp_start_timeout_s()
|
||||
)
|
||||
if mcp_mounted:
|
||||
logger.info("MCP server mounted at /mcp")
|
||||
# Startup finished — disarm the hang watchdog before serving (#632).
|
||||
if _watchdog_armed:
|
||||
try:
|
||||
import faulthandler
|
||||
faulthandler.cancel_dump_traceback_later()
|
||||
except Exception:
|
||||
pass
|
||||
yield
|
||||
# ── Graceful shutdown (SIGTERM from Tauri, Ctrl+C, etc.) ────────────
|
||||
logger.info("Shutdown: cleaning up…")
|
||||
# Stop MCP first — signal its task to exit its own anyio context (correct
|
||||
# task-affinity), then bound the wait so a wedged manager can't hang exit.
|
||||
mcp_stop.set()
|
||||
if mcp_task is not None:
|
||||
try:
|
||||
await asyncio.wait_for(mcp_task, timeout=5.0)
|
||||
except (asyncio.TimeoutError, asyncio.CancelledError):
|
||||
pass
|
||||
except Exception:
|
||||
pass
|
||||
idle_task.cancel()
|
||||
worker_task.cancel()
|
||||
# Wait for tasks to finish their current iteration
|
||||
@@ -740,6 +848,20 @@ app.add_middleware(NetworkAccessMiddleware)
|
||||
# keyed non-loopback client must reach them.
|
||||
app.add_middleware(BearerKeyMiddleware)
|
||||
|
||||
# Register canonical audio MIME types before any StaticFiles mount.
|
||||
# Python's `mimetypes.guess_type()` returns `audio/x-wav` for `.wav` and
|
||||
# `audio/x-flac` for `.flac` on most platforms — these are vendor-experimental
|
||||
# (x- prefix, never IANA-registered). macOS Chrome/Safari MIME-sniff leniently
|
||||
# via CoreAudio so playback works there, but Linux Chrome/Firefox (FFmpeg) and
|
||||
# Android Chrome (ExoPlayer) strictly honor the declared type and treat the
|
||||
# x- variants as download-only — manifesting as the play button silently
|
||||
# doing nothing in the browser app while working in the Tauri desktop shell.
|
||||
# `audio/wav` / `audio/flac` are the IANA-canonical types.
|
||||
# Ref: https://www.iana.org/assignments/media-types/media-types.xhtml#audio
|
||||
import mimetypes as _mimetypes
|
||||
_mimetypes.add_type("audio/wav", ".wav")
|
||||
_mimetypes.add_type("audio/flac", ".flac")
|
||||
|
||||
app.mount("/audio", StaticFiles(directory=OUTPUTS_DIR), name="audio")
|
||||
app.mount("/voice_audio", StaticFiles(directory=VOICES_DIR), name="voice_audio")
|
||||
|
||||
@@ -789,6 +911,7 @@ app.include_router(watermark.router)
|
||||
app.include_router(events.router)
|
||||
app.include_router(capture.router)
|
||||
app.include_router(capture_ws.router)
|
||||
app.include_router(dictation.router)
|
||||
app.include_router(openai_compat.router)
|
||||
app.include_router(tts_stream.router)
|
||||
app.include_router(marketplace.router)
|
||||
@@ -796,6 +919,7 @@ app.include_router(personas.router)
|
||||
app.include_router(sonitranslate.router)
|
||||
app.include_router(audiobook.router)
|
||||
app.include_router(longform_jobs.router)
|
||||
app.include_router(pronunciation.router) # Expressive-TTS Spec 01: pronunciation dictionary
|
||||
app.include_router(settings_router.router) # Phase 1 AUTH-03 endpoints
|
||||
from api.routers import mcp_bindings as _mcp_bindings_router # noqa: E402
|
||||
app.include_router(_mcp_bindings_router.router) # Wave 2.2 per-agent voice bindings
|
||||
|
||||
@@ -0,0 +1,36 @@
|
||||
"""Heal voice_profiles.instruct poisoned with the "[object Object]" sentinel.
|
||||
|
||||
Revision ID: 0006_strip_object_object_instruct
|
||||
Revises: 0005_unified_profiles
|
||||
Create Date: 2026-06-20 00:00:00.000000
|
||||
|
||||
A pre-fix Voice Studio build ("Save design as profile") passed the
|
||||
``buildDesignInstruct()`` *object* straight to FormData, which string-coerced it
|
||||
to the literal ``"[object Object]"`` and persisted that into
|
||||
``voice_profiles.instruct`` (#550 #545 #542 #537 #530 #525). On first
|
||||
preview/use that value fails the engine instruct validator with a 400. The
|
||||
frontend + backend fixes stop any NEW poisoned rows; this migration heals the
|
||||
ones already saved on the buggy build (the local-first backward-compat rule —
|
||||
existing project data must keep working without manual migration).
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
from alembic import op
|
||||
from sqlalchemy import inspect
|
||||
|
||||
revision: str = "0006_strip_object_object_instruct"
|
||||
down_revision: Union[str, None] = "0005_unified_profiles"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
bind = op.get_bind()
|
||||
if "voice_profiles" in inspect(bind).get_table_names():
|
||||
# Idempotent: only touches rows whose instruct is literally the sentinel.
|
||||
op.execute("UPDATE voice_profiles SET instruct='' WHERE instruct='[object Object]'")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
# Irreversible heal — the original garbage sentinel is not worth restoring.
|
||||
pass
|
||||
@@ -0,0 +1,118 @@
|
||||
"""Rebuild design-profile instructs poisoned with prose / "[object Object]".
|
||||
|
||||
Revision ID: 0007_rebuild_poisoned_design_instruct
|
||||
Revises: 0006_strip_object_object_instruct
|
||||
Create Date: 2026-06-22 00:00:00.000000
|
||||
|
||||
Migration 0006 *blanked* the literal ``"[object Object]"`` sentinel. That stops
|
||||
the 400 on use, but it also throws away the designed voice: a row that read
|
||||
``"[object Object]"`` (or freeform prose like "A gentle, quiet male voice…")
|
||||
becomes ``instruct=''`` and then renders with the engine's neutral default —
|
||||
which is why an Indonesian *female* designed voice came out *male* (#594), and
|
||||
why prose-poisoned designs still 400 (#571 #596).
|
||||
|
||||
This migration heals it properly: for every design profile it recomputes a
|
||||
validator-safe instruct, preferring any whitelist tags already in the stored
|
||||
value and otherwise rebuilding the tags from ``vd_states`` (the authoritative
|
||||
category→pick map the Voice Design picker persists). Non-design rows simply get
|
||||
their instruct sanitized (poison dropped). Idempotent — a healthy row is left
|
||||
byte-for-byte unchanged, so re-running is a no-op.
|
||||
|
||||
Self-contained by design: alembic migrations must not import evolving app code
|
||||
(``omnivoice`` would also drag in torch at startup), so the tag whitelist is a
|
||||
frozen snapshot of ``omnivoice.utils.voice_design._INSTRUCT_ALL_VALID``.
|
||||
``tests/test_migration_0007_instruct_rebuild.py`` asserts the snapshot stays in
|
||||
sync with the canonical set.
|
||||
"""
|
||||
import json
|
||||
import re
|
||||
from typing import Sequence, Union
|
||||
|
||||
from alembic import op
|
||||
from sqlalchemy import inspect
|
||||
|
||||
revision: str = "0007_rebuild_poisoned_design_instruct"
|
||||
down_revision: Union[str, None] = "0006_strip_object_object_instruct"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
# Frozen snapshot of the design-instruct whitelist + mutually-exclusive
|
||||
# categories (omnivoice/utils/voice_design.py). Kept self-contained so the
|
||||
# migration's behaviour is pinned to the data it heals, not to future vocab
|
||||
# edits. Parity is guarded by the migration test.
|
||||
_CATEGORIES = [
|
||||
{"male", "女", "female", "男"},
|
||||
{"child", "teenager", "young adult", "middle-aged", "elderly",
|
||||
"儿童", "少年", "青年", "中年", "老年"},
|
||||
{"very low pitch", "low pitch", "moderate pitch", "high pitch", "very high pitch",
|
||||
"极低音调", "低音调", "中音调", "高音调", "极高音调"},
|
||||
{"whisper", "耳语"},
|
||||
{"american accent", "british accent", "australian accent", "chinese accent",
|
||||
"canadian accent", "indian accent", "korean accent", "portuguese accent",
|
||||
"russian accent", "japanese accent"},
|
||||
{"河南话", "陕西话", "四川话", "贵州话", "云南话", "桂林话",
|
||||
"济南话", "石家庄话", "甘肃话", "宁夏话", "青岛话", "东北话"},
|
||||
]
|
||||
_ALL_VALID = set().union(*_CATEGORIES)
|
||||
|
||||
|
||||
def _valid_from_items(items) -> str:
|
||||
"""One whitelist tag per category, first-seen order; everything else dropped."""
|
||||
seen = set()
|
||||
out = []
|
||||
for raw in items:
|
||||
tag = str(raw if raw is not None else "").strip().lower()
|
||||
if not tag or tag not in _ALL_VALID:
|
||||
continue
|
||||
ci = next((i for i, c in enumerate(_CATEGORIES) if tag in c), -1)
|
||||
if ci in seen:
|
||||
continue
|
||||
seen.add(ci)
|
||||
out.append(tag)
|
||||
return ", ".join(out)
|
||||
|
||||
|
||||
def _heal(instruct, vd_states, is_design) -> str:
|
||||
healed = _valid_from_items(re.split(r"\s*[,,]\s*", str(instruct or "").strip()))
|
||||
if healed or not is_design:
|
||||
return healed
|
||||
# Stored instruct was all-poison — recover the design from vd_states.
|
||||
if not vd_states:
|
||||
return ""
|
||||
try:
|
||||
vd = json.loads(vd_states)
|
||||
except (ValueError, TypeError):
|
||||
return ""
|
||||
return _valid_from_items(vd.values()) if isinstance(vd, dict) else ""
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
bind = op.get_bind()
|
||||
insp = inspect(bind)
|
||||
if "voice_profiles" not in insp.get_table_names():
|
||||
return
|
||||
cols = {c["name"] for c in insp.get_columns("voice_profiles")}
|
||||
has_kind = "kind" in cols
|
||||
has_vd = "vd_states" in cols
|
||||
|
||||
select = "SELECT id, instruct"
|
||||
select += ", kind" if has_kind else ""
|
||||
select += ", vd_states" if has_vd else ""
|
||||
select += " FROM voice_profiles"
|
||||
|
||||
for row in bind.exec_driver_sql(select).mappings().all():
|
||||
instruct = row["instruct"] or ""
|
||||
is_design = (row["kind"] == "design") if has_kind else bool(instruct)
|
||||
vd = row["vd_states"] if has_vd else None
|
||||
healed = _heal(instruct, vd, is_design)
|
||||
if healed != instruct:
|
||||
bind.exec_driver_sql(
|
||||
"UPDATE voice_profiles SET instruct = ? WHERE id = ?",
|
||||
(healed, row["id"]),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
# Irreversible heal — the original poisoned value isn't worth restoring.
|
||||
pass
|
||||
@@ -0,0 +1,67 @@
|
||||
"""Expressive-TTS Spec 01 Phase 1: user pronunciation dictionary
|
||||
|
||||
Revision ID: 0008_pronunciation_dictionary
|
||||
Revises: 0007_rebuild_poisoned_design_instruct
|
||||
Create Date: 2026-06-25 00:00:00.000000
|
||||
|
||||
Adds the ``pronunciation_entries`` table backing the user-editable, per-language
|
||||
pronunciation dictionary (Settings → Pronunciation). Each row maps a ``term`` to
|
||||
a ``replacement`` the engine pronounces correctly, scoped global (``language='*'``)
|
||||
or to a 2-letter language. Applied as pure text substitution before synthesis, so
|
||||
every engine honors it.
|
||||
|
||||
* ``id`` TEXT PRIMARY KEY — stable row id.
|
||||
* ``term`` TEXT — the word/phrase to match (whole-word, case-insensitive).
|
||||
* ``replacement`` TEXT — the respelling (or, for phoneme rows, the markup).
|
||||
* ``type`` TEXT — 'respelling' | 'ipa' | 'cmu'.
|
||||
* ``language`` TEXT — '*' = global, else a language code (e.g. 'en', 'de').
|
||||
* ``enabled`` INTEGER — 1 = applied, 0 = parked.
|
||||
* ``created_at`` REAL.
|
||||
|
||||
Additive + idempotent (guarded by sqlite_master), matching 0002/0003/0004, so
|
||||
re-running on a fresh-install DB where ``_BASE_SCHEMA`` already created the table
|
||||
is a no-op (Backward-compatible project data constraint). The same table is
|
||||
mirrored into ``core/db.py::_BASE_SCHEMA`` so fresh installs and migrated DBs
|
||||
converge on an identical end-state (the dual-path discipline).
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
from alembic import op
|
||||
import sqlalchemy as sa
|
||||
|
||||
|
||||
revision: str = "0008_pronunciation_dictionary"
|
||||
down_revision: Union[str, None] = "0007_rebuild_poisoned_design_instruct"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def _has_table(name: str) -> bool:
|
||||
bind = op.get_bind()
|
||||
row = bind.execute(
|
||||
sa.text("SELECT name FROM sqlite_master WHERE type='table' AND name=:n"),
|
||||
{"n": name},
|
||||
).fetchone()
|
||||
return row is not None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
if _has_table("pronunciation_entries"):
|
||||
return
|
||||
op.create_table(
|
||||
"pronunciation_entries",
|
||||
sa.Column("id", sa.Text(), primary_key=True),
|
||||
sa.Column("term", sa.Text(), nullable=False),
|
||||
sa.Column("replacement", sa.Text(), nullable=False, server_default=""),
|
||||
sa.Column("type", sa.Text(), nullable=False, server_default="respelling"),
|
||||
sa.Column("language", sa.Text(), nullable=False, server_default="*"),
|
||||
sa.Column("enabled", sa.Integer(), nullable=False, server_default="1"),
|
||||
sa.Column("created_at", sa.Float(), nullable=True),
|
||||
)
|
||||
op.create_index("idx_pron_lang", "pronunciation_entries", ["language"])
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
if _has_table("pronunciation_entries"):
|
||||
op.drop_index("idx_pron_lang", table_name="pronunciation_entries")
|
||||
op.drop_table("pronunciation_entries")
|
||||
@@ -129,7 +129,7 @@ class TranslateRequest(BaseModel):
|
||||
provider: Optional[str] = None
|
||||
source_lang: Optional[str] = None # ISO 639-1; overrides job detection
|
||||
job_id: Optional[str] = None # Dub job id, used to resolve detected source_lang
|
||||
quality: Optional[str] = "fast" # "fast" (one-shot) | "cinematic" (reflect → adapt)
|
||||
quality: Optional[str] = "fast" # "fast" (one-shot) | "cinematic" (reflect→adapt) | "autofit" (cinematic + strict fit-to-slot)
|
||||
glossary: Optional[List[dict]] = None # [{"source": "...", "target": "...", "note": "..."}]
|
||||
# Optional regional dialect (BCP-47, e.g. "es-AR", "pt-BR") — #280 item 2.
|
||||
# Applied by LLM-backed paths (provider="openai" or quality="cinematic"):
|
||||
|
||||
+525
-25
@@ -23,6 +23,7 @@ faster-whisper because it's available on every platform we ship to).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
@@ -30,6 +31,90 @@ from abc import ABC, abstractmethod
|
||||
|
||||
logger = logging.getLogger("omnivoice.asr")
|
||||
|
||||
# A single ASR transcribe must never block a request indefinitely. The chunked
|
||||
# dub pipeline already bounds each chunk (OMNIVOICE_TRANSCRIBE_CHUNK_TIMEOUT_S);
|
||||
# the *whole-file* paths (dub QC re-transcribe, dictation, OpenAI-compat) ran
|
||||
# unbounded, so a slow/stuck transcribe — e.g. large-v3 on a VRAM-starved GPU
|
||||
# where the resident TTS model contends for memory — hung the request *and* tied
|
||||
# up a GPU-pool worker, surfacing in the UI as the misleading "can't reach the
|
||||
# local backend" (TamKieu / Vietnam report). Bound them so a hang becomes a fast,
|
||||
# actionable error instead. Generous default (whole-file large-v3 on CPU is slow
|
||||
# but valid); override with the env var for very long single files.
|
||||
ASR_TRANSCRIBE_TIMEOUT_S = float(os.environ.get("OMNIVOICE_ASR_TRANSCRIBE_TIMEOUT_S", "300.0"))
|
||||
|
||||
|
||||
class ASRTimeoutError(TimeoutError):
|
||||
"""Raised when a whole-file transcribe exceeds ASR_TRANSCRIBE_TIMEOUT_S.
|
||||
|
||||
Carries a user-actionable message: the backend is alive (this is not a
|
||||
connection failure) — the ASR model is too heavy for the available compute.
|
||||
"""
|
||||
|
||||
|
||||
async def run_transcribe_guarded(executor, fn, *, what: str = "ASR",
|
||||
timeout: float = ASR_TRANSCRIBE_TIMEOUT_S):
|
||||
"""Run a blocking transcribe ``fn`` in ``executor`` with a hard wall-clock
|
||||
bound. On timeout, raise :class:`ASRTimeoutError` with guidance instead of
|
||||
letting the request hang forever.
|
||||
|
||||
``run_in_executor`` cannot cancel the underlying thread, so a wedged
|
||||
transcribe (a CTranslate2 / whisperx / VAD hang seen on some Windows + CUDA
|
||||
setups, #730) keeps occupying its GPU-pool worker. With a 1–2 worker pool
|
||||
that starves every *other* request — including TTS generate — and the next
|
||||
thing the user does surfaces as "Can't reach the local backend" even though
|
||||
the process is alive. So on timeout we also ``reset()`` the pool when it
|
||||
supports it (``_ResilientGpuPool``): the wedged thread is abandoned and the
|
||||
next submit gets a fresh worker, restoring capacity without an app restart.
|
||||
The orphaned thread still holds its VRAM until the process exits, which is
|
||||
why the message still recommends a smaller ASR model / Flush as the durable
|
||||
fix. Executors without ``reset`` (a plain ThreadPoolExecutor in tests) just
|
||||
get the bound + actionable error.
|
||||
"""
|
||||
loop = asyncio.get_running_loop()
|
||||
fut = loop.run_in_executor(executor, fn)
|
||||
try:
|
||||
return await asyncio.wait_for(fut, timeout=timeout)
|
||||
except asyncio.TimeoutError:
|
||||
# Free the poisoned pool so a hung transcribe can't keep starving TTS /
|
||||
# other ASR work (the "can't reach backend" symptom, #730).
|
||||
_reset = getattr(executor, "reset", None)
|
||||
if callable(_reset):
|
||||
try:
|
||||
_reset()
|
||||
logger.warning(
|
||||
"%s transcription exceeded %.0fs — abandoned the GPU-pool "
|
||||
"worker to restore capacity (#730).", what, timeout,
|
||||
)
|
||||
except Exception:
|
||||
logger.exception("GPU pool reset after ASR timeout failed")
|
||||
raise ASRTimeoutError(
|
||||
f"{what} transcription exceeded {timeout:.0f}s and was abandoned — "
|
||||
"the backend is running, but the ASR model is too heavy for the "
|
||||
"available compute. Most often the GPU is VRAM-starved: the resident "
|
||||
"TTS model and a large ASR model (large-v3) contend for memory. "
|
||||
"Capacity was restored automatically, but for a durable fix Flush the "
|
||||
"TTS model to free VRAM, pick a smaller ASR model in Settings → "
|
||||
"Models, or set ASR to CPU. (Raise OMNIVOICE_ASR_TRANSCRIBE_TIMEOUT_S "
|
||||
"for very long single files.)"
|
||||
)
|
||||
|
||||
|
||||
def _compute_type_candidates(device: str) -> list[str]:
|
||||
"""Per-device compute_type fallback chain. int8 is supported by every
|
||||
CTranslate2 CUDA+CPU build; float16/int8_float16 only on GPUs with efficient
|
||||
fp16 — so degrade rather than crash (#551). Honors an ASR_COMPUTE_TYPE env
|
||||
override (power users on exotic hardware can pin int8/float32)."""
|
||||
import os
|
||||
override = os.environ.get("ASR_COMPUTE_TYPE")
|
||||
if override:
|
||||
return [override]
|
||||
return ["float16", "int8_float16", "int8"] if device == "cuda" else ["int8", "float32"]
|
||||
|
||||
|
||||
def _is_compute_type_error(msg: str) -> bool:
|
||||
low = msg.lower()
|
||||
return "compute type" in low or "efficient float16" in low
|
||||
|
||||
|
||||
def _decode_audio_16k_mono(audio_path: str):
|
||||
"""Decode `audio_path` to a 16 kHz mono float32 waveform using OmniVoice's
|
||||
@@ -112,6 +197,22 @@ class ASRBackend(ABC):
|
||||
that already speak the shape plug in with zero adapter work.
|
||||
"""
|
||||
|
||||
def ensure_loaded(self) -> None:
|
||||
"""Eagerly load the model weights, raising the real cause on failure.
|
||||
|
||||
Backends load lazily inside ``transcribe()`` by default, so a load
|
||||
failure (missing weights, CUDA/cuDNN mismatch, torch-2.6 weights-only
|
||||
VAD regression, import error) first surfaces buried in per-chunk
|
||||
errors — and is retried on *every* chunk. The transcribe preflight
|
||||
calls this so the genuine cause is surfaced once, up front, as a clean
|
||||
terminal error event instead of N cryptic per-chunk failures (#578).
|
||||
|
||||
Default is a no-op; backends that hold a heavy model override it to
|
||||
trigger their lazy loader. It MUST raise the underlying exception (not
|
||||
swallow it) so the caller can classify and surface it.
|
||||
"""
|
||||
pass
|
||||
|
||||
def unload(self) -> None:
|
||||
"""Release the model from memory."""
|
||||
pass
|
||||
@@ -120,6 +221,75 @@ class ASRBackend(ABC):
|
||||
# ── WhisperX (cross-platform default — forced-alignment word timing) ────────
|
||||
|
||||
|
||||
def _harden_speechbrain_lazy_imports() -> None:
|
||||
"""Make speechbrain 1.x's lazy-import guard fire on Windows too (#630/#611/#647).
|
||||
|
||||
speechbrain 1.x exposes optional integrations (``k2_fsa``, ``numba`` losses,
|
||||
``spacy``/``flair`` nlp) as ``LazyModule`` redirects living in ``sys.modules``.
|
||||
Stray introspection — PyTorch's op-registration machinery, pickling, a
|
||||
``dir()``/``hasattr`` walk — touches one of these during ``whisperx.load_model``
|
||||
(pyannote → speechbrain), which would *actually* import the optional package.
|
||||
speechbrain guards against that by suppressing the import when the triggering
|
||||
frame is the stdlib ``inspect`` module — but the check is
|
||||
``filename.endswith("/inspect.py")``, a hardcoded POSIX separator. On Windows
|
||||
the frame filename uses backslashes (``...\\Lib\\inspect.py``), so the guard
|
||||
misses, the redirect imports ``speechbrain.integrations.k2_fsa`` → ``import k2``
|
||||
→ k2 isn't installed → ``ImportError: Lazy import of LazyModule(...k2_fsa...)
|
||||
failed``. That bubbles out of WhisperX and aborts transcription with zero
|
||||
segments. WhisperX is the *default* ASR, so this is a Windows-only break of a
|
||||
cross-platform-default feature (P0 parity).
|
||||
|
||||
Fix the whole class — every optional-integration redirect, not just k2 — by
|
||||
re-implementing ``LazyModule.ensure_module`` with an ``os.sep``-agnostic
|
||||
basename check. Idempotent and a no-op on macOS/Linux (basename match is a
|
||||
strict superset of the old forward-slash check) and when speechbrain is
|
||||
absent. A genuine access from real user code with k2 missing still raises
|
||||
ImportError unchanged — only inspect-triggered spurious imports are
|
||||
suppressed, on every platform.
|
||||
"""
|
||||
try:
|
||||
from speechbrain.utils import importutils as _iu
|
||||
except Exception: # speechbrain not installed / import side-effect — nothing to harden
|
||||
return
|
||||
if getattr(_iu.LazyModule, "_omnivoice_xplat_guard", False):
|
||||
return
|
||||
import importlib as _importlib
|
||||
import inspect as _inspect
|
||||
import sys as _sys
|
||||
import warnings as _warnings
|
||||
|
||||
def ensure_module(self, stacklevel):
|
||||
importer_frame = None
|
||||
try:
|
||||
importer_frame = _inspect.getframeinfo(_sys._getframe(stacklevel + 1))
|
||||
except AttributeError:
|
||||
_warnings.warn(
|
||||
"Failed to inspect frame to check if we should ignore importing a "
|
||||
"module lazily (OmniVoice cross-platform guard)."
|
||||
)
|
||||
if importer_frame is not None:
|
||||
# Normalise BOTH separators explicitly (not os.path.basename, which is
|
||||
# host-dependent) so the guard is correct regardless of which os.path
|
||||
# flavour is active. Upstream's `.endswith("/inspect.py")` matched only
|
||||
# POSIX paths — that is the Windows-only bug (#630/#611/#647).
|
||||
base = importer_frame.filename.replace("\\", "/").rsplit("/", 1)[-1]
|
||||
if base == "inspect.py":
|
||||
raise AttributeError()
|
||||
if self.lazy_module is None:
|
||||
try:
|
||||
if self.package is None:
|
||||
self.lazy_module = _importlib.import_module(self.target)
|
||||
else:
|
||||
self.lazy_module = _importlib.import_module(f".{self.target}", self.package)
|
||||
except Exception as e: # noqa: BLE001 — match upstream: wrap as ImportError
|
||||
raise ImportError(f"Lazy import of {repr(self)} failed") from e
|
||||
return self.lazy_module
|
||||
|
||||
_iu.LazyModule.ensure_module = ensure_module
|
||||
_iu.LazyModule._omnivoice_xplat_guard = True
|
||||
logger.debug("speechbrain LazyModule guard hardened for cross-platform inspect.py check")
|
||||
|
||||
|
||||
class WhisperXBackend(ASRBackend):
|
||||
id = "whisperx"
|
||||
display_name = "WhisperX (faster-whisper + wav2vec2 forced alignment)"
|
||||
@@ -153,10 +323,28 @@ class WhisperXBackend(ASRBackend):
|
||||
return True, "ready"
|
||||
except ImportError as e:
|
||||
return False, f"whisperx not installed: {e}"
|
||||
except Exception as e: # noqa: BLE001
|
||||
# The import can fail while loading a native dep — CTranslate2's .so
|
||||
# is rejected by hardened kernels / newer glibc with "cannot enable
|
||||
# executable stack" (#692), an OSError, not an ImportError. An
|
||||
# availability probe must REPORT 'unusable here', never raise, so
|
||||
# engine selection falls back instead of crashing the ASR preflight.
|
||||
return False, f"whisperx failed to load ({type(e).__name__}): {e}"
|
||||
|
||||
def ensure_loaded(self) -> None:
|
||||
# Surface a whisperx/CTranslate2/torch load failure at preflight (once,
|
||||
# with the real cause) instead of buried per-chunk and retried N times
|
||||
# (#578). Re-raises whatever `_ensure_asr` raises after its fp16→int8
|
||||
# and OOM→CPU fallbacks are exhausted.
|
||||
self._ensure_asr()
|
||||
|
||||
def _ensure_asr(self):
|
||||
if self._asr is not None:
|
||||
return
|
||||
# Patch speechbrain's lazy-import guard BEFORE whisperx pulls in pyannote
|
||||
# → speechbrain, or a stray k2_fsa redirect import aborts ASR on Windows
|
||||
# (#630/#611/#647). No-op on macOS/Linux and when speechbrain is absent.
|
||||
_harden_speechbrain_lazy_imports()
|
||||
import whisperx
|
||||
logger.info(
|
||||
"whisperx loading ASR %s on %s (%s)",
|
||||
@@ -189,7 +377,40 @@ class WhisperXBackend(ASRBackend):
|
||||
# vad_method="silero" is the default; keep it so short gaps
|
||||
# get cleaned up before transcription.
|
||||
)
|
||||
except RuntimeError as e:
|
||||
except (ValueError, RuntimeError) as e:
|
||||
# #551: GPUs without efficient fp16 (older Maxwell/Pascal, GTX 16xx)
|
||||
# or a CTranslate2/cuDNN binary mismatch raise a *ValueError*
|
||||
# ("Requested float16 compute type, but the target device or backend
|
||||
# do not support efficient float16 computation") at load — not an
|
||||
# OOM, not a RuntimeError. Retry on the SAME device with the next
|
||||
# compute_type candidate (cuda: int8_float16 → int8) before touching
|
||||
# the OOM→CPU path, so we degrade rather than crash every chunk.
|
||||
if _is_compute_type_error(str(e)):
|
||||
candidates = _compute_type_candidates(self._device)
|
||||
try:
|
||||
nxt = candidates[candidates.index(self._compute_type) + 1:]
|
||||
except ValueError:
|
||||
nxt = [c for c in candidates if c != self._compute_type]
|
||||
for ct in nxt:
|
||||
logger.warning(
|
||||
"whisperx %s unsupported on %s — retrying with %s. Detail: %s",
|
||||
self._compute_type, self._device, ct, e,
|
||||
)
|
||||
self._compute_type = ct
|
||||
try:
|
||||
self._asr = whisperx.load_model(
|
||||
self._model_name,
|
||||
device=self._device,
|
||||
compute_type=self._compute_type,
|
||||
)
|
||||
return
|
||||
except (ValueError, RuntimeError) as e2:
|
||||
if _is_compute_type_error(str(e2)):
|
||||
e = e2
|
||||
continue
|
||||
raise
|
||||
# Exhausted compute-type candidates on this device — re-raise.
|
||||
raise
|
||||
# CUDA OOM: a resident TTS model + the GPU worker pool can starve
|
||||
# VRAM on small (e.g. 8 GB laptop) GPUs, so loading large-v3 on
|
||||
# CUDA dies here — which previously surfaced as a bare 500 from
|
||||
@@ -462,6 +683,10 @@ class FasterWhisperBackend(ASRBackend):
|
||||
"ASR_MODEL_FASTER", "Systran/faster-whisper-large-v3"
|
||||
)
|
||||
self._model = None # lazy — first transcribe() loads weights
|
||||
# Set by _ensure_model() to the device/compute_type that actually loaded
|
||||
# (after the #551 compute_type / #255 OOM→CPU fallback chain).
|
||||
self._device: str | None = None
|
||||
self._compute_type: str | None = None
|
||||
|
||||
@classmethod
|
||||
def is_available(cls) -> tuple[bool, str]:
|
||||
@@ -470,6 +695,11 @@ class FasterWhisperBackend(ASRBackend):
|
||||
return True, "ready"
|
||||
except ImportError as e:
|
||||
return False, f"faster-whisper not installed: {e}"
|
||||
except Exception as e: # noqa: BLE001
|
||||
# faster-whisper pulls in CTranslate2, whose .so is rejected by
|
||||
# hardened kernels / newer glibc ("cannot enable executable stack",
|
||||
# #692) — an OSError. Report unavailable so we fall back, not crash.
|
||||
return False, f"faster-whisper failed to load ({type(e).__name__}): {e}"
|
||||
|
||||
def _ensure_model(self):
|
||||
if self._model is not None:
|
||||
@@ -490,9 +720,58 @@ class FasterWhisperBackend(ASRBackend):
|
||||
"faster-whisper loading %s on %s (%s)",
|
||||
self._model_name, device, compute_type,
|
||||
)
|
||||
self._model = WhisperModel(
|
||||
self._model_name, device=device, compute_type=compute_type
|
||||
)
|
||||
# Try the per-device compute_type chain (cuda: float16 → int8_float16 →
|
||||
# int8; cpu: int8 → float32). A GPU without efficient fp16 (older
|
||||
# Maxwell/Pascal, GTX 16xx, or a CTranslate2/cuDNN mismatch) raises a
|
||||
# *ValueError* at construction (#551) — degrade to the next candidate
|
||||
# instead of failing every chunk. A genuine CUDA OOM falls back to CPU
|
||||
# (slower, same model/accuracy), preserving the existing #255 behaviour.
|
||||
candidates = _compute_type_candidates(device)
|
||||
if compute_type in candidates:
|
||||
candidates = candidates[candidates.index(compute_type):]
|
||||
last_err: Exception | None = None
|
||||
while True:
|
||||
for ct in candidates:
|
||||
try:
|
||||
self._model = WhisperModel(
|
||||
self._model_name, device=device, compute_type=ct
|
||||
)
|
||||
self._device, self._compute_type = device, ct
|
||||
return
|
||||
except (ValueError, RuntimeError) as e:
|
||||
last_err = e
|
||||
if _is_compute_type_error(str(e)):
|
||||
logger.warning(
|
||||
"faster-whisper %s unsupported on %s — trying next "
|
||||
"compute_type. Detail: %s", ct, device, e,
|
||||
)
|
||||
continue
|
||||
if device == "cuda" and "out of memory" in str(e).lower():
|
||||
# Stop scanning GPU candidates; fall back to CPU below.
|
||||
break
|
||||
raise
|
||||
# Exhausted candidates for this device. If we were on CUDA and the
|
||||
# last failure was an OOM, retry on CPU with its candidates (#255).
|
||||
if device == "cuda" and last_err is not None and (
|
||||
"out of memory" in str(last_err).lower()
|
||||
):
|
||||
logger.warning(
|
||||
"faster-whisper CUDA OOM loading %s — retrying on CPU "
|
||||
"(slower). Free VRAM (Flush the TTS model) for GPU-speed "
|
||||
"ASR. Detail: %s", self._model_name, last_err,
|
||||
)
|
||||
try:
|
||||
import torch
|
||||
torch.cuda.empty_cache()
|
||||
except Exception: # noqa: BLE001 — cache clear is best-effort
|
||||
pass
|
||||
device = "cpu"
|
||||
candidates = _compute_type_candidates(device)
|
||||
compute_type = candidates[0]
|
||||
continue
|
||||
# All candidates exhausted (and no OOM→CPU retry available) — surface
|
||||
# the last error.
|
||||
raise last_err
|
||||
|
||||
def transcribe(self, audio_path: str, *, word_timestamps: bool = True) -> dict:
|
||||
self._ensure_model()
|
||||
@@ -681,12 +960,25 @@ class PyTorchWhisperBackend(ASRBackend):
|
||||
"PyTorchWhisperBackend: loading standalone ASR pipeline %s on %s",
|
||||
model_name, device,
|
||||
)
|
||||
self._pipe = hf_pipeline(
|
||||
"automatic-speech-recognition",
|
||||
model=model_name,
|
||||
dtype=asr_dtype,
|
||||
device_map=device,
|
||||
)
|
||||
try:
|
||||
self._pipe = hf_pipeline(
|
||||
"automatic-speech-recognition",
|
||||
model=model_name,
|
||||
dtype=asr_dtype,
|
||||
device_map=device,
|
||||
)
|
||||
except Exception as e:
|
||||
# #549: an incomplete transformers install fails to build the ASR
|
||||
# pipeline (e.g. "Could not import module 'AutoFeatureExtractor'").
|
||||
# The raw error is opaque; re-raise with an actionable next step so
|
||||
# the toast tells the user how to recover instead of "no segments".
|
||||
raise RuntimeError(
|
||||
"transformers ASR pipeline failed to import (AutoFeatureExtractor) "
|
||||
"— your transformers install is incomplete; reinstall with "
|
||||
"`uv pip install --reinstall transformers`, or use faster-whisper "
|
||||
"(OmniVoice's default ASR) which avoids the transformers pipeline. "
|
||||
f"Underlying: {e}"
|
||||
) from e
|
||||
|
||||
def transcribe(self, audio_path: str, *, word_timestamps: bool = True) -> dict:
|
||||
import soundfile as sf
|
||||
@@ -908,6 +1200,148 @@ class MoonshineASRBackend(ASRBackend):
|
||||
self._transcriber = None
|
||||
|
||||
|
||||
# ── sherpa-onnx live dictation (ONNX, CPU, streaming + offline) ─────────────
|
||||
|
||||
|
||||
def _load_audio_16k_mono_f32(audio_path: str):
|
||||
"""Decode any audio file to 16 kHz mono float32 in [-1, 1] for sherpa.
|
||||
|
||||
Prefers soundfile (WAV/FLAC — the dictation buffers are already WAV) and
|
||||
resamples to 16 kHz when needed; falls back to OmniVoice's validated ffmpeg
|
||||
for containers soundfile can't read (WebM/Opus). 16 kHz is sherpa's cheapest
|
||||
feed; it resamples internally too, but doing it here keeps the contract tight.
|
||||
"""
|
||||
import numpy as np
|
||||
try:
|
||||
import soundfile as sf
|
||||
data, sr = sf.read(audio_path, dtype="float32", always_2d=False)
|
||||
if getattr(data, "ndim", 1) > 1:
|
||||
data = data.mean(axis=1)
|
||||
data = np.ascontiguousarray(data, dtype=np.float32)
|
||||
if sr != 16000:
|
||||
# Lightweight linear resample — adequate for ASR features.
|
||||
n = int(round(len(data) * 16000 / sr))
|
||||
if n > 0:
|
||||
xp = np.linspace(0.0, 1.0, num=len(data), endpoint=False)
|
||||
x = np.linspace(0.0, 1.0, num=n, endpoint=False)
|
||||
data = np.interp(x, xp, data).astype(np.float32)
|
||||
sr = 16000
|
||||
return data, sr
|
||||
except Exception:
|
||||
# Container soundfile can't read (WebM/Opus) — use the validated ffmpeg
|
||||
# path, which already yields 16 kHz mono float32.
|
||||
return _decode_audio_16k_mono(audio_path), 16000
|
||||
|
||||
|
||||
class SherpaDictationBackend(ASRBackend):
|
||||
"""k2-fsa/sherpa-onnx ONNX dictation engine (CPU, live + offline).
|
||||
|
||||
One :class:`ASRBackend` instance is bound to one of the seven sherpa
|
||||
dictation models (see :mod:`services.sherpa_dictation`). For the offline
|
||||
``transcribe(path)`` contract it runs an ``OfflineRecognizer`` for offline
|
||||
models and a one-shot ``OnlineRecognizer`` decode for streaming models
|
||||
(so ``POST /transcribe`` works for every sherpa model). The *live* WS path
|
||||
drives the streaming recognizer incrementally — see ``capture_ws.py``.
|
||||
|
||||
CPU provider only (cross-platform default-parity rule); no CUDA dep.
|
||||
"""
|
||||
id = "sherpa-onnx-asr"
|
||||
display_name = "Sherpa-ONNX dictation (live, CPU — streaming + offline)"
|
||||
gpu_compat = ("cpu",)
|
||||
|
||||
def __init__(self, model_id: str | None = None):
|
||||
from services import sherpa_dictation as _sd
|
||||
mid = model_id or os.environ.get(
|
||||
"OMNIVOICE_SHERPA_ASR_MODEL", _sd.DEFAULT_MODEL_ID
|
||||
)
|
||||
spec = _sd.get_spec(mid)
|
||||
if spec is None:
|
||||
raise ValueError(
|
||||
f"Unknown sherpa dictation model {mid!r}. Known: "
|
||||
f"{[s.id for s in _sd.list_specs()]}"
|
||||
)
|
||||
self._spec = spec
|
||||
self._rec = None # lazy OfflineRecognizer / OnlineRecognizer
|
||||
|
||||
@property
|
||||
def spec(self):
|
||||
return self._spec
|
||||
|
||||
@property
|
||||
def streaming(self) -> bool:
|
||||
return self._spec.streaming
|
||||
|
||||
@classmethod
|
||||
def is_available(cls) -> tuple[bool, str]:
|
||||
from services.sherpa_dictation import sherpa_available
|
||||
return sherpa_available()
|
||||
|
||||
def ensure_loaded(self) -> None:
|
||||
self._ensure_rec()
|
||||
|
||||
def _ensure_rec(self):
|
||||
if self._rec is not None:
|
||||
return
|
||||
from services import sherpa_dictation as _sd
|
||||
if self._spec.streaming:
|
||||
self._rec = _sd.build_online_recognizer(self._spec)
|
||||
else:
|
||||
self._rec = _sd.build_offline_recognizer(self._spec)
|
||||
|
||||
def transcribe(self, audio_path: str, *, word_timestamps: bool = True) -> dict:
|
||||
self._ensure_rec()
|
||||
logger.info(
|
||||
"sherpa-onnx dictation transcribing %s (model=%s, kind=%s)",
|
||||
audio_path, self._spec.id, self._spec.kind,
|
||||
)
|
||||
samples, sr = _load_audio_16k_mono_f32(audio_path)
|
||||
if self._spec.streaming:
|
||||
text = self._decode_online_oneshot(samples, sr)
|
||||
else:
|
||||
text = self._decode_offline(samples, sr)
|
||||
return _sherpa_result(text, samples, sr)
|
||||
|
||||
def _decode_offline(self, samples, sr) -> str:
|
||||
s = self._rec.create_stream()
|
||||
s.accept_waveform(sr, samples)
|
||||
self._rec.decode_stream(s)
|
||||
return (s.result.text or "").strip()
|
||||
|
||||
def _decode_online_oneshot(self, samples, sr) -> str:
|
||||
"""One-shot decode of a whole buffer through the streaming recognizer
|
||||
(for the non-streaming ``transcribe()`` / partial re-decode path)."""
|
||||
import numpy as np
|
||||
s = self._rec.create_stream()
|
||||
s.accept_waveform(sr, samples)
|
||||
tail = np.zeros(int(0.5 * sr), dtype=np.float32)
|
||||
s.accept_waveform(sr, tail)
|
||||
s.input_finished()
|
||||
while self._rec.is_ready(s):
|
||||
self._rec.decode_stream(s)
|
||||
return (self._rec.get_result(s) or "").strip()
|
||||
|
||||
def unload(self) -> None:
|
||||
self._rec = None
|
||||
import gc
|
||||
gc.collect()
|
||||
|
||||
|
||||
def _sherpa_result(text: str, samples, sr) -> dict:
|
||||
"""Normalise a sherpa decode to OmniVoice's ``{chunks, segments, language,
|
||||
text}`` contract. sherpa gives plain text (no VAD split), so emit a single
|
||||
segment spanning the buffer — same shape Moonshine uses."""
|
||||
text = (text or "").strip()
|
||||
try:
|
||||
duration = round(len(samples) / float(sr), 3)
|
||||
except Exception:
|
||||
duration = None
|
||||
segments = []
|
||||
if text:
|
||||
segments.append({"text": text, "start": 0.0, "end": duration, "words": []})
|
||||
chunks = [{"text": s["text"], "timestamp": (s["start"], s["end"])} for s in segments]
|
||||
return {"chunks": chunks, "segments": segments, "language": "auto", "text": text}
|
||||
|
||||
|
||||
# ── Registry ────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
@@ -1070,6 +1504,7 @@ _REGISTRY: dict[str, type[ASRBackend]] = _LazyASRRegistry({
|
||||
"nemo-parakeet": NeMoASRBackend,
|
||||
"moonshine": MoonshineASRBackend,
|
||||
"funasr": FunASRBackend,
|
||||
"sherpa-onnx-asr": SherpaDictationBackend,
|
||||
# "faster-whisper-isolated": resolved lazily (crash-isolated subprocess).
|
||||
})
|
||||
|
||||
@@ -1084,6 +1519,7 @@ _INSTALL_HINTS: dict[str, str] = {
|
||||
"nemo-parakeet": "pip install nemo_toolkit[asr] (NVIDIA Parakeet; CUDA or CPU)",
|
||||
"moonshine": "pip install useful-moonshine (edge/CPU-optimized ASR)",
|
||||
"funasr": "pip install funasr (SenseVoiceSmall + FSMN-VAD; CUDA or CPU)",
|
||||
"sherpa-onnx-asr": "uv add sherpa-onnx (ONNX live dictation; CPU, cross-platform)",
|
||||
}
|
||||
|
||||
# Most-recent failure per backend, so a transient probe error survives between
|
||||
@@ -1138,6 +1574,22 @@ def list_backends() -> list[dict]:
|
||||
return out
|
||||
|
||||
|
||||
def _probe_available(cls) -> bool:
|
||||
"""``is_available()`` that never raises. A probe that explodes (e.g. a native
|
||||
lib that refuses to load — CTranslate2's exec-stack rejection, #692) means the
|
||||
engine is unusable on this host, so treat it as unavailable and fall through
|
||||
to the next candidate rather than crash engine selection."""
|
||||
try:
|
||||
ok, _ = cls.is_available()
|
||||
return bool(ok)
|
||||
except Exception: # noqa: BLE001
|
||||
logger.warning(
|
||||
"ASR auto-detect: %s.is_available() raised — treating as unavailable",
|
||||
cls.__name__, exc_info=True,
|
||||
)
|
||||
return False
|
||||
|
||||
|
||||
def _auto_detect() -> str:
|
||||
"""Pick the best available ASR engine for the current hardware.
|
||||
|
||||
@@ -1156,17 +1608,14 @@ def _auto_detect() -> str:
|
||||
4. pytorch-whisper — last resort; requires the TTS model to be loaded
|
||||
so it can reuse `_asr_pipe`.
|
||||
"""
|
||||
ok, _ = WhisperXBackend.is_available()
|
||||
if ok:
|
||||
if _probe_available(WhisperXBackend):
|
||||
return "whisperx"
|
||||
ok, _ = FasterWhisperBackend.is_available()
|
||||
if ok:
|
||||
if _probe_available(FasterWhisperBackend):
|
||||
return "faster-whisper"
|
||||
try:
|
||||
import torch
|
||||
if hasattr(torch.backends, "mps") and torch.backends.mps.is_available():
|
||||
ok, _ = MLXWhisperBackend.is_available()
|
||||
if ok:
|
||||
if _probe_available(MLXWhisperBackend):
|
||||
return "mlx-whisper"
|
||||
except Exception:
|
||||
pass
|
||||
@@ -1238,40 +1687,91 @@ def transcribe_reference(audio_path: str) -> str | None:
|
||||
|
||||
|
||||
_capture_backend: ASRBackend | None = None
|
||||
# The sherpa model id the cached capture backend was built for, so a model
|
||||
# switch in Settings rebuilds the singleton instead of serving the old model.
|
||||
_capture_backend_key: str | None = None
|
||||
|
||||
|
||||
def dictation_model_id() -> str | None:
|
||||
"""The selected sherpa dictation model id, or None when dictation is off /
|
||||
no sherpa model is chosen. Env var wins (power-user pin), then prefs."""
|
||||
explicit = os.environ.get("OMNIVOICE_SHERPA_ASR_MODEL")
|
||||
if explicit:
|
||||
return explicit
|
||||
try:
|
||||
from core import prefs
|
||||
if not prefs.get("dictation.enabled", True):
|
||||
return None
|
||||
mid = prefs.get("dictation.model_id")
|
||||
except Exception:
|
||||
return None
|
||||
from services.sherpa_dictation import is_sherpa_model
|
||||
return mid if is_sherpa_model(mid) else None
|
||||
|
||||
|
||||
def get_capture_asr_backend() -> ASRBackend:
|
||||
"""Pick the fastest ASR engine for capture / dictation.
|
||||
|
||||
Priority order (speed-first — word alignment is unnecessary for
|
||||
dictation, so we skip WhisperX's forced-alignment overhead):
|
||||
Selection order:
|
||||
|
||||
1. mlx-whisper Turbo — Apple Silicon, ~5× faster than large-v3
|
||||
2. mlx-whisper large — still native Metal, faster than CPU int8
|
||||
3. faster-whisper — cross-platform CTranslate2 fallback
|
||||
4. pytorch-whisper — last resort
|
||||
0. sherpa-onnx dictation — when ``dictation.model_id`` names one of the
|
||||
seven sherpa models (live/CPU; the new live-dictation path).
|
||||
1. mlx-whisper Turbo — Apple Silicon, ~5× faster than large-v3
|
||||
2. mlx-whisper large — still native Metal, faster than CPU int8
|
||||
3. faster-whisper — cross-platform CTranslate2 fallback
|
||||
4. pytorch-whisper — last resort
|
||||
|
||||
The caller should also pass ``word_timestamps=False`` to the returned
|
||||
backend to skip per-word timing and shave another ~30% latency.
|
||||
|
||||
Returns a cached singleton so the model stays warm between calls.
|
||||
Returns a cached singleton so the model stays warm between calls; the
|
||||
singleton is rebuilt if the selected sherpa model changes.
|
||||
"""
|
||||
global _capture_backend
|
||||
if _capture_backend is not None:
|
||||
global _capture_backend, _capture_backend_key
|
||||
|
||||
# 0. Honor an explicit sherpa dictation model selection.
|
||||
sherpa_id = dictation_model_id()
|
||||
if sherpa_id:
|
||||
ok, _ = SherpaDictationBackend.is_available()
|
||||
if ok:
|
||||
if not (isinstance(_capture_backend, SherpaDictationBackend)
|
||||
and _capture_backend_key == sherpa_id):
|
||||
try:
|
||||
_capture_backend = SherpaDictationBackend(model_id=sherpa_id)
|
||||
_capture_backend_key = sherpa_id
|
||||
except Exception as e: # noqa: BLE001 — fall through to Whisper
|
||||
logger.warning(
|
||||
"sherpa dictation model %r unavailable (%s) — falling "
|
||||
"back to Whisper capture engine", sherpa_id, e,
|
||||
)
|
||||
_capture_backend = None
|
||||
_capture_backend_key = None
|
||||
if _capture_backend is not None:
|
||||
return _capture_backend
|
||||
else:
|
||||
logger.info(
|
||||
"dictation.model_id=%r selected but sherpa-onnx not installed — "
|
||||
"falling back to Whisper capture engine", sherpa_id,
|
||||
)
|
||||
|
||||
if _capture_backend is not None and _capture_backend_key is None:
|
||||
return _capture_backend
|
||||
|
||||
# Prefer MLX Turbo on Apple Silicon
|
||||
ok, _ = MLXWhisperBackend.is_available()
|
||||
if ok:
|
||||
_capture_backend = MLXWhisperBackend(model_name=_MLX_MODEL_TURBO)
|
||||
_capture_backend_key = None
|
||||
return _capture_backend
|
||||
|
||||
# Fall back to faster-whisper (CPU int8 on non-Apple)
|
||||
ok, _ = FasterWhisperBackend.is_available()
|
||||
if ok:
|
||||
_capture_backend = FasterWhisperBackend()
|
||||
_capture_backend_key = None
|
||||
return _capture_backend
|
||||
|
||||
# Last resort
|
||||
_capture_backend = PyTorchWhisperBackend()
|
||||
_capture_backend_key = None
|
||||
return _capture_backend
|
||||
|
||||
@@ -65,10 +65,14 @@ class AudiobookPlan:
|
||||
def char_count(self) -> int:
|
||||
return sum(c.char_count for c in self.chapters)
|
||||
|
||||
@property
|
||||
def chapter_count(self) -> int:
|
||||
return len(self.chapters)
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return {
|
||||
"chapters": [c.to_dict() for c in self.chapters],
|
||||
"chapter_count": len(self.chapters),
|
||||
"chapter_count": self.chapter_count,
|
||||
"char_count": self.char_count,
|
||||
}
|
||||
|
||||
|
||||
@@ -42,6 +42,46 @@ _ABBREVIATIONS = frozenset({
|
||||
# [pause 300ms] markers). The splitter must never cut inside one.
|
||||
_BRACKET_TAG_RE = re.compile(r"\[[^\]]*\]")
|
||||
|
||||
# Dense scripts (CJK ideographs, kana, Hangul) where ~1 character = 1 syllable,
|
||||
# so an N-char chunk is far more *speech* than N Latin chars. Counted by code
|
||||
# point (see _dense_char_count) so there are no literal CJK chars in source.
|
||||
def _dense_char_count(text: str) -> int:
|
||||
"""Number of CJK / kana / Hangul characters in *text* (dense scripts)."""
|
||||
n = 0
|
||||
for ch in text:
|
||||
o = ord(ch)
|
||||
if (0x3040 <= o <= 0x30FF or 0x3400 <= o <= 0x4DBF
|
||||
or 0x4E00 <= o <= 0x9FFF or 0xAC00 <= o <= 0xD7AF
|
||||
or 0xF900 <= o <= 0xFAFF):
|
||||
n += 1
|
||||
return n
|
||||
|
||||
# A chunk that is predominantly dense-script (>= this fraction) gets the smaller
|
||||
# limit; below it, the text is mostly spaced/Latin and the full limit applies.
|
||||
_DENSE_FRACTION_THRESHOLD = 0.3
|
||||
# Speech-per-char multiplier for dense scripts vs Latin (~1 ideograph ≈ 2.5
|
||||
# Latin chars of audio). Used to scale the char limit down.
|
||||
_DENSE_SPEECH_FACTOR = 2.5
|
||||
|
||||
|
||||
def _effective_max_chars(text: str, max_chars: int) -> int:
|
||||
"""Scale *max_chars* down for dense-script text (#505).
|
||||
|
||||
Long-form (5+ min) generation degrades — repeated / skipped / mispronounced
|
||||
words — when a single chunk's acoustic sequence gets too long. With CJK /
|
||||
kana / Hangul, ~1 char = 1 syllable, so an 800-char chunk is ~4-5 minutes of
|
||||
audio in one shot, well past the model's reliable range. When a chunk is
|
||||
predominantly dense-script, cap it to ``max_chars / _DENSE_SPEECH_FACTOR``
|
||||
(floored) so each chunk's spoken length stays bounded. Latin / spaced text
|
||||
is unchanged. ``max_chars <= 0`` (chunking disabled) is left untouched.
|
||||
"""
|
||||
if max_chars <= 0 or not text:
|
||||
return max_chars
|
||||
dense = _dense_char_count(text)
|
||||
if dense and dense / len(text) >= _DENSE_FRACTION_THRESHOLD:
|
||||
return max(120, min(max_chars, round(max_chars / _DENSE_SPEECH_FACTOR)))
|
||||
return max_chars
|
||||
|
||||
|
||||
def split_text_into_chunks(text: str, max_chars: int = DEFAULT_MAX_CHUNK_CHARS) -> List[str]:
|
||||
"""Split *text* at natural boundaries into chunks of at most *max_chars*.
|
||||
@@ -54,6 +94,9 @@ def split_text_into_chunks(text: str, max_chars: int = DEFAULT_MAX_CHUNK_CHARS)
|
||||
text = text.strip()
|
||||
if not text:
|
||||
return []
|
||||
# #505: dense-script text packs far more speech per char, so cap the chunk
|
||||
# smaller to keep each chunk's spoken length in the model's reliable range.
|
||||
max_chars = _effective_max_chars(text, max_chars)
|
||||
if max_chars <= 0 or len(text) <= max_chars:
|
||||
return [text]
|
||||
|
||||
|
||||
@@ -435,6 +435,60 @@ def _ensure_browser_playable_mp4(video_path: str) -> str:
|
||||
return video_path
|
||||
|
||||
|
||||
# Bounded retry for transient download failures (#579/#598). yt-dlp's own
|
||||
# `retries`/`fragment_retries` cover per-fragment HTTP flakes, but a broken
|
||||
# pipe ([Errno 32]) raised while the write side of a pipe closes mid-stream
|
||||
# (a killed ffmpeg merge child, a CDN reset during muxing) aborts the whole
|
||||
# `extract_info` call and is NOT covered by them — so a single transient blip
|
||||
# failed the entire ingest with a raw "Broken pipe". We add a small download-
|
||||
# level retry on top, cleaning up the partial download between attempts so a
|
||||
# half-written `original.*` can't poison the next try.
|
||||
_YT_DOWNLOAD_RETRIES = 2 # total attempts = 1 + retries = 3
|
||||
|
||||
|
||||
def _is_transient_download_error(exc: BaseException) -> bool:
|
||||
"""True when a download failure is worth retrying (broken pipe / net drop).
|
||||
|
||||
Reuses the single failure taxonomy (`VIDEO_DOWNLOAD_NETWORK`) rather than a
|
||||
parallel keyword list, so "what counts as transient" stays single-sourced
|
||||
with the error-hint classification. ``BrokenPipeError``/``ConnectionError``
|
||||
are matched by class too, since a bare instance may be wrapped or re-raised
|
||||
with a stripped message that no longer contains "broken pipe".
|
||||
"""
|
||||
if isinstance(exc, (BrokenPipeError, ConnectionError)):
|
||||
return True
|
||||
return failure.classify(str(exc)) == "VIDEO_DOWNLOAD_NETWORK"
|
||||
|
||||
|
||||
# YouTube serves some videos' high-quality formats signature-protected to the
|
||||
# default player client, so the media download 403s even though extraction
|
||||
# worked. Forcing an alternate client commonly bypasses it; on a 403 we escalate
|
||||
# through these (in order) before giving up (#625).
|
||||
_YT_PLAYER_CLIENTS = ["tv", "android", "web_safari"]
|
||||
|
||||
|
||||
def _is_forbidden_download_error(exc: BaseException) -> bool:
|
||||
"""True for an HTTP 403 — not transient (the same client keeps 403ing), but
|
||||
often fixable by switching the YouTube player client."""
|
||||
s = str(exc)
|
||||
return "403" in s or "Forbidden" in s
|
||||
|
||||
|
||||
def _cleanup_partial_download(job_dir: str) -> None:
|
||||
"""Remove any half-written `original.*` files before a retry.
|
||||
|
||||
A partial download left on disk would otherwise be picked up as a "finished"
|
||||
file by the post-download codec probe, or collide with the next attempt's
|
||||
output. Best-effort — never raises on the failure path.
|
||||
"""
|
||||
import glob
|
||||
for stale in glob.glob(os.path.join(job_dir, "original.*")):
|
||||
try:
|
||||
os.remove(stale)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def yt_download_sync(
|
||||
url: str,
|
||||
job_dir: str,
|
||||
@@ -480,6 +534,12 @@ def yt_download_sync(
|
||||
"quiet": True,
|
||||
"no_warnings": True,
|
||||
"restrictfilenames": True,
|
||||
# Don't stamp the downloaded file's mtime with the video's upload date
|
||||
# (#642): on Windows an out-of-range/invalid timestamp makes the os.utime
|
||||
# call raise `[Errno 22] Invalid argument`, failing the whole ingest. We
|
||||
# download to a throwaway `original.*` and never use its mtime, so skip
|
||||
# it entirely (equivalent to yt-dlp's --no-mtime).
|
||||
"updatetime": False,
|
||||
"socket_timeout": 30,
|
||||
# Resilience against YouTube CDN flakes: a single empty fragment
|
||||
# (commonly the very last one — "Did not get any data blocks")
|
||||
@@ -491,17 +551,64 @@ def yt_download_sync(
|
||||
"extractor_retries": 5,
|
||||
"skip_unavailable_fragments": True,
|
||||
}
|
||||
# #712: the format selector above pulls separate video+audio streams, so
|
||||
# yt-dlp muxes them via ffmpeg (merge_output_format=mp4). yt-dlp only looks
|
||||
# for ffmpeg on PATH and aborts with "you have requested merging of multiple
|
||||
# formats but ffmpeg is not installed" — but OmniVoice's ffmpeg is often a
|
||||
# bundled Tauri sidecar / imageio-ffmpeg binary that isn't on PATH (common on
|
||||
# Windows). Point yt-dlp at the exact ffmpeg we resolve so the merge works.
|
||||
_ffmpeg_bin = find_ffmpeg()
|
||||
if _ffmpeg_bin:
|
||||
ydl_opts["ffmpeg_location"] = _ffmpeg_bin
|
||||
if progress_hook is not None:
|
||||
ydl_opts["progress_hooks"] = [progress_hook]
|
||||
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
|
||||
info = ydl.extract_info(url, download=True)
|
||||
path = ydl.prepare_filename(info)
|
||||
root, _ = os.path.splitext(path)
|
||||
mp4 = root + ".mp4"
|
||||
if os.path.exists(mp4):
|
||||
video_path = mp4
|
||||
else:
|
||||
video_path = path
|
||||
|
||||
# Download with a bounded retry on transient/broken-pipe-class failures
|
||||
# (#579/#598). A broken pipe mid-mux isn't recoverable inside yt-dlp's own
|
||||
# fragment retries, but a fresh `extract_info` usually succeeds. Between
|
||||
# attempts we wipe the partial `original.*` so a half-written file can't be
|
||||
# mistaken for a finished download.
|
||||
info = None
|
||||
path = None
|
||||
transient_used = 0
|
||||
client_idx = 0
|
||||
while True:
|
||||
try:
|
||||
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
|
||||
info = ydl.extract_info(url, download=True)
|
||||
path = ydl.prepare_filename(info)
|
||||
break
|
||||
except Exception as exc:
|
||||
_cleanup_partial_download(job_dir)
|
||||
# 403 Forbidden: not transient — escalate the YouTube player client,
|
||||
# which commonly bypasses a signature-protected format set (#625).
|
||||
if _is_forbidden_download_error(exc) and client_idx < len(_YT_PLAYER_CLIENTS):
|
||||
client = _YT_PLAYER_CLIENTS[client_idx]
|
||||
client_idx += 1
|
||||
ydl_opts = {**ydl_opts, "extractor_args": {"youtube": {"player_client": [client]}}}
|
||||
logger.warning(
|
||||
"Download 403 for %s — retrying with player_client=%s (#625)", url, client,
|
||||
)
|
||||
continue
|
||||
# Transient/broken-pipe: a fresh extract_info usually succeeds
|
||||
# (#579/#598). A 403 never counts here — it's escalated above.
|
||||
if (transient_used < _YT_DOWNLOAD_RETRIES
|
||||
and _is_transient_download_error(exc)
|
||||
and not _is_forbidden_download_error(exc)):
|
||||
transient_used += 1
|
||||
logger.warning(
|
||||
"Transient download failure for %s (attempt %d/%d): %s — retrying",
|
||||
url, transient_used, _YT_DOWNLOAD_RETRIES, exc,
|
||||
)
|
||||
time.sleep(2 * transient_used) # brief, increasing backoff
|
||||
continue
|
||||
raise
|
||||
root, _ = os.path.splitext(path)
|
||||
mp4 = root + ".mp4"
|
||||
if os.path.exists(mp4):
|
||||
video_path = mp4
|
||||
else:
|
||||
video_path = path
|
||||
# Browser-playability guard: WKWebView (Tauri on macOS) refuses to
|
||||
# decode VP9/AV1 video and Opus audio even when they're wrapped in an
|
||||
# mp4 container, and refuses .webm/.mkv outright. We probe the actual
|
||||
|
||||
@@ -76,33 +76,42 @@ class OpenAICompatBackend(LLMBackend):
|
||||
import openai # noqa: F401
|
||||
except ImportError:
|
||||
return False, "openai package missing (install with `pip install openai`)."
|
||||
base_url = os.environ.get("TRANSLATE_BASE_URL")
|
||||
api_key = (
|
||||
os.environ.get("TRANSLATE_API_KEY")
|
||||
or os.environ.get("OPENAI_API_KEY")
|
||||
or ("local" if base_url else None)
|
||||
)
|
||||
if not api_key:
|
||||
# Resolve through the provider registry — the active provider carries
|
||||
# its own base_url/key/model. Legacy single-endpoint setups (a lone
|
||||
# TRANSLATE_BASE_URL) resolve to the "custom" provider, so this stays
|
||||
# backward-compatible with pre-registry configs.
|
||||
from services import llm_providers
|
||||
p = llm_providers.active_provider()
|
||||
if p is None:
|
||||
return False, (
|
||||
"No LLM configured. Set TRANSLATE_BASE_URL (+ TRANSLATE_API_KEY) to "
|
||||
"point at OpenAI, Ollama (http://localhost:11434/v1), or any compatible host."
|
||||
"No LLM configured. Add a provider key in Settings → LLM Providers "
|
||||
"(OpenAI/OpenRouter/Groq/… or a local Ollama), or set "
|
||||
"TRANSLATE_BASE_URL (+ TRANSLATE_API_KEY)."
|
||||
)
|
||||
return True, "ready"
|
||||
if not llm_providers.resolve_base_url(p):
|
||||
return False, f"{p.display_name}: set a Base URL in Settings → LLM Providers."
|
||||
if not llm_providers.has_key(p):
|
||||
return False, f"{p.display_name}: add an API key in Settings → LLM Providers."
|
||||
return True, f"ready ({p.display_name})"
|
||||
|
||||
@property
|
||||
def model_name(self) -> str:
|
||||
from services import llm_providers
|
||||
p = llm_providers.active_provider()
|
||||
if p is not None:
|
||||
return llm_providers.resolve_model(p)
|
||||
return os.environ.get("TRANSLATE_MODEL", "gpt-4o-mini")
|
||||
|
||||
def _get_client(self):
|
||||
if self._client is not None:
|
||||
return self._client
|
||||
from openai import OpenAI
|
||||
base_url = os.environ.get("TRANSLATE_BASE_URL")
|
||||
api_key = (
|
||||
os.environ.get("TRANSLATE_API_KEY")
|
||||
or os.environ.get("OPENAI_API_KEY")
|
||||
or ("local" if base_url else None)
|
||||
)
|
||||
from services import llm_providers
|
||||
p = llm_providers.active_provider()
|
||||
if p is None:
|
||||
raise RuntimeError("LLM not configured. See `is_available()` for the hint.")
|
||||
base_url = llm_providers.resolve_base_url(p)
|
||||
api_key = llm_providers.resolve_api_key(p)
|
||||
if not api_key:
|
||||
raise RuntimeError("LLM not configured. See `is_available()` for the hint.")
|
||||
kw = {"api_key": api_key}
|
||||
|
||||
@@ -0,0 +1,311 @@
|
||||
"""LLM provider registry — the OpenAI-compatible providers OmniVoice can use
|
||||
for Cinematic / Autofit translation (and any future LLM feature).
|
||||
|
||||
Every provider here speaks the OpenAI chat-completions shape, so a single
|
||||
client (`llm_backend.OpenAICompatBackend`) drives all of them — the only
|
||||
per-provider differences are ``base_url``, ``model``, and the API key. This
|
||||
module is the one place that knows those defaults and resolves the live value
|
||||
for the *active* provider.
|
||||
|
||||
Resolution precedence for every field (key / base_url / model), highest first:
|
||||
1. Environment variable — power-user / `.env` override, wins always.
|
||||
2. Encrypted settings store (UI-entered) — `settings_store.get_secret` for
|
||||
keys, `get_text` for base_url/model overrides.
|
||||
3. Built-in default from the table below.
|
||||
|
||||
Local providers (Ollama, LM Studio) need no key — a "local" sentinel is used
|
||||
so the OpenAI client is happy. This keeps the local-first path fully offline:
|
||||
nothing is sent anywhere unless the user picks a remote provider *and* a
|
||||
feature gate (quality="cinematic"/"autofit") fires.
|
||||
|
||||
Keys entered in the UI are stored **encrypted** (never in `.env`, never
|
||||
returned to the client). `.env` keys remain a valid override for CI / power
|
||||
users.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from dataclasses import dataclass
|
||||
from typing import Optional
|
||||
|
||||
# Settings-store row names (non-secret overrides live in the plaintext table;
|
||||
# keys live in the encrypted secret table under ``llm_key.<id>``).
|
||||
_ACTIVE_PROVIDER_KEY = "llm.active_provider"
|
||||
_BASE_URL_KEY = "llm.base_url." # + provider id
|
||||
_MODEL_KEY = "llm.model." # + provider id
|
||||
SECRET_PREFIX = "llm_key." # + provider id → settings_store secret name
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Provider:
|
||||
id: str
|
||||
display_name: str
|
||||
default_base_url: str
|
||||
default_model: str
|
||||
# Env var names checked (in order) for the API key. First one set wins.
|
||||
key_envs: tuple[str, ...] = ()
|
||||
base_url_env: Optional[str] = None
|
||||
model_env: Optional[str] = None
|
||||
local: bool = False # runs on the user's machine → no key, offline
|
||||
# Key optional when a base_url is set (self-hosted OpenAI-compatible servers
|
||||
# — vLLM, LM Studio behind a custom URL — often ignore the key). Preserves
|
||||
# the pre-registry behaviour where a lone TRANSLATE_BASE_URL was usable
|
||||
# keyless.
|
||||
key_optional: bool = False
|
||||
needs_account: bool = False # Cloudflare: base_url needs an account id
|
||||
account_env: Optional[str] = None
|
||||
signup_url: str = ""
|
||||
notes: str = ""
|
||||
|
||||
|
||||
# Order here is the display order in the settings page. OpenAI first (the
|
||||
# canonical), then the free/fast cloud providers from the shipped .env, then
|
||||
# the local engines, then Custom.
|
||||
_PROVIDERS: tuple[Provider, ...] = (
|
||||
Provider("openai", "OpenAI", "https://api.openai.com/v1", "gpt-4o-mini",
|
||||
key_envs=("OPENAI_API_KEY", "TRANSLATE_API_KEY"),
|
||||
base_url_env="OPENAI_BASE_URL", model_env="OPENAI_MODEL",
|
||||
signup_url="https://platform.openai.com/api-keys",
|
||||
notes="GPT-4o / o-series. Highest quality; paid."),
|
||||
Provider("openrouter", "OpenRouter", "https://openrouter.ai/api/v1",
|
||||
"openai/gpt-4o-mini",
|
||||
key_envs=("OPENROUTER_API_KEY",), base_url_env="OPENROUTER_BASE_URL",
|
||||
model_env="OPENROUTER_MODEL",
|
||||
signup_url="https://openrouter.ai/keys",
|
||||
notes="One key, hundreds of models incl. free tiers."),
|
||||
Provider("groq", "Groq", "https://api.groq.com/openai/v1",
|
||||
"llama-3.3-70b-versatile",
|
||||
key_envs=("GROQ_API_KEY",), base_url_env="GROQ_BASE_URL",
|
||||
model_env="GROQ_MODEL", signup_url="https://console.groq.com/keys",
|
||||
notes="Very fast Llama/Mixtral inference. Generous free tier."),
|
||||
Provider("cerebras", "Cerebras", "https://api.cerebras.ai/v1",
|
||||
"llama-3.3-70b",
|
||||
key_envs=("CEREBRAS_API_KEY",), base_url_env="CEREBRAS_BASE_URL",
|
||||
model_env="CEREBRAS_MODEL", signup_url="https://cloud.cerebras.ai",
|
||||
notes="Fastest Llama inference. Free tier."),
|
||||
Provider("google-ai", "Google AI (Gemini)",
|
||||
"https://generativelanguage.googleapis.com/v1beta/openai",
|
||||
"gemini-2.0-flash",
|
||||
key_envs=("GOOGLE_AI_API_KEY",), base_url_env="GOOGLE_AI_BASE_URL",
|
||||
model_env="GOOGLE_AI_MODEL",
|
||||
signup_url="https://aistudio.google.com/app/apikey",
|
||||
notes="Gemini via OpenAI-compatible endpoint. Free tier."),
|
||||
Provider("mistral", "Mistral", "https://api.mistral.ai/v1",
|
||||
"mistral-small-latest",
|
||||
key_envs=("MISTRAL_API_KEY",), base_url_env="MISTRAL_BASE_URL",
|
||||
model_env="MISTRAL_MODEL", signup_url="https://console.mistral.ai/api-keys",
|
||||
notes="Strong multilingual models. Free tier."),
|
||||
Provider("cohere", "Cohere", "https://api.cohere.ai/compatibility/v1",
|
||||
"command-r-08-2024",
|
||||
key_envs=("COHERE_API_KEY",), base_url_env="COHERE_BASE_URL",
|
||||
model_env="COHERE_MODEL", signup_url="https://dashboard.cohere.com/api-keys",
|
||||
notes="Command models; good for RAG/translation. Free trial keys."),
|
||||
Provider("nvidia", "NVIDIA NIM", "https://integrate.api.nvidia.com/v1",
|
||||
"meta/llama-3.3-70b-instruct",
|
||||
key_envs=("NVIDIA_API_KEY",), base_url_env="NVIDIA_BASE_URL",
|
||||
model_env="NVIDIA_MODEL", signup_url="https://build.nvidia.com",
|
||||
notes="NIM-hosted open models. Free credits."),
|
||||
Provider("github-models", "GitHub Models",
|
||||
"https://models.github.ai/inference", "openai/gpt-4o-mini",
|
||||
key_envs=("GITHUB_MODELS_API_KEY",), base_url_env="GITHUB_MODELS_BASE_URL",
|
||||
model_env="GITHUB_MODELS_MODEL",
|
||||
signup_url="https://github.com/settings/tokens",
|
||||
notes="Uses a GitHub PAT. Free for dev, rate-limited."),
|
||||
Provider("cloudflare", "Cloudflare Workers AI",
|
||||
"https://api.cloudflare.com/client/v4/accounts/{account_id}/ai/v1",
|
||||
"@cf/meta/llama-3.3-70b-instruct-fp8-fast",
|
||||
key_envs=("CLOUDFLARE_API_KEY",), base_url_env="CLOUDFLARE_BASE_URL",
|
||||
model_env="CLOUDFLARE_MODEL", needs_account=True,
|
||||
account_env="CLOUDFLARE_ACCOUNT_ID",
|
||||
signup_url="https://dash.cloudflare.com/profile/api-tokens",
|
||||
notes="Needs an Account ID. Free tier."),
|
||||
Provider("huggingface", "Hugging Face", "https://router.huggingface.co/v1",
|
||||
"meta-llama/Llama-3.3-70B-Instruct",
|
||||
key_envs=("HUGGINGFACE_API_KEY", "HF_TOKEN"),
|
||||
base_url_env="HUGGINGFACE_BASE_URL", model_env="HUGGINGFACE_MODEL",
|
||||
signup_url="https://huggingface.co/settings/tokens",
|
||||
notes="HF Inference router. Reuses your HF token."),
|
||||
Provider("sambanova", "SambaNova", "https://api.sambanova.ai/v1",
|
||||
"Meta-Llama-3.3-70B-Instruct",
|
||||
key_envs=("SAMBANOVA_API_KEY",), base_url_env="SAMBANOVA_BASE_URL",
|
||||
model_env="SAMBANOVA_MODEL", signup_url="https://cloud.sambanova.ai",
|
||||
notes="Fast open models. Free tier."),
|
||||
Provider("siliconflow", "SiliconFlow", "https://api.siliconflow.com/v1",
|
||||
"Qwen/Qwen2.5-7B-Instruct",
|
||||
key_envs=("SILICONFLOW_API_KEY",), base_url_env="SILICONFLOW_BASE_URL",
|
||||
model_env="SILICONFLOW_MODEL", signup_url="https://siliconflow.com",
|
||||
notes="Qwen/DeepSeek and more. Strong for CJK."),
|
||||
Provider("ollama", "Ollama (local)", "http://localhost:11434/v1",
|
||||
"llama3.1", local=True,
|
||||
base_url_env="OLLAMA_BASE_URL", model_env="OLLAMA_MODEL",
|
||||
signup_url="https://ollama.com",
|
||||
notes="Fully offline. Run `ollama pull llama3.1` first."),
|
||||
Provider("lmstudio", "LM Studio (local)", "http://localhost:1234/v1",
|
||||
"local-model", local=True,
|
||||
base_url_env="LMSTUDIO_BASE_URL", model_env="LMSTUDIO_MODEL",
|
||||
signup_url="https://lmstudio.ai",
|
||||
notes="Fully offline. Start the LM Studio local server."),
|
||||
Provider("custom", "Custom (OpenAI-compatible)", "", "",
|
||||
key_envs=("TRANSLATE_API_KEY",), base_url_env="TRANSLATE_BASE_URL",
|
||||
model_env="TRANSLATE_MODEL", key_optional=True,
|
||||
notes="Any OpenAI-compatible host. Set Base URL + Model (+ key)."),
|
||||
)
|
||||
|
||||
_BY_ID: dict[str, Provider] = {p.id: p for p in _PROVIDERS}
|
||||
|
||||
|
||||
def all_providers() -> tuple[Provider, ...]:
|
||||
return _PROVIDERS
|
||||
|
||||
|
||||
def get_provider(pid: str) -> Optional[Provider]:
|
||||
return _BY_ID.get(pid)
|
||||
|
||||
|
||||
# ── Field resolution (env → store → default) ──────────────────────────────
|
||||
|
||||
def _env_first(names: tuple[str, ...]) -> Optional[str]:
|
||||
for n in names:
|
||||
v = os.environ.get(n)
|
||||
if v:
|
||||
return v
|
||||
return None
|
||||
|
||||
|
||||
def resolve_base_url(p: Provider) -> str:
|
||||
from services import settings_store
|
||||
val = (
|
||||
(p.base_url_env and os.environ.get(p.base_url_env))
|
||||
or settings_store.get_text(_BASE_URL_KEY + p.id)
|
||||
or p.default_base_url
|
||||
)
|
||||
if p.needs_account and val and "{account_id}" in val:
|
||||
acct = (p.account_env and os.environ.get(p.account_env)) or \
|
||||
settings_store.get_text(f"llm.account.{p.id}") or ""
|
||||
val = val.replace("{account_id}", acct)
|
||||
return val or ""
|
||||
|
||||
|
||||
def resolve_model(p: Provider) -> str:
|
||||
from services import settings_store
|
||||
return (
|
||||
(p.model_env and os.environ.get(p.model_env))
|
||||
or settings_store.get_text(_MODEL_KEY + p.id)
|
||||
or p.default_model
|
||||
)
|
||||
|
||||
|
||||
def resolve_api_key(p: Provider) -> Optional[str]:
|
||||
"""Env key → encrypted stored key → 'local' sentinel for local/keyless."""
|
||||
from services import settings_store
|
||||
env_key = _env_first(p.key_envs)
|
||||
if env_key:
|
||||
return env_key
|
||||
stored = settings_store.get_secret(SECRET_PREFIX + p.id)
|
||||
if stored:
|
||||
return stored
|
||||
if p.local or (p.key_optional and resolve_base_url(p)):
|
||||
return "local" # self-hosted OpenAI-compatible servers ignore the key
|
||||
return None
|
||||
|
||||
|
||||
def has_key(p: Provider) -> bool:
|
||||
"""True if a usable key is resolvable (local, or keyless-with-base_url)."""
|
||||
if p.local:
|
||||
return True
|
||||
if _env_first(p.key_envs) or _key_in_store(p.id):
|
||||
return True
|
||||
return bool(p.key_optional and resolve_base_url(p))
|
||||
|
||||
|
||||
def _key_in_store(pid: str) -> bool:
|
||||
from services import settings_store
|
||||
return (SECRET_PREFIX + pid) in settings_store.list_secret_names()
|
||||
|
||||
|
||||
def is_configured(p: Provider) -> bool:
|
||||
"""Usable end-to-end: has a base_url (custom needs one set) and a key."""
|
||||
if not resolve_base_url(p):
|
||||
return False
|
||||
return has_key(p)
|
||||
|
||||
|
||||
# ── Active provider selection ─────────────────────────────────────────────
|
||||
|
||||
def active_provider_id() -> Optional[str]:
|
||||
"""The provider Cinematic/Autofit should use.
|
||||
|
||||
Precedence: env ``LLM_DEFAULT_PROVIDER`` → stored selection → first
|
||||
configured provider → None. Legacy ``TRANSLATE_BASE_URL`` users with no
|
||||
explicit selection resolve to ``custom`` (its envs are TRANSLATE_*).
|
||||
"""
|
||||
from services import settings_store
|
||||
env_pick = os.environ.get("LLM_DEFAULT_PROVIDER")
|
||||
if env_pick and env_pick in _BY_ID:
|
||||
return env_pick
|
||||
stored = settings_store.get_text(_ACTIVE_PROVIDER_KEY)
|
||||
if stored and stored in _BY_ID:
|
||||
return stored
|
||||
# Legacy: a lone TRANSLATE_BASE_URL means the old single-endpoint setup.
|
||||
if os.environ.get("TRANSLATE_BASE_URL"):
|
||||
return "custom"
|
||||
# Auto-select only a provider with a real key. Local providers (Ollama/
|
||||
# LM Studio) are *always* "configured" (no key needed) but we must NOT
|
||||
# assume their server is running — they require an explicit selection.
|
||||
for p in _PROVIDERS:
|
||||
if not p.local and is_configured(p):
|
||||
return p.id
|
||||
return None
|
||||
|
||||
|
||||
def set_active_provider(pid: str) -> None:
|
||||
from services import settings_store
|
||||
if pid not in _BY_ID:
|
||||
raise ValueError(f"unknown provider {pid!r}")
|
||||
settings_store.set_text(_ACTIVE_PROVIDER_KEY, pid)
|
||||
|
||||
|
||||
def active_provider() -> Optional[Provider]:
|
||||
pid = active_provider_id()
|
||||
return _BY_ID.get(pid) if pid else None
|
||||
|
||||
|
||||
# ── UI + persistence helpers ──────────────────────────────────────────────
|
||||
|
||||
def save_key(pid: str, api_key: str) -> None:
|
||||
"""Persist (encrypted) or clear an API key for a provider."""
|
||||
from services import settings_store
|
||||
if pid not in _BY_ID:
|
||||
raise ValueError(f"unknown provider {pid!r}")
|
||||
settings_store.set_secret(SECRET_PREFIX + pid, api_key or "")
|
||||
|
||||
|
||||
def save_overrides(pid: str, *, base_url: Optional[str] = None,
|
||||
model: Optional[str] = None,
|
||||
account_id: Optional[str] = None) -> None:
|
||||
from services import settings_store
|
||||
if pid not in _BY_ID:
|
||||
raise ValueError(f"unknown provider {pid!r}")
|
||||
if base_url is not None:
|
||||
settings_store.set_text(_BASE_URL_KEY + pid, base_url.strip())
|
||||
if model is not None:
|
||||
settings_store.set_text(_MODEL_KEY + pid, model.strip())
|
||||
if account_id is not None:
|
||||
settings_store.set_text(f"llm.account.{pid}", account_id.strip())
|
||||
|
||||
|
||||
def describe(p: Provider) -> dict:
|
||||
"""Client-safe provider descriptor — NEVER includes the key material."""
|
||||
return {
|
||||
"id": p.id,
|
||||
"display_name": p.display_name,
|
||||
"local": p.local,
|
||||
"needs_account": p.needs_account,
|
||||
"signup_url": p.signup_url,
|
||||
"notes": p.notes,
|
||||
"base_url": resolve_base_url(p),
|
||||
"model": resolve_model(p),
|
||||
"has_key": has_key(p),
|
||||
"key_from_env": bool(_env_first(p.key_envs)),
|
||||
"configured": is_configured(p),
|
||||
}
|
||||
@@ -60,7 +60,7 @@ def list_loaded() -> dict:
|
||||
models.append({
|
||||
"id": "tts",
|
||||
"name": "OmniVoice TTS",
|
||||
"checkpoint": os.environ.get("OMNIVOICE_MODEL", "k2-fsa/OmniVoice"),
|
||||
"checkpoint": mm.resolve_omnivoice_checkpoint(), # #693: effective checkpoint, not a leaked raw value
|
||||
"device": device,
|
||||
"vram_mb": round(_tts_vram_mb(), 1),
|
||||
"unloadable": True,
|
||||
|
||||
@@ -3,7 +3,7 @@ import time
|
||||
import asyncio
|
||||
import logging
|
||||
import threading
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from concurrent.futures import ThreadPoolExecutor, Executor
|
||||
|
||||
# ── Lazy imports ─────────────────────────────────────────────────────
|
||||
# torch and OmniVoice are heavy (~2-3s import on Apple Silicon).
|
||||
@@ -25,7 +25,16 @@ def _lazy_torch():
|
||||
def _lazy_omnivoice():
|
||||
global _OmniVoice
|
||||
if _OmniVoice is None:
|
||||
from omnivoice.models.omnivoice import OmniVoice as _OV
|
||||
try:
|
||||
from omnivoice.models.omnivoice import OmniVoice as _OV
|
||||
except ModuleNotFoundError:
|
||||
# The venv's editable install is missing/broken (#564). main.py wires
|
||||
# the source fallback at startup, but resolve it here too so the
|
||||
# model-load path self-heals and logs the paths it searched.
|
||||
from core.omnivoice_path import ensure_omnivoice_importable
|
||||
_backend_dir = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
ensure_omnivoice_importable(_backend_dir, logger)
|
||||
from omnivoice.models.omnivoice import OmniVoice as _OV
|
||||
_OmniVoice = _OV
|
||||
return _OmniVoice
|
||||
|
||||
@@ -35,17 +44,30 @@ from core.config import IDLE_TIMEOUT_SECONDS, CPU_POOL_WORKERS
|
||||
logger = logging.getLogger("omnivoice.model")
|
||||
|
||||
# Per-TTS-job VRAM headroom estimate. OmniVoice's forward + autoregressive
|
||||
# decode peaks around 1.6 GB on a 24 kHz 8-second utterance; we budget 2.5 GB
|
||||
# to leave room for the ASR/diarization pipelines that run concurrently in
|
||||
# the same process. Tuned empirically — bumps to 3 GB if anyone reports OOM
|
||||
# at 16 GB on a multi-segment dub.
|
||||
_GPU_VRAM_PER_JOB_GB = 2.5
|
||||
# decode peaks around 1.6 GB, but the interactive clone path co-loads WhisperX
|
||||
# large-v3 ASR (~3 GB) to transcribe the reference, so a *concurrent* clone job
|
||||
# is realistically ~5 GB. The old 2.5 GB budget over-committed: an 8 GB card
|
||||
# (~7 GB free) got 2 workers, and two concurrent clone jobs blew past VRAM into
|
||||
# a sticky CUDA "illegal memory access" that aborts the whole backend process —
|
||||
# the wave of "Can't reach the local backend" crash reports on 8 GB GPUs
|
||||
# (#567/#570/#571/#580/#582/#583/#584). Budgeting 5 GB serializes to 1 worker on
|
||||
# ≤10 GB cards (no contention → no crash) while 16/24 GB cards still parallelize.
|
||||
# Power users override with OMNIVOICE_GPU_WORKERS.
|
||||
_GPU_VRAM_PER_JOB_GB = 5.0
|
||||
_GPU_WORKER_CAP = 4
|
||||
|
||||
_gpu_pool_singleton: "ThreadPoolExecutor | None" = None
|
||||
_gpu_pool_singleton: "_ResilientGpuPool | None" = None
|
||||
_cpu_pool = ThreadPoolExecutor(max_workers=CPU_POOL_WORKERS)
|
||||
|
||||
|
||||
def _workers_for_free_vram(free_gb: float) -> int:
|
||||
"""GPU worker count for a given free-VRAM figure: free // per-job budget,
|
||||
floored at 1 and capped at _GPU_WORKER_CAP. Pure so the sizing policy is
|
||||
unit-tested without a GPU (the #567 crash hinged on this returning >1 on
|
||||
8 GB cards)."""
|
||||
return max(1, min(_GPU_WORKER_CAP, int(free_gb // _GPU_VRAM_PER_JOB_GB)))
|
||||
|
||||
|
||||
def _pick_gpu_workers() -> int:
|
||||
"""Pick a sensible GPU worker count from the runtime environment.
|
||||
|
||||
@@ -68,7 +90,7 @@ def _pick_gpu_workers() -> int:
|
||||
if hasattr(torch, "cuda") and torch.cuda.is_available():
|
||||
free_bytes, _total = torch.cuda.mem_get_info()
|
||||
free_gb = free_bytes / (1024 ** 3)
|
||||
workers = max(1, min(_GPU_WORKER_CAP, int(free_gb // _GPU_VRAM_PER_JOB_GB)))
|
||||
workers = _workers_for_free_vram(free_gb)
|
||||
logger.info(
|
||||
"GPU pool sized to %d worker(s) — %.1f GB free / %.1f GB per job (cap %d)",
|
||||
workers, free_gb, _GPU_VRAM_PER_JOB_GB, _GPU_WORKER_CAP,
|
||||
@@ -87,14 +109,82 @@ def _build_gpu_pool() -> ThreadPoolExecutor:
|
||||
return ThreadPoolExecutor(max_workers=workers, thread_name_prefix="gpu-pool")
|
||||
|
||||
|
||||
def _get_gpu_pool() -> ThreadPoolExecutor:
|
||||
"""Internal accessor. Same singleton as the module-level `_gpu_pool`
|
||||
attribute, but resolvable from inside this module (Python's module
|
||||
`__getattr__` only fires for unresolved lookups from *outside*).
|
||||
class _ResilientGpuPool(Executor):
|
||||
"""A stable, self-healing wrapper around the GPU `ThreadPoolExecutor`.
|
||||
|
||||
The crash this fixes (#589 #599): `_reset_gpu_pool()` shuts the pool down on
|
||||
a model-load timeout, but consumers that captured the executor *object* at
|
||||
import time (`from services.model_manager import _gpu_pool` at module level —
|
||||
generation, dub_generate, dub_core, dub_translate, openai_compat) kept
|
||||
submitting to the dead pool and got `RuntimeError: cannot schedule new
|
||||
futures after shutdown` on the next generate/dub/translate.
|
||||
|
||||
Making `_gpu_pool` a single long-lived wrapper whose *inner* pool is swapped
|
||||
means those references never go stale: every `submit()` resolves the live
|
||||
pool, and a submit that races a shutdown rebuilds once and retries. Building
|
||||
the inner pool stays lazy so we still size workers after torch's device
|
||||
probe (the reason for the original `__getattr__` indirection).
|
||||
"""
|
||||
|
||||
def __init__(self):
|
||||
self._pool: "ThreadPoolExecutor | None" = None
|
||||
self._lock = threading.Lock()
|
||||
|
||||
def _live_pool(self) -> ThreadPoolExecutor:
|
||||
pool = self._pool
|
||||
if pool is None:
|
||||
with self._lock:
|
||||
if self._pool is None:
|
||||
self._pool = _build_gpu_pool()
|
||||
pool = self._pool
|
||||
return pool
|
||||
|
||||
def submit(self, fn, /, *args, **kwargs):
|
||||
try:
|
||||
return self._live_pool().submit(fn, *args, **kwargs)
|
||||
except RuntimeError as e:
|
||||
# "cannot schedule new futures after shutdown": the inner pool was
|
||||
# reset (or torn down) under us. Rebuild once and retry so a stale
|
||||
# caller self-heals instead of 500-ing. (Interpreter-shutdown races
|
||||
# re-raise on the retry — we don't loop.)
|
||||
if "shutdown" not in str(e).lower():
|
||||
raise
|
||||
with self._lock:
|
||||
self._pool = _build_gpu_pool()
|
||||
pool = self._pool
|
||||
return pool.submit(fn, *args, **kwargs)
|
||||
|
||||
def reset(self) -> None:
|
||||
"""Abandon the current worker pool; the next submit builds a fresh one.
|
||||
|
||||
Python can't kill a thread wedged in a timed-out load, but dropping the
|
||||
poisoned pool means a retry gets a clean worker instead of queueing
|
||||
behind the wedged one. The wrapper identity is preserved, so references
|
||||
held by importers stay valid.
|
||||
"""
|
||||
with self._lock:
|
||||
pool, self._pool = self._pool, None
|
||||
if pool is not None:
|
||||
try:
|
||||
pool.shutdown(wait=False, cancel_futures=True)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
def shutdown(self, wait=True, *, cancel_futures=False):
|
||||
with self._lock:
|
||||
pool, self._pool = self._pool, None
|
||||
if pool is not None:
|
||||
pool.shutdown(wait=wait, cancel_futures=cancel_futures)
|
||||
|
||||
|
||||
def _get_gpu_pool() -> "_ResilientGpuPool":
|
||||
"""Internal accessor for the GPU pool singleton. Same object as the
|
||||
module-level `_gpu_pool` attribute, but resolvable from inside this module
|
||||
(Python's module `__getattr__` only fires for lookups from *outside*).
|
||||
"""
|
||||
global _gpu_pool_singleton
|
||||
if _gpu_pool_singleton is None:
|
||||
_gpu_pool_singleton = _build_gpu_pool()
|
||||
_gpu_pool_singleton = _ResilientGpuPool()
|
||||
return _gpu_pool_singleton
|
||||
|
||||
|
||||
@@ -108,6 +198,68 @@ def __getattr__(name: str):
|
||||
return _get_gpu_pool()
|
||||
raise AttributeError(f"module 'services.model_manager' has no attribute {name!r}")
|
||||
|
||||
|
||||
# ── GPU-job timeout guard (#730 class; residual #850/#802/#755 …) ─────
|
||||
# A blocking GPU job that wedges on a Windows+CUDA hang keeps occupying its
|
||||
# worker forever — run_in_executor can't cancel the thread. With a 1–2 worker
|
||||
# pool that starves *every* other request, so the next user action surfaces as
|
||||
# the misleading "Can't reach the local backend" even though the process is
|
||||
# alive. ASR/dub/model-load already bound+reset on hang (run_transcribe_guarded,
|
||||
# _reset_pool_on_wedge, _load_model_with_timeout); the TTS **generate** paths
|
||||
# (generation.py, tts_stream.py) were the last unguarded dispatch — and the
|
||||
# residual on-main reports all fail on generate:start (audio). This is the same
|
||||
# guard generalised so every GPU dispatch shares one recovery path.
|
||||
GPU_JOB_TIMEOUT_S = float(os.environ.get("OMNIVOICE_GENERATE_TIMEOUT_S", "300.0"))
|
||||
|
||||
|
||||
class GpuJobTimeoutError(TimeoutError):
|
||||
"""A GPU-pool job exceeded its wall-clock bound and was abandoned.
|
||||
|
||||
The backend is alive — the job was too heavy for the available compute
|
||||
(most often a VRAM-starved GPU). Pool capacity is restored automatically by
|
||||
resetting the pool; the message carries the durable fix.
|
||||
"""
|
||||
|
||||
|
||||
async def run_on_gpu_pool_guarded(fn, *, what: str = "GPU job",
|
||||
timeout: float = GPU_JOB_TIMEOUT_S,
|
||||
executor=None):
|
||||
"""Run blocking ``fn`` on the GPU pool with a hard wall-clock bound.
|
||||
|
||||
On timeout, ``reset()`` the pool (abandon the wedged worker so the next
|
||||
submit gets a fresh one) and raise :class:`GpuJobTimeoutError`. ``fn`` must
|
||||
be a zero-arg callable — wrap args with ``functools.partial`` at the call
|
||||
site. Deliberately mirrors ``asr_backend.run_transcribe_guarded`` so every
|
||||
GPU dispatch shares one bound+recover path (#730 class). Executors without
|
||||
``reset`` (a plain ThreadPoolExecutor in tests) still get the bound + error.
|
||||
"""
|
||||
loop = asyncio.get_running_loop()
|
||||
ex = executor if executor is not None else _get_gpu_pool()
|
||||
fut = loop.run_in_executor(ex, fn)
|
||||
try:
|
||||
return await asyncio.wait_for(fut, timeout=timeout)
|
||||
except asyncio.TimeoutError:
|
||||
_reset = getattr(ex, "reset", None)
|
||||
if callable(_reset):
|
||||
try:
|
||||
_reset()
|
||||
logger.warning(
|
||||
"%s exceeded %.0fs — abandoned the GPU-pool worker to "
|
||||
"restore capacity (#730).", what, timeout,
|
||||
)
|
||||
except Exception:
|
||||
logger.exception("GPU pool reset after %s timeout failed", what)
|
||||
raise GpuJobTimeoutError(
|
||||
f"{what} exceeded {timeout:.0f}s and was abandoned — the backend is "
|
||||
"running, but the job was too heavy for the available compute. Most "
|
||||
"often the GPU is VRAM-starved (a resident model and this job contend "
|
||||
"for memory). Capacity was restored automatically; for a durable fix "
|
||||
"try shorter text, a lighter engine, or set the engine to CPU in "
|
||||
"Settings → Models. (Raise OMNIVOICE_GENERATE_TIMEOUT_S for very long "
|
||||
"single generations.)"
|
||||
)
|
||||
|
||||
|
||||
model = None # type: ignore
|
||||
_model_lock = asyncio.Lock()
|
||||
_last_used = time.time()
|
||||
@@ -218,6 +370,19 @@ def get_best_device():
|
||||
compatible, warning = check_device_compatibility()
|
||||
if not compatible:
|
||||
logger.warning(warning)
|
||||
# #756: the GPU's compute capability isn't in this torch build's arch
|
||||
# list, so CUDA kernels can't launch ("no kernel image is available
|
||||
# for execution") — every generate would 500. Too-old (Pascal sm_61)
|
||||
# and too-new (Blackwell sm_120 on pre-cu128 wheels) both land here.
|
||||
# Fall back to CPU so the app WORKS (slowly) instead of dead-ending;
|
||||
# OMNIVOICE_FORCE_CUDA=1 overrides for users who installed a matching
|
||||
# torch and know the arch_list probe is wrong for their setup.
|
||||
if not _env_flag("OMNIVOICE_FORCE_CUDA"):
|
||||
logger.warning(
|
||||
"Falling back to CPU: this GPU is unsupported by the installed "
|
||||
"PyTorch build (set OMNIVOICE_FORCE_CUDA=1 to force CUDA anyway)."
|
||||
)
|
||||
return "cpu"
|
||||
return "cuda"
|
||||
|
||||
# ── Intel Arc / discrete GPU via IPEX ────────────────────────────
|
||||
@@ -449,6 +614,138 @@ def should_preload_tts_asr() -> bool:
|
||||
return _env_flag("OMNIVOICE_PRELOAD_TTS_ASR")
|
||||
|
||||
|
||||
def _is_incomplete_cache_error(exc: BaseException) -> bool:
|
||||
"""True when `exc` is the truncated-HF-cache class (#352 / #581).
|
||||
|
||||
transformers raises an OSError whose message contains "does not appear to
|
||||
have a file named …" when the on-disk snapshot has config/tokenizer files
|
||||
but no weight shard — the signature of an interrupted download. We match on
|
||||
that phrase (stable across transformers 4.x/5.x) rather than the error type,
|
||||
since the same OSError type covers unrelated I/O failures."""
|
||||
return "does not appear to have a file named" in str(exc)
|
||||
|
||||
|
||||
def _hf_offline() -> bool:
|
||||
"""Respect HF's offline switches so repair never makes a network call the
|
||||
user opted out of. `snapshot_download` would itself raise offline, but
|
||||
checking up front lets us skip straight to the actionable message."""
|
||||
return _env_flag("HF_HUB_OFFLINE") or _env_flag("TRANSFORMERS_OFFLINE")
|
||||
|
||||
|
||||
def _repair_model_cache(checkpoint: str, *, force: bool = False) -> bool:
|
||||
"""Re-fetch a checkpoint's missing files in place and report success.
|
||||
|
||||
An interrupted download leaves the cache missing only some files;
|
||||
`snapshot_download` resumes/fills exactly those (already-present, correctly
|
||||
sized blobs are skipped by hash, so a near-complete cache repairs in
|
||||
seconds and a complete one would no-op). Returns False — leaving the caller
|
||||
to surface the actionable delete-and-reinstall message — when repair is
|
||||
impossible (offline) or the re-fetch itself fails (no network, gated repo,
|
||||
full disk). Never raises; repair is best-effort.
|
||||
|
||||
``force=True`` passes ``force_download`` so the re-fetch replaces files that
|
||||
are *present but corrupt* — a truncated/garbled blob that still has the right
|
||||
size won't be re-fetched by the default resume (#739). It re-downloads the
|
||||
whole snapshot, so it's the last resort the load path only reaches after a
|
||||
plain resume-repair didn't fix the cache."""
|
||||
if _hf_offline():
|
||||
logger.warning(
|
||||
"Model cache for %s is incomplete but HF offline mode is set — "
|
||||
"cannot auto-repair.", checkpoint,
|
||||
)
|
||||
return False
|
||||
try:
|
||||
from huggingface_hub import snapshot_download
|
||||
except Exception as imp_err: # pragma: no cover - huggingface_hub is a hard dep
|
||||
logger.warning("Cannot import snapshot_download to repair cache: %s", imp_err)
|
||||
return False
|
||||
dl_kwargs: dict = {"repo_id": checkpoint}
|
||||
endpoint = os.environ.get("HF_ENDPOINT")
|
||||
if endpoint:
|
||||
dl_kwargs["endpoint"] = endpoint
|
||||
if force:
|
||||
# Replace present-but-corrupt blobs that resume would trust by size.
|
||||
dl_kwargs["force_download"] = True
|
||||
if os.name == "nt":
|
||||
# Match the install path (download.py): avoid symlinks on Windows.
|
||||
dl_kwargs["local_dir_use_symlinks"] = False
|
||||
|
||||
def _attempt() -> None:
|
||||
"""One snapshot_download, tolerating an hf_hub that rejects the optional
|
||||
symlink knob. Lets real failures (network, gated repo, disk) propagate."""
|
||||
try:
|
||||
snapshot_download(**dl_kwargs)
|
||||
except TypeError:
|
||||
# Older/newer huggingface_hub may not accept local_dir_use_symlinks
|
||||
# on a cache-only call — retry without the optional knob.
|
||||
dl_kwargs.pop("local_dir_use_symlinks", None)
|
||||
snapshot_download(**dl_kwargs)
|
||||
|
||||
# Bounded retries (#739): an incomplete cache *is* an interrupted download, so
|
||||
# a single transient blip mid-repair shouldn't drop the user back to a manual
|
||||
# delete-and-reinstall. snapshot_download resumes between attempts (present,
|
||||
# correctly-sized blobs are skipped by hash), so each retry continues where
|
||||
# the last left off — cheap and idempotent. Counts/backoff are env-tunable
|
||||
# for restricted networks and kept fast (backoff=0) in tests.
|
||||
try:
|
||||
retries = max(1, int(os.environ.get("OMNIVOICE_MODEL_REPAIR_RETRIES", "3")))
|
||||
except ValueError:
|
||||
retries = 3
|
||||
try:
|
||||
backoff = max(0.0, float(os.environ.get("OMNIVOICE_MODEL_REPAIR_BACKOFF_S", "2")))
|
||||
except ValueError:
|
||||
backoff = 2.0
|
||||
|
||||
logger.info(
|
||||
"Auto-repairing incomplete model cache for %s (up to %d attempt(s)) …",
|
||||
checkpoint, retries,
|
||||
)
|
||||
for attempt in range(1, retries + 1):
|
||||
try:
|
||||
_attempt()
|
||||
logger.info("Auto-repair of %s completed; retrying model load.", checkpoint)
|
||||
return True
|
||||
except Exception as e:
|
||||
logger.warning(
|
||||
"Auto-repair of %s attempt %d/%d failed: %s",
|
||||
checkpoint, attempt, retries, e,
|
||||
)
|
||||
if attempt < retries and backoff:
|
||||
time.sleep(backoff * attempt)
|
||||
return False
|
||||
|
||||
|
||||
_DEFAULT_OMNIVOICE_CHECKPOINT = "k2-fsa/OmniVoice"
|
||||
|
||||
|
||||
def resolve_omnivoice_checkpoint() -> str:
|
||||
"""Resolve the OmniVoice TTS checkpoint from ``OMNIVOICE_MODEL``, self-healing
|
||||
a misconfigured value.
|
||||
|
||||
A valid checkpoint is either a HuggingFace repo id (``org/repo`` — contains a
|
||||
``/``) or an existing local directory. A bare token like ``"omnivoice"`` — a
|
||||
TTS *engine id* that leaked into ``OMNIVOICE_MODEL`` (e.g. a stale pref/env) —
|
||||
is neither, and would crash model load with *"omnivoice is not a local folder
|
||||
and is not a valid model identifier listed on huggingface.co/models"* (#693).
|
||||
Fall back to the default rather than 500 on every launch.
|
||||
"""
|
||||
checkpoint = os.environ.get("OMNIVOICE_MODEL", _DEFAULT_OMNIVOICE_CHECKPOINT).strip()
|
||||
if not checkpoint:
|
||||
return _DEFAULT_OMNIVOICE_CHECKPOINT
|
||||
# Honor a HF repo id (org/repo) or an EXPLICIT local path (absolute, or with
|
||||
# a path separator). A bare token like "omnivoice" must NOT be treated as a
|
||||
# local dir even if a cwd-relative folder happens to share its name — that
|
||||
# is exactly the engine-id leak (#693), so self-heal to the default.
|
||||
if "/" in checkpoint or "\\" in checkpoint or os.path.isabs(checkpoint):
|
||||
return checkpoint
|
||||
logger.warning(
|
||||
"OMNIVOICE_MODEL=%r is not a HuggingFace repo id (org/repo) or a local "
|
||||
"path — falling back to %s (#693).",
|
||||
checkpoint, _DEFAULT_OMNIVOICE_CHECKPOINT,
|
||||
)
|
||||
return _DEFAULT_OMNIVOICE_CHECKPOINT
|
||||
|
||||
|
||||
def _load_model_sync():
|
||||
global model
|
||||
from utils.hf_progress import register_listener, unregister_listener
|
||||
@@ -475,7 +772,7 @@ def _load_model_sync():
|
||||
OmniVoice = _lazy_omnivoice()
|
||||
device = get_best_device()
|
||||
|
||||
checkpoint = os.environ.get("OMNIVOICE_MODEL", "k2-fsa/OmniVoice")
|
||||
checkpoint = resolve_omnivoice_checkpoint()
|
||||
_set_loading("loading_weights", f"Loading TTS weights on {device}…")
|
||||
logger.info("Loading OmniVoice model on device: %s", device)
|
||||
preload_asr = should_preload_tts_asr()
|
||||
@@ -483,23 +780,65 @@ def _load_model_sync():
|
||||
logger.info("Preloading PyTorch Whisper with TTS model.")
|
||||
else:
|
||||
logger.info("Skipping PyTorch Whisper preload; ASR will load on demand.")
|
||||
try:
|
||||
_model = OmniVoice.from_pretrained(
|
||||
def _load():
|
||||
return OmniVoice.from_pretrained(
|
||||
checkpoint, device_map=device, dtype=torch.float16, load_asr=preload_asr,
|
||||
)
|
||||
|
||||
try:
|
||||
_model = _load()
|
||||
except OSError as e:
|
||||
# #352: a truncated HF cache surfaces here as "does not appear to
|
||||
# have a file named pytorch_model.bin or model.safetensors".
|
||||
# Translate to an actionable message instead of the raw
|
||||
# transformers error.
|
||||
if "does not appear to have a file named" in str(e):
|
||||
# #352 / #581: a truncated HF cache surfaces here as "does not
|
||||
# appear to have a file named pytorch_model.bin or
|
||||
# model.safetensors". Instead of dead-ending the user with a
|
||||
# manual delete-and-reinstall instruction, try to self-repair: an
|
||||
# interrupted download leaves the cache missing only some files,
|
||||
# and snapshot_download() resumes/fills exactly those (a complete
|
||||
# cache never reaches this branch, so the fast path is untouched).
|
||||
if not _is_incomplete_cache_error(e):
|
||||
raise
|
||||
_set_loading("loading_weights", "Repairing incomplete model cache…")
|
||||
if not _repair_model_cache(checkpoint):
|
||||
raise RuntimeError(
|
||||
f"The TTS model cache for {checkpoint} is incomplete "
|
||||
"(weights missing — usually an interrupted download). "
|
||||
"Open Settings → Models, delete the OmniVoice TTS model, "
|
||||
"and install it again."
|
||||
) from e
|
||||
raise
|
||||
_set_loading("loading_weights", f"Loading TTS weights on {device}…")
|
||||
try:
|
||||
_model = _load()
|
||||
except OSError as e2:
|
||||
# Resume-repair ran but the cache is still unusable. The usual
|
||||
# cause beyond "repo genuinely lacks weights" is a blob that's
|
||||
# present with the right size but corrupt — snapshot_download's
|
||||
# resume trusts it and never re-fetches it (#739). Force a full
|
||||
# re-download (replaces corrupt blobs) and retry once more before
|
||||
# falling back to the manual delete-and-reinstall message.
|
||||
if _is_incomplete_cache_error(e2):
|
||||
_set_loading("loading_weights", "Re-downloading model files…")
|
||||
if _repair_model_cache(checkpoint, force=True):
|
||||
try:
|
||||
_model = _load()
|
||||
except OSError as e3:
|
||||
raise RuntimeError(
|
||||
f"The TTS model cache for {checkpoint} is incomplete "
|
||||
"and could not be auto-repaired. Open Settings → "
|
||||
"Models, delete the OmniVoice TTS model, and install "
|
||||
"it again."
|
||||
) from e3
|
||||
else:
|
||||
raise RuntimeError(
|
||||
f"The TTS model cache for {checkpoint} is incomplete and "
|
||||
"could not be auto-repaired. Open Settings → Models, "
|
||||
"delete the OmniVoice TTS model, and install it again."
|
||||
) from e2
|
||||
else:
|
||||
raise RuntimeError(
|
||||
f"The TTS model cache for {checkpoint} is incomplete and "
|
||||
"could not be auto-repaired. Open Settings → Models, delete "
|
||||
"the OmniVoice TTS model, and install it again."
|
||||
) from e2
|
||||
|
||||
try:
|
||||
# plan-02 (#65): gate on Triton availability (+ user setting), not
|
||||
@@ -549,9 +888,19 @@ def _load_model_sync():
|
||||
logger.info("OmniVoice model loaded successfully.")
|
||||
return _model
|
||||
except Exception as exc:
|
||||
err_msg = str(exc)
|
||||
# Surface an ACTIONABLE, sanitized error in /model/status (it's shown in
|
||||
# the first-run System Check). build_failure classifies the cause and
|
||||
# attaches a fix hint — e.g. a corrupted transformers install
|
||||
# ([Errno 2] … modeling_*.py) now says "reinstall transformers" instead
|
||||
# of an unhelpful raw path + "try restarting" — and strips the home dir.
|
||||
try:
|
||||
from core.failure import build_failure
|
||||
_f = build_failure(exc, stage="model-load", include_diagnostic=False)
|
||||
err_msg = _f["reason"] + (f" — {_f['hint']}" if _f.get("hint") else "")
|
||||
except Exception: # never let failure-formatting mask the real error
|
||||
err_msg = str(exc)
|
||||
_set_loading("error", "Model loading failed", error=err_msg)
|
||||
logger.error("Model loading failed: %s", err_msg)
|
||||
logger.error("Model loading failed: %s", str(exc))
|
||||
raise
|
||||
finally:
|
||||
unregister_listener(lid)
|
||||
@@ -571,19 +920,15 @@ def _model_load_timeout() -> float:
|
||||
|
||||
|
||||
def _reset_gpu_pool() -> None:
|
||||
"""Drop the GPU pool singleton so the next access builds a fresh one.
|
||||
"""Recover from a wedged/timed-out load by abandoning the GPU worker pool.
|
||||
|
||||
Python can't kill the thread stuck in a timed-out load, but abandoning the
|
||||
poisoned single-worker pool means a *retry* gets a clean worker instead of
|
||||
queueing forever behind the wedged one.
|
||||
The resilient wrapper is kept (its identity is shared by every importer);
|
||||
only its inner `ThreadPoolExecutor` is dropped, so the next submit builds a
|
||||
fresh worker. This is what stops stale references from raising "cannot
|
||||
schedule new futures after shutdown" after a reset (#589 #599).
|
||||
"""
|
||||
global _gpu_pool_singleton
|
||||
pool, _gpu_pool_singleton = _gpu_pool_singleton, None
|
||||
if pool is not None:
|
||||
try:
|
||||
pool.shutdown(wait=False, cancel_futures=True)
|
||||
except Exception:
|
||||
pass
|
||||
if _gpu_pool_singleton is not None:
|
||||
_gpu_pool_singleton.reset()
|
||||
|
||||
|
||||
async def _load_model_with_timeout():
|
||||
@@ -634,8 +979,11 @@ async def preload_model():
|
||||
return # already loaded
|
||||
try:
|
||||
# Check if the required model checkpoint exists before attempting
|
||||
# a heavy load that would fail and pollute startup logs.
|
||||
checkpoint = os.environ.get("OMNIVOICE_MODEL", "k2-fsa/OmniVoice")
|
||||
# a heavy load that would fail and pollute startup logs. Use the same
|
||||
# resolver as the load path (#693) so a leaked engine id in
|
||||
# OMNIVOICE_MODEL can't make this model_info() probe fail and silently
|
||||
# disable warm-up (then the first /generate eats the full load).
|
||||
checkpoint = resolve_omnivoice_checkpoint()
|
||||
try:
|
||||
from huggingface_hub import model_info
|
||||
model_info(checkpoint, timeout=5)
|
||||
|
||||
@@ -145,3 +145,179 @@ def save_lexicon(path, lexicon: Optional[dict]) -> dict[str, str]:
|
||||
encoding="utf-8",
|
||||
)
|
||||
return clean
|
||||
|
||||
|
||||
# ── DB-backed global / per-language dictionary (Expressive-TTS Spec 01) ───────
|
||||
#
|
||||
# The JSON ``load_lexicon``/``save_lexicon`` above stay the per-project audiobook
|
||||
# override. THIS layer is the user-editable, DB-persisted, per-language default
|
||||
# dictionary surfaced in Settings → Pronunciation. Rows scoped ``language="*"``
|
||||
# apply to every request; a 2-letter language row applies only when the request
|
||||
# language's prefix matches (case-insensitive), so a German entry never fires on
|
||||
# an English render. Both layers are pure text substitution — they ride the same
|
||||
# ReDoS-safe ``apply_lexicon`` matcher, so every engine honors them.
|
||||
|
||||
_ALL_LANG = "*"
|
||||
|
||||
|
||||
def _lang_prefix(language: Optional[str]) -> Optional[str]:
|
||||
"""Normalize a request language to a lowercase 2-letter prefix.
|
||||
|
||||
``"Auto"``/``None``/``""`` → ``None`` (means "no language pin": only global
|
||||
``*`` rows apply, language-tagged rows are skipped, mirroring how the engines
|
||||
treat an unset language). A value like ``"en-US"`` / ``"English"`` →
|
||||
``"en"`` (first two letters); matching against entries is on this prefix.
|
||||
"""
|
||||
if not language:
|
||||
return None
|
||||
s = str(language).strip().lower()
|
||||
if not s or s == "auto":
|
||||
return None
|
||||
return s[:2]
|
||||
|
||||
|
||||
def entries_for_language(entries, language: Optional[str]) -> dict[str, str]:
|
||||
"""Collapse DB rows into a ``{term: replacement}`` map for ``apply_lexicon``.
|
||||
|
||||
Filters to ``enabled`` rows whose scope is global (``*``) OR whose language
|
||||
prefix matches the request language. Only the **respelling** path produces a
|
||||
plain substitution here (Phase 1); IPA/CMU rows that carry no respelling are
|
||||
skipped at this layer (they're handled — or honestly degraded — by the
|
||||
engine-markup path, never silently mangling text). A language-specific row
|
||||
overrides a global row with the same (case-folded) term, so a per-language
|
||||
pronunciation can refine the global default.
|
||||
|
||||
``entries`` is any iterable of mappings/rows with ``term``, ``replacement``,
|
||||
``type``, ``language``, ``enabled`` keys (a ``sqlite3.Row`` works directly).
|
||||
"""
|
||||
req_prefix = _lang_prefix(language)
|
||||
# Two passes so language rows win over global rows on the same term: collect
|
||||
# global first, then overlay matching-language rows.
|
||||
glob: dict[str, str] = {}
|
||||
lang: dict[str, str] = {}
|
||||
for e in entries:
|
||||
try:
|
||||
if not int(e["enabled"]):
|
||||
continue
|
||||
except (KeyError, IndexError, TypeError, ValueError):
|
||||
continue
|
||||
term = (e["term"] or "").strip()
|
||||
if not term:
|
||||
continue
|
||||
etype = (e["type"] or "respelling").strip().lower()
|
||||
replacement = e["replacement"] if e["replacement"] is not None else ""
|
||||
# Phase 1: only respelling rows substitute text. IPA/CMU rows without a
|
||||
# respelling fall through (Phase 2 lowers them to engine markup); we do
|
||||
# NOT feed a raw IPA string into the grapheme stream.
|
||||
if etype != "respelling":
|
||||
continue
|
||||
scope = (e["language"] or _ALL_LANG).strip() or _ALL_LANG
|
||||
if scope == _ALL_LANG:
|
||||
glob[term] = str(replacement)
|
||||
else:
|
||||
if req_prefix is not None and scope[:2].lower() == req_prefix:
|
||||
lang[term] = str(replacement)
|
||||
merged = dict(glob)
|
||||
merged.update(lang) # language rows override global on the same term
|
||||
return merged
|
||||
|
||||
|
||||
# ── Inline one-off override: [[term|replacement]] / [[replacement]] ─────────
|
||||
#
|
||||
# Double brackets are unambiguous against the single-bracket grammar
|
||||
# (``[voice:]``/``[pause]``/SSML-lite/``[Name]``): ``_VOICE_RE`` is
|
||||
# ``\[voice:([^\]\[]*)\]`` — it forbids inner brackets, so it can't span a
|
||||
# ``[[…]]``; the SSML-lite / pause vocabularies are closed literal sets that
|
||||
# ``[[…]]`` is not a member of. We resolve ``[[…]]`` BEFORE chunking so the
|
||||
# splitter never sees it. ReDoS-safe: ``\[\[[^\]]*\]\]`` is a bounded literal
|
||||
# class, no nested quantifier.
|
||||
#
|
||||
# [[gif|jiff]] → replaces the literal "gif" → "jiff" for this occurrence
|
||||
# [[Nuh-VAD-uh]] → the bracket content itself is spoken (brackets stripped)
|
||||
# Bounded inner repetition ({0,256}) keeps this strictly linear: ``[^\]]`` also
|
||||
# matches ``[``, so an unbounded run of ``[`` with no closing ``]]`` would let the
|
||||
# engine re-scan O(n) content from O(n) start positions (polynomial ReDoS). The
|
||||
# bound caps per-position work; an inline override is a short respelling, so 256
|
||||
# chars is far more than any real ``[[term|replacement]]`` needs.
|
||||
_INLINE_RE = re.compile(r"\[\[([^\]]{0,256})\]\]")
|
||||
|
||||
|
||||
def apply_inline_overrides(text: str) -> str:
|
||||
"""Resolve ``[[…]]`` one-off pronunciation overrides to plain spoken text.
|
||||
|
||||
``[[term|replacement]]`` → ``replacement`` (the ``term`` half is a label for
|
||||
the author; only the replacement is spoken). ``[[replacement]]`` (no pipe) →
|
||||
``replacement`` with the brackets stripped. Empty ``[[]]`` collapses away.
|
||||
Applied once per occurrence; nothing persists. Single ``[…]`` tags are left
|
||||
untouched (the regex requires a double bracket on both sides).
|
||||
"""
|
||||
if not text or "[[" not in text:
|
||||
return text or ""
|
||||
|
||||
def _repl(m: re.Match) -> str:
|
||||
inner = m.group(1)
|
||||
if "|" in inner:
|
||||
inner = inner.split("|", 1)[1]
|
||||
return inner
|
||||
|
||||
return _INLINE_RE.sub(_repl, text)
|
||||
|
||||
|
||||
def apply_pronunciation(
|
||||
text: str,
|
||||
entries=None,
|
||||
language: Optional[str] = None,
|
||||
*,
|
||||
lexicon: Optional[dict] = None,
|
||||
) -> str:
|
||||
"""Apply the pronunciation dictionary + inline overrides to ``text``.
|
||||
|
||||
Order (load-bearing):
|
||||
1. DB dictionary rows (``entries``) filtered to ``language`` + an optional
|
||||
per-project ``lexicon`` JSON overlay (project wins on term conflict,
|
||||
matching the audiobook layering). Both go through one ``apply_lexicon``
|
||||
pass (longest-term-first, word-boundary aware, idempotent).
|
||||
2. Inline ``[[…]]`` one-off overrides resolved last, so an inline override
|
||||
always wins over any dictionary entry for that occurrence.
|
||||
|
||||
A falsy ``text`` / empty dictionary / no inline markers is a pass-through, so
|
||||
legacy plain text is byte-identical.
|
||||
"""
|
||||
if not text:
|
||||
return text or ""
|
||||
merged = entries_for_language(entries or [], language)
|
||||
if lexicon:
|
||||
# Project-local JSON overlays the DB defaults; project wins on conflict.
|
||||
merged.update(normalize_lexicon(lexicon))
|
||||
out = apply_lexicon(text, merged) if merged else text
|
||||
return apply_inline_overrides(out)
|
||||
|
||||
|
||||
# ── DB load/save ──────────────────────────────────────────────────────────────
|
||||
|
||||
def load_entries_from_db() -> list[dict]:
|
||||
"""Return every pronunciation_entries row as a list of plain dicts.
|
||||
|
||||
Import-light: the DB module is imported lazily so the pure-parser path (and
|
||||
the audiobook JSON path) never pull in sqlite/config.
|
||||
"""
|
||||
from core.db import db_conn
|
||||
|
||||
with db_conn() as conn:
|
||||
rows = conn.execute(
|
||||
"SELECT id, term, replacement, type, language, enabled, created_at "
|
||||
"FROM pronunciation_entries ORDER BY created_at ASC, id ASC"
|
||||
).fetchall()
|
||||
return [dict(r) for r in rows]
|
||||
|
||||
|
||||
def load_dict_for_request(language: Optional[str] = None) -> dict[str, str]:
|
||||
"""Convenience: DB rows → ``{term: replacement}`` for a request language.
|
||||
|
||||
Returns ``{}`` (a no-op for ``apply_pronunciation``) if the table is absent
|
||||
or the DB can't be opened — pronunciation is never allowed to break synth.
|
||||
"""
|
||||
try:
|
||||
return entries_for_language(load_entries_from_db(), language)
|
||||
except Exception: # noqa: BLE001 — table missing / DB locked → no-op
|
||||
return {}
|
||||
|
||||
@@ -542,3 +542,146 @@ def assign_speakers_heuristic(segments: List[dict]) -> List[dict]:
|
||||
s["speaker_id"] = f"Speaker {current}"
|
||||
last_end = s["end"]
|
||||
return segments
|
||||
|
||||
|
||||
# ── Speaker-aware re-split (#486) ────────────────────────────────────────────
|
||||
#
|
||||
# Segmentation runs BEFORE diarization and groups words by sentence/duration
|
||||
# only, so one segment can span two speakers' turns. assign_speakers_* then only
|
||||
# *relabels* each segment with its majority speaker — the boundary is lost and a
|
||||
# two-speaker exchange reads as one line. This pass re-splits such a segment at
|
||||
# the word-level speaker boundary, after diarization.
|
||||
#
|
||||
# Hard invariant (the single-speaker no-regression guarantee): a segment whose
|
||||
# words all map to ONE speaker is returned byte-for-byte unchanged — same dict,
|
||||
# id, text, start, end — so single-speaker dubs and their timing never move.
|
||||
|
||||
def _word_speaker(w: "Word", turns: Sequence[tuple]) -> Optional[str]:
|
||||
"""Majority-overlap speaker label for a word; midpoint membership as a
|
||||
fallback; ``None`` when the word has no diarization coverage at all."""
|
||||
acc: dict = {}
|
||||
for ts, te, label in turns:
|
||||
left = max(w.start, ts)
|
||||
right = min(w.end, te)
|
||||
if right > left:
|
||||
acc[label] = acc.get(label, 0.0) + (right - left)
|
||||
if acc:
|
||||
return max(acc.items(), key=lambda kv: kv[1])[0]
|
||||
mid = (w.start + w.end) / 2.0
|
||||
for ts, te, label in turns:
|
||||
if ts <= mid <= te:
|
||||
return label
|
||||
return None
|
||||
|
||||
|
||||
def _fill_and_smooth(labels: List[Optional[str]]) -> List[Optional[str]]:
|
||||
"""Forward/back-fill gaps (words with no coverage inherit a neighbor) and
|
||||
smooth single-word flips, so one mis-attributed word inside a speaker's run
|
||||
(diarization noise) doesn't trigger a spurious split."""
|
||||
out = list(labels)
|
||||
n = len(out)
|
||||
last = None
|
||||
for i in range(n):
|
||||
if out[i] is None:
|
||||
out[i] = last
|
||||
else:
|
||||
last = out[i]
|
||||
nxt = None
|
||||
for i in range(n - 1, -1, -1):
|
||||
if out[i] is None:
|
||||
out[i] = nxt
|
||||
else:
|
||||
nxt = out[i]
|
||||
for i in range(1, n - 1):
|
||||
if out[i] != out[i - 1] and out[i - 1] == out[i + 1]:
|
||||
out[i] = out[i - 1]
|
||||
return out
|
||||
|
||||
|
||||
def _resplit_core(
|
||||
segments: List[dict], words: Sequence["Word"], turns: Sequence[tuple],
|
||||
) -> List[dict]:
|
||||
"""Split each segment that spans >1 speaker at the word-level boundary.
|
||||
|
||||
``turns`` is a normalised list of ``(start, end, speaker_label)``. Single-
|
||||
speaker segments are passed through untouched. Pieces keep the segment's
|
||||
outer start/end (preserving any onset-snap) and use word times for interior
|
||||
boundaries, so the pieces exactly cover the original span.
|
||||
"""
|
||||
if not turns or not words:
|
||||
return segments
|
||||
ordered = sorted(words, key=lambda w: (w.start, w.end))
|
||||
out: List[dict] = []
|
||||
for seg in segments:
|
||||
s0, s1 = seg["start"], seg["end"]
|
||||
seg_words = [w for w in ordered if min(w.end, s1) - max(w.start, s0) > 1e-6]
|
||||
if len(seg_words) < 2:
|
||||
out.append(seg)
|
||||
continue
|
||||
labels = _fill_and_smooth([_word_speaker(w, turns) for w in seg_words])
|
||||
if len({l for l in labels if l is not None}) <= 1:
|
||||
out.append(seg) # single speaker (or unknown) → byte-for-byte unchanged
|
||||
continue
|
||||
runs: List[tuple] = []
|
||||
for w, label in zip(seg_words, labels):
|
||||
if runs and runs[-1][0] == label:
|
||||
runs[-1][1].append(w)
|
||||
else:
|
||||
runs.append((label, [w]))
|
||||
n_runs = len(runs)
|
||||
piece_no = 0
|
||||
for k, (label, ws) in enumerate(runs):
|
||||
text = _clean(" ".join(w.text for w in ws))
|
||||
if not text:
|
||||
continue
|
||||
piece = dict(seg)
|
||||
piece["text"] = text
|
||||
piece["start"] = s0 if k == 0 else ws[0].start
|
||||
piece["end"] = s1 if k == n_runs - 1 else ws[-1].end
|
||||
if label:
|
||||
piece["speaker_id"] = label
|
||||
if piece_no > 0:
|
||||
piece["id"] = f"{seg.get('id', 'seg')}-{piece_no}"
|
||||
if "text_original" in piece:
|
||||
piece["text_original"] = text
|
||||
elif "text_original" in piece:
|
||||
piece["text_original"] = text
|
||||
out.append(piece)
|
||||
piece_no += 1
|
||||
return out
|
||||
|
||||
|
||||
def _diar_speaker_label(raw) -> str:
|
||||
"""``SPEAKER_00`` → ``Speaker 1`` (mirrors assign_speakers_from_diarization)."""
|
||||
try:
|
||||
return f"Speaker {int(str(raw).split('_')[-1]) + 1}"
|
||||
except (ValueError, AttributeError):
|
||||
return str(raw)
|
||||
|
||||
|
||||
def resplit_segments_by_diarization(
|
||||
segments: List[dict], words: Sequence["Word"], diarization,
|
||||
) -> List[dict]:
|
||||
"""Speaker-aware re-split using a pyannote diarization result (#486)."""
|
||||
turns = [
|
||||
(turn.start, turn.end, _diar_speaker_label(spk))
|
||||
for turn, _, spk in diarization.itertracks(yield_label=True)
|
||||
]
|
||||
return _resplit_core(segments, words, turns)
|
||||
|
||||
|
||||
def resplit_segments_by_turns(
|
||||
segments: List[dict], words: Sequence["Word"], turns: Sequence[dict],
|
||||
) -> List[dict]:
|
||||
"""Speaker-aware re-split using inline ASR speaker turns (FunASR cam++).
|
||||
|
||||
``speaker`` is used verbatim (FunASR already labels ``"Speaker N"``), matching
|
||||
:func:`assign_speakers_from_turns`."""
|
||||
norm = [
|
||||
(t["start"], t["end"], t["speaker"])
|
||||
for t in (turns or [])
|
||||
if t.get("speaker") is not None
|
||||
and t.get("start") is not None
|
||||
and t.get("end") is not None
|
||||
]
|
||||
return _resplit_core(segments, words, norm)
|
||||
|
||||
@@ -108,6 +108,107 @@ def clear_hf_token() -> None:
|
||||
conn.execute("DELETE FROM settings WHERE key = ?", (_TOKEN_KEY,))
|
||||
|
||||
|
||||
# ── Generic encrypted secrets (LLM provider API keys, future tokens) ───────
|
||||
# The HF token got the first bespoke encrypted row; the LLM-providers feature
|
||||
# needs the *same* at-rest protection for a dozen provider keys. Rather than
|
||||
# copy the Fernet dance per provider, expose generic secret helpers. Rows are
|
||||
# namespaced with the ``secret.`` prefix so a misrouted ``get_text`` on a
|
||||
# secret key returns opaque ciphertext (defence in depth), and so plaintext
|
||||
# ``settings`` rows can never collide with a secret. Same InvalidToken →
|
||||
# None degrade as the HF path (install moved across machines → fall back to
|
||||
# env), same per-install key.
|
||||
_SECRET_PREFIX = "secret."
|
||||
|
||||
|
||||
def _secret_key_name(name: str) -> str:
|
||||
if not name or not isinstance(name, str):
|
||||
raise ValueError(f"secret name must be a non-empty string, got {name!r}")
|
||||
if name == _TOKEN_KEY or name.startswith(_SECRET_PREFIX):
|
||||
raise ValueError(f"invalid secret name {name!r}")
|
||||
return f"{_SECRET_PREFIX}{name}"
|
||||
|
||||
|
||||
def get_secret(name: str) -> Optional[str]:
|
||||
"""Return a decrypted secret (e.g. an LLM provider API key), or None.
|
||||
|
||||
Mirrors :func:`get_hf_token`: on decrypt failure (install migrated across
|
||||
machines) or any SQLite error, log and return None so callers fall back to
|
||||
env / provider defaults instead of crashing.
|
||||
"""
|
||||
from core.db import db_conn
|
||||
|
||||
key = _secret_key_name(name)
|
||||
try:
|
||||
with db_conn() as conn:
|
||||
row = conn.execute(
|
||||
"SELECT value FROM settings WHERE key = ?", (key,)
|
||||
).fetchone()
|
||||
if row is None or not row[0]:
|
||||
return None
|
||||
try:
|
||||
from cryptography.fernet import InvalidToken
|
||||
except ImportError: # pragma: no cover — dep should always be present
|
||||
logger.error("cryptography unavailable; cannot decrypt secret %s", name)
|
||||
return None
|
||||
try:
|
||||
return _fernet().decrypt(row[0].encode("ascii")).decode("utf-8")
|
||||
except InvalidToken:
|
||||
logger.warning(
|
||||
"Stored secret %r failed to decrypt (install moved across "
|
||||
"machines or salt tampered) — falling back to env/default.", name,
|
||||
)
|
||||
return None
|
||||
except Exception:
|
||||
logger.exception("settings_store.get_secret(%s): SQLite read failed", name)
|
||||
return None
|
||||
|
||||
|
||||
def set_secret(name: str, value: str) -> None:
|
||||
"""Persist an encrypted secret. Empty value clears the row."""
|
||||
if not value:
|
||||
clear_secret(name)
|
||||
return
|
||||
from core.db import db_conn
|
||||
|
||||
key = _secret_key_name(name)
|
||||
blob = _fernet().encrypt(value.encode("utf-8")).decode("ascii")
|
||||
with db_conn() as conn:
|
||||
conn.execute(
|
||||
"INSERT OR REPLACE INTO settings(key, value, updated_at) "
|
||||
"VALUES (?, ?, ?)",
|
||||
(key, blob, time.time()),
|
||||
)
|
||||
|
||||
|
||||
def clear_secret(name: str) -> None:
|
||||
"""Remove a secret row (salt row preserved, like clear_hf_token)."""
|
||||
from core.db import db_conn
|
||||
|
||||
key = _secret_key_name(name)
|
||||
with db_conn() as conn:
|
||||
conn.execute("DELETE FROM settings WHERE key = ?", (key,))
|
||||
|
||||
|
||||
def list_secret_names() -> list[str]:
|
||||
"""Return the bare names of all stored secrets (no values, no ciphertext).
|
||||
|
||||
Lets the LLM-providers settings API report *which* providers have a key
|
||||
configured without ever decrypting or returning the key material.
|
||||
"""
|
||||
from core.db import db_conn
|
||||
|
||||
try:
|
||||
with db_conn() as conn:
|
||||
rows = conn.execute(
|
||||
"SELECT key FROM settings WHERE key LIKE ?",
|
||||
(f"{_SECRET_PREFIX}%",),
|
||||
).fetchall()
|
||||
return [r[0][len(_SECRET_PREFIX):] for r in rows if r and r[0]]
|
||||
except Exception:
|
||||
logger.exception("settings_store.list_secret_names: SQLite read failed")
|
||||
return []
|
||||
|
||||
|
||||
# ── Non-secret text settings ──────────────────────────────────────────────
|
||||
# Plan 01-02 Task 4 (INST-12): the Performance panel needs to persist a
|
||||
# boolean toggle (`perf.torch_compile_disabled`). It is NOT a secret — no
|
||||
@@ -128,7 +229,8 @@ def get_text(key: str, default: Optional[str] = None) -> Optional[str]:
|
||||
looking like opaque bytes — callers MUST use `get_hf_token()` for
|
||||
secrets and only ever pass non-secret keys to `get_text()`.
|
||||
"""
|
||||
if key == _TOKEN_KEY: # defence in depth — never let a misrouted call leak ciphertext
|
||||
if key == _TOKEN_KEY or key.startswith(_SECRET_PREFIX):
|
||||
# defence in depth — never let a misrouted call leak ciphertext
|
||||
return default
|
||||
from core.db import db_conn
|
||||
|
||||
@@ -150,10 +252,10 @@ def set_text(key: str, value: str) -> None:
|
||||
|
||||
Use for non-secret config only. For tokens, use `set_hf_token()`.
|
||||
"""
|
||||
if key == _TOKEN_KEY:
|
||||
if key == _TOKEN_KEY or key.startswith(_SECRET_PREFIX):
|
||||
raise ValueError(
|
||||
"set_text refuses to write to the encrypted hf_token row; "
|
||||
"use set_hf_token() for secrets"
|
||||
"set_text refuses to write to an encrypted secret row; "
|
||||
"use set_hf_token()/set_secret() for secrets"
|
||||
)
|
||||
from core.db import db_conn
|
||||
|
||||
|
||||
@@ -0,0 +1,316 @@
|
||||
"""
|
||||
sherpa-onnx live-dictation ASR backend.
|
||||
|
||||
Adds the k2-fsa/sherpa-onnx ONNX runtime as a *dictation* engine alongside the
|
||||
existing Whisper/NeMo family — without touching any of them. The whole point of
|
||||
this engine is **live, faster-than-real-time dictation on CPU**:
|
||||
|
||||
• STREAMING models (OnlineRecognizer) emit partial text frame-by-frame as the
|
||||
user speaks, finalising on sherpa's built-in endpoint (silence) detection.
|
||||
• OFFLINE models (OfflineRecognizer) re-transcribe a growing buffer on a short
|
||||
cadence so the user still sees live partials, finalising on EOF/silence.
|
||||
|
||||
CPU provider only (strict cross-platform-default parity rule): identical
|
||||
behaviour on macOS arm64+x86_64, Windows x64, Linux. No CUDA dependency.
|
||||
|
||||
Model weights are the small int8 ONNX checkpoints published under
|
||||
``csukuangfj/`` on HuggingFace; they download on first use through the same HF
|
||||
cache the rest of the app uses (``snapshot_download``). Exact asset filenames
|
||||
were verified against the live HF repo trees (see ``_MODELS`` below) — the
|
||||
streaming zipformer repos use the plain ``encoder-epoch-99-avg-1.int8.onnx``
|
||||
naming, NOT a ``-chunk-16-left-64`` variant.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
logger = logging.getLogger("omnivoice.asr.sherpa")
|
||||
|
||||
# CPU only — strict cross-platform default-parity rule. Overridable for
|
||||
# power users on a verified GPU build, but the default never diverges.
|
||||
_PROVIDER = os.environ.get("OMNIVOICE_SHERPA_ASR_PROVIDER", "cpu")
|
||||
_NUM_THREADS = int(os.environ.get("OMNIVOICE_SHERPA_ASR_THREADS", "2"))
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SherpaModelSpec:
|
||||
"""One downloadable sherpa-onnx dictation model.
|
||||
|
||||
``files`` maps a logical role (encoder/decoder/joiner/tokens) to the EXACT
|
||||
asset filename in the HF repo. ``kind`` selects the recognizer factory:
|
||||
``offline-transducer`` | ``offline-whisper`` | ``online-transducer`` |
|
||||
``online-paraformer``. ``tag`` is the frontend-facing "offline"/"streaming".
|
||||
"""
|
||||
id: str
|
||||
repo_id: str
|
||||
label: str
|
||||
tag: str # "offline" | "streaming"
|
||||
kind: str # recognizer factory selector
|
||||
size_gb: float
|
||||
languages: str
|
||||
files: dict[str, str]
|
||||
recommended: bool = False
|
||||
model_type: str = "" # offline transducer only (nemo_transducer)
|
||||
extra: dict = field(default_factory=dict)
|
||||
|
||||
@property
|
||||
def streaming(self) -> bool:
|
||||
return self.tag == "streaming"
|
||||
|
||||
|
||||
# ── The 7 models (HF repo ids under csukuangfj/, filenames VERIFIED against the
|
||||
# live HF /api/models/<repo>/tree/main on 2026-06-25; int8 variants pinned).
|
||||
_MODELS: dict[str, SherpaModelSpec] = {
|
||||
"sherpa-parakeet-tdt-v3": SherpaModelSpec(
|
||||
id="sherpa-parakeet-tdt-v3",
|
||||
repo_id="csukuangfj/sherpa-onnx-nemo-parakeet-tdt-0.6b-v3-int8",
|
||||
label="Parakeet TDT v3",
|
||||
tag="offline",
|
||||
kind="offline-transducer",
|
||||
size_gb=0.18,
|
||||
languages="25 European languages",
|
||||
recommended=True,
|
||||
model_type="nemo_transducer",
|
||||
files={
|
||||
"encoder": "encoder.int8.onnx",
|
||||
"decoder": "decoder.int8.onnx",
|
||||
"joiner": "joiner.int8.onnx",
|
||||
"tokens": "tokens.txt",
|
||||
},
|
||||
),
|
||||
"sherpa-parakeet-tdt-v2": SherpaModelSpec(
|
||||
id="sherpa-parakeet-tdt-v2",
|
||||
repo_id="csukuangfj/sherpa-onnx-nemo-parakeet-tdt-0.6b-v2-int8",
|
||||
label="Parakeet TDT v2",
|
||||
tag="offline",
|
||||
kind="offline-transducer",
|
||||
size_gb=0.17,
|
||||
languages="English",
|
||||
model_type="nemo_transducer",
|
||||
files={
|
||||
"encoder": "encoder.int8.onnx",
|
||||
"decoder": "decoder.int8.onnx",
|
||||
"joiner": "joiner.int8.onnx",
|
||||
"tokens": "tokens.txt",
|
||||
},
|
||||
),
|
||||
"sherpa-zipformer-bilingual-zh-en": SherpaModelSpec(
|
||||
id="sherpa-zipformer-bilingual-zh-en",
|
||||
repo_id="csukuangfj/sherpa-onnx-streaming-zipformer-bilingual-zh-en-2023-02-20",
|
||||
label="Zipformer Bilingual",
|
||||
tag="streaming",
|
||||
kind="online-transducer",
|
||||
size_gb=0.13,
|
||||
languages="Chinese + English",
|
||||
files={
|
||||
"encoder": "encoder-epoch-99-avg-1.int8.onnx",
|
||||
"decoder": "decoder-epoch-99-avg-1.int8.onnx",
|
||||
"joiner": "joiner-epoch-99-avg-1.int8.onnx",
|
||||
"tokens": "tokens.txt",
|
||||
},
|
||||
),
|
||||
"sherpa-paraformer-bilingual-zh-en": SherpaModelSpec(
|
||||
id="sherpa-paraformer-bilingual-zh-en",
|
||||
repo_id="csukuangfj/sherpa-onnx-streaming-paraformer-bilingual-zh-en",
|
||||
label="Paraformer Bilingual",
|
||||
tag="streaming",
|
||||
kind="online-paraformer",
|
||||
size_gb=0.115,
|
||||
languages="Chinese + English",
|
||||
files={
|
||||
"encoder": "encoder.int8.onnx",
|
||||
"decoder": "decoder.int8.onnx",
|
||||
"tokens": "tokens.txt",
|
||||
},
|
||||
),
|
||||
"sherpa-zipformer-en-20m": SherpaModelSpec(
|
||||
id="sherpa-zipformer-en-20m",
|
||||
repo_id="csukuangfj/sherpa-onnx-streaming-zipformer-en-20M-2023-02-17",
|
||||
label="Zipformer Streaming EN",
|
||||
tag="streaming",
|
||||
kind="online-transducer",
|
||||
size_gb=0.128,
|
||||
languages="English",
|
||||
files={
|
||||
"encoder": "encoder-epoch-99-avg-1.int8.onnx",
|
||||
"decoder": "decoder-epoch-99-avg-1.int8.onnx",
|
||||
"joiner": "joiner-epoch-99-avg-1.int8.onnx",
|
||||
"tokens": "tokens.txt",
|
||||
},
|
||||
),
|
||||
"sherpa-zipformer-zh-14m": SherpaModelSpec(
|
||||
id="sherpa-zipformer-zh-14m",
|
||||
repo_id="csukuangfj/sherpa-onnx-streaming-zipformer-zh-14M-2023-02-23",
|
||||
label="Zipformer Streaming ZH",
|
||||
tag="streaming",
|
||||
kind="online-transducer",
|
||||
size_gb=0.074,
|
||||
languages="Chinese",
|
||||
files={
|
||||
"encoder": "encoder-epoch-99-avg-1.int8.onnx",
|
||||
"decoder": "decoder-epoch-99-avg-1.int8.onnx",
|
||||
"joiner": "joiner-epoch-99-avg-1.int8.onnx",
|
||||
"tokens": "tokens.txt",
|
||||
},
|
||||
),
|
||||
"sherpa-whisper-tiny": SherpaModelSpec(
|
||||
id="sherpa-whisper-tiny",
|
||||
repo_id="csukuangfj/sherpa-onnx-whisper-tiny",
|
||||
label="Whisper Tiny",
|
||||
tag="offline",
|
||||
kind="offline-whisper",
|
||||
size_gb=0.116,
|
||||
languages="90+ languages (auto-detect)",
|
||||
files={
|
||||
"encoder": "tiny-encoder.int8.onnx",
|
||||
"decoder": "tiny-decoder.int8.onnx",
|
||||
"tokens": "tiny-tokens.txt",
|
||||
},
|
||||
),
|
||||
}
|
||||
|
||||
DEFAULT_MODEL_ID = "sherpa-parakeet-tdt-v3"
|
||||
|
||||
# repo_id → model id, so the model-store list (keyed by repo_id) can be
|
||||
# enriched with the dictation metadata, and so capture can map either key.
|
||||
_REPO_TO_ID: dict[str, str] = {m.repo_id: mid for mid, m in _MODELS.items()}
|
||||
|
||||
|
||||
def list_specs() -> list[SherpaModelSpec]:
|
||||
return list(_MODELS.values())
|
||||
|
||||
|
||||
def get_spec(model_id: str) -> SherpaModelSpec | None:
|
||||
"""Look up a spec by its dictation id OR its HF repo_id."""
|
||||
if model_id in _MODELS:
|
||||
return _MODELS[model_id]
|
||||
if model_id in _REPO_TO_ID:
|
||||
return _MODELS[_REPO_TO_ID[model_id]]
|
||||
return None
|
||||
|
||||
|
||||
def is_sherpa_model(model_id: str | None) -> bool:
|
||||
return bool(model_id) and get_spec(model_id) is not None
|
||||
|
||||
|
||||
def sherpa_available() -> tuple[bool, str]:
|
||||
try:
|
||||
import sherpa_onnx # noqa: F401
|
||||
return True, "ready"
|
||||
except ImportError as e:
|
||||
return False, f"sherpa-onnx not installed: {e}. Install with: uv add sherpa-onnx"
|
||||
|
||||
|
||||
def _resolve_model_dir(spec: SherpaModelSpec, *, download: bool = True) -> str:
|
||||
"""Return the local directory containing this model's ONNX assets.
|
||||
|
||||
Tries the HF cache offline first (``local_files_only=True``); on a miss,
|
||||
downloads on first use (like every other engine) unless ``download=False``.
|
||||
Restricts the fetch to the exact int8 assets we pin via ``allow_patterns``
|
||||
so we never pull the bundled fp32 weights or test wavs.
|
||||
"""
|
||||
from huggingface_hub import snapshot_download
|
||||
|
||||
wanted = list(spec.files.values())
|
||||
try:
|
||||
return snapshot_download(
|
||||
repo_id=spec.repo_id,
|
||||
local_files_only=True,
|
||||
allow_patterns=wanted,
|
||||
)
|
||||
except Exception:
|
||||
if not download:
|
||||
raise
|
||||
logger.info("sherpa dictation: downloading %s on first use", spec.repo_id)
|
||||
return snapshot_download(repo_id=spec.repo_id, allow_patterns=wanted)
|
||||
|
||||
|
||||
def is_installed(spec: SherpaModelSpec) -> bool:
|
||||
"""True if every pinned asset is already present in the HF cache."""
|
||||
try:
|
||||
d = _resolve_model_dir(spec, download=False)
|
||||
except Exception:
|
||||
return False
|
||||
return all(os.path.isfile(os.path.join(d, f)) for f in spec.files.values())
|
||||
|
||||
|
||||
# ── Recognizers ──────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def build_offline_recognizer(spec: SherpaModelSpec, *, download: bool = True):
|
||||
"""Construct an ``OfflineRecognizer`` for an offline transducer/whisper model."""
|
||||
import sherpa_onnx
|
||||
|
||||
d = _resolve_model_dir(spec, download=download)
|
||||
|
||||
def p(role: str) -> str:
|
||||
return os.path.join(d, spec.files[role])
|
||||
|
||||
if spec.kind == "offline-transducer":
|
||||
return sherpa_onnx.OfflineRecognizer.from_transducer(
|
||||
encoder=p("encoder"),
|
||||
decoder=p("decoder"),
|
||||
joiner=p("joiner"),
|
||||
tokens=p("tokens"),
|
||||
num_threads=_NUM_THREADS,
|
||||
provider=_PROVIDER,
|
||||
decoding_method="greedy_search",
|
||||
model_type=spec.model_type or "nemo_transducer",
|
||||
)
|
||||
if spec.kind == "offline-whisper":
|
||||
return sherpa_onnx.OfflineRecognizer.from_whisper(
|
||||
encoder=p("encoder"),
|
||||
decoder=p("decoder"),
|
||||
tokens=p("tokens"),
|
||||
num_threads=_NUM_THREADS,
|
||||
provider=_PROVIDER,
|
||||
language="", # auto-detect
|
||||
task="transcribe",
|
||||
)
|
||||
raise ValueError(f"{spec.id} is not an offline model (kind={spec.kind})")
|
||||
|
||||
|
||||
def build_online_recognizer(spec: SherpaModelSpec, *, download: bool = True):
|
||||
"""Construct an ``OnlineRecognizer`` (true streaming) with endpoint detection.
|
||||
|
||||
Endpoint (silence) detection drives the live "final" boundary: sherpa
|
||||
commits a sentence after trailing silence so we can flush a ``final`` and
|
||||
reset the stream for the next utterance — all within one WS session.
|
||||
"""
|
||||
import sherpa_onnx
|
||||
|
||||
d = _resolve_model_dir(spec, download=download)
|
||||
|
||||
def p(role: str) -> str:
|
||||
return os.path.join(d, spec.files[role])
|
||||
|
||||
if spec.kind == "online-transducer":
|
||||
return sherpa_onnx.OnlineRecognizer.from_transducer(
|
||||
tokens=p("tokens"),
|
||||
encoder=p("encoder"),
|
||||
decoder=p("decoder"),
|
||||
joiner=p("joiner"),
|
||||
num_threads=_NUM_THREADS,
|
||||
provider=_PROVIDER,
|
||||
decoding_method="greedy_search",
|
||||
enable_endpoint_detection=True,
|
||||
rule1_min_trailing_silence=2.4,
|
||||
rule2_min_trailing_silence=1.2,
|
||||
rule3_min_utterance_length=20,
|
||||
)
|
||||
if spec.kind == "online-paraformer":
|
||||
return sherpa_onnx.OnlineRecognizer.from_paraformer(
|
||||
tokens=p("tokens"),
|
||||
encoder=p("encoder"),
|
||||
decoder=p("decoder"),
|
||||
num_threads=_NUM_THREADS,
|
||||
provider=_PROVIDER,
|
||||
decoding_method="greedy_search",
|
||||
enable_endpoint_detection=True,
|
||||
rule1_min_trailing_silence=2.4,
|
||||
rule2_min_trailing_silence=1.2,
|
||||
rule3_min_utterance_length=20,
|
||||
)
|
||||
raise ValueError(f"{spec.id} is not a streaming model (kind={spec.kind})")
|
||||
@@ -97,13 +97,22 @@ def adjust_for_slot(
|
||||
slot_seconds: float,
|
||||
target_lang: str,
|
||||
source_text: Optional[str] = None,
|
||||
strict: bool = False,
|
||||
) -> dict:
|
||||
"""Return `{text, rate_ratio, attempts, error?}`.
|
||||
|
||||
Falls back to the input text if the LLM is off or the loop gives up.
|
||||
|
||||
``strict`` (Autofit mode) caps the accepted upper bound at 1.0 instead of
|
||||
``TOL_HIGH`` — i.e. the line must fit *within* the slot, never overrun it —
|
||||
so the target-language reading time can't exceed the segment and push the
|
||||
video timing out. A too-short line is still accepted down to ``TOL_LOW`` (we
|
||||
don't pad just to fill silence). Best-effort: after ``MAX_ATTEMPTS`` it
|
||||
returns the closest candidate seen, so a stubborn line degrades gracefully.
|
||||
"""
|
||||
tol_high = 1.0 if strict else TOL_HIGH
|
||||
initial_ratio = rate_ratio(text, slot_seconds, target_lang)
|
||||
if TOL_LOW <= initial_ratio <= TOL_HIGH:
|
||||
if TOL_LOW <= initial_ratio <= tol_high:
|
||||
return {"text": text, "rate_ratio": initial_ratio, "attempts": 0}
|
||||
|
||||
llm = get_active_llm_backend()
|
||||
@@ -119,7 +128,7 @@ def adjust_for_slot(
|
||||
best = (current, initial_ratio)
|
||||
for attempt in range(1, MAX_ATTEMPTS + 1):
|
||||
r = rate_ratio(current, slot_seconds, target_lang)
|
||||
if TOL_LOW <= r <= TOL_HIGH:
|
||||
if TOL_LOW <= r <= tol_high:
|
||||
return {"text": current, "rate_ratio": r, "attempts": attempt - 1}
|
||||
|
||||
system = _TRIM_PROMPT if r > 1.0 else _EXPAND_PROMPT
|
||||
|
||||
@@ -120,6 +120,23 @@ def _probe(entry: dict) -> tuple[bool, str]:
|
||||
return False, f"import {mod!r} failed: {e}"
|
||||
|
||||
|
||||
def install_command(engine: "str | dict | None") -> str | None:
|
||||
"""The exact shell command that makes this engine importable, or None.
|
||||
|
||||
Single source of truth for the install string. BOTH the proactive Install
|
||||
affordance in the Engine selector (via list_engines' ``install_command``
|
||||
field) AND the translate-time 400 error (dub_translate.py) read from here,
|
||||
so the command a user is told to run can never drift between the two
|
||||
surfaces. Returns None when the engine needs no separate install — either
|
||||
it's unknown or its dependency is a core dep already pinned in
|
||||
``pyproject.toml`` (e.g. NLLB → transformers), in which case a
|
||||
``uv pip install`` line would be misleading.
|
||||
"""
|
||||
entry = engine if isinstance(engine, dict) else REGISTRY.get(engine) if engine else None
|
||||
pkg = entry.get("pip_package") if entry else None
|
||||
return f"uv pip install {pkg}" if pkg else None
|
||||
|
||||
|
||||
def list_engines() -> list[dict]:
|
||||
"""Return a UI-ready list with per-engine availability stamped in."""
|
||||
out = []
|
||||
@@ -129,6 +146,7 @@ def list_engines() -> list[dict]:
|
||||
**e,
|
||||
"installed": installed,
|
||||
"availability_reason": reason,
|
||||
"install_command": install_command(e),
|
||||
})
|
||||
return out
|
||||
|
||||
@@ -177,6 +195,16 @@ async def run_pip(args: list[str], timeout: float = 600.0) -> tuple[int, str]:
|
||||
"""
|
||||
base = _installer_cmd()
|
||||
using_uv = base[:1] == ["uv"]
|
||||
# Pin `uv pip` to the interpreter the backend ACTUALLY runs under. The desktop
|
||||
# spawns `<venv>/bin/python -m uvicorn` WITHOUT exporting VIRTUAL_ENV, so bare
|
||||
# `uv pip install` finds no venv and 500s with "No virtual environment found"
|
||||
# (#529/#527) — and the `--system` branch below never fires, because the
|
||||
# running interpreter genuinely IS in a venv (uv just can't auto-discover it).
|
||||
# `--python sys.executable` targets the same interpreter _probe()/is_installed()
|
||||
# import from, and takes precedence when both flags are present, so the Docker
|
||||
# `--system` path is unaffected.
|
||||
if using_uv and args and args[0] in ("install", "uninstall") and "--python" not in args:
|
||||
args = [args[0], "--python", sys.executable, *args[1:]]
|
||||
if using_uv and not _in_virtualenv() and args and args[0] in ("install", "uninstall") and "--system" not in args:
|
||||
args = [args[0], "--system", *args[1:]]
|
||||
cmd = base + args
|
||||
|
||||
@@ -1151,6 +1151,13 @@ _LAZY_REGISTRY: dict[str, tuple[str, str]] = {
|
||||
# this module for TTSBackend). The class is resolved on first
|
||||
# attribute access via the LazyRegistry below.
|
||||
"supertonic3": ("engines.supertonic3", "Supertonic3Backend"),
|
||||
# Issue #498: MOSS-TTS-v1.5 (8B) and dots.tts (2B) — both opt-in,
|
||||
# subprocess-isolated with their own venv because each pins a
|
||||
# transformers version that conflicts with the parent's >=5.3
|
||||
# (MOSS == 5.0.0, dots.tts == 4.57.0). Same dedicated-venv pattern as
|
||||
# IndexTTS2. Lazy for the same import-cycle reason as the entries above.
|
||||
"moss-tts-v15": ("engines.moss_tts_v15", "MossTTSV15Backend"),
|
||||
"dots-tts": ("engines.dots_tts", "DotsTTSBackend"),
|
||||
}
|
||||
|
||||
|
||||
@@ -1244,6 +1251,8 @@ _INSTALL_HINTS: dict[str, str] = {
|
||||
"sherpa-onnx": "pip install sherpa-onnx (universal ONNX runtime, WASM-ready)",
|
||||
"omnivoice-gguf":"Bundled — runs the C++ omnivoice-tts binary in bin/. Quants download lazily from Serveurperso/OmniVoice-GGUF on first generate.",
|
||||
"supertonic3": "uv sync --extra supertonic (CPU-only ONNX, 31 langs, ~400 MB model on first use; OpenRAIL-M model license)",
|
||||
"moss-tts-v15": "git clone OpenMOSS/MOSS-TTS + set OMNIVOICE_MOSS_TTS_V15_DIR (own venv, transformers==5.0; 8B, ~16 GB weights; CUDA/CPU, no MPS; Apache-2.0)",
|
||||
"dots-tts": "git clone rednote-hilab/dots.tts + set OMNIVOICE_DOTS_TTS_DIR (own venv, transformers==4.57; 2B, ~9 GB weights; CUDA/CPU, Linux/macOS only — no Windows; Apache-2.0)",
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -25,7 +25,17 @@ sys.modules["core.config"] = _config
|
||||
|
||||
whisperx = pytest.importorskip("whisperx")
|
||||
|
||||
from services.asr_backend import WhisperXBackend # noqa: E402
|
||||
from services.asr_backend import ( # noqa: E402
|
||||
WhisperXBackend,
|
||||
_is_compute_type_error,
|
||||
)
|
||||
|
||||
# The exact ValueError CTranslate2 raises at model construction on a GPU
|
||||
# without efficient fp16 (older Maxwell/Pascal, GTX 16xx) or a cuDNN mismatch.
|
||||
_FP16_ERR = (
|
||||
"Requested float16 compute type, but the target device or backend do not "
|
||||
"support efficient float16 computation"
|
||||
)
|
||||
|
||||
|
||||
def test_cuda_oom_falls_back_to_cpu(monkeypatch):
|
||||
@@ -52,8 +62,13 @@ def test_cuda_oom_falls_back_to_cpu(monkeypatch):
|
||||
|
||||
|
||||
def test_non_oom_runtime_error_still_raises(monkeypatch):
|
||||
msg = "some other failure"
|
||||
# A generic non-OOM, non-compute-type RuntimeError must still propagate —
|
||||
# the new compute_type fallback must NOT swallow it.
|
||||
assert _is_compute_type_error(msg) is False
|
||||
|
||||
def fake_load_model(name, device, compute_type, **kw):
|
||||
raise RuntimeError("some other failure") # not an OOM → must propagate
|
||||
raise RuntimeError(msg) # not an OOM, not compute-type → must propagate
|
||||
|
||||
monkeypatch.setattr(whisperx, "load_model", fake_load_model)
|
||||
|
||||
@@ -62,3 +77,63 @@ def test_non_oom_runtime_error_still_raises(monkeypatch):
|
||||
be._allow_vad_pickle_globals = lambda: None
|
||||
with pytest.raises(RuntimeError, match="some other failure"):
|
||||
be._ensure_asr()
|
||||
|
||||
|
||||
def test_float16_unsupported_falls_back_to_int8(monkeypatch):
|
||||
"""#551: a GPU without efficient fp16 raises a ValueError at load for both
|
||||
float16 AND int8_float16; the backend must degrade to int8 on the SAME
|
||||
device (cuda) without raising — not fall to CPU and not crash."""
|
||||
calls = []
|
||||
|
||||
def fake_load_model(name, device, compute_type, **kw):
|
||||
calls.append((device, compute_type))
|
||||
if device == "cuda" and compute_type in ("float16", "int8_float16"):
|
||||
raise ValueError(_FP16_ERR)
|
||||
return object() # cuda int8 succeeds
|
||||
|
||||
monkeypatch.setattr(whisperx, "load_model", fake_load_model)
|
||||
|
||||
be = WhisperXBackend()
|
||||
be._device, be._compute_type = "cuda", "float16"
|
||||
be._allow_vad_pickle_globals = lambda: None
|
||||
|
||||
be._ensure_asr()
|
||||
|
||||
assert be._asr is not None # recovered, no raise
|
||||
assert be._device == "cuda" and be._compute_type == "int8" # same device, int8
|
||||
assert calls == [("cuda", "float16"), ("cuda", "int8_float16"), ("cuda", "int8")]
|
||||
|
||||
|
||||
def test_faster_whisper_float16_unsupported_falls_back_to_int8(monkeypatch):
|
||||
"""Mirror for FasterWhisperBackend: float16 + int8_float16 raise the fp16
|
||||
ValueError, int8 succeeds → loads on (cuda, int8) without raising."""
|
||||
import services.asr_backend as asr_backend
|
||||
from services.asr_backend import FasterWhisperBackend
|
||||
|
||||
calls = []
|
||||
|
||||
class FakeWhisperModel:
|
||||
def __init__(self, name, device, compute_type, **kw):
|
||||
calls.append((device, compute_type))
|
||||
if device == "cuda" and compute_type in ("float16", "int8_float16"):
|
||||
raise ValueError(_FP16_ERR)
|
||||
# cuda int8 succeeds
|
||||
|
||||
fake_fw = types.ModuleType("faster_whisper")
|
||||
fake_fw.WhisperModel = FakeWhisperModel
|
||||
monkeypatch.setitem(sys.modules, "faster_whisper", fake_fw)
|
||||
|
||||
# Force the CUDA starting point regardless of the CI host's hardware by
|
||||
# making torch.cuda.is_available() return True inside _ensure_model.
|
||||
fake_torch = types.ModuleType("torch")
|
||||
fake_torch.cuda = types.SimpleNamespace(
|
||||
is_available=lambda: True, empty_cache=lambda: None
|
||||
)
|
||||
monkeypatch.setitem(sys.modules, "torch", fake_torch)
|
||||
|
||||
be = FasterWhisperBackend()
|
||||
be._ensure_model()
|
||||
|
||||
assert be._model is not None # recovered, no raise
|
||||
assert be._device == "cuda" and be._compute_type == "int8"
|
||||
assert calls == [("cuda", "float16"), ("cuda", "int8_float16"), ("cuda", "int8")]
|
||||
|
||||
@@ -0,0 +1,84 @@
|
||||
"""speechbrain LazyModule cross-platform guard (#630/#611/#647).
|
||||
|
||||
speechbrain 1.x suppresses optional-integration imports (k2_fsa, numba, …) that
|
||||
are triggered merely by introspection from the stdlib `inspect` module. Its
|
||||
guard checked `filename.endswith("/inspect.py")` — a hardcoded POSIX separator —
|
||||
so on Windows (backslash paths) the guard MISSED and a stray access to the
|
||||
`speechbrain.k2_integration` redirect actually imported the (absent) k2 package,
|
||||
raising `ImportError: Lazy import of LazyModule(...k2_fsa...) failed` that aborted
|
||||
WhisperX transcription with zero segments.
|
||||
|
||||
`_harden_speechbrain_lazy_imports()` re-implements `ensure_module` with an
|
||||
`os.path.basename` check so the guard fires on every platform. These tests fake
|
||||
the importer frame (both Windows- and POSIX-style `inspect.py` paths, plus a
|
||||
real-caller path) so they pin the behaviour regardless of the host OS.
|
||||
"""
|
||||
import os
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))))
|
||||
|
||||
importutils = pytest.importorskip(
|
||||
"speechbrain.utils.importutils",
|
||||
reason="speechbrain not installed in this environment",
|
||||
)
|
||||
from services.asr_backend import _harden_speechbrain_lazy_imports # noqa: E402
|
||||
|
||||
|
||||
class _FakeFrameInfo:
|
||||
def __init__(self, filename):
|
||||
self.filename = filename
|
||||
|
||||
|
||||
def _bogus_lazy_module():
|
||||
# A LazyModule whose target can never import — so we can observe whether the
|
||||
# inspect.py guard fired (AttributeError) or the import was attempted (ImportError).
|
||||
return importutils.LazyModule(
|
||||
"omnivoice_nonexistent_zzz",
|
||||
"omnivoice_nonexistent_zzz_target",
|
||||
None,
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"inspect_path",
|
||||
[
|
||||
r"C:\Python311\Lib\inspect.py", # Windows — the case the old guard missed
|
||||
"/usr/lib/python3.11/inspect.py", # POSIX — already worked, must keep working
|
||||
],
|
||||
)
|
||||
def test_guard_fires_for_inspect_frame_on_any_separator(monkeypatch, inspect_path):
|
||||
_harden_speechbrain_lazy_imports()
|
||||
lm = _bogus_lazy_module()
|
||||
monkeypatch.setattr(
|
||||
importutils.inspect, "getframeinfo",
|
||||
lambda *_a, **_k: _FakeFrameInfo(inspect_path),
|
||||
)
|
||||
# Guard must treat an inspect.py-triggered access as "attribute absent"
|
||||
# (AttributeError) rather than attempting the doomed import (ImportError).
|
||||
with pytest.raises(AttributeError):
|
||||
lm.ensure_module(0)
|
||||
|
||||
|
||||
def test_real_caller_still_surfaces_import_error(monkeypatch):
|
||||
"""A genuine access from real user code (not inspect.py) with the target
|
||||
missing must still raise ImportError — we only suppress inspect-triggered
|
||||
spurious imports, never legitimate failures."""
|
||||
_harden_speechbrain_lazy_imports()
|
||||
lm = _bogus_lazy_module()
|
||||
monkeypatch.setattr(
|
||||
importutils.inspect, "getframeinfo",
|
||||
lambda *_a, **_k: _FakeFrameInfo(r"C:\Users\me\app\real_caller.py"),
|
||||
)
|
||||
with pytest.raises(ImportError):
|
||||
lm.ensure_module(0)
|
||||
|
||||
|
||||
def test_patch_is_idempotent():
|
||||
_harden_speechbrain_lazy_imports()
|
||||
first = importutils.LazyModule.ensure_module
|
||||
_harden_speechbrain_lazy_imports()
|
||||
assert importutils.LazyModule.ensure_module is first
|
||||
assert getattr(importutils.LazyModule, "_omnivoice_xplat_guard", False) is True
|
||||
@@ -0,0 +1,115 @@
|
||||
"""Whole-file ASR transcribe must be wall-clock bounded (TamKieu / Vietnam report).
|
||||
|
||||
The chunked dub pipeline already bounds each chunk, but the whole-file paths
|
||||
(dub QC re-transcribe, dictation, OpenAI-compat) ran unbounded — a slow/stuck
|
||||
transcribe (e.g. large-v3 on a VRAM-starved GPU) hung the request *and* held a
|
||||
GPU-pool worker, surfacing in the UI as the misleading "can't reach the local
|
||||
backend". `run_transcribe_guarded` bounds them and raises `ASRTimeoutError` with
|
||||
actionable guidance. These tests pin the timeout path, the pass-through path, and
|
||||
that the error message tells the user what to do.
|
||||
"""
|
||||
import asyncio
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))))
|
||||
|
||||
from services.asr_backend import ( # noqa: E402
|
||||
ASRTimeoutError,
|
||||
ASR_TRANSCRIBE_TIMEOUT_S,
|
||||
run_transcribe_guarded,
|
||||
)
|
||||
from concurrent.futures import ThreadPoolExecutor # noqa: E402
|
||||
|
||||
|
||||
def test_default_timeout_is_env_overridable(monkeypatch):
|
||||
# The constant is read at import; just assert it's a sane positive default.
|
||||
assert ASR_TRANSCRIBE_TIMEOUT_S > 0
|
||||
|
||||
|
||||
def test_slow_transcribe_raises_actionable_timeout():
|
||||
pool = ThreadPoolExecutor(max_workers=1)
|
||||
|
||||
def _hang():
|
||||
time.sleep(5) # would block far past our tiny timeout
|
||||
return "never"
|
||||
|
||||
async def _go():
|
||||
with pytest.raises(ASRTimeoutError) as ei:
|
||||
await run_transcribe_guarded(pool, _hang, what="QC", timeout=0.2)
|
||||
msg = str(ei.value)
|
||||
# Message must reassure (backend alive) + give concrete remedies.
|
||||
assert "backend is running" in msg
|
||||
assert "Settings → Models" in msg
|
||||
assert "CPU" in msg
|
||||
|
||||
asyncio.run(_go())
|
||||
pool.shutdown(wait=False)
|
||||
|
||||
|
||||
def test_fast_transcribe_passes_through():
|
||||
pool = ThreadPoolExecutor(max_workers=1)
|
||||
|
||||
def _quick():
|
||||
return {"segments": [{"text": "hi"}]}, "whisperx"
|
||||
|
||||
async def _go():
|
||||
out = await run_transcribe_guarded(pool, _quick, what="Dictation", timeout=5.0)
|
||||
assert out == ({"segments": [{"text": "hi"}]}, "whisperx")
|
||||
|
||||
asyncio.run(_go())
|
||||
pool.shutdown(wait=True)
|
||||
|
||||
|
||||
def test_timeout_error_is_a_timeouterror_subclass():
|
||||
# Routers that catch broad TimeoutError (openai_compat) must also catch ours.
|
||||
assert issubclass(ASRTimeoutError, TimeoutError)
|
||||
|
||||
|
||||
def test_timeout_resets_a_resilient_pool_to_restore_capacity():
|
||||
# #730: a wedged transcribe holds its GPU-pool worker forever; with a 1-2
|
||||
# worker pool that starves TTS generate and surfaces as "can't reach
|
||||
# backend". On timeout, run_transcribe_guarded must reset() a pool that
|
||||
# supports it (the real _ResilientGpuPool) so the next submit gets a fresh
|
||||
# worker — capacity restored without an app restart.
|
||||
class _FakePool(ThreadPoolExecutor):
|
||||
def __init__(self):
|
||||
super().__init__(max_workers=1)
|
||||
self.reset_calls = 0
|
||||
|
||||
def reset(self):
|
||||
self.reset_calls += 1
|
||||
|
||||
pool = _FakePool()
|
||||
|
||||
def _hang():
|
||||
time.sleep(5)
|
||||
return "never"
|
||||
|
||||
async def _go():
|
||||
with pytest.raises(ASRTimeoutError):
|
||||
await run_transcribe_guarded(pool, _hang, what="Dub", timeout=0.2)
|
||||
|
||||
asyncio.run(_go())
|
||||
assert pool.reset_calls == 1
|
||||
pool.shutdown(wait=False)
|
||||
|
||||
|
||||
def test_timeout_without_reset_capable_pool_does_not_crash():
|
||||
# A plain ThreadPoolExecutor (no reset) must still bound + raise cleanly —
|
||||
# the reset() is best-effort, never required.
|
||||
pool = ThreadPoolExecutor(max_workers=1)
|
||||
|
||||
def _hang():
|
||||
time.sleep(5)
|
||||
return "never"
|
||||
|
||||
async def _go():
|
||||
with pytest.raises(ASRTimeoutError):
|
||||
await run_transcribe_guarded(pool, _hang, what="QC", timeout=0.2)
|
||||
|
||||
asyncio.run(_go())
|
||||
pool.shutdown(wait=False)
|
||||
@@ -8,6 +8,7 @@
|
||||
"concurrently": "^9.2.1",
|
||||
"kill-port-process": "^4.0.2",
|
||||
"playwright": "^1.60.0",
|
||||
"taze": "^19.14.1",
|
||||
"turbo": "^2.9.18",
|
||||
"typescript": "^6.0.3",
|
||||
"wait-on": "^9.0.10",
|
||||
@@ -15,18 +16,19 @@
|
||||
},
|
||||
"frontend": {
|
||||
"name": "omnivoice-studio",
|
||||
"version": "0.3.5",
|
||||
"version": "0.3.8",
|
||||
"dependencies": {
|
||||
"@fontsource-variable/inter": "^5.2.8",
|
||||
"@fontsource-variable/source-serif-4": "^5.2.9",
|
||||
"@fontsource/ibm-plex-mono": "^5.2.7",
|
||||
"@radix-ui/react-dialog": "^1.1.17",
|
||||
"@radix-ui/react-dropdown-menu": "^2.1.18",
|
||||
"@radix-ui/react-popover": "^1.1.17",
|
||||
"@radix-ui/react-progress": "^1.1.10",
|
||||
"@radix-ui/react-select": "^2.3.1",
|
||||
"@radix-ui/react-slider": "^1.4.1",
|
||||
"@radix-ui/react-slot": "^1.3.0",
|
||||
"@radix-ui/react-tabs": "^1.1.15",
|
||||
"@radix-ui/react-toggle": "^1.1.12",
|
||||
"@radix-ui/react-toggle-group": "^1.1.13",
|
||||
"@radix-ui/react-tooltip": "^1.2.10",
|
||||
"@tailwindcss/vite": "^4.3.1",
|
||||
@@ -38,6 +40,8 @@
|
||||
"@tauri-apps/plugin-process": "^2.3.1",
|
||||
"@tauri-apps/plugin-updater": "^2.10.1",
|
||||
"@tauri-apps/plugin-window-state": "^2.4.1",
|
||||
"class-variance-authority": "^0.7.1",
|
||||
"clsx": "^2.1.1",
|
||||
"country-flag-icons": "^1.6.17",
|
||||
"i18next": "^26.3.1",
|
||||
"i18next-browser-languagedetector": "^8.2.1",
|
||||
@@ -48,12 +52,13 @@
|
||||
"react-hot-toast": "^2.6.0",
|
||||
"react-i18next": "^17.0.8",
|
||||
"react-window": "^2.2.7",
|
||||
"tailwind-merge": "^3.6.0",
|
||||
"tailwindcss": "^4.3.1",
|
||||
"tw-animate-css": "^1.4.0",
|
||||
"wavesurfer.js": "^7.12.8",
|
||||
"zustand": "^5.0.14",
|
||||
},
|
||||
"devDependencies": {
|
||||
"@eslint/js": "^10.0.1",
|
||||
"@playwright/test": "^1.61.0",
|
||||
"@tauri-apps/api": "^2.11.0",
|
||||
"@tauri-apps/cli": "^2.11.2",
|
||||
@@ -64,9 +69,11 @@
|
||||
"@vitejs/plugin-react": "^6.0.2",
|
||||
"eslint": "^10.5.0",
|
||||
"eslint-plugin-react-hooks": "^7.1.1",
|
||||
"eslint-plugin-react-refresh": "^0.5.3",
|
||||
"globals": "^17.6.0",
|
||||
"jsdom": "^29.1.1",
|
||||
"knip": "^6.23.0",
|
||||
"oxfmt": "^0.57.0",
|
||||
"oxlint": "^1.71.0",
|
||||
"playwright-core": "1.61.0",
|
||||
"typescript": "^6.0.3",
|
||||
"vite": "^8.0.16",
|
||||
@@ -77,6 +84,8 @@
|
||||
"packages": {
|
||||
"@adobe/css-tools": ["@adobe/css-tools@4.4.4", "", {}, "sha512-Elp+iwUx5rN5+Y8xLt5/GRoG20WGoDCQ/1Fb+1LiGtvwbDavuSk0jhD/eZdckHAuzcDzccnkv+rEjyWfRx18gg=="],
|
||||
|
||||
"@antfu/ni": ["@antfu/ni@30.2.0", "", { "dependencies": { "fzf": "^0.5.2", "package-manager-detector": "^1.6.0", "tinyexec": "^1.2.4", "tinyglobby": "^0.2.17" }, "bin": { "ni": "bin/ni.mjs", "nci": "bin/nci.mjs", "nr": "bin/nr.mjs", "nup": "bin/nup.mjs", "nd": "bin/nd.mjs", "nlx": "bin/nlx.mjs", "na": "bin/na.mjs", "nun": "bin/nun.mjs" } }, "sha512-/FOdAP1w8COnANVD3TtNj/tnpt/36RkU/ysKZTqx86x9acdhCqTFjDXNYVDyBg6UzcrTwWPUeY75ng7CWLNr+g=="],
|
||||
|
||||
"@asamuzakjp/css-color": ["@asamuzakjp/css-color@5.1.11", "", { "dependencies": { "@asamuzakjp/generational-cache": "^1.0.1", "@csstools/css-calc": "^3.2.0", "@csstools/css-color-parser": "^4.1.0", "@csstools/css-parser-algorithms": "^4.0.0", "@csstools/css-tokenizer": "^4.0.0" } }, "sha512-KVw6qIiCTUQhByfTd78h2yD1/00waTmm9uy/R7Ck/ctUyAPj+AEDLkQIdJW0T8+qGgj3j5bpNKK7Q3G+LedJWg=="],
|
||||
|
||||
"@asamuzakjp/dom-selector": ["@asamuzakjp/dom-selector@7.1.1", "", { "dependencies": { "@asamuzakjp/generational-cache": "^1.0.1", "@asamuzakjp/nwsapi": "^2.3.9", "bidi-js": "^1.0.3", "css-tree": "^3.2.1", "is-potential-custom-element-name": "^1.0.1" } }, "sha512-67RZDnYRc8H/8MLDgQCDE//zoqVFwajkepHZgmXrbwybzXOEwOWGPYGmALYl9J2DOLfFPPs6kKCqmbzV895hTQ=="],
|
||||
@@ -133,11 +142,11 @@
|
||||
|
||||
"@csstools/css-tokenizer": ["@csstools/css-tokenizer@4.0.0", "", {}, "sha512-QxULHAm7cNu72w97JUNCBFODFaXpbDg+dP8b/oWFAZ2MTRppA3U00Y2L1HqaS4J6yBqxwa/Y3nMBaxVKbB/NsA=="],
|
||||
|
||||
"@emnapi/core": ["@emnapi/core@1.10.0", "", { "dependencies": { "@emnapi/wasi-threads": "1.2.1", "tslib": "^2.4.0" } }, "sha512-yq6OkJ4p82CAfPl0u9mQebQHKPJkY7WrIuk205cTYnYe+k2Z8YBh11FrbRG/H6ihirqcacOgl2BIO8oyMQLeXw=="],
|
||||
"@emnapi/core": ["@emnapi/core@1.11.1", "", { "dependencies": { "@emnapi/wasi-threads": "1.2.2", "tslib": "^2.4.0" } }, "sha512-RSvbQmHzdKzNsLYa/wHrbc3KN4sYLKAdPZxqiM2HATqv/SBk2/ENSHpvXGaLOMcsAyz0poEGqkmmKYG3OWiJEQ=="],
|
||||
|
||||
"@emnapi/runtime": ["@emnapi/runtime@1.10.0", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-ewvYlk86xUoGI0zQRNq/mC+16R1QeDlKQy21Ki3oSYXNgLb45GV1P6A0M+/s6nyCuNDqe5VpaY84BzXGwVbwFA=="],
|
||||
"@emnapi/runtime": ["@emnapi/runtime@1.11.1", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-vgj7R3y3Wgx24IQaGPA/R6YFXLHVMOZ0uVEyIQPaWs+rd1AzfEMXlAC22FYwO1XkKR6NPsq7mUandH8oIRdZFw=="],
|
||||
|
||||
"@emnapi/wasi-threads": ["@emnapi/wasi-threads@1.2.1", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-uTII7OYF+/Mes/MrcIOYp5yOtSMLBWSIoLPpcgwipoiKbli6k322tcoFsxoIIxPDqW01SQGAgko4EzZi2BNv2w=="],
|
||||
"@emnapi/wasi-threads": ["@emnapi/wasi-threads@1.2.2", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-c95qOXkHdydNKhscBTebqEC1CVAZpyqOfVfBzQ1qgzyl3gfeldUjIggDbIZgDKsHLgnsM+igH7TJ/eAasaVuMA=="],
|
||||
|
||||
"@eslint-community/eslint-utils": ["@eslint-community/eslint-utils@4.9.1", "", { "dependencies": { "eslint-visitor-keys": "^3.4.3" }, "peerDependencies": { "eslint": "^6.0.0 || ^7.0.0 || >=8.0.0" } }, "sha512-phrYmNiYppR7znFEdqgfWHXR6NCkZEK7hwWDHZUjit/2/U0r6XvkDl0SYnoM51Hq7FhCGdLDT6zxCCOY1hexsQ=="],
|
||||
|
||||
@@ -149,8 +158,6 @@
|
||||
|
||||
"@eslint/core": ["@eslint/core@1.2.1", "", { "dependencies": { "@types/json-schema": "^7.0.15" } }, "sha512-MwcE1P+AZ4C6DWlpin/OmOA54mmIZ/+xZuJiQd4SyB29oAJjN30UW9wkKNptW2ctp4cEsvhlLY/CsQ1uoHDloQ=="],
|
||||
|
||||
"@eslint/js": ["@eslint/js@10.0.1", "", { "peerDependencies": { "eslint": "^10.0.0" }, "optionalPeers": ["eslint"] }, "sha512-zeR9k5pd4gxjZ0abRoIaxdc7I3nDktoXZk2qOv9gCNWx3mVwEn32VRhyLaRsDiJjTs0xq/T8mfPtyuXu7GWBcA=="],
|
||||
|
||||
"@eslint/object-schema": ["@eslint/object-schema@3.0.5", "", {}, "sha512-vqTaUEgxzm+YDSdElad6PiRoX4t8VGDjCtt05zn4nU810UIx/uNEV7/lZJ6KwFThKZOzOxzXy48da+No7HZaMw=="],
|
||||
|
||||
"@eslint/plugin-kit": ["@eslint/plugin-kit@0.7.2", "", { "dependencies": { "@eslint/core": "^1.2.1", "levn": "^0.4.1" } }, "sha512-+CNAzxglkrpNf/kKywqQfk74QjtceuOE7Qm+AF8miRvPF/wmmK5+OJOgVh3AVTT3RP2mH3+FOaxlE5v72owk0A=="],
|
||||
@@ -183,6 +190,8 @@
|
||||
|
||||
"@hapi/topo": ["@hapi/topo@6.0.2", "", { "dependencies": { "@hapi/hoek": "^11.0.2" } }, "sha512-KR3rD5inZbGMrHmgPxsJ9dbi6zEK+C3ZwUwTa+eMwWLz7oijWUTWD2pMSNNYJAU6Qq+65NkxXjqHr/7LM2Xkqg=="],
|
||||
|
||||
"@henrygd/queue": ["@henrygd/queue@1.2.0", "", {}, "sha512-jW/BLSTpcvExDhqJGxtIPgGr2O0IFF8XUNDwEbfCfhrXT8a4xztQ9Lv6U/vbYzYC0xVWn+3zv6YnLUh3bEFUKA=="],
|
||||
|
||||
"@humanfs/core": ["@humanfs/core@0.19.1", "", {}, "sha512-5DyQ4+1JEUzejeK1JGICcideyfUbGixgS9jNgex5nqkW+cY7WZhxBigmieN5Qnw9ZosSNVC9KQKyb+GUaGyKUA=="],
|
||||
|
||||
"@humanfs/node": ["@humanfs/node@0.16.7", "", { "dependencies": { "@humanfs/core": "^0.19.1", "@humanwhocodes/retry": "^0.4.0" } }, "sha512-/zUx+yOsIrG4Y43Eh2peDeKCxlRt/gET6aHfaKpuq267qXdYDFViVHfMaLyygZOnl0kGWxFIgsBy8QFuTLUXEQ=="],
|
||||
@@ -201,12 +210,168 @@
|
||||
|
||||
"@jridgewell/trace-mapping": ["@jridgewell/trace-mapping@0.3.31", "", { "dependencies": { "@jridgewell/resolve-uri": "^3.1.0", "@jridgewell/sourcemap-codec": "^1.4.14" } }, "sha512-zzNR+SdQSDJzc8joaeP8QQoCQr8NuYx2dIIytl1QeBEZHJ9uW6hebsrYgbz8hJwUQao3TWCMtmfV8Nu1twOLAw=="],
|
||||
|
||||
"@napi-rs/wasm-runtime": ["@napi-rs/wasm-runtime@1.1.4", "", { "dependencies": { "@tybys/wasm-util": "^0.10.1" }, "peerDependencies": { "@emnapi/core": "^1.7.1", "@emnapi/runtime": "^1.7.1" } }, "sha512-3NQNNgA1YSlJb/kMH1ildASP9HW7/7kYnRI2szWJaofaS1hWmbGI4H+d3+22aGzXXN9IJ+n+GiFVcGipJP18ow=="],
|
||||
"@napi-rs/wasm-runtime": ["@napi-rs/wasm-runtime@1.1.6", "", { "dependencies": { "@tybys/wasm-util": "^0.10.3" }, "peerDependencies": { "@emnapi/core": "^1.7.1", "@emnapi/runtime": "^1.7.1" } }, "sha512-ZLv/JdUfkvOy9eCnnBaGfiO+XimbjebAeO+MRQqD/B+FR1tnRN0tpKSJHRbE8sFfS6aqsXZ67TQjfwfsxULVbg=="],
|
||||
|
||||
"@oxc-project/types": ["@oxc-project/types@0.133.0", "", {}, "sha512-KzkdCd6Uxqnf6l3HOw1xfatAlUURA0g14cvBYFyJ5SaNOQbOUvBr9PKArcPcrNIeRsBdgcUzOGrhKveVpvOIGA=="],
|
||||
"@oxc-parser/binding-android-arm-eabi": ["@oxc-parser/binding-android-arm-eabi@0.137.0", "", { "os": "android", "cpu": "arm" }, "sha512-KDs+0VPdEmasOkpuJHW9V5WCF+cvYdMQv2Jd+aJXt+cxIx12NToRQRbXaRwUEDsZw+/jMk81Ve8ZFbjUkJTOwA=="],
|
||||
|
||||
"@oxc-parser/binding-android-arm64": ["@oxc-parser/binding-android-arm64@0.137.0", "", { "os": "android", "cpu": "arm64" }, "sha512-WhALNzfy3x/RfC6bsqX+csavuUY0yHHE7XfgPE5M542uhoBZUUoGTPG+nkMbGoG4+gcfss5s7urMyn5QBHu0sw=="],
|
||||
|
||||
"@oxc-parser/binding-darwin-arm64": ["@oxc-parser/binding-darwin-arm64@0.137.0", "", { "os": "darwin", "cpu": "arm64" }, "sha512-bFPr5hgmNMOMoyPTGtdsK4Ug21RovIPojRMgDDhSp1LtCnc/DkLwGONKjgRjszg677RlGnkYSviQ8hHaUPOVYA=="],
|
||||
|
||||
"@oxc-parser/binding-darwin-x64": ["@oxc-parser/binding-darwin-x64@0.137.0", "", { "os": "darwin", "cpu": "x64" }, "sha512-CL5dMm1asqXIDZHg14FLxj3Mc36w8PI7xCWh1uA4is6z8g2XrIILoTcQYOxDbwzuk34RDPX5IAGUxZr6LA9KAg=="],
|
||||
|
||||
"@oxc-parser/binding-freebsd-x64": ["@oxc-parser/binding-freebsd-x64@0.137.0", "", { "os": "freebsd", "cpu": "x64" }, "sha512-79h8rYGnSlKPGWo7mHr2ixO6ea7aW8B0CT965SZ8SLbNnCOH5aOYBTeVXUY6eMvEaiLyWr8Skuiugr5pDYgLGw=="],
|
||||
|
||||
"@oxc-parser/binding-linux-arm-gnueabihf": ["@oxc-parser/binding-linux-arm-gnueabihf@0.137.0", "", { "os": "linux", "cpu": "arm" }, "sha512-ASgmlSimhGyr0lksgVIo6hibz1obnDq4qJbiMX/AzltfgPnanRrzG1Q+23g8ljOHOjv6dsznkUuCYL3gg0sY1Q=="],
|
||||
|
||||
"@oxc-parser/binding-linux-arm-musleabihf": ["@oxc-parser/binding-linux-arm-musleabihf@0.137.0", "", { "os": "linux", "cpu": "arm" }, "sha512-AU2J9aa22Sx32wRGnDjybOU9TQXXQUud5sdUi+ZB0XxwM8aToWLweV+yA0wlQm0yIUVqljquqoHCYEq9II8gJQ=="],
|
||||
|
||||
"@oxc-parser/binding-linux-arm64-gnu": ["@oxc-parser/binding-linux-arm64-gnu@0.137.0", "", { "os": "linux", "cpu": "arm64" }, "sha512-GdEtiG89yMr7XkUGxifgodXEEm2f+xW2f9CpDjlgAnBOwhTmrpQMvhOGobLVKUyzf/qHBXW16smk5zbF3nZU6w=="],
|
||||
|
||||
"@oxc-parser/binding-linux-arm64-musl": ["@oxc-parser/binding-linux-arm64-musl@0.137.0", "", { "os": "linux", "cpu": "arm64" }, "sha512-EGJ+Bs8iXx8KBH8DQ5BLoEm5lnHaYjlh4/8j8vFhrr/6z4tqONy5BZDzLpKmmNWlN6Hlc5r8YOuBVHqZ9vRFEQ=="],
|
||||
|
||||
"@oxc-parser/binding-linux-ppc64-gnu": ["@oxc-parser/binding-linux-ppc64-gnu@0.137.0", "", { "os": "linux", "cpu": "ppc64" }, "sha512-vzFUQENy/fnbSe5DZWovq6tIBc1uhuMztanSW6rz1e9WdQE4gHwYuD7ZII6JnrJifd1R3RSoqiZbgRFlVL2tYQ=="],
|
||||
|
||||
"@oxc-parser/binding-linux-riscv64-gnu": ["@oxc-parser/binding-linux-riscv64-gnu@0.137.0", "", { "os": "linux", "cpu": "none" }, "sha512-SfVI14HBQs9gtLcUD5hTt5hsNbdrqSUNg9S8muN+LhVQ5nf1WwH3hAoK6B9NKgdYgWAQSXFXGiiBedQ4r/BKuw=="],
|
||||
|
||||
"@oxc-parser/binding-linux-riscv64-musl": ["@oxc-parser/binding-linux-riscv64-musl@0.137.0", "", { "os": "linux", "cpu": "none" }, "sha512-e7Ppy4FCIFNQxT/ikSeIWFoQ0l+N9vgtRBtLcyZXeolTzApyVoPqEXsYPrcdM/9i0Bwk8knvYd37vaEMxHyi6g=="],
|
||||
|
||||
"@oxc-parser/binding-linux-s390x-gnu": ["@oxc-parser/binding-linux-s390x-gnu@0.137.0", "", { "os": "linux", "cpu": "s390x" }, "sha512-Bho5qFwdhqsIFR7gipYEUlqvi3SRrY8sugxXig380MIaakBB1PyU9+7dBiBVScfImTNWhijUxdBwqrprGdq5WA=="],
|
||||
|
||||
"@oxc-parser/binding-linux-x64-gnu": ["@oxc-parser/binding-linux-x64-gnu@0.137.0", "", { "os": "linux", "cpu": "x64" }, "sha512-36mGWtg7PyFzjJwGDkH6/F4o2nIDEoKXLPr/X/lwqklkomQwJJt1I5GJVmGhovUEmgPK5WAeAZMqlFCehwiy9Q=="],
|
||||
|
||||
"@oxc-parser/binding-linux-x64-musl": ["@oxc-parser/binding-linux-x64-musl@0.137.0", "", { "os": "linux", "cpu": "x64" }, "sha512-/Jqx6+N7A44n2BdvUr7pXhVr2vFjs6WGH3unZRczwrfiH0H1zY0QwKQMG/dtRiTlKGDKGukznPT8lx84/oEsZg=="],
|
||||
|
||||
"@oxc-parser/binding-openharmony-arm64": ["@oxc-parser/binding-openharmony-arm64@0.137.0", "", { "os": "none", "cpu": "arm64" }, "sha512-9Uj0qHNNl+OgT1UTGwF7ixIXU6T1u2SbMidmgPy/h1h/fl2gRS6YpAxxY1gwHofcWjoTwkoMFd8xs5Vuj6GOFA=="],
|
||||
|
||||
"@oxc-parser/binding-wasm32-wasi": ["@oxc-parser/binding-wasm32-wasi@0.137.0", "", { "dependencies": { "@emnapi/core": "1.11.1", "@emnapi/runtime": "1.11.1", "@napi-rs/wasm-runtime": "^1.1.5" }, "cpu": "none" }, "sha512-gW2vfkytNGgMVADiuzdvOfw0mWG9za20F/1fCJsif5aBMAvWJTSbpIXbIe0XkOe0VENk+PadpQ7cZgUy2sUJcA=="],
|
||||
|
||||
"@oxc-parser/binding-win32-arm64-msvc": ["@oxc-parser/binding-win32-arm64-msvc@0.137.0", "", { "os": "win32", "cpu": "arm64" }, "sha512-x+pFANF0yL5uK/6T7lu6SlR5qid6sp//eZXKLq5iNsIE+EQg6EaS8/wsW7E91nXXjpnPhSoMOHXShSVhGRdn8w=="],
|
||||
|
||||
"@oxc-parser/binding-win32-ia32-msvc": ["@oxc-parser/binding-win32-ia32-msvc@0.137.0", "", { "os": "win32", "cpu": "ia32" }, "sha512-sQUqym80PFi6McRsIqfJrSu2JrSClEZIXXD+/FjAFoULEKzOPsldIdFBG96xdX8aVMzCNQ9792FPx3MfkEIrFA=="],
|
||||
|
||||
"@oxc-parser/binding-win32-x64-msvc": ["@oxc-parser/binding-win32-x64-msvc@0.137.0", "", { "os": "win32", "cpu": "x64" }, "sha512-2AsevxlvNN4WKxpEn3RtqD5zbqMaXF+T7JXblsP4gVuY+vC9dXS4ED/PwfRCliFqoeisYS3Iro4DHzxr0TEvVA=="],
|
||||
|
||||
"@oxc-project/types": ["@oxc-project/types@0.137.0", "", {}, "sha512-WT+Gb24i8hmvo85AIv2oEYouEXkRlKAlT9WaCa3TfLgNCN+GhrJOGZuIlMouAh38Qe4QOx26eUOVsq70qXrywA=="],
|
||||
|
||||
"@oxc-resolver/binding-android-arm-eabi": ["@oxc-resolver/binding-android-arm-eabi@11.21.3", "", { "os": "android", "cpu": "arm" }, "sha512-eNU11A2WNizh04v3uyaJCootrHIaS0B9aHYXvAvVnPNk4xYSjMUjHnhQ6dewPN2MRYDskV85d1N0Aw0WNWhcyg=="],
|
||||
|
||||
"@oxc-resolver/binding-android-arm64": ["@oxc-resolver/binding-android-arm64@11.21.3", "", { "os": "android", "cpu": "arm64" }, "sha512-8Q+ZjTLvn2dIcWsrmhdrEihm7q+ag/k+mkry7Z+t0QbbHaVxXQfvH9AewyVMh/WrpEKhQ3DDgx9fYbqeCpeOEw=="],
|
||||
|
||||
"@oxc-resolver/binding-darwin-arm64": ["@oxc-resolver/binding-darwin-arm64@11.21.3", "", { "os": "darwin", "cpu": "arm64" }, "sha512-wkh0qKZGHXVUDxFw3oA1TXnU2BDYY/r775oJflGeIr8uDPPoN2pk8gijQIzYRT6hoql/lg3+Tx/SaTn9e2/aGg=="],
|
||||
|
||||
"@oxc-resolver/binding-darwin-x64": ["@oxc-resolver/binding-darwin-x64@11.21.3", "", { "os": "darwin", "cpu": "x64" }, "sha512-HbNc23FAQYbuyDV2vBWMez4u4mrsm5RAkniGZAWqr6lYZ3N4beeqIb776jzwRl8qL2zRhHVXpUj97X0QgogVzg=="],
|
||||
|
||||
"@oxc-resolver/binding-freebsd-x64": ["@oxc-resolver/binding-freebsd-x64@11.21.3", "", { "os": "freebsd", "cpu": "x64" }, "sha512-K6xNsTUPEUdfrn0+kbMq5nOUB5w1C5pavPQngt4TM2FpN91lP0PBe2srSpamb4d69O7h86oAi/qWX/kZNRSjkw=="],
|
||||
|
||||
"@oxc-resolver/binding-linux-arm-gnueabihf": ["@oxc-resolver/binding-linux-arm-gnueabihf@11.21.3", "", { "os": "linux", "cpu": "arm" }, "sha512-VcFmOpcpWX1zoEy8M58tR2M9YxM+Z9RuQhqAx5q0CTmrruaP7Gveejg75hzd/5sg5nk9G3aLALEa3hE2FsmmTQ=="],
|
||||
|
||||
"@oxc-resolver/binding-linux-arm-musleabihf": ["@oxc-resolver/binding-linux-arm-musleabihf@11.21.3", "", { "os": "linux", "cpu": "arm" }, "sha512-quVoxFLBy43hWaQbbDtQNRwAX5vX76mv7n64icAtQcJ3eNgVeblqmkupF/hAneNthdqSlnd1sTjb3aQSaDPaCQ=="],
|
||||
|
||||
"@oxc-resolver/binding-linux-arm64-gnu": ["@oxc-resolver/binding-linux-arm64-gnu@11.21.3", "", { "os": "linux", "cpu": "arm64" }, "sha512-X0AqNZgcD07Q4V3RDK18/vYOj/HQT/FnmEFGYS2jTWqY7JO13ryE3TEs3eAIgUJhBnNkpEaiXqz3VK8M7qQhWQ=="],
|
||||
|
||||
"@oxc-resolver/binding-linux-arm64-musl": ["@oxc-resolver/binding-linux-arm64-musl@11.21.3", "", { "os": "linux", "cpu": "arm64" }, "sha512-YkaQnaKYdbuaXvRt5Qd0GpbihzVnyfR6z1SpYfIUC6RTu4NF7lDKPjVkYb+jRI2gedVO2rVpN35Y6akG6ud4Lw=="],
|
||||
|
||||
"@oxc-resolver/binding-linux-ppc64-gnu": ["@oxc-resolver/binding-linux-ppc64-gnu@11.21.3", "", { "os": "linux", "cpu": "ppc64" }, "sha512-gB9HwhrPiFqUzDeEq+y/CgAijz1YdI6BnXz5GaH2Pa9cWdutchlkGFAiAuGb/PjVQpiK6NFKzFuztxrweoit7A=="],
|
||||
|
||||
"@oxc-resolver/binding-linux-riscv64-gnu": ["@oxc-resolver/binding-linux-riscv64-gnu@11.21.3", "", { "os": "linux", "cpu": "none" }, "sha512-zjDWBlYk8QGv0H8dsPUWqkfjYIIjG2TvspGkzXL0eImbgxtZorA/klKeHyolevoT3Kvbi+1iMr9Lhrh7jf54Og=="],
|
||||
|
||||
"@oxc-resolver/binding-linux-riscv64-musl": ["@oxc-resolver/binding-linux-riscv64-musl@11.21.3", "", { "os": "linux", "cpu": "none" }, "sha512-4UfsQvacV388y1zpXL7C1x1FNYaV52JtuNRiuzrfQA2z1z6ElVrsidkGsrvQ5EgeSq1Pj7kaKqrgGkvFuxJ/tw=="],
|
||||
|
||||
"@oxc-resolver/binding-linux-s390x-gnu": ["@oxc-resolver/binding-linux-s390x-gnu@11.21.3", "", { "os": "linux", "cpu": "s390x" }, "sha512-b5uH+HKH0MP5mNBYaK75SKsJbw52URqrx2LavYdq6wb0l3ExAG5niYRP9DWUNHdKilpaBVM2bXk9HNWrH3ew7Q=="],
|
||||
|
||||
"@oxc-resolver/binding-linux-x64-gnu": ["@oxc-resolver/binding-linux-x64-gnu@11.21.3", "", { "os": "linux", "cpu": "x64" }, "sha512-PjYlmilBpNRh2ntXNYAK3Am5w/nPfEpnU/96iNx7CI8EzAn12J4JRiec63wHJTH31nLoCNxBg/829pN+3CfG3Q=="],
|
||||
|
||||
"@oxc-resolver/binding-linux-x64-musl": ["@oxc-resolver/binding-linux-x64-musl@11.21.3", "", { "os": "linux", "cpu": "x64" }, "sha512-QTBAb7JuHlZ7JUEyM8UiQi2f7m/L4swBhP2TNpYIDc9Wp/wRw1G/8sl6i13aIzQAXH7LKIm294LeOHd0lQR8zA=="],
|
||||
|
||||
"@oxc-resolver/binding-openharmony-arm64": ["@oxc-resolver/binding-openharmony-arm64@11.21.3", "", { "os": "none", "cpu": "arm64" }, "sha512-4j1DFwjwv36ec9kds0jU/ucQ5Ha4ERO/H95BxR5JFf0kqUUAJ1kwII7XhTc1vZrkdJkvLGC9Q2MbpObpum8RBg=="],
|
||||
|
||||
"@oxc-resolver/binding-wasm32-wasi": ["@oxc-resolver/binding-wasm32-wasi@11.21.3", "", { "dependencies": { "@emnapi/core": "1.11.0", "@emnapi/runtime": "1.11.0", "@napi-rs/wasm-runtime": "^1.1.5" }, "cpu": "none" }, "sha512-i8oluoel5kru/j1WNrjmQSiA3GQ7wvIYVR1IwIoZtKogAhya2iub+ZKIeSIkcJOrnzQ18Tzl/F+kL3fYOxZLvA=="],
|
||||
|
||||
"@oxc-resolver/binding-win32-arm64-msvc": ["@oxc-resolver/binding-win32-arm64-msvc@11.21.3", "", { "os": "win32", "cpu": "arm64" }, "sha512-M/8dw8dD6aOs+NlPJax401CZB9I7Aut84isQLgALGGwke4Afvw+/7yYhZb94yXf6t2sPLhQLmSmtSV+2FhsOWg=="],
|
||||
|
||||
"@oxc-resolver/binding-win32-x64-msvc": ["@oxc-resolver/binding-win32-x64-msvc@11.21.3", "", { "os": "win32", "cpu": "x64" }, "sha512-H7BCt/VnS9hnmMp42eGhZ99izSCRvlnWwy/N71K1/J8QoExwY4262Z8QiEkMDtduRJrztayDxETTckmUuAVL9Q=="],
|
||||
|
||||
"@oxfmt/binding-android-arm-eabi": ["@oxfmt/binding-android-arm-eabi@0.57.0", "", { "os": "android", "cpu": "arm" }, "sha512-qVBsEO+KugOsCmUHcO8iqNnqc65p7PCKpCs8M66mPZ+Ri+CWbcpoQOEJBg2OTu03+0qu++NK1jj6IzvQVs0Sig=="],
|
||||
|
||||
"@oxfmt/binding-android-arm64": ["@oxfmt/binding-android-arm64@0.57.0", "", { "os": "android", "cpu": "arm64" }, "sha512-mp6PibWbao3aizijcheOeHQaYEhcUAt8pwLniYbtLfHxL/psFF0BykAwCj+s3c6qIpa8yN8keZICWrqtZ70w8g=="],
|
||||
|
||||
"@oxfmt/binding-darwin-arm64": ["@oxfmt/binding-darwin-arm64@0.57.0", "", { "os": "darwin", "cpu": "arm64" }, "sha512-T+0stuCBqmUVY+aMIvrgXhzGhHO3sD5tNiiEcYqgSdPsnukskQqn2u5qOVD0sv1l7RLdFS5Z/f5Wi9Ktyjr3Eg=="],
|
||||
|
||||
"@oxfmt/binding-darwin-x64": ["@oxfmt/binding-darwin-x64@0.57.0", "", { "os": "darwin", "cpu": "x64" }, "sha512-O+3JbqWs/mCI2oi4xfhRO2IVPFJNDDEBV8Odo+ZpmsUOeKJfjXoNH7nDmBEQcDgK7NfjDIyE7kRgYSZcTLDO0A=="],
|
||||
|
||||
"@oxfmt/binding-freebsd-x64": ["@oxfmt/binding-freebsd-x64@0.57.0", "", { "os": "freebsd", "cpu": "x64" }, "sha512-pxwhxVC+JkLX9twOQ/8C/vbuOQcMZyKIDmiRDZfO7yITuVcIdZCiLRqqf4QOxb2+8FWrRXzQpm+1DBKcMpHSSQ=="],
|
||||
|
||||
"@oxfmt/binding-linux-arm-gnueabihf": ["@oxfmt/binding-linux-arm-gnueabihf@0.57.0", "", { "os": "linux", "cpu": "arm" }, "sha512-pxBU4zH2imB/MDBfth2rOMeVxXUMjRQLCazagwLARIFH3hVlxZJBlM4nSnHXaIHJK4/qezoFCIORN6AY8Mra4A=="],
|
||||
|
||||
"@oxfmt/binding-linux-arm-musleabihf": ["@oxfmt/binding-linux-arm-musleabihf@0.57.0", "", { "os": "linux", "cpu": "arm" }, "sha512-JAprOzt8tycYou36ZgEw14DlRHTiN8qdtKANdV3VZIRIvTI/lh/cX13c9pJ/EnDk2GT3FASH7KvCgQ2AufAifQ=="],
|
||||
|
||||
"@oxfmt/binding-linux-arm64-gnu": ["@oxfmt/binding-linux-arm64-gnu@0.57.0", "", { "os": "linux", "cpu": "arm64" }, "sha512-ajtjaxSaj9xl4BW7REt+Cef/ttzbAq00Bq4z7JUDZEfgFXdwSjH8K9bF+IcIJzZB9lKqMfQ4eHuSFOvvlvtqOg=="],
|
||||
|
||||
"@oxfmt/binding-linux-arm64-musl": ["@oxfmt/binding-linux-arm64-musl@0.57.0", "", { "os": "linux", "cpu": "arm64" }, "sha512-p4Y/+RYk9Bk5WO+zHSUXAClRmZ2fbJCejMuCAsU2HhyME4jqf6Ftt/mJYEwIah1wGCBDYOB7wEGV1x5bCEZ6hA=="],
|
||||
|
||||
"@oxfmt/binding-linux-ppc64-gnu": ["@oxfmt/binding-linux-ppc64-gnu@0.57.0", "", { "os": "linux", "cpu": "ppc64" }, "sha512-By6tRALAZsno0F4zedmtG+wdMvJiJmJoXM4d3+A9zHE4HRXLqXITwRH8mgrlcXc5yJM2g2W3riRPwTYdgemZLQ=="],
|
||||
|
||||
"@oxfmt/binding-linux-riscv64-gnu": ["@oxfmt/binding-linux-riscv64-gnu@0.57.0", "", { "os": "linux", "cpu": "none" }, "sha512-skYeG+RgvyzspqVEBsEprL90OYYZfoVNqB3HcCNR6QDJyXKOzfDRT3zncnHmUaFluIlBHuY23mU1b5WGgR98hA=="],
|
||||
|
||||
"@oxfmt/binding-linux-riscv64-musl": ["@oxfmt/binding-linux-riscv64-musl@0.57.0", "", { "os": "linux", "cpu": "none" }, "sha512-FFgACrZOXAXUh5KQh2mt1CDOVOZmn+QzHP71wM9QobNwyQvoFfyAeefVUltW83g3sm7LTiH3yfFqLLVUpA5ZFQ=="],
|
||||
|
||||
"@oxfmt/binding-linux-s390x-gnu": ["@oxfmt/binding-linux-s390x-gnu@0.57.0", "", { "os": "linux", "cpu": "s390x" }, "sha512-Nm/BAOfQeFiiKd502mZn/GAVKJwtd0RdCg17G3Wz/WSOIQmDi3+7/SZH4BHn1Ye5KvTVH3ua8WvfwLLycNIuvA=="],
|
||||
|
||||
"@oxfmt/binding-linux-x64-gnu": ["@oxfmt/binding-linux-x64-gnu@0.57.0", "", { "os": "linux", "cpu": "x64" }, "sha512-BiSy5Ku3mQqyxS6YIqAJgd403wEUWvI7kerfzPxc2l/txZVmZM0pSj7oDM+4bGBExowxOi7o73jEam1W0EDTZg=="],
|
||||
|
||||
"@oxfmt/binding-linux-x64-musl": ["@oxfmt/binding-linux-x64-musl@0.57.0", "", { "os": "linux", "cpu": "x64" }, "sha512-BCRkJiotz5s9afLYD2LuMvzAoDYx9H17E/YbDyu4xK7l4zHDPeny9ErSXL//i/nJyaOwRk08x4b8cgJC00+JDg=="],
|
||||
|
||||
"@oxfmt/binding-openharmony-arm64": ["@oxfmt/binding-openharmony-arm64@0.57.0", "", { "os": "none", "cpu": "arm64" }, "sha512-4Oaxe1qrGgXfpCJ1C/ERJ2iCtV2rN1R79ga9fsfyVHfSQRu/hVW780u2KDqZWFZ/iGTHODJji0JemxqFZ63eIQ=="],
|
||||
|
||||
"@oxfmt/binding-win32-arm64-msvc": ["@oxfmt/binding-win32-arm64-msvc@0.57.0", "", { "os": "win32", "cpu": "arm64" }, "sha512-MYLAsDnhdNsSGheLYhWgbk0vfIrlS84iQYun/y21fX6u0jj8iBtYtbpZMdiqYeuf8U12eVPUjVY2xE2NrCfJ0g=="],
|
||||
|
||||
"@oxfmt/binding-win32-ia32-msvc": ["@oxfmt/binding-win32-ia32-msvc@0.57.0", "", { "os": "win32", "cpu": "ia32" }, "sha512-PBwdzZALJY/jcCx2E6is0yu+cuVXeySTDmwuseD+9j0mHqlRNxwlKgsyRTBed/woPeqfVfuXfWjoq4Cx2Zt3Eg=="],
|
||||
|
||||
"@oxfmt/binding-win32-x64-msvc": ["@oxfmt/binding-win32-x64-msvc@0.57.0", "", { "os": "win32", "cpu": "x64" }, "sha512-bQJdH9i4RRfw55jm7+8/xS7GzHLLTbHx4huhrrDxQJaJtbSDbsyOnODvP1ftT7EG0KFKAYO2S+q6AcioXODx8w=="],
|
||||
|
||||
"@oxlint/binding-android-arm-eabi": ["@oxlint/binding-android-arm-eabi@1.71.0", "", { "os": "android", "cpu": "arm" }, "sha512-ImGmd1njEg4FEJH03jhRnveEegtO3czCtfptvaHivKAZQIYATbVFBrrzbaYMYv0oJioTnxZAZVSyV+oL7W8S2g=="],
|
||||
|
||||
"@oxlint/binding-android-arm64": ["@oxlint/binding-android-arm64@1.71.0", "", { "os": "android", "cpu": "arm64" }, "sha512-4A5BEexBrwY1YFF8Kiq/lp/wQPRG79G3BWIE1FuWaM5MvmpYSd+7ZySVcKkHdwo0UDzdQGddp6pD9mpctMqLnw=="],
|
||||
|
||||
"@oxlint/binding-darwin-arm64": ["@oxlint/binding-darwin-arm64@1.71.0", "", { "os": "darwin", "cpu": "arm64" }, "sha512-9wJA9GJulLwS2usU3CEisI/ESDO1n1z9eyTCvApMDrAkbJ1ve0mORgTMjcWWsKxkzkeZ2N/Gpra5IQE7x8tYgQ=="],
|
||||
|
||||
"@oxlint/binding-darwin-x64": ["@oxlint/binding-darwin-x64@1.71.0", "", { "os": "darwin", "cpu": "x64" }, "sha512-PlLCjS06V0PeJMAJwzjrExw1sYNW9Gch3JtNlcwwZDXGlTYDuwHNN89zYH8LTXFfgkVtsYvs2nv0FqrzyuFDzg=="],
|
||||
|
||||
"@oxlint/binding-freebsd-x64": ["@oxlint/binding-freebsd-x64@1.71.0", "", { "os": "freebsd", "cpu": "x64" }, "sha512-Lhil7bWre0ncxbUoDoxfS0JzpTz17BRQKW7iwoAUY8GJ66+WwJEfYPCFJ1P0WgVZR5/O/b3Q2pENlHOjeXLOGQ=="],
|
||||
|
||||
"@oxlint/binding-linux-arm-gnueabihf": ["@oxlint/binding-linux-arm-gnueabihf@1.71.0", "", { "os": "linux", "cpu": "arm" }, "sha512-Oo9/L58PYD3RC0x05d2upAPLllHytTjHQGsnC06P6Ynn7jKkp5mdImQxXdJ3+FnBaKspNpGogzgVsi6g872LiA=="],
|
||||
|
||||
"@oxlint/binding-linux-arm-musleabihf": ["@oxlint/binding-linux-arm-musleabihf@1.71.0", "", { "os": "linux", "cpu": "arm" }, "sha512-mSHfyfgJrEbyIR29ejaeS50BdPk+GoNPlC1dckpDiUZbJAIel68sjSMdOt4WY0/gva+ECC7FNITQkxMJU+vSBw=="],
|
||||
|
||||
"@oxlint/binding-linux-arm64-gnu": ["@oxlint/binding-linux-arm64-gnu@1.71.0", "", { "os": "linux", "cpu": "arm64" }, "sha512-n9yY4M2tiy3aij4AqtlnspzpfdpeT5JQfK2/w2d8oyp5W0FRwOb1dIeX99nORNcxGr08iD9bH8N5XFz3I2iy8w=="],
|
||||
|
||||
"@oxlint/binding-linux-arm64-musl": ["@oxlint/binding-linux-arm64-musl@1.71.0", "", { "os": "linux", "cpu": "arm64" }, "sha512-fJZrs5sDZtTaPIOiemRQQmo82Ezy+vOGXemPc4Ok7iVVsYsFa7SlW6Z5XN819VfsqBHRm3NJ3rTdnR8+bJYJdQ=="],
|
||||
|
||||
"@oxlint/binding-linux-ppc64-gnu": ["@oxlint/binding-linux-ppc64-gnu@1.71.0", "", { "os": "linux", "cpu": "ppc64" }, "sha512-cwl7VKGERIy9p+G+AvZdfy/06q0aHXaTt/mMRReC751iuNYJgqKjB7NydXSS30nBT9vtr2tunciOtrR4fD6FUA=="],
|
||||
|
||||
"@oxlint/binding-linux-riscv64-gnu": ["@oxlint/binding-linux-riscv64-gnu@1.71.0", "", { "os": "linux", "cpu": "none" }, "sha512-eZ8ieVXvzGi8jr7+ybQGPK2STw3mldfxZlgA2738iflfB/rzA69sE6m5rDRpQaxC7dpm745Enlh1Tod0QAk9Gg=="],
|
||||
|
||||
"@oxlint/binding-linux-riscv64-musl": ["@oxlint/binding-linux-riscv64-musl@1.71.0", "", { "os": "linux", "cpu": "none" }, "sha512-puMDbQYe6+NXwfMusojoA7CXGn2b3utukmd23PQqc1E3XhVCwyZ+FueSMzDYeNgDV2dUfIVXAAKZBcFDeCL6sA=="],
|
||||
|
||||
"@oxlint/binding-linux-s390x-gnu": ["@oxlint/binding-linux-s390x-gnu@1.71.0", "", { "os": "linux", "cpu": "s390x" }, "sha512-4NJLxBs1ujISCt3L/1FcywLs73PWtJuw+piD6feK2V6h6OS6P7xu9/sWt1DTRLibe6QCzmfZzmM/2HPORoV/Lg=="],
|
||||
|
||||
"@oxlint/binding-linux-x64-gnu": ["@oxlint/binding-linux-x64-gnu@1.71.0", "", { "os": "linux", "cpu": "x64" }, "sha512-cFDaiR8L3430qp88tfZnvFlt3KotFhR/DlbIL0nHOMMYiG/9Wy4l+6f7t8G8pTa9bd8Lt8+M0y/qjRQ/xcB74g=="],
|
||||
|
||||
"@oxlint/binding-linux-x64-musl": ["@oxlint/binding-linux-x64-musl@1.71.0", "", { "os": "linux", "cpu": "x64" }, "sha512-orfixdt76KlpNly9z0PkWBBNfwjKz+JFVLP/7wnVchlKNU9Dpt9InU/ZggeSej6fC7qwHmHNOGlhLnQXcYoGuA=="],
|
||||
|
||||
"@oxlint/binding-openharmony-arm64": ["@oxlint/binding-openharmony-arm64@1.71.0", "", { "os": "none", "cpu": "arm64" }, "sha512-9emQu2lAp6yhPB3XuI+++vR+l/o6JR1X+EpxwcumPdQXBWXEPAsquPGL7l158EqU8SebQMXTUa/S5zN98juyHw=="],
|
||||
|
||||
"@oxlint/binding-win32-arm64-msvc": ["@oxlint/binding-win32-arm64-msvc@1.71.0", "", { "os": "win32", "cpu": "arm64" }, "sha512-bd5kI8spYwTm3BILDtGhi73zoup5dw8MlPQNT8YB3BD5UIsjNe3K9/4ctrzQMX4SZMoK5HgzVLkLJzacEXB7fA=="],
|
||||
|
||||
"@oxlint/binding-win32-ia32-msvc": ["@oxlint/binding-win32-ia32-msvc@1.71.0", "", { "os": "win32", "cpu": "ia32" }, "sha512-W4HvOHGzVLHcrmFu+bMrJlho+/yrlX5ZNdJZqGe8MEldkQG+RHYhxxad9P4jvWAYFmIqUA5i9DQ8QsJqSU9GIw=="],
|
||||
|
||||
"@oxlint/binding-win32-x64-msvc": ["@oxlint/binding-win32-x64-msvc@1.71.0", "", { "os": "win32", "cpu": "x64" }, "sha512-D2kyEIPHk/G/wiZLnwTVC/sVst+T/lKldVOjAFpgTIBUAOlry72e5OiapDbDBF4LfJLkN5ypJb/8Eu6yJzkveQ=="],
|
||||
|
||||
"@playwright/test": ["@playwright/test@1.61.0", "", { "dependencies": { "playwright": "1.61.0" }, "bin": { "playwright": "cli.js" } }, "sha512-cKA5B6lpFEMyMGjxF54QihfYpB4FkEGH+qZhtArDEG+wezQAJY8Pq6C7T1SjWz+FFzt3TbyoXBQYk/0292TdJA=="],
|
||||
|
||||
"@quansync/fs": ["@quansync/fs@1.0.0", "", { "dependencies": { "quansync": "^1.0.0" } }, "sha512-4TJ3DFtlf1L5LDMaM6CanJ/0lckGNtJcMjQ1NAV6zDmA0tEHKZtxNKin8EgPaVX1YzljbxckyT2tJrpQKAtngQ=="],
|
||||
|
||||
"@radix-ui/number": ["@radix-ui/number@1.1.2", "", {}, "sha512-ceTwaxc4I5IOi97DgCotl3pqiyRGvffcc0oOsE2dQYaJOFIDsDt4VWG6xEbg1QePv9QWausCEIppud/tJ1wNig=="],
|
||||
|
||||
"@radix-ui/primitive": ["@radix-ui/primitive@1.1.4", "", {}, "sha512-7AdCK9PQyiljKoBDbN8OuctCbd/esdwZPQ8RtOE3SsyQtUpiPb+ND75q0jEhC1m1ecBI0MFNeLJvwIh9iKHRcQ=="],
|
||||
@@ -235,8 +400,6 @@
|
||||
|
||||
"@radix-ui/react-menu": ["@radix-ui/react-menu@2.1.18", "", { "dependencies": { "@radix-ui/primitive": "1.1.4", "@radix-ui/react-collection": "1.1.10", "@radix-ui/react-compose-refs": "1.1.3", "@radix-ui/react-context": "1.1.4", "@radix-ui/react-direction": "1.1.2", "@radix-ui/react-dismissable-layer": "1.1.13", "@radix-ui/react-focus-guards": "1.1.4", "@radix-ui/react-focus-scope": "1.1.10", "@radix-ui/react-id": "1.1.2", "@radix-ui/react-popper": "1.3.1", "@radix-ui/react-portal": "1.1.12", "@radix-ui/react-presence": "1.1.6", "@radix-ui/react-primitive": "2.1.6", "@radix-ui/react-roving-focus": "1.1.13", "@radix-ui/react-slot": "1.3.0", "@radix-ui/react-use-callback-ref": "1.1.2", "aria-hidden": "^1.2.4", "react-remove-scroll": "^2.7.2" }, "peerDependencies": { "@types/react": "*", "@types/react-dom": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" }, "optionalPeers": ["@types/react", "@types/react-dom"] }, "sha512-lj8Rxjtn6zJq1oSbE/uDtAwCbB9BnxgHD+8MwJMuTh6u1dPamYhW9iuELr/Z8d0D/UysFblYYHeBPwi7T4k0YQ=="],
|
||||
|
||||
"@radix-ui/react-popover": ["@radix-ui/react-popover@1.1.17", "", { "dependencies": { "@radix-ui/primitive": "1.1.4", "@radix-ui/react-compose-refs": "1.1.3", "@radix-ui/react-context": "1.1.4", "@radix-ui/react-dismissable-layer": "1.1.13", "@radix-ui/react-focus-guards": "1.1.4", "@radix-ui/react-focus-scope": "1.1.10", "@radix-ui/react-id": "1.1.2", "@radix-ui/react-popper": "1.3.1", "@radix-ui/react-portal": "1.1.12", "@radix-ui/react-presence": "1.1.6", "@radix-ui/react-primitive": "2.1.6", "@radix-ui/react-slot": "1.3.0", "@radix-ui/react-use-controllable-state": "1.2.3", "aria-hidden": "^1.2.4", "react-remove-scroll": "^2.7.2" }, "peerDependencies": { "@types/react": "*", "@types/react-dom": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" }, "optionalPeers": ["@types/react", "@types/react-dom"] }, "sha512-/YSAOdJ7YJvdn7bn5sdSx2egW+SKY+u7O5RyAVs94Ymrg2fg5QTSFPMRkzvhGyFuE4/qsmPBdrwYoZMZh/4f+g=="],
|
||||
|
||||
"@radix-ui/react-popper": ["@radix-ui/react-popper@1.3.1", "", { "dependencies": { "@floating-ui/react-dom": "^2.0.0", "@radix-ui/react-arrow": "1.1.10", "@radix-ui/react-compose-refs": "1.1.3", "@radix-ui/react-context": "1.1.4", "@radix-ui/react-primitive": "2.1.6", "@radix-ui/react-use-callback-ref": "1.1.2", "@radix-ui/react-use-layout-effect": "1.1.2", "@radix-ui/react-use-rect": "1.1.2", "@radix-ui/react-use-size": "1.1.2", "@radix-ui/rect": "1.1.2" }, "peerDependencies": { "@types/react": "*", "@types/react-dom": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" }, "optionalPeers": ["@types/react", "@types/react-dom"] }, "sha512-bhnq/0DEPTi2lsOD3J5rTL65qUKHbKbhqHsmN9TMiclSXpipi651ooUKPPp6G5lF/WiHBdn1s0Wuqsn+myVAvw=="],
|
||||
|
||||
"@radix-ui/react-portal": ["@radix-ui/react-portal@1.1.12", "", { "dependencies": { "@radix-ui/react-primitive": "2.1.6", "@radix-ui/react-use-layout-effect": "1.1.2" }, "peerDependencies": { "@types/react": "*", "@types/react-dom": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" }, "optionalPeers": ["@types/react", "@types/react-dom"] }, "sha512-m309havGzsjLHHaIX50G5PlvRs3xkgPCsGk/5PTvYm8D5q33yG0J7w/712PTOhid7NTaFETtnSXjngHQavvhVw=="],
|
||||
@@ -417,7 +580,7 @@
|
||||
|
||||
"@turbo/windows-arm64": ["@turbo/windows-arm64@2.9.18", "", { "os": "win32", "cpu": "arm64" }, "sha512-nUdR8WqoomUys9iIQmG45TMiizJ+5BV8egSeLLZba/AWblyp3fVBcIH1kSE58OtK4g2YzbMJEth6Ttv9w5rqMA=="],
|
||||
|
||||
"@tybys/wasm-util": ["@tybys/wasm-util@0.10.1", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-9tTaPJLSiejZKx+Bmog4uSubteqTvFrVrURwkmHixBo0G4seD0zUxp98E1DzUBJxLQ3NPwXrGKDiVjwx/DpPsg=="],
|
||||
"@tybys/wasm-util": ["@tybys/wasm-util@0.10.3", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-F3fo1MYrRJYL3zER0OUOmkutjr1Vp23m7OsSgp7nq4SP6OqX6C/56XFIPAl5bt3zaBRjmW7SGz3u/6LwFpYcOg=="],
|
||||
|
||||
"@types/aria-query": ["@types/aria-query@5.0.4", "", {}, "sha512-rfT93uj5s0PRL7EzccGMs3brplhcrghnDoV26NqKhCAS1hVo+WdNsPvE/yb6ilfr5hi2MEk6d5EWJTKdxg8jVw=="],
|
||||
|
||||
@@ -483,6 +646,8 @@
|
||||
|
||||
"browserslist": ["browserslist@4.28.2", "", { "dependencies": { "baseline-browser-mapping": "^2.10.12", "caniuse-lite": "^1.0.30001782", "electron-to-chromium": "^1.5.328", "node-releases": "^2.0.36", "update-browserslist-db": "^1.2.3" }, "bin": { "browserslist": "cli.js" } }, "sha512-48xSriZYYg+8qXna9kwqjIVzuQxi+KYWp2+5nCYnYKPTr0LvD89Jqk2Or5ogxz0NUMfIjhh2lIUX/LyX9B4oIg=="],
|
||||
|
||||
"cac": ["cac@7.0.0", "", {}, "sha512-tixWYgm5ZoOD+3g6UTea91eow5z6AAHaho3g0V9CNSNb45gM8SmflpAc+GRd1InC4AqN/07Unrgp56Y94N9hJQ=="],
|
||||
|
||||
"call-bind-apply-helpers": ["call-bind-apply-helpers@1.0.2", "", { "dependencies": { "es-errors": "^1.3.0", "function-bind": "^1.1.2" } }, "sha512-Sp1ablJ0ivDkSzjcaJdxEunN5/XvksFJ2sMBFfq6x0ryhQV/2b/KwFe21cMpmHtPOSij8K99/wSfoEuTObmuMQ=="],
|
||||
|
||||
"camelcase": ["camelcase@5.3.1", "", {}, "sha512-L28STB170nwWS63UjtlEOE3dldQApaJXZkOI1uMFfzf3rRuPegHaHesyee+YxQ+W6SvRDQV6UrdOdRiR153wJg=="],
|
||||
@@ -493,8 +658,12 @@
|
||||
|
||||
"chalk": ["chalk@4.1.2", "", { "dependencies": { "ansi-styles": "^4.1.0", "supports-color": "^7.1.0" } }, "sha512-oKnbhFyRIXpUuez8iBMmyEa4nbj4IOQyuhc/wy9kY7/WVPcwIO9VA668Pu8RkO7+0G76SLROeyw9CpQ061i4mA=="],
|
||||
|
||||
"class-variance-authority": ["class-variance-authority@0.7.1", "", { "dependencies": { "clsx": "^2.1.1" } }, "sha512-Ka+9Trutv7G8M6WT6SeiRWz792K5qEqIGEGzXKhAE6xOWAY6pPH8U+9IY3oCMv6kqTmLsv7Xh/2w2RigkePMsg=="],
|
||||
|
||||
"cliui": ["cliui@8.0.1", "", { "dependencies": { "string-width": "^4.2.0", "strip-ansi": "^6.0.1", "wrap-ansi": "^7.0.0" } }, "sha512-BSeNnyus75C4//NQ9gQt1/csTXyo/8Sb+afLAkzAptFuMsod9HFokGNudZpi/oQV73hnVK+sR+5PVRMd+Dr7YQ=="],
|
||||
|
||||
"clsx": ["clsx@2.1.1", "", {}, "sha512-eYm0QWBtUrBWZWG0d386OGAw16Z995PiOVo2B7bjWSbHedGl5e0ZWaq65kOGgUSNesEIDkB9ISbTg/JK9dhCZA=="],
|
||||
|
||||
"color-convert": ["color-convert@2.0.1", "", { "dependencies": { "color-name": "~1.1.4" } }, "sha512-RRECPsj7iu/xb5oKYcsFHSppFNnsj/52OVTRKb4zP5onXwVF3zVmmToNcOfGC+CRDpfK/U584fMg38ZHCaElKQ=="],
|
||||
|
||||
"color-name": ["color-name@1.1.4", "", {}, "sha512-dOy+3AuW3a2wNbZHIuMZpTcgjGuLU/uBL/ubcZF9OXbDo8ff4O8yVp5Bf0efS8uEoYo5q4Fx7dY9OgQGXgAsQA=="],
|
||||
@@ -525,10 +694,14 @@
|
||||
|
||||
"deep-is": ["deep-is@0.1.4", "", {}, "sha512-oIPzksmTg4/MriiaYGO+okXDT7ztn/w3Eptv/+gSIdMdKsJo0u4CfYNFJPy+4SKMuCqGw2wxnA+URMg3t8a/bQ=="],
|
||||
|
||||
"defu": ["defu@6.1.7", "", {}, "sha512-7z22QmUWiQ/2d0KkdYmANbRUVABpZ9SNYyH5vx6PZ+nE5bcC0l7uFvEfHlyld/HcGBFTL536ClDt3DEcSlEJAQ=="],
|
||||
|
||||
"delayed-stream": ["delayed-stream@1.0.0", "", {}, "sha512-ZySD7Nf91aLB0RxL4KGrKHBXl7Eds1DAmEdcoVawXnLD7SDhpNgtuII2aAkg7a7QS41jxPSZ17p4VdGnMHk3MQ=="],
|
||||
|
||||
"dequal": ["dequal@2.0.3", "", {}, "sha512-0je+qPKHEMohvfRTCEo3CrPG6cAzAYgmzKyxRiYSSDkS6eGJdyVJm7WaYA5ECaAD9wLB2T4EEeymA5aFVcYXCA=="],
|
||||
|
||||
"destr": ["destr@2.0.5", "", {}, "sha512-ugFTXCtDZunbzasqBxrK93Ik/DRYsO6S/fedkWEMKqt04xZ4csmnmwGDBAb07QWNaGMAmnTIemsYZCksjATwsA=="],
|
||||
|
||||
"detect-libc": ["detect-libc@2.1.2", "", {}, "sha512-Btj2BOOO83o3WyH59e8MgXsxEQVcarkUOpEYrubB0urwnN10yQ364rsiByU11nZlqWYZm05i/of7io4mzihBtQ=="],
|
||||
|
||||
"detect-node-es": ["detect-node-es@1.1.0", "", {}, "sha512-ypdmJU/TbBby2Dxibuv7ZLW3Bs1QEmM7nHjEANfohJLvE0XVujisn1qPJcZxg+qDucsr+bP6fLD1rPS3AhJ7EQ=="],
|
||||
@@ -565,8 +738,6 @@
|
||||
|
||||
"eslint-plugin-react-hooks": ["eslint-plugin-react-hooks@7.1.1", "", { "dependencies": { "@babel/core": "^7.24.4", "@babel/parser": "^7.24.4", "hermes-parser": "^0.25.1", "zod": "^3.25.0 || ^4.0.0", "zod-validation-error": "^3.5.0 || ^4.0.0" }, "peerDependencies": { "eslint": "^3.0.0 || ^4.0.0 || ^5.0.0 || ^6.0.0 || ^7.0.0 || ^8.0.0-0 || ^9.0.0 || ^10.0.0" } }, "sha512-f2I7Gw6JbvCexzIInuSbZpfdQ44D7iqdWX01FKLvrPgqxoE7oMj8clOfto8U6vYiz4yd5oKu39rRSVOe1zRu0g=="],
|
||||
|
||||
"eslint-plugin-react-refresh": ["eslint-plugin-react-refresh@0.5.3", "", { "peerDependencies": { "eslint": "^9 || ^10" } }, "sha512-5EMmLCV98Pi4o/f/3DP/v/tNqLHMIc9I8LKClNDWhZ9JTho89/kQcitCXQBMG7sAfVRK0Ie3T2EDOzp1YXYiVA=="],
|
||||
|
||||
"eslint-scope": ["eslint-scope@9.1.2", "", { "dependencies": { "@types/esrecurse": "^4.3.1", "@types/estree": "^1.0.8", "esrecurse": "^4.3.0", "estraverse": "^5.2.0" } }, "sha512-xS90H51cKw0jltxmvmHy2Iai1LIqrfbw57b79w/J7MfvDfkIkFZ+kj6zC3BjtUwh150HsSSdxXZcsuv72miDFQ=="],
|
||||
|
||||
"eslint-visitor-keys": ["eslint-visitor-keys@5.0.1", "", {}, "sha512-tD40eHxA35h0PEIZNeIjkHoDR4YjjJp34biM0mDvplBe//mB+IHCqHDGV7pxF+7MklTvighcCPPZC7ynWyjdTA=="],
|
||||
@@ -593,6 +764,8 @@
|
||||
|
||||
"fast-levenshtein": ["fast-levenshtein@2.0.6", "", {}, "sha512-DCXu6Ifhqcks7TZKY3Hxp3y6qphY5SJZmrWMDrKcERSOXWQdMhU9Ig/PYrzyw/ul9jOIyh0N4M0tbC5hodg8dw=="],
|
||||
|
||||
"fd-package-json": ["fd-package-json@2.0.0", "", { "dependencies": { "walk-up-path": "^4.0.0" } }, "sha512-jKmm9YtsNXN789RS/0mSzOC1NUq9mkVd65vbSSVsKdjGvYXBuE4oWe2QOEoFeRmJg+lPuZxpmrfFclNhoRMneQ=="],
|
||||
|
||||
"fdir": ["fdir@6.5.0", "", { "peerDependencies": { "picomatch": "^3 || ^4" }, "optionalPeers": ["picomatch"] }, "sha512-tIbYtZbucOs0BRGqPJkshJUYdL+SDH7dVM8gjy+ERp3WAUjLEFJE+02kanyHtwjWOnwrKYBiwAmM0p4kLJAnXg=="],
|
||||
|
||||
"figures": ["figures@6.1.0", "", { "dependencies": { "is-unicode-supported": "^2.0.0" } }, "sha512-d+l3qxjSesT4V7v2fh+QnmFnUWv9lSpjarhShNTgBOfA0ttejbQUAlHLitbjkoRiDulW0OPoQPYIGhIC8ohejg=="],
|
||||
@@ -609,10 +782,14 @@
|
||||
|
||||
"form-data": ["form-data@4.0.5", "", { "dependencies": { "asynckit": "^0.4.0", "combined-stream": "^1.0.8", "es-set-tostringtag": "^2.1.0", "hasown": "^2.0.2", "mime-types": "^2.1.12" } }, "sha512-8RipRLol37bNs2bhoV67fiTEvdTrbMUYcFTiy3+wuuOnUog2QBHCZWXDRijWQfAkhBj2Uf5UnVaiWwA5vdd82w=="],
|
||||
|
||||
"formatly": ["formatly@0.3.0", "", { "dependencies": { "fd-package-json": "^2.0.0" }, "bin": { "formatly": "bin/index.mjs" } }, "sha512-9XNj/o4wrRFyhSMJOvsuyMwy8aUfBaZ1VrqHVfohyXf0Sw0e+yfKG+xZaY3arGCOMdwFsqObtzVOc1gU9KiT9w=="],
|
||||
|
||||
"fsevents": ["fsevents@2.3.2", "", { "os": "darwin" }, "sha512-xiqMQR4xAeHTuB9uWm+fFRcIOgKBMiOBP+eXiyT7jsgVCq1bkVygt00oASowB7EdtpOHaaPgKt812P9ab+DDKA=="],
|
||||
|
||||
"function-bind": ["function-bind@1.1.2", "", {}, "sha512-7XHNxH7qX9xG5mIwxkhumTox/MIRNcOgDrxWsMt2pAr23WHp6MrRlN7FBSFpCpr+oVO0F744iUgR82nJMfG2SA=="],
|
||||
|
||||
"fzf": ["fzf@0.5.2", "", {}, "sha512-Tt4kuxLXFKHy8KT40zwsUPUkg1CrsgY25FxA2U/j/0WgEDCk3ddc/zLTCCcbSHX9FcKtLuVaDGtGE/STWC+j3Q=="],
|
||||
|
||||
"gensync": ["gensync@1.0.0-beta.2", "", {}, "sha512-3hN7NaskYvMDLQY55gnW3NQ+mesEAepTqlg+VEbj7zzqEMBVNhzcGYYeqFo/TlYz6eQiFcp1HcsCZO+nGgS8zg=="],
|
||||
|
||||
"get-caller-file": ["get-caller-file@2.0.5", "", {}, "sha512-DyFP3BM/3YHTQOCUL/w0OZHR0lpKeGrxotcHWcqNEdnltqFwXVfhEBQ94eIo34AfQpo0rGki4cyIiftY06h2Fg=="],
|
||||
@@ -627,6 +804,8 @@
|
||||
|
||||
"get-them-args": ["get-them-args@1.3.2", "", {}, "sha512-LRn8Jlk+DwZE4GTlDbT3Hikd1wSHgLMme/+7ddlqKd7ldwR6LjJgTVWzBnR01wnYGe4KgrXjg287RaI22UHmAw=="],
|
||||
|
||||
"get-tsconfig": ["get-tsconfig@4.14.0", "", { "dependencies": { "resolve-pkg-maps": "^1.0.0" } }, "sha512-yTb+8DXzDREzgvYmh6s9vHsSVCHeC0G3PI5bEXNBHtmshPnO+S5O7qgLEOn0I5QvMy6kpZN8K1NKGyilLb93wA=="],
|
||||
|
||||
"glob-parent": ["glob-parent@6.0.2", "", { "dependencies": { "is-glob": "^4.0.3" } }, "sha512-XxwI8EOhVQgWp6iDL+3b0r86f4d6AX6zSU55HfB4ydCEuXLXc5FcYeOu+nnGftS4TEju/11rt4KJPTMgbfmv4A=="],
|
||||
|
||||
"globals": ["globals@17.6.0", "", {}, "sha512-sepffkT8stwnIYbsMBpoCHJuJM5l98FUF2AnE07hfvE0m/qp3R586hw4jF4uadbhvg1ooIdzuu7CsfD2jzCaNA=="],
|
||||
@@ -705,6 +884,8 @@
|
||||
|
||||
"kill-port-process": ["kill-port-process@4.0.2", "", { "dependencies": { "get-them-args": "1.3.2", "pid-port": "2.0.1" }, "bin": { "kill-port": "dist/bin/kill-port-process.js" } }, "sha512-fO8gc45EYJQUQWozPBmdTpsR0GDvldsmrhP2I4FPoNejwyBY4Liiwj9Is7P/5rj6k07ZQ5Ob0g0k2dqQcslW/w=="],
|
||||
|
||||
"knip": ["knip@6.23.0", "", { "dependencies": { "fdir": "^6.5.0", "formatly": "^0.3.0", "get-tsconfig": "4.14.0", "jiti": "^2.7.0", "oxc-parser": "^0.137.0", "oxc-resolver": "11.21.3", "picomatch": "^4.0.4", "smol-toml": "^1.6.1", "strip-json-comments": "5.0.3", "tinyglobby": "^0.2.17", "unbash": "^4.0.1", "yaml": "^2.9.0", "zod": "^4.1.11" }, "bin": { "knip": "bin/knip.js", "knip-bun": "bin/knip-bun.js" } }, "sha512-2DvAOX2pZWiG4SLvRRxOAU0aWGEn1ZoVblI541xIoXFdHqq2THMZXy66/qcY5WGuW3TXhb9T1x1zd/Hd1u+yqg=="],
|
||||
|
||||
"levn": ["levn@0.4.1", "", { "dependencies": { "prelude-ls": "^1.2.1", "type-check": "~0.4.0" } }, "sha512-+bT2uH4E5LGE7h/n3evcS/sQlJXCpIp6ym8OWJ5eV6+67Dsql/LaaT7qJBAt2rzfoa/5QBGBhxDix1dMt2kQKQ=="],
|
||||
|
||||
"lightningcss": ["lightningcss@1.32.0", "", { "dependencies": { "detect-libc": "^2.0.3" }, "optionalDependencies": { "lightningcss-android-arm64": "1.32.0", "lightningcss-darwin-arm64": "1.32.0", "lightningcss-darwin-x64": "1.32.0", "lightningcss-freebsd-x64": "1.32.0", "lightningcss-linux-arm-gnueabihf": "1.32.0", "lightningcss-linux-arm64-gnu": "1.32.0", "lightningcss-linux-arm64-musl": "1.32.0", "lightningcss-linux-x64-gnu": "1.32.0", "lightningcss-linux-x64-musl": "1.32.0", "lightningcss-win32-arm64-msvc": "1.32.0", "lightningcss-win32-x64-msvc": "1.32.0" } }, "sha512-NXYBzinNrblfraPGyrbPoD19C1h9lfI/1mzgWYvXUTe414Gz/X1FD2XBZSZM7rRTrMA8JL3OtAaGifrIKhQ5yQ=="],
|
||||
@@ -751,6 +932,8 @@
|
||||
|
||||
"mime-types": ["mime-types@2.1.35", "", { "dependencies": { "mime-db": "1.52.0" } }, "sha512-ZDY+bPm5zTTF+YpCrAU9nK0UgICYPT0QtT1NZWFv4s++TNkcgVaT0g6+4R2uI4MjQjzysHB1zxuWL50hzaeXiw=="],
|
||||
|
||||
"mimic-function": ["mimic-function@5.0.1", "", {}, "sha512-VP79XUPxV2CigYP3jWwAUFSku2aKqBH7uTAapFWCBqutsbmDo96KY5o8uh6U+/YSIn5OxJnXp73beVkpqMIGhA=="],
|
||||
|
||||
"min-indent": ["min-indent@1.0.1", "", {}, "sha512-I9jwMn07Sy/IwOj3zVkVik2JTvgpaykDZEigL6Rx6N9LbMywwUSMtxET+7lVoDLLd3O3IXwJwvuuns8UB/HeAg=="],
|
||||
|
||||
"minimatch": ["minimatch@10.2.5", "", { "dependencies": { "brace-expansion": "^5.0.5" } }, "sha512-MULkVLfKGYDFYejP07QOurDLLQpcjk7Fw+7jXS2R2czRQzR56yHRveU5NDJEOviH+hETZKSkIk5c+T23GjFUMg=="],
|
||||
@@ -763,22 +946,38 @@
|
||||
|
||||
"natural-compare": ["natural-compare@1.4.0", "", {}, "sha512-OWND8ei3VtNC9h7V60qff3SVobHr996CTwgxubgyQYEpg290h9J0buyECNNJexkFm5sOajh5G116RYA1c8ZMSw=="],
|
||||
|
||||
"node-fetch-native": ["node-fetch-native@1.6.7", "", {}, "sha512-g9yhqoedzIUm0nTnTqAQvueMPVOuIY16bqgAJJC8XOOubYFNwz6IER9qs0Gq2Xd0+CecCKFjtdDTMA4u4xG06Q=="],
|
||||
|
||||
"node-releases": ["node-releases@2.0.37", "", {}, "sha512-1h5gKZCF+pO/o3Iqt5Jp7wc9rH3eJJ0+nh/CIoiRwjRxde/hAHyLPXYN4V3CqKAbiZPSeJFSWHmJsbkicta0Eg=="],
|
||||
|
||||
"npm-run-path": ["npm-run-path@6.0.0", "", { "dependencies": { "path-key": "^4.0.0", "unicorn-magic": "^0.3.0" } }, "sha512-9qny7Z9DsQU8Ou39ERsPU4OZQlSTP47ShQzuKZ6PRXpYLtIFgl/DEBYEXKlvcEa+9tHVcK8CF81Y2V72qaZhWA=="],
|
||||
|
||||
"obug": ["obug@2.1.1", "", {}, "sha512-uTqF9MuPraAQ+IsnPf366RG4cP9RtUi7MLO1N3KEc+wb0a6yKpeL0lmk2IB1jY5KHPAlTc6T/JRdC/YqxHNwkQ=="],
|
||||
|
||||
"ofetch": ["ofetch@1.5.1", "", { "dependencies": { "destr": "^2.0.5", "node-fetch-native": "^1.6.7", "ufo": "^1.6.1" } }, "sha512-2W4oUZlVaqAPAil6FUg/difl6YhqhUR7x2eZY4bQCko22UXg3hptq9KLQdqFClV+Wu85UX7hNtdGTngi/1BxcA=="],
|
||||
|
||||
"omnivoice-studio": ["omnivoice-studio@workspace:frontend"],
|
||||
|
||||
"onetime": ["onetime@7.0.0", "", { "dependencies": { "mimic-function": "^5.0.0" } }, "sha512-VXJjc87FScF88uafS3JllDgvAm+c/Slfz06lorj2uAY34rlUu0Nt+v8wreiImcrgAjjIHp1rXpTDlLOGw29WwQ=="],
|
||||
|
||||
"optionator": ["optionator@0.9.4", "", { "dependencies": { "deep-is": "^0.1.3", "fast-levenshtein": "^2.0.6", "levn": "^0.4.1", "prelude-ls": "^1.2.1", "type-check": "^0.4.0", "word-wrap": "^1.2.5" } }, "sha512-6IpQ7mKUxRcZNLIObR0hz7lxsapSSIYNZJwXPGeF0mTVqGKFIXj1DQcMoT22S3ROcLyY/rz0PWaWZ9ayWmad9g=="],
|
||||
|
||||
"oxc-parser": ["oxc-parser@0.137.0", "", { "dependencies": { "@oxc-project/types": "^0.137.0" }, "optionalDependencies": { "@oxc-parser/binding-android-arm-eabi": "0.137.0", "@oxc-parser/binding-android-arm64": "0.137.0", "@oxc-parser/binding-darwin-arm64": "0.137.0", "@oxc-parser/binding-darwin-x64": "0.137.0", "@oxc-parser/binding-freebsd-x64": "0.137.0", "@oxc-parser/binding-linux-arm-gnueabihf": "0.137.0", "@oxc-parser/binding-linux-arm-musleabihf": "0.137.0", "@oxc-parser/binding-linux-arm64-gnu": "0.137.0", "@oxc-parser/binding-linux-arm64-musl": "0.137.0", "@oxc-parser/binding-linux-ppc64-gnu": "0.137.0", "@oxc-parser/binding-linux-riscv64-gnu": "0.137.0", "@oxc-parser/binding-linux-riscv64-musl": "0.137.0", "@oxc-parser/binding-linux-s390x-gnu": "0.137.0", "@oxc-parser/binding-linux-x64-gnu": "0.137.0", "@oxc-parser/binding-linux-x64-musl": "0.137.0", "@oxc-parser/binding-openharmony-arm64": "0.137.0", "@oxc-parser/binding-wasm32-wasi": "0.137.0", "@oxc-parser/binding-win32-arm64-msvc": "0.137.0", "@oxc-parser/binding-win32-ia32-msvc": "0.137.0", "@oxc-parser/binding-win32-x64-msvc": "0.137.0" } }, "sha512-yFImD+WLElJpLKy8llG1qe4DCmMsL18peRp8XP1JKfig/gISbJkglnpDtX2aTmAn10kZF7164HbN2H8QPsXxGg=="],
|
||||
|
||||
"oxc-resolver": ["oxc-resolver@11.21.3", "", { "optionalDependencies": { "@oxc-resolver/binding-android-arm-eabi": "11.21.3", "@oxc-resolver/binding-android-arm64": "11.21.3", "@oxc-resolver/binding-darwin-arm64": "11.21.3", "@oxc-resolver/binding-darwin-x64": "11.21.3", "@oxc-resolver/binding-freebsd-x64": "11.21.3", "@oxc-resolver/binding-linux-arm-gnueabihf": "11.21.3", "@oxc-resolver/binding-linux-arm-musleabihf": "11.21.3", "@oxc-resolver/binding-linux-arm64-gnu": "11.21.3", "@oxc-resolver/binding-linux-arm64-musl": "11.21.3", "@oxc-resolver/binding-linux-ppc64-gnu": "11.21.3", "@oxc-resolver/binding-linux-riscv64-gnu": "11.21.3", "@oxc-resolver/binding-linux-riscv64-musl": "11.21.3", "@oxc-resolver/binding-linux-s390x-gnu": "11.21.3", "@oxc-resolver/binding-linux-x64-gnu": "11.21.3", "@oxc-resolver/binding-linux-x64-musl": "11.21.3", "@oxc-resolver/binding-openharmony-arm64": "11.21.3", "@oxc-resolver/binding-wasm32-wasi": "11.21.3", "@oxc-resolver/binding-win32-arm64-msvc": "11.21.3", "@oxc-resolver/binding-win32-x64-msvc": "11.21.3" } }, "sha512-2Mx3fKQz7+xgrBONjsxOgCGtMHOn38/HxMzW1I5efwXB5a4lRN0Vp40gYUJFBWJslcrvwoofTrqoTnLbwTd3pA=="],
|
||||
|
||||
"oxfmt": ["oxfmt@0.57.0", "", { "dependencies": { "tinypool": "2.1.0" }, "optionalDependencies": { "@oxfmt/binding-android-arm-eabi": "0.57.0", "@oxfmt/binding-android-arm64": "0.57.0", "@oxfmt/binding-darwin-arm64": "0.57.0", "@oxfmt/binding-darwin-x64": "0.57.0", "@oxfmt/binding-freebsd-x64": "0.57.0", "@oxfmt/binding-linux-arm-gnueabihf": "0.57.0", "@oxfmt/binding-linux-arm-musleabihf": "0.57.0", "@oxfmt/binding-linux-arm64-gnu": "0.57.0", "@oxfmt/binding-linux-arm64-musl": "0.57.0", "@oxfmt/binding-linux-ppc64-gnu": "0.57.0", "@oxfmt/binding-linux-riscv64-gnu": "0.57.0", "@oxfmt/binding-linux-riscv64-musl": "0.57.0", "@oxfmt/binding-linux-s390x-gnu": "0.57.0", "@oxfmt/binding-linux-x64-gnu": "0.57.0", "@oxfmt/binding-linux-x64-musl": "0.57.0", "@oxfmt/binding-openharmony-arm64": "0.57.0", "@oxfmt/binding-win32-arm64-msvc": "0.57.0", "@oxfmt/binding-win32-ia32-msvc": "0.57.0", "@oxfmt/binding-win32-x64-msvc": "0.57.0" }, "peerDependencies": { "svelte": "^5.0.0", "vite-plus": "*" }, "optionalPeers": ["svelte", "vite-plus"], "bin": { "oxfmt": "bin/oxfmt" } }, "sha512-ZB7Bi+rGDSqmVIo9jwcLyFgjxXvQhDdU+jx+ZrVy6VRiVXK2+CHc4hO3J4dUQjHe7V0ymHB+MDuv5z+NhK07HA=="],
|
||||
|
||||
"oxlint": ["oxlint@1.71.0", "", { "optionalDependencies": { "@oxlint/binding-android-arm-eabi": "1.71.0", "@oxlint/binding-android-arm64": "1.71.0", "@oxlint/binding-darwin-arm64": "1.71.0", "@oxlint/binding-darwin-x64": "1.71.0", "@oxlint/binding-freebsd-x64": "1.71.0", "@oxlint/binding-linux-arm-gnueabihf": "1.71.0", "@oxlint/binding-linux-arm-musleabihf": "1.71.0", "@oxlint/binding-linux-arm64-gnu": "1.71.0", "@oxlint/binding-linux-arm64-musl": "1.71.0", "@oxlint/binding-linux-ppc64-gnu": "1.71.0", "@oxlint/binding-linux-riscv64-gnu": "1.71.0", "@oxlint/binding-linux-riscv64-musl": "1.71.0", "@oxlint/binding-linux-s390x-gnu": "1.71.0", "@oxlint/binding-linux-x64-gnu": "1.71.0", "@oxlint/binding-linux-x64-musl": "1.71.0", "@oxlint/binding-openharmony-arm64": "1.71.0", "@oxlint/binding-win32-arm64-msvc": "1.71.0", "@oxlint/binding-win32-ia32-msvc": "1.71.0", "@oxlint/binding-win32-x64-msvc": "1.71.0" }, "peerDependencies": { "oxlint-tsgolint": ">=0.22.1", "vite-plus": "*" }, "optionalPeers": ["oxlint-tsgolint", "vite-plus"], "bin": { "oxlint": "bin/oxlint" } }, "sha512-U1m1X+C0vDj7DC1e13IoZULzEcPczE7UOMTs8VlZGHUEIUaSTZKo5qkPsQEfzpgnQ29Pea/w3Xntk62UCecxZw=="],
|
||||
|
||||
"p-limit": ["p-limit@3.1.0", "", { "dependencies": { "yocto-queue": "^0.1.0" } }, "sha512-TYOanM3wGwNGsZN2cVTYPArw454xnXj5qmWF1bEoAc4+cU/ol7GVh7odevjp1FNHduHc3KZMcFduxU5Xc6uJRQ=="],
|
||||
|
||||
"p-locate": ["p-locate@5.0.0", "", { "dependencies": { "p-limit": "^3.0.2" } }, "sha512-LaNjtRWUBY++zB5nE/NwcaoMylSPk+S+ZHNB1TzdbMJMny6dynpAGt7X/tl/QYq3TIeE6nxHppbo2LGymrG5Pw=="],
|
||||
|
||||
"p-try": ["p-try@2.2.0", "", {}, "sha512-R4nPAVTAU0B9D35/Gk3uJf/7XYbQcyohSKdvAxIRSNghFl4e71hVoGnBNQz9cWaXxO2I10KTC+3jMdvvoKw6dQ=="],
|
||||
|
||||
"package-manager-detector": ["package-manager-detector@1.6.0", "", {}, "sha512-61A5ThoTiDG/C8s8UMZwSorAGwMJ0ERVGj2OjoW5pAalsNOg15+iQiPzrLJ4jhZ1HJzmC2PIHT2oEiH3R5fzNA=="],
|
||||
|
||||
"parse-ms": ["parse-ms@4.0.0", "", {}, "sha512-TXfryirbmq34y8QBwgqCVLi+8oA3oWx2eAnSn62ITyEhEYaWRlVZ2DvMM9eZbMs/RfxPu/PK/aBLyGj4IrqMHw=="],
|
||||
|
||||
"parse5": ["parse5@8.0.1", "", { "dependencies": { "entities": "^8.0.0" } }, "sha512-z1e/HMG90obSGeidlli3hj7cbocou0/wa5HacvI3ASx34PecNjNQeaHNo5WIZpWofN9kgkqV1q5YvXe3F0FoPw=="],
|
||||
@@ -801,6 +1000,8 @@
|
||||
|
||||
"pngjs": ["pngjs@5.0.0", "", {}, "sha512-40QW5YalBNfQo5yRYmiw7Yz6TKKVr3h6970B2YE+3fQpsWcrbj1PzJgxeJ19DRQjhMbKPIuMY8rFaXc8moolVw=="],
|
||||
|
||||
"pnpm-workspace-yaml": ["pnpm-workspace-yaml@1.6.1", "", { "dependencies": { "yaml": "^2.9.0" } }, "sha512-yTeZntGWi8m9WNuhoVsP0DpFc4sC1U0+rr/qR6Zi9n2g3sxXY+JfccjXjjruNz96tM8I09yaJUA86doRnNLkbg=="],
|
||||
|
||||
"postcss": ["postcss@8.5.15", "", { "dependencies": { "nanoid": "^3.3.12", "picocolors": "^1.1.1", "source-map-js": "^1.2.1" } }, "sha512-FfR8sjd4em2T6fb3I2MwAJU7HWVMr9zba+enmQeeWFfCbm+UOC/0X4DS8XtpUTMwWMGbjKYP7xjfNekzyGmB3A=="],
|
||||
|
||||
"prelude-ls": ["prelude-ls@1.2.1", "", {}, "sha512-vkcDPrRZo1QZLbn5RLGPpg/WmIQ65qoWWhcGKf/b5eplkkarX0m9z8ppCat4mlOqUsWpyNuYgO3VRyrYHSzX5g=="],
|
||||
@@ -815,6 +1016,8 @@
|
||||
|
||||
"qrcode": ["qrcode@1.5.4", "", { "dependencies": { "dijkstrajs": "^1.0.1", "pngjs": "^5.0.0", "yargs": "^15.3.1" }, "bin": { "qrcode": "bin/qrcode" } }, "sha512-1ca71Zgiu6ORjHqFBDpnSMTR2ReToX4l1Au1VFLyVeBTFavzQnv5JxMFr3ukHVKpSrSA2MCk0lNJSykjUfz7Zg=="],
|
||||
|
||||
"quansync": ["quansync@1.0.0", "", {}, "sha512-5xZacEEufv3HSTPQuchrvV6soaiACMFnq1H8wkVioctoH3TRha9Sz66lOxRwPK/qZj7HPiSveih9yAyh98gvqA=="],
|
||||
|
||||
"react": ["react@19.2.7", "", {}, "sha512-HNe9WslTbXmFK8o8cmwgAeJFSBvt1bPdHCVKtaaV+WlAN36mpT4hcRpwbf3fY56ar2oIXzsBpOAiIRHAdY0OlQ=="],
|
||||
|
||||
"react-dom": ["react-dom@19.2.7", "", { "dependencies": { "scheduler": "^0.27.0" }, "peerDependencies": { "react": "^19.2.7" } }, "sha512-t0BRVXvbiE/o20Hfw669rLbMCDWtYZLvmJigy2f0MxsXF+71pxhR3xOkspmsO8h3ZlNzyibAmtCa3l4lYKk6gQ=="],
|
||||
@@ -841,6 +1044,10 @@
|
||||
|
||||
"require-main-filename": ["require-main-filename@2.0.0", "", {}, "sha512-NKN5kMDylKuldxYLSUfrbo5Tuzh4hd+2E8NPPX02mZtn1VuREQToYe/ZdlJy+J3uCpfaiGF05e7B8W0iXbQHmg=="],
|
||||
|
||||
"resolve-pkg-maps": ["resolve-pkg-maps@1.0.0", "", {}, "sha512-seS2Tj26TBVOC2NIc2rOe2y2ZO7efxITtLZcGSOnHHNOQ7CkiUBfw0Iw2ck6xkIhPwLhKNLS8BO+hEpngQlqzw=="],
|
||||
|
||||
"restore-cursor": ["restore-cursor@5.1.0", "", { "dependencies": { "onetime": "^7.0.0", "signal-exit": "^4.1.0" } }, "sha512-oMA2dcrw6u0YfxJQXm342bFKX/E4sG9rbTzO9ptUcR/e8A33cHuvStiYOwH7fszkZlZ1z/ta9AAoPk2F4qIOHA=="],
|
||||
|
||||
"rolldown": ["rolldown@1.0.3", "", { "dependencies": { "@oxc-project/types": "=0.133.0", "@rolldown/pluginutils": "^1.0.0" }, "optionalDependencies": { "@rolldown/binding-android-arm64": "1.0.3", "@rolldown/binding-darwin-arm64": "1.0.3", "@rolldown/binding-darwin-x64": "1.0.3", "@rolldown/binding-freebsd-x64": "1.0.3", "@rolldown/binding-linux-arm-gnueabihf": "1.0.3", "@rolldown/binding-linux-arm64-gnu": "1.0.3", "@rolldown/binding-linux-arm64-musl": "1.0.3", "@rolldown/binding-linux-ppc64-gnu": "1.0.3", "@rolldown/binding-linux-s390x-gnu": "1.0.3", "@rolldown/binding-linux-x64-gnu": "1.0.3", "@rolldown/binding-linux-x64-musl": "1.0.3", "@rolldown/binding-openharmony-arm64": "1.0.3", "@rolldown/binding-wasm32-wasi": "1.0.3", "@rolldown/binding-win32-arm64-msvc": "1.0.3", "@rolldown/binding-win32-x64-msvc": "1.0.3" }, "bin": { "rolldown": "./bin/cli.mjs" } }, "sha512-i00lAJ2ks1BYr7rjNjKC7BcqAS7nVfiT3QX1SI5aY+AFHblCmaUf9OE9dbdzDvW6dJxbi2ZCZiy9v3CcwOiX3g=="],
|
||||
|
||||
"rxjs": ["rxjs@7.8.2", "", { "dependencies": { "tslib": "^2.1.0" } }, "sha512-dhKf903U/PQZY6boNNtAGdWbG85WAbjT/1xYoZIC7FAY0yWapOBQVsVrDl58W86//e1VpMNBtRV4MaXfdMySFA=="],
|
||||
@@ -863,6 +1070,8 @@
|
||||
|
||||
"signal-exit": ["signal-exit@4.1.0", "", {}, "sha512-bzyZ1e88w9O1iNJbKnOlvYTrWPDl46O1bG0D3XInv+9tkPrxrN8jUUTiFlDkkmKWgn1M6CfIA13SuGqOa9Korw=="],
|
||||
|
||||
"smol-toml": ["smol-toml@1.7.0", "", {}, "sha512-aqVvWoyO21L23mb+drl4RmMXbf6N7FdHjAhTRA9ZBL7apWBgfWC16KjrASI+1p9GAroljyMHj6fK67i0UiTNvQ=="],
|
||||
|
||||
"source-map-js": ["source-map-js@1.2.1", "", {}, "sha512-UXWMKhLOwVKb728IUtQPXxfYU+usdybtUrK/8uGE8CQMvrhOpwvzDBwj0QhSL7MQc7vIsISBG8VQ8+IDQxpfQA=="],
|
||||
|
||||
"stackback": ["stackback@0.0.2", "", {}, "sha512-1XMJE5fQo1jGH6Y/7ebnwPOBEkIEnT4QF32d5R1+VXdXveM0IBMJt8zfaxX1P3QhVwrYe+576+jkANtSS2mBbw=="],
|
||||
@@ -877,20 +1086,28 @@
|
||||
|
||||
"strip-indent": ["strip-indent@3.0.0", "", { "dependencies": { "min-indent": "^1.0.0" } }, "sha512-laJTa3Jb+VQpaC6DseHhF7dXVqHTfJPCRDaEbid/drOhgitgYku/letMUqOXFoWV0zIIUbjpdH2t+tYj4bQMRQ=="],
|
||||
|
||||
"strip-json-comments": ["strip-json-comments@5.0.3", "", {}, "sha512-1tB5mhVo7U+ETBKNf92xT4hrQa3pm0MZ0PQvuDnWgAAGHDsfp4lPSpiS6psrSiet87wyGPh9ft6wmhOMQ0hDiw=="],
|
||||
|
||||
"supports-color": ["supports-color@8.1.1", "", { "dependencies": { "has-flag": "^4.0.0" } }, "sha512-MpUEN2OodtUzxvKQl72cUF7RQ5EiHsGvSsVG0ia9c5RbWGL2CI4C7EpPS8UTBIplnlzZiNuV56w+FuNxy3ty2Q=="],
|
||||
|
||||
"symbol-tree": ["symbol-tree@3.2.4", "", {}, "sha512-9QNk5KwDF+Bvz+PyObkmSYjI5ksVUYtjW7AU22r2NKcfLJcXp96hkDWU3+XndOsUb+AQ9QhfzfCT2O+CNWT5Tw=="],
|
||||
|
||||
"tailwind-merge": ["tailwind-merge@3.6.0", "", {}, "sha512-uxL7qAVQriqRQPAyK3pj66VqskWqoZ37PW94jwOTwNfq/z9oyu1V+eqrZqtR2+fCiXdYOZe/Modt8GtvqNzu+w=="],
|
||||
|
||||
"tailwindcss": ["tailwindcss@4.3.1", "", {}, "sha512-hk+TB1m+K8CYNrP6rjQaq/Y+4Zylwpa87mLYBKCunwnnQ9p+fHb7kmSfGqyEJoxF/O6CDyABWVFEafNSYKll+Q=="],
|
||||
|
||||
"tapable": ["tapable@2.3.3", "", {}, "sha512-uxc/zpqFg6x7C8vOE7lh6Lbda8eEL9zmVm/PLeTPBRhh1xCgdWaQ+J1CUieGpIfm2HdtsUpRv+HshiasBMcc6A=="],
|
||||
|
||||
"taze": ["taze@19.14.1", "", { "dependencies": { "@antfu/ni": "^30.1.0", "@henrygd/queue": "^1.2.0", "cac": "^7.0.0", "ofetch": "^1.5.1", "package-manager-detector": "^1.6.0", "pathe": "^2.0.3", "pnpm-workspace-yaml": "^1.6.1", "restore-cursor": "^5.1.0", "tinyexec": "^1.2.2", "tinyglobby": "^0.2.16", "unconfig": "^7.5.0", "yaml": "^2.9.0" }, "bin": { "taze": "bin/taze.mjs" } }, "sha512-+wf/IqGReU68vBE/iJ7JCuV5QeD6zQBp9MI6YphN7bT2vf/YIHd0oVA4AJiX3uANI1hQY58MrVmDwLv0x/q3BA=="],
|
||||
|
||||
"tinybench": ["tinybench@2.9.0", "", {}, "sha512-0+DUvqWMValLmha6lr4kD8iAMK1HzV0/aKnCtWb9v9641TnP/MFb7Pc2bxoxQjTXAErryXVgUOfv2YqNllqGeg=="],
|
||||
|
||||
"tinyexec": ["tinyexec@1.1.2", "", {}, "sha512-dAqSqE/RabpBKI8+h26GfLq6Vb3JVXs30XYQjdMjaj/c2tS8IYYMbIzP599KtRj7c57/wYApb3QjgRgXmrCukA=="],
|
||||
"tinyexec": ["tinyexec@1.2.4", "", {}, "sha512-SHf/r48b7vOrjve9PxJo3MN5v5yuyjHvdUcrQffT3WXMUfnGmHDVbC4k3sHJaJTgZCwpUplIaAo5ANtMyp3YHg=="],
|
||||
|
||||
"tinyglobby": ["tinyglobby@0.2.17", "", { "dependencies": { "fdir": "^6.5.0", "picomatch": "^4.0.4" } }, "sha512-wXR/dYpcqKmfWpEdZjiKJOwCNFndD0DMnrW/cYjVGttEkBfVgcLFHoNrlj47mjOVic9yyNu65alsgF4NQyTa2g=="],
|
||||
|
||||
"tinypool": ["tinypool@2.1.0", "", {}, "sha512-Pugqs6M0m7Lv1I7FtxN4aoyToKg1C4tu+/381vH35y8oENM/Ai7f7C4StcoK4/+BSw9ebcS8jRiVrORFKCALLw=="],
|
||||
|
||||
"tinyrainbow": ["tinyrainbow@3.1.0", "", {}, "sha512-Bf+ILmBgretUrdJxzXM0SgXLZ3XfiaUuOj/IKQHuTXip+05Xn+uyEYdVg0kYDipTBcLrCVyUzAPz7QmArb0mmw=="],
|
||||
|
||||
"tldts": ["tldts@7.0.30", "", { "dependencies": { "tldts-core": "^7.0.30" }, "bin": { "tldts": "bin/cli.js" } }, "sha512-ELrFxuqsDdHUwoh0XxDbxuLD3Wnz49Z57IFvTtvWy1hJdcMZjXLIuonjilCiWHlT2GbE4Wlv1wKVTzDFnXH1aw=="],
|
||||
@@ -907,10 +1124,20 @@
|
||||
|
||||
"turbo": ["turbo@2.9.18", "", { "optionalDependencies": { "@turbo/darwin-64": "2.9.18", "@turbo/darwin-arm64": "2.9.18", "@turbo/linux-64": "2.9.18", "@turbo/linux-arm64": "2.9.18", "@turbo/windows-64": "2.9.18", "@turbo/windows-arm64": "2.9.18" }, "bin": { "turbo": "bin/turbo" } }, "sha512-bwabv6PupzeavybzEoArBAkwq5fnzwf8OFnRtpHwnviFWuwJPFxtyH+aVp36TmIqK3aYYgtTJ3J0m2ysxxSzQg=="],
|
||||
|
||||
"tw-animate-css": ["tw-animate-css@1.4.0", "", {}, "sha512-7bziOlRqH0hJx80h/3mbicLW7o8qLsH5+RaLR2t+OHM3D0JlWGODQKQ4cxbK7WlvmUxpcj6Kgu6EKqjrGFe3QQ=="],
|
||||
|
||||
"type-check": ["type-check@0.4.0", "", { "dependencies": { "prelude-ls": "^1.2.1" } }, "sha512-XleUoc9uwGXqjWwXaUTZAmzMcFZ5858QA2vvx1Ur5xIcixXIP+8LnFDgRplU30us6teqdlskFfu+ae4K79Ooew=="],
|
||||
|
||||
"typescript": ["typescript@6.0.3", "", { "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" } }, "sha512-y2TvuxSZPDyQakkFRPZHKFm+KKVqIisdg9/CZwm9ftvKXLP8NRWj38/ODjNbr43SsoXqNuAisEf1GdCxqWcdBw=="],
|
||||
|
||||
"ufo": ["ufo@1.6.4", "", {}, "sha512-JFNbkD1Svwe0KvGi8GOeLcP4kAWQ609twvCdcHxq1oSL8svv39ZuSvajcD8B+5D0eL4+s1Is2D/O6KN3qcTeRA=="],
|
||||
|
||||
"unbash": ["unbash@4.0.2", "", {}, "sha512-8gwNZ29+0/3zmXw7ToIHZtg6wK37xnniRUdBt7B27xZxaxfgR5tGMaGHT0t0dLtBV9fXE7zurh0s6Z1DHVjfWg=="],
|
||||
|
||||
"unconfig": ["unconfig@7.5.0", "", { "dependencies": { "@quansync/fs": "^1.0.0", "defu": "^6.1.4", "jiti": "^2.6.1", "quansync": "^1.0.0", "unconfig-core": "7.5.0" } }, "sha512-oi8Qy2JV4D3UQ0PsopR28CzdQ3S/5A1zwsUwp/rosSbfhJ5z7b90bIyTwi/F7hCLD4SGcZVjDzd4XoUQcEanvA=="],
|
||||
|
||||
"unconfig-core": ["unconfig-core@7.5.0", "", { "dependencies": { "@quansync/fs": "^1.0.0", "quansync": "^1.0.0" } }, "sha512-Su3FauozOGP44ZmKdHy2oE6LPjk51M/TRRjHv2HNCWiDvfvCoxC2lno6jevMA91MYAdCdwP05QnWdWpSbncX/w=="],
|
||||
|
||||
"undici": ["undici@7.25.0", "", {}, "sha512-xXnp4kTyor2Zq+J1FfPI6Eq3ew5h6Vl0F/8d9XU5zZQf1tX9s2Su1/3PiMmUANFULpmksxkClamIZcaUqryHsQ=="],
|
||||
|
||||
"unicorn-magic": ["unicorn-magic@0.3.0", "", {}, "sha512-+QBBXBCvifc56fsbuxZQ6Sic3wqqc3WWaqxs58gvJrcOuN83HGTCwz3oS5phzU9LthRNE9VrJCFCLUgHeeFnfA=="],
|
||||
@@ -935,6 +1162,8 @@
|
||||
|
||||
"wait-on": ["wait-on@9.0.10", "", { "dependencies": { "axios": "^1.16.0", "joi": "^18.2.1", "lodash": "^4.18.1", "minimist": "^1.2.8", "rxjs": "^7.8.2" }, "bin": { "wait-on": "bin/wait-on" } }, "sha512-rCoJEhvMr0X6alHmwc9abbrA5ZrLZFKpFQVKPNFwl2h7DapXOGdmimIHDtLOWhT4PjhZhxFEtZoQgEXbkDWdZw=="],
|
||||
|
||||
"walk-up-path": ["walk-up-path@4.0.0", "", {}, "sha512-3hu+tD8YzSLGuFYtPRb48vdhKMi0KQV5sn+uWr8+7dMEq/2G/dtLrdDinkLjqq5TIbIBjYJ4Ax/n3YiaW7QM8A=="],
|
||||
|
||||
"wavesurfer.js": ["wavesurfer.js@7.12.8", "", {}, "sha512-G3nxzcC4X+ZWrLtcIV17kCWHVq3ysJCS4dS0YkGKILrQ2esAb8cScw965zKNKYxUvpiZsPK93KLWgWTYdIBQiw=="],
|
||||
|
||||
"webidl-conversions": ["webidl-conversions@8.0.1", "", {}, "sha512-BMhLD/Sw+GbJC21C/UgyaZX41nPt8bUTg+jWyDeg7e7YN4xOM05YPSIXceACnXVtqyEw/LMClUQMtMZ+PGGpqQ=="],
|
||||
@@ -961,6 +1190,8 @@
|
||||
|
||||
"yallist": ["yallist@3.1.1", "", {}, "sha512-a4UGQaWPH59mOXUYnAG2ewncQS4i4F43Tv3JoAM+s2VDAmS9NsK8GpDMLrCHPksFT7h3K6TOoUNn2pb7RoXx4g=="],
|
||||
|
||||
"yaml": ["yaml@2.9.0", "", { "bin": { "yaml": "bin.mjs" } }, "sha512-2AvhNX3mb8zd6Zy7INTtSpl1F15HW6Wnqj0srWlkKLcpYl/gMIMJiyuGq2KeI2YFxUPjdlB+3Lc10seMLtL4cA=="],
|
||||
|
||||
"yargs": ["yargs@17.7.2", "", { "dependencies": { "cliui": "^8.0.1", "escalade": "^3.1.1", "get-caller-file": "^2.0.5", "require-directory": "^2.1.1", "string-width": "^4.2.3", "y18n": "^5.0.5", "yargs-parser": "^21.1.1" } }, "sha512-7dSzzRQ++CKnNI/krKnYRV7JKKPUXMEh61soaHKg9mrWEhzFWhFnxPxGl+69cD1Ou63C13NUPCnmIcrvqCuM6w=="],
|
||||
|
||||
"yargs-parser": ["yargs-parser@21.1.1", "", {}, "sha512-tVpsJW7DdjecAiFpbIB1e3qxIQsE6NoPc5/eTdrbbIC4h0LVsWhnoa3g+m2HclBIujHzsxZ4VJVA+GUuc2/LBw=="],
|
||||
@@ -979,8 +1210,18 @@
|
||||
|
||||
"@eslint-community/eslint-utils/eslint-visitor-keys": ["eslint-visitor-keys@3.4.3", "", {}, "sha512-wpc+LXeiyiisxPlEkUzU6svyS1frIO3Mgxj1fdy7Pm8Ygzguax2N3Fa/D/ag1WqbOprdI+uY6wMUl8/a2G+iag=="],
|
||||
|
||||
"@oxc-resolver/binding-wasm32-wasi/@emnapi/core": ["@emnapi/core@1.11.0", "", { "dependencies": { "@emnapi/wasi-threads": "1.2.2", "tslib": "^2.4.0" } }, "sha512-l9Oo58x0HOP5znGzVhYW9U3e5wVuA4LAZU2AGezTmkhO1CgQRFDhDg4nneHsu/t3WniXg9QrG2nIXL/ZS8ln8Q=="],
|
||||
|
||||
"@oxc-resolver/binding-wasm32-wasi/@emnapi/runtime": ["@emnapi/runtime@1.11.0", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-55coeOFKHv1ywEcUXJtWU5f+Jr/W5tZDvZig8DLKSwUN1JpROQ4rk/SNOQiFWmaR/VKF4zuFyW1B8JduOSv6Pg=="],
|
||||
|
||||
"@playwright/test/playwright": ["playwright@1.61.0", "", { "dependencies": { "playwright-core": "1.61.0" }, "optionalDependencies": { "fsevents": "2.3.2" }, "bin": { "playwright": "cli.js" } }, "sha512-Z+7BeeqQPRRzklHsVFP4KTGIyMxKUmfeRA4WisM6G3/XW6nwGeX6fX9qYaDa+CiUqpOkb2f6X3nar05R3kSuJQ=="],
|
||||
|
||||
"@rolldown/binding-wasm32-wasi/@emnapi/core": ["@emnapi/core@1.10.0", "", { "dependencies": { "@emnapi/wasi-threads": "1.2.1", "tslib": "^2.4.0" } }, "sha512-yq6OkJ4p82CAfPl0u9mQebQHKPJkY7WrIuk205cTYnYe+k2Z8YBh11FrbRG/H6ihirqcacOgl2BIO8oyMQLeXw=="],
|
||||
|
||||
"@rolldown/binding-wasm32-wasi/@emnapi/runtime": ["@emnapi/runtime@1.10.0", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-ewvYlk86xUoGI0zQRNq/mC+16R1QeDlKQy21Ki3oSYXNgLb45GV1P6A0M+/s6nyCuNDqe5VpaY84BzXGwVbwFA=="],
|
||||
|
||||
"@rolldown/binding-wasm32-wasi/@napi-rs/wasm-runtime": ["@napi-rs/wasm-runtime@1.1.4", "", { "dependencies": { "@tybys/wasm-util": "^0.10.1" }, "peerDependencies": { "@emnapi/core": "^1.7.1", "@emnapi/runtime": "^1.7.1" } }, "sha512-3NQNNgA1YSlJb/kMH1ildASP9HW7/7kYnRI2szWJaofaS1hWmbGI4H+d3+22aGzXXN9IJ+n+GiFVcGipJP18ow=="],
|
||||
|
||||
"@tailwindcss/oxide-wasm32-wasi/@emnapi/core": ["@emnapi/core@1.10.0", "", { "dependencies": { "@emnapi/wasi-threads": "1.2.1", "tslib": "^2.4.0" }, "bundled": true }, "sha512-yq6OkJ4p82CAfPl0u9mQebQHKPJkY7WrIuk205cTYnYe+k2Z8YBh11FrbRG/H6ihirqcacOgl2BIO8oyMQLeXw=="],
|
||||
|
||||
"@tailwindcss/oxide-wasm32-wasi/@emnapi/runtime": ["@emnapi/runtime@1.10.0", "", { "dependencies": { "tslib": "^2.4.0" }, "bundled": true }, "sha512-ewvYlk86xUoGI0zQRNq/mC+16R1QeDlKQy21Ki3oSYXNgLb45GV1P6A0M+/s6nyCuNDqe5VpaY84BzXGwVbwFA=="],
|
||||
@@ -1013,10 +1254,18 @@
|
||||
|
||||
"qrcode/yargs": ["yargs@15.4.1", "", { "dependencies": { "cliui": "^6.0.0", "decamelize": "^1.2.0", "find-up": "^4.1.0", "get-caller-file": "^2.0.1", "require-directory": "^2.1.1", "require-main-filename": "^2.0.0", "set-blocking": "^2.0.0", "string-width": "^4.2.0", "which-module": "^2.0.0", "y18n": "^4.0.0", "yargs-parser": "^18.1.2" } }, "sha512-aePbxDmcYW++PaqBsJ+HYUFwCdv4LVvdnhBy78E57PIor8/OVvhMrADFFEDh8DHDFRv/O9i3lPhsENjO7QX0+A=="],
|
||||
|
||||
"rolldown/@oxc-project/types": ["@oxc-project/types@0.133.0", "", {}, "sha512-KzkdCd6Uxqnf6l3HOw1xfatAlUURA0g14cvBYFyJ5SaNOQbOUvBr9PKArcPcrNIeRsBdgcUzOGrhKveVpvOIGA=="],
|
||||
|
||||
"vite/fsevents": ["fsevents@2.3.3", "", { "os": "darwin" }, "sha512-5xoDfX+fL7faATnagmWPpbFtwh/R77WmMMqqHGS65C3vvB0YHrgF+B1YmZ3441tMj5n63k0212XNoJwzlhffQw=="],
|
||||
|
||||
"vitest/tinyexec": ["tinyexec@1.1.2", "", {}, "sha512-dAqSqE/RabpBKI8+h26GfLq6Vb3JVXs30XYQjdMjaj/c2tS8IYYMbIzP599KtRj7c57/wYApb3QjgRgXmrCukA=="],
|
||||
|
||||
"vitest/tinyglobby": ["tinyglobby@0.2.16", "", { "dependencies": { "fdir": "^6.5.0", "picomatch": "^4.0.4" } }, "sha512-pn99VhoACYR8nFHhxqix+uvsbXineAasWm5ojXoN8xEwK5Kd3/TrhNn1wByuD52UxWRLy8pu+kRMniEi6Eq9Zg=="],
|
||||
|
||||
"@rolldown/binding-wasm32-wasi/@emnapi/core/@emnapi/wasi-threads": ["@emnapi/wasi-threads@1.2.1", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-uTII7OYF+/Mes/MrcIOYp5yOtSMLBWSIoLPpcgwipoiKbli6k322tcoFsxoIIxPDqW01SQGAgko4EzZi2BNv2w=="],
|
||||
|
||||
"@rolldown/binding-wasm32-wasi/@napi-rs/wasm-runtime/@tybys/wasm-util": ["@tybys/wasm-util@0.10.1", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-9tTaPJLSiejZKx+Bmog4uSubteqTvFrVrURwkmHixBo0G4seD0zUxp98E1DzUBJxLQ3NPwXrGKDiVjwx/DpPsg=="],
|
||||
|
||||
"@tailwindcss/oxide-wasm32-wasi/@napi-rs/wasm-runtime/@tybys/wasm-util": ["@tybys/wasm-util@0.10.1", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-9tTaPJLSiejZKx+Bmog4uSubteqTvFrVrURwkmHixBo0G4seD0zUxp98E1DzUBJxLQ3NPwXrGKDiVjwx/DpPsg=="],
|
||||
|
||||
"qrcode/yargs/cliui": ["cliui@6.0.0", "", { "dependencies": { "string-width": "^4.2.0", "strip-ansi": "^6.0.0", "wrap-ansi": "^6.2.0" } }, "sha512-t6wbgtoCXvAzst7QgXxJYqPt0usEfbgQdftEPbLL/cvv6HPE5VgvqCuAIDR0NgU52ds6rFwqrgakNLrHEjCbrQ=="],
|
||||
|
||||
@@ -0,0 +1,352 @@
|
||||
# Migration — Per-Component `.css` → Tailwind v4 Utilities
|
||||
|
||||
**Status:** Plan (not yet executed) · **Drafted:** 2026-06-30 · **Type:** Incremental styling migration, no intended visual change
|
||||
**Owner stance:** leans toward a full migration but values not breaking the UI · **This plan's recommendation:** *bounded* migration (utilities for layout/spacing/typography everywhere; keep CSS for the hard stuff). See §8.
|
||||
|
||||
## Why
|
||||
|
||||
`frontend/src` carries **74 `.css` files / 16,615 lines** of global, BEM-ish CSS
|
||||
(`dub-col`, `models-row__role`, `readiness-checklist__title`, …). Tailwind v4 is
|
||||
**already wired** — `src/index.css` imports `tailwindcss/theme.css` +
|
||||
`tailwindcss/utilities.css`, has an `@theme` block, and `vite.config.js` runs
|
||||
`@tailwindcss/vite`. So the runtime cost of utilities is already paid; we are just
|
||||
not using them. Editing a layout today means hunting a class across a 989-line
|
||||
file and a JSX `className`. Utilities put the layout where it's read — in the JSX —
|
||||
and shrink the per-component CSS to only what utilities can't express.
|
||||
|
||||
This is **not** a redesign. Every step must render pixel-identical. The honest
|
||||
blocker is that **there are zero visual-regression tests** — the prior page
|
||||
refactors (see `docs/maintenance-pages-modularization.md`) verified "no change"
|
||||
by diffing `className` strings, and *that trick is useless here because the whole
|
||||
point is that class names change*. Closing that gap is the first real task (§4),
|
||||
not an afterthought.
|
||||
|
||||
## Current state (measured 2026-06-30)
|
||||
|
||||
| Metric | Value |
|
||||
|--------|------:|
|
||||
| `.css` files | 74 |
|
||||
| Total CSS lines | 16,615 |
|
||||
| `var(--…)` token references across CSS | ~3,200 |
|
||||
| Files using `display:flex` | 64 |
|
||||
| Files using `display:grid` / `grid-template` | 27 |
|
||||
| Files using `transition:` | 46 |
|
||||
| Files using `box-shadow` | 43 |
|
||||
| Files using `@media` | 25 |
|
||||
| Files using `linear/radial-gradient` | 22 |
|
||||
| Files using `@keyframes` (73 blocks total) | 30 |
|
||||
| Files using `animation:` | 34 |
|
||||
| Files using `backdrop-filter`/glass blur | 11 |
|
||||
| Files using `::before`/`::after` | 11 |
|
||||
| Files using `:has()` | 3 |
|
||||
| Files using `!important` | 14 |
|
||||
|
||||
Biggest files (conversion ROI ranked by layout density, not raw size):
|
||||
`index.css` 2532 · `FirstRunSetup.css` 1020 · `DubTab.css` 989 ·
|
||||
`VoiceGallery.css` 541 · `StoriesEditor.css` 525 · `LogsFooter.css` 507 ·
|
||||
`Settings.css` 469 · `CloneDesignTab.css` 458 · `settings/primitives/primitives.css` 368.
|
||||
|
||||
The token system (do **not** redesign it):
|
||||
|
||||
- `src/ui/tokens.css` (157 lines, ~82 custom props): the declared "single source of
|
||||
truth" — colors, a 4px spacing scale (`--space-0..9`), radius, fonts, type scale,
|
||||
weights, shadows, motion, z-index, focus ring, glass blur. Imported via `src/ui/index.js`.
|
||||
- `src/ui/themes.css` (188 lines): per-theme overrides of the semantic color tokens,
|
||||
keyed on `[data-theme="midnight|nord|solarized|…"]` on `<html>`. Default (no attribute)
|
||||
= Gruvbox Dark.
|
||||
- `src/index.css` `@theme { … }`: maps a subset of tokens into Tailwind's theme
|
||||
namespace (`--color-*`, `--radius-*`, `--font-*`) so utilities like `bg-bg`,
|
||||
`text-fg`, `rounded-lg`, `font-mono` exist. **It hardcodes hex literals that
|
||||
duplicate `tokens.css`** — the known drift bug (see §2).
|
||||
|
||||
Load order today: `index.css` (`@theme` → `theme` layer, lowest priority) is
|
||||
imported in `App.jsx`; `tokens.css` + `themes.css` are **unlayered** `:root` /
|
||||
`[data-theme]` rules imported via `ui/index.js`. Because unlayered CSS outranks
|
||||
`@layer theme`, **`tokens.css` already wins for the default values and theming
|
||||
already works** — the `@theme` hex literals are effectively a *losing duplicate*
|
||||
that exists only so Tailwind knows the utility names. That is precisely why they
|
||||
drift silently: nothing at runtime reads them, so a stale value never shows up.
|
||||
|
||||
## Strategy (the shape of the whole thing)
|
||||
|
||||
1. **Incremental, component-by-component — never big-bang.** One component (or one
|
||||
small cluster) per PR. Each PR is independently shippable and CI-green. A
|
||||
half-migrated component is fine; a half-migrated *codebase* is the steady state
|
||||
for months and that's acceptable.
|
||||
2. **Utilities-first for the mechanical 80%:** flexbox, grid, gap, padding/margin,
|
||||
width/height, `text-*`/`font-*`, `rounded-*`, `border`, simple `bg-*`/`text-*`
|
||||
color, `hidden`, `truncate`, basic `hover:`/`focus:` color states. These map 1:1
|
||||
to utilities and are where the line-count win lives.
|
||||
3. **Keep `.css` for the hard 20%:** glassmorphism (layered gradients +
|
||||
`backdrop-filter`), `::before`/`::after`, `@keyframes`, `:has()` and other complex
|
||||
combinators, `[data-theme]`-specific rules, and anything with `!important`
|
||||
fighting specificity. Utilities don't express these cleanly and forcing them
|
||||
(arbitrary-value soup, `[&::before]:…`) trades readable CSS for unreadable JSX.
|
||||
4. **One source of truth via the token bridge (§2):** utilities reference the same
|
||||
CSS vars the remaining `.css` reads, so a value lives in exactly one place and
|
||||
`data-theme` switching keeps working for both.
|
||||
5. **No file is "done" until it's deleted or demonstrably minimal.** Success is
|
||||
measured in CSS LOC removed and `.css` files deleted, not files "touched."
|
||||
|
||||
## 2. Token-bridge prerequisite (P0 — gates everything)
|
||||
|
||||
The migration is only safe if a utility and the leftover CSS in the same component
|
||||
resolve a token to the *same* value, including after a theme switch. Today the
|
||||
`@theme` literals duplicate `tokens.css`; once components start mixing `bg-bg`
|
||||
(utility) with `background: var(--color-bg)` (CSS), any drift becomes a visible,
|
||||
theme-dependent bug. Fix the source-of-truth **before** converting anything.
|
||||
|
||||
**Recommended fix — Solution A (lowest churn, no rename):** Make `@theme` the
|
||||
single declared home for the **already-overlapping** groups only — colors, radius,
|
||||
fonts — and **delete those default declarations from `tokens.css`** (leave a
|
||||
one-line pointer comment). Everything else in `tokens.css` (spacing, type scale,
|
||||
weights, shadows, motion, z-index, focus ring, glass blur) stays put.
|
||||
|
||||
Why this is correct and safe:
|
||||
|
||||
- Tailwind needs the keys present in `@theme` to generate the utility names
|
||||
(`--color-fg` → `text-fg`/`bg-fg`; `--radius-lg` → `rounded-lg`; `--font-mono` →
|
||||
`font-mono`). Keeping the keys there is non-negotiable.
|
||||
- `@theme` emits `:root { --color-fg: … }` into the low-priority `theme` layer.
|
||||
`themes.css` `[data-theme]` rules are unlayered and still outrank it, so
|
||||
**theme switching is unchanged** — verify with a quick manual cycle through all
|
||||
themes after the edit.
|
||||
- Removing the duplicate `:root` color/radius/font lines from `tokens.css` leaves
|
||||
exactly one literal per value. All ~3,200 existing `var(--…)` references keep
|
||||
resolving (the var still exists on `:root`, now sourced from `@theme`).
|
||||
|
||||
**Guard against recurrence (required, per the "fix the class" rule):** add
|
||||
`frontend/src/__tests__/theme-token-parity.test.js` (vitest, no browser) that
|
||||
parses `index.css` `@theme` + `tokens.css` + `themes.css` and asserts:
|
||||
(a) no token key is declared with a literal in **both** `@theme` and `tokens.css`
|
||||
(catches re-introduced duplication), and (b) every `@theme` color key is overridden
|
||||
by every `[data-theme]` block in `themes.css` (catches a theme that forgot a color).
|
||||
This test is the thing that makes the de-dup *stay* de-duped.
|
||||
|
||||
**Rejected alternative — Solution B (purist):** rename source tokens to a private
|
||||
namespace (`--ov-color-fg`) and bridge with `@theme inline { --color-fg:
|
||||
var(--ov-color-fg) }`. This honors "`tokens.css` is the source" literally and is
|
||||
the textbook Tailwind pattern, **but** it forces renaming all ~3,200 `var(--color-*)`
|
||||
references across 74 files in one shot — a massive, high-risk diff that violates
|
||||
"low-risk, incremental." Not worth it. (`@theme inline` referencing the *same* name
|
||||
is circular and is not an option.)
|
||||
|
||||
**Optionally, later:** add `--spacing` to `@theme` so `p-*`/`gap-*`/`m-*` map onto
|
||||
the existing 4px scale (`--space-1 = 2px` … `--space-9 = 44px`). Tailwind's default
|
||||
spacing is a 0.25rem multiplier; OmniVoice's scale is custom, so without this,
|
||||
`gap-3` ≠ `var(--space-3)`. Two choices, decide in P0:
|
||||
- **Map to the scale:** set `--spacing: 2px` won't reproduce the non-linear steps;
|
||||
instead define explicit `--spacing-1..9` in `@theme` mirroring `--space-1..9`,
|
||||
and use `gap-2`/`p-5` etc. Cleanest for readers, but utility numbers won't match
|
||||
Tailwind defaults — document it.
|
||||
- **Use arbitrary values bridged to the var:** `gap-[var(--space-3)]`,
|
||||
`p-[var(--space-5)]`. Zero ambiguity, slightly noisier JSX, guarantees identical
|
||||
pixels. **Recommended for P1–P2** (safest for "no visual change"); revisit named
|
||||
spacing once confidence is high.
|
||||
|
||||
## 3. What converts cleanly vs. what stays CSS
|
||||
|
||||
**Converts cleanly → utilities** (concrete, from real files):
|
||||
|
||||
- `ReadinessChecklist.css` `.readiness-checklist { display:flex; flex-direction:column;
|
||||
gap:var(--space-3); padding:var(--space-5); border:1px solid var(--color-border);
|
||||
border-radius:var(--radius-lg); font-size:var(--text-sm); }`
|
||||
→ `className="flex flex-col gap-[var(--space-3)] p-[var(--space-5)] border
|
||||
border-border rounded-lg text-sm"` (or mapped `text-sm` if the type scale is
|
||||
bridged). The `backdrop-filter` line on the same selector **stays in CSS** (see below).
|
||||
- `.readiness-checklist__title { font-weight:var(--weight-semibold);
|
||||
color:var(--color-fg); display:flex; align-items:center; gap:var(--space-3); }`
|
||||
→ `font-semibold text-fg flex items-center gap-[var(--space-3)]`.
|
||||
- Generic layout rows/cols (`dub-col`, `models-row`) — flex/grid/gap/padding → utilities.
|
||||
|
||||
**Stays in `.css`** (criteria + real examples):
|
||||
|
||||
- **Glassmorphism / layered backgrounds.** `Panel.css` `.ui-panel--glass` stacks two
|
||||
`radial-gradient`s + a `linear-gradient` + `backdrop-filter: var(--glass-blur-md)`.
|
||||
Leave entirely in CSS. (11 files use glass blur.)
|
||||
- **Pseudo-elements.** `Panel.css` `.ui-panel--glass::before` (top hairline gradient);
|
||||
`DubTab.css` `.dub-stepper__step::before` (connector line). 11 files. Stay.
|
||||
- **Keyframes + animations.** 73 `@keyframes` blocks across 30 files
|
||||
(`@keyframes mesh/spin/pulse/shimmer` in `index.css`; `dub-pulse`,
|
||||
`dub-stepper-spin`, `dub-skel-shimmer` in `DubTab.css`). Keep the `@keyframes` and
|
||||
the `animation:` shorthand in CSS; a `className="animate-…"` only helps if you
|
||||
register the animation in `@theme`, which isn't worth it for one-off effects.
|
||||
- **`:has()` and complex combinators** (3 files), **`[data-theme]`-specific rules**
|
||||
(all of `themes.css` + scattered overrides), **`!important` blocks** (14 files,
|
||||
e.g. `DubTab.css` `.dub-footer-panel::before { display:none !important; }`).
|
||||
- **Media queries** (25 files): convertible to `sm:`/`md:`/`lg:` **only** if the
|
||||
breakpoints match Tailwind's; OmniVoice's are custom, so leave responsive blocks in
|
||||
CSS unless a component's breakpoints are first added to `@theme`. Low priority.
|
||||
|
||||
Rule of thumb for a reviewer: *if a declaration reads a single token and sets one
|
||||
box/text/flex property, it's a utility; if it composes multiple values, targets a
|
||||
pseudo-element/state combinator, or animates, it stays.*
|
||||
|
||||
## 4. Risk mitigation — the no-visual-test gap (the gating risk)
|
||||
|
||||
This is the make-or-break item. Be honest: **without a visual baseline, "no change"
|
||||
is unverifiable**, and `className`-diffing (what the page refactors relied on) cannot
|
||||
work when class names are the thing changing. Two layers, do both:
|
||||
|
||||
**(a) Establish a screenshot baseline before touching components (part of P0).**
|
||||
Add Playwright component/page screenshots for the surfaces being migrated. The repo
|
||||
already references Playwright tooling in its docs stack; wire a minimal
|
||||
`tests/visual/` that boots the Vite app (or Storybook-less direct route renders) and
|
||||
captures per-component PNGs at a fixed viewport for **the default theme + one dark +
|
||||
one light theme** (catches token-bridge regressions specifically). Commit baselines.
|
||||
Each migration PR runs `playwright test --update-snapshots=none` and **fails on any
|
||||
pixel diff above a tiny threshold**. This converts "did it change?" from a human
|
||||
guess into a CI gate. Capture baselines *first*, on `main`, so they reflect
|
||||
pre-migration truth.
|
||||
|
||||
- Scope realistically: snapshotting all 74 surfaces up front is its own project.
|
||||
Snapshot **per phase, just-in-time** — before P1 leaf work, baseline the leaf
|
||||
components; before P3, baseline the big pages. Baselines for a component land in
|
||||
the same PR that prepares to migrate it (separate from the conversion PR so the
|
||||
baseline diff is reviewable on its own).
|
||||
|
||||
**(b) A per-component manual checklist** (belt-and-suspenders, and the fallback for
|
||||
surfaces that are hard to screenshot deterministically — anything with animation,
|
||||
canvas/waveform, or live backend data):
|
||||
1. Default theme: side-by-side before/after at the same viewport.
|
||||
2. Cycle every `[data-theme]` — confirm colors still swap (token-bridge check).
|
||||
3. Hover/focus/active/disabled states on interactive elements.
|
||||
4. The component's `@keyframes`/animation still runs.
|
||||
5. `prefers-reduced-motion` path unaffected (e.g. `#root` launch animation).
|
||||
6. No console warnings; `bun run build` + `bun run lint` clean.
|
||||
|
||||
If neither (a) nor (b) is in place for a surface, **do not migrate it** — defer it to
|
||||
the "leave as CSS" bucket rather than fly blind.
|
||||
|
||||
## 5. Phasing
|
||||
|
||||
Each phase = one or more independently shippable, CI-green PRs. Ordered
|
||||
leaf-inward so blast radius grows only as confidence does.
|
||||
|
||||
### P0 — Token bridge + tooling + visual baseline (no component conversions)
|
||||
- De-dup `@theme` ↔ `tokens.css` (§2 Solution A) + the parity test.
|
||||
- Decide + document the spacing approach (arbitrary-value bridge recommended).
|
||||
- Add `prettier-plugin-tailwindcss` (or confirm oxlint/oxfmt class-sort) and wire
|
||||
class sorting (§6).
|
||||
- Update `CONTRIBUTING.md` (§6 — currently says *"Vanilla CSS … no Tailwind"*, which
|
||||
now contradicts reality and **must** change in this same PR per the docs-sync rule).
|
||||
- Stand up `tests/visual/` Playwright harness (no per-component baselines yet — just
|
||||
the runner + theme matrix).
|
||||
- **Effort:** ~1–2 days. **Success:** parity test green; theme switch verified across
|
||||
all themes; CI gains a class-sort check; zero pixels changed (this PR ships no
|
||||
component edits).
|
||||
|
||||
### P1 — Leaf / presentational components (lowest risk)
|
||||
Targets: small `ui/` primitives and stateless components where CSS is mostly
|
||||
flex/grid/spacing/type — e.g. `Badge`, `UpdateStatusChip`, `NetworkToggle`,
|
||||
`ReadinessChecklist`, `ReadinessChecklist`, `DemoPresetGrid`, `KeyboardCheatsheet`,
|
||||
`MultiLangPicker`. Skip glass-heavy ones for now.
|
||||
- Per component: baseline screenshot PR → conversion PR. Convert layout/spacing/type
|
||||
to utilities; keep any glass/`::before`/animation lines in a now-tiny `.css`; delete
|
||||
the `.css` entirely if nothing remains and remove its import.
|
||||
- **Effort:** ~3–5 days across ~10–15 components. **Success:** ~10 `.css` files deleted
|
||||
or reduced >70%; visual diffs clean; a repeatable per-component recipe proven.
|
||||
|
||||
### P2 — Panels & mid-size components
|
||||
Targets: `settings/*Panel.css`, `Sidebar`, `NotificationPanel`, `CastingView`,
|
||||
`ExportModal`, `EngineCompatibilityMatrix`, `donate/Postcard`, etc. More state,
|
||||
some glass — convert the layout skeleton, leave glass/pseudo/animation.
|
||||
- **Effort:** ~1–1.5 weeks. **Success:** settings panels are thin utility JSX + a
|
||||
shared `primitives.css` for the glass/control look; CSS LOC down materially.
|
||||
|
||||
### P3 — Big pages
|
||||
Targets in ROI order: `DubTab` (989), `VoiceGallery` (541), `StoriesEditor` (525),
|
||||
`LogsFooter` (507), `Settings` (469), `CloneDesignTab` (458), `FirstRunSetup` (1020).
|
||||
These pair naturally with the already-planned page modularization
|
||||
(`docs/maintenance-pages-modularization.md`) — **sequence the modularization first**,
|
||||
then migrate the smaller extracted components (P3 becomes "P1 again" on the pieces).
|
||||
Convert layout/spacing; the pipeline steppers, overlays, gradients, and keyframes
|
||||
stay as CSS.
|
||||
- **Effort:** ~2–3 weeks. **Success:** each page's `.css` drops to the
|
||||
glass/animation/pseudo residue; biggest single LOC reductions land here.
|
||||
|
||||
### P4 — Retire `index.css` globals last
|
||||
`index.css` (2532 lines) is foundation: `@theme`, `@keyframes`, `::selection`, root
|
||||
rendering, base resets, and shared global classes. Convert only the **global utility
|
||||
classes** that components reuse into real utilities or component-scoped CSS; **keep**
|
||||
the `@theme`, keyframes, resets, and `::selection`. Do this last because everything
|
||||
depends on it.
|
||||
- **Effort:** ~1 week. **Success:** `index.css` shrinks to foundation only; no
|
||||
orphaned global classes.
|
||||
|
||||
## 6. Tooling
|
||||
|
||||
- **Class sorting / formatting.** The repo lints with **oxlint** (`bun run lint`,
|
||||
gate) and an advisory ESLint for hooks. For Tailwind class ordering, add
|
||||
**`prettier-plugin-tailwindcss`** (canonical, understands `@theme`) wired to run on
|
||||
`*.jsx`, *or* adopt oxfmt's Tailwind class-sorting if the team prefers a single
|
||||
formatter. Either way the goal is deterministic class order so diffs stay readable
|
||||
and merge-clean.
|
||||
- **Regression prevention.** Add an oxlint/convention guard so new components don't
|
||||
reintroduce sprawling CSS: a soft rule (warn-only first, per "keep main green") that
|
||||
flags new `.css` files over a small line budget for components that should be
|
||||
utility-first, and the §2 parity test as a hard gate on token drift.
|
||||
- **CONTRIBUTING update (required).** `CONTRIBUTING.md` currently states *"CSS:
|
||||
Vanilla CSS in component-level files — no Tailwind."* That is now false. Replace it
|
||||
with the utilities-first standard: *layout/spacing/typography/simple color via
|
||||
Tailwind utilities; component `.css` only for glass, pseudo-elements, keyframes,
|
||||
`:has()`, `[data-theme]` rules, and `!important` overrides; tokens live in
|
||||
`tokens.css`/`@theme`, never hardcoded.* Per the docs-sync hard rule this lands in
|
||||
the **same PR** as P0.
|
||||
- **No new build infra** — `@tailwindcss/vite` already does everything; no PostCSS
|
||||
config, no Tailwind config file (v4 is CSS-first via `@theme`).
|
||||
|
||||
## 7. Non-goals / when to stop
|
||||
|
||||
- **No 100% conversion target.** ~20% of the CSS (the 11 glass files, 30 keyframe
|
||||
files, 11 pseudo-element files, 3 `:has()` files, 14 `!important` files, custom-
|
||||
breakpoint media queries) is **genuinely better as CSS** and should stay. Forcing it
|
||||
into arbitrary-value utilities makes JSX unreadable for zero benefit.
|
||||
- **No token-system redesign.** `tokens.css`/`themes.css` and the `data-theme` model
|
||||
stay as-is (only the §2 de-dup).
|
||||
- **No visual redesign.** Pixel-identical is the contract; restyling is a separate task.
|
||||
- **No `.jsx` → `.tsx`**, no engine/backend/Tauri/Python surface, no version bump,
|
||||
no dependency change beyond the dev-only formatter plugin + Playwright (frontend-only).
|
||||
- **Stop conditions for an individual file:** if after pulling out layout/spacing the
|
||||
remaining CSS is all glass/animation/pseudo, it's *done* — don't chase the last 10%.
|
||||
- **Hands off** `BootstrapSplash.css`, `WaveformPlayer.css`/`SegmentTrack.css`
|
||||
(canvas-adjacent), and other animation/`::before`-dominated files unless a clear
|
||||
layout win exists.
|
||||
|
||||
## 8. Effort + recommendation
|
||||
|
||||
**Total rough effort:** ~5–7 focused weeks for P0–P4 at the *bounded* scope below,
|
||||
spread across many small PRs (it parallelizes and pauses cleanly — it never has to be
|
||||
one big push).
|
||||
|
||||
**Recommendation — bounded migration, not 100%.** The owner leans full-migration and
|
||||
prizes not breaking things; those two goals partly conflict, and the honest call is:
|
||||
|
||||
- **Do** convert layout/spacing/typography/simple color **everywhere** — that's the
|
||||
real maintainability win, it's where ~80% of the 16.6k lines live, and it's the
|
||||
low-risk part.
|
||||
- **Keep ~15–25% as CSS** (glass, keyframes, pseudo-elements, `:has()`,
|
||||
`[data-theme]`, `!important`, custom-breakpoint media). Converting these buys
|
||||
unreadable JSX and *raises* visual-regression risk on exactly the components where
|
||||
diffs are hardest to verify.
|
||||
- **Gate on the visual baseline (§4).** This is the single most important decision: if
|
||||
the Playwright screenshot harness doesn't ship in P0, do **not** start P1 — without
|
||||
it the "won't break the UI" requirement is unmet by construction. The token-bridge
|
||||
de-dup (§2) is the other hard prerequisite; both are cheap and both are P0.
|
||||
|
||||
A realistic end state: ~60 `.css` files deleted or reduced >70%, perhaps ~10–12k of
|
||||
the 16.6k CSS lines removed, the rest a deliberate, documented residue of effects
|
||||
utilities can't express. That delivers nearly all the maintainability benefit of a
|
||||
"full" migration at a fraction of the regression risk.
|
||||
|
||||
## Constraints honored
|
||||
|
||||
- **Keep main green** — every phase is an independently CI-green PR; lint/format and
|
||||
parity-test guards are warn-first where they'd otherwise churn.
|
||||
- **Docs-sync** — the `CONTRIBUTING.md` rewrite lands in the same PR as P0.
|
||||
- **No versioning/Docker/Tauri/Python impact** — frontend-only; dev-dependency-only
|
||||
tooling additions; no `package.json` *version* bump (a devDependency add still
|
||||
requires regenerating root `bun.lock` and confirming `bun install --frozen-lockfile`
|
||||
per the Docker-green rule).
|
||||
- **Local-first / cross-platform parity** — pure styling; no behavior, no platform
|
||||
divergence.
|
||||
@@ -14,6 +14,15 @@ download UI couldn't show real bytes/speed. Until a proper Xet progress hook
|
||||
lands, the app forces the **classic LFS path**, which streams through the
|
||||
standard progress reporter and gives accurate downloaded/remaining/speed.
|
||||
|
||||
To keep that path **fast** despite Xet being off, the app runs a built-in
|
||||
**multi-connection (segmented) downloader on by default** — it fetches each file
|
||||
over parallel byte-ranges (IDM/uGet style), so the legacy-LFS path is no longer
|
||||
single-stream. It reports real live speed/ETA and **falls back to the normal
|
||||
download on any error**, so it can never compromise a correct install. Adding a
|
||||
free Hugging Face token (first-run setup, or Settings → Credentials) makes this
|
||||
faster still — authenticated downloads get higher rate limits and fewer stalls.
|
||||
To force the old single-stream path, set `OMNIVOICE_SEGMENTED_DOWNLOAD=0`.
|
||||
|
||||
State is reported at **Settings → About** / `GET /system/info`:
|
||||
|
||||
- `fast_download.xet_installed` — `hf_xet` present (true)
|
||||
@@ -54,12 +63,13 @@ When a download starts you'll see, in order:
|
||||
|
||||
## Advanced / opt-in tuning
|
||||
|
||||
All of these default **off** and apply to every platform identically. Set them
|
||||
as environment variables (or via **Settings → API keys / environment**).
|
||||
These apply to every platform identically. Set them as environment variables (or
|
||||
via **Settings → API keys / environment**). The segmented accelerator is **on by
|
||||
default** (set its var to `0` to disable); the rest default **off**.
|
||||
|
||||
| Setting | Env var | Effect |
|
||||
|---|---|---|
|
||||
| Segmented accelerator | `OMNIVOICE_SEGMENTED_DOWNLOAD=1` | Multi-connection downloader (parallel byte-ranges) for the legacy-LFS path — restores parallel speed **and** shows live byte speed/ETA. Falls back to the normal download on any error; files land in the standard cache. Best paired with Xet disabled (the default). |
|
||||
| Segmented accelerator | `OMNIVOICE_SEGMENTED_DOWNLOAD=0` | **On by default** (see above). Set to `0` to force the old single-stream legacy-LFS download instead of the parallel byte-range one. |
|
||||
| Max parallel files | `OMNIVOICE_DOWNLOAD_MAX_WORKERS` (default 8) | Files fetched at once. Xet already parallelises *within* a file, so raising this rarely helps and uses more memory. |
|
||||
| High-performance mode | `HF_XET_HIGH_PERFORMANCE=1` | Maximum throughput. Needs lots of RAM and bandwidth — can **hurt** low-RAM machines. Leave off unless you have headroom. |
|
||||
| Spinning-disk (HDD) | `HF_XET_RECONSTRUCT_WRITE_SEQUENTIALLY=1` | Sequential writes; avoids parallel-write thrash on HDDs. Leave off on SSD/NVMe. |
|
||||
|
||||
@@ -0,0 +1,124 @@
|
||||
# Translation engines (Dub tab)
|
||||
|
||||
OmniVoice dubs in two steps: **transcribe → translate → speak**. The *translate*
|
||||
step is pluggable — pick the engine in the Dub tab's **Engine** dropdown. Two
|
||||
engines are **built in** and always available offline; the rest need a small
|
||||
optional Python package.
|
||||
|
||||
| Engine | Category | Needs a package? | Key needed? |
|
||||
|--------|----------|------------------|-------------|
|
||||
| **Argos** (Local, Fast) | offline | `argostranslate` (bundled) | no |
|
||||
| **NLLB-200** (Local, Heavy) | offline | none (uses core `transformers`) | no |
|
||||
| Google Translate (Free) | online | `deep_translator` | no |
|
||||
| DeepL | online | `deep_translator` | yes (`DEEPL_API_KEY`) |
|
||||
| Microsoft Translator | online | `deep_translator` | yes (`MICROSOFT_API_KEY`) |
|
||||
| MyMemory | online | `deep_translator` | no |
|
||||
| LLM (OpenAI-compatible) | llm | `openai` | usually yes |
|
||||
|
||||
If you pick an engine whose package isn't importable yet, the Engine label shows
|
||||
a **highlighted Install affordance**, and — if you try to translate anyway — the
|
||||
backend returns a single, actionable error telling you exactly what to install
|
||||
(the install command is single-sourced, so the button and the error never
|
||||
disagree).
|
||||
|
||||
## Installing optional translation engines (from-source vs packaged build)
|
||||
|
||||
How you add an engine depends on **how you installed OmniVoice**.
|
||||
|
||||
### From-source / dev install (one-click)
|
||||
|
||||
If you cloned the repo and run OmniVoice from source (`uv sync` + the dev
|
||||
launcher) or via Docker, the app can install engines for you:
|
||||
|
||||
1. In the Dub tab, open the translation settings and pick the engine you want
|
||||
(e.g. **Google Translate**) from the **Engine** dropdown.
|
||||
2. A highlighted **Install** button appears next to the *Engine* label. Click it.
|
||||
3. OmniVoice runs the install into the **same** Python environment the backend
|
||||
is using (`uv pip install <package> --python <backend-interpreter>`), then
|
||||
re-probes. When it reports *"restart the backend to load it"*, restart so the
|
||||
freshly-installed module is importable.
|
||||
|
||||
You can also install by hand into the backend venv:
|
||||
|
||||
```
|
||||
uv pip install deep_translator # Google / DeepL / Microsoft / MyMemory
|
||||
uv pip install argostranslate # Argos (already bundled; rarely needed)
|
||||
uv pip install openai # LLM (OpenAI-compatible) provider
|
||||
```
|
||||
|
||||
Then restart the backend.
|
||||
|
||||
### Packaged / installer build (read-only — use the popover)
|
||||
|
||||
The signed desktop installers (`.dmg`, `.msi`, AppImage, `.deb`) ship a
|
||||
**read-only, code-signed Python environment**. Installing extra packages into it
|
||||
would break the signature, so **in-app install is intentionally disabled** on
|
||||
these builds. Selecting an uninstalled engine there shows a highlighted button
|
||||
that opens a small popover with everything you need:
|
||||
|
||||
- **The exact command** to run (with a copy-to-clipboard button) if you *do*
|
||||
have a from-source checkout somewhere and want the online engines there.
|
||||
- **Switch to Argos (bundled, offline)** — one click. Argos and NLLB are always
|
||||
importable in every build, so this is the guaranteed escape hatch: you can
|
||||
keep dubbing immediately, fully offline, no install required.
|
||||
- A link back to this page.
|
||||
|
||||
**Recommendation for packaged builds:** just use **Argos** (fast, offline) or
|
||||
**NLLB-200** (heavier, higher quality, offline). They need nothing installed and
|
||||
never leave your machine. Reach for the online engines only from a from-source
|
||||
install where you can add their package.
|
||||
|
||||
## Translation quality: Fast, Autofit, Cinematic
|
||||
|
||||
The **Quality** control in the Dub tab (and Settings → Translation) picks how the
|
||||
translation is produced:
|
||||
|
||||
- **Fast** — a direct one-shot translation from the selected engine (Argos, NLLB,
|
||||
Google, …). No LLM, no timing awareness.
|
||||
- **Cinematic** — an LLM refines the literal translation (reflect → adapt) for
|
||||
natural, in-context phrasing.
|
||||
- **Autofit** — Cinematic **plus** a strict fit-to-time pass: the LLM rewrites
|
||||
each line so its target-language reading time fits **within** the segment's
|
||||
slot (never overruns it). This keeps the video timing intact and avoids the
|
||||
stressed audio time-stretch you get when a translation is too long for its
|
||||
slot. Fit is per-language pronunciation-speed aware.
|
||||
|
||||
Cinematic and Autofit **require an LLM** (below). If none is configured, they
|
||||
fall back to Fast with a notice.
|
||||
|
||||
## LLM Providers (for Cinematic / Autofit)
|
||||
|
||||
**Settings → System → LLM Providers** is the one place to set up the LLM. Pick a
|
||||
provider, paste its API key, choose a model, **Test** it, and "use for
|
||||
translation." Supported: OpenAI, OpenRouter, Groq, Cerebras, Google AI (Gemini),
|
||||
Mistral, Cohere, NVIDIA, GitHub Models, Cloudflare, Hugging Face, SambaNova,
|
||||
SiliconFlow, **local Ollama / LM Studio** (offline, no key), and a **Custom**
|
||||
OpenAI-compatible endpoint.
|
||||
|
||||
Keys entered here are stored **encrypted** on your machine and never returned to
|
||||
the UI. For a fully offline setup, pick **Ollama** (`ollama pull llama3.1`) or
|
||||
**LM Studio** — nothing leaves the machine. Power users can still override any
|
||||
provider via environment variables (e.g. `GROQ_API_KEY`, or the legacy
|
||||
`TRANSLATE_BASE_URL` / `TRANSLATE_API_KEY` / `TRANSLATE_MODEL`, which map to the
|
||||
**Custom** provider).
|
||||
|
||||
## API keys (online MT engines)
|
||||
|
||||
The non-LLM online engines need a key, set as an environment variable before
|
||||
launching the backend (or in **Settings → Credentials**):
|
||||
|
||||
- **DeepL:** `DEEPL_API_KEY` (optionally `DEEPL_BASE_URL` for a self-hosted /
|
||||
pro endpoint).
|
||||
- **Microsoft Translator:** `MICROSOFT_API_KEY` (optionally `MICROSOFT_BASE_URL`).
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- **"The 'google' translation engine needs the optional deep_translator Python
|
||||
package…"** — the package isn't installed. On a from-source install, click the
|
||||
Install button (or run the command above) and restart. On a packaged build,
|
||||
switch to Argos/NLLB via the popover.
|
||||
- **Install button does nothing / says "disabled in packaged builds"** — you're
|
||||
on a signed installer build (expected). Use Argos/NLLB, or add the package in a
|
||||
from-source checkout.
|
||||
- **Installed it but still "needs install"** — restart the backend so Python
|
||||
picks up the newly-installed module.
|
||||
@@ -1,10 +1,11 @@
|
||||
# Engine venvs & disk usage
|
||||
|
||||
Most engines run in-process in OmniVoice's main environment. A few
|
||||
(**IndexTTS2**, and any engine whose dependencies conflict with the parent's
|
||||
`torch`/`transformers` pins) run in a **dedicated sidecar venv** so their pins
|
||||
can't break the rest of the app. Those sidecars are where disk adds up — this
|
||||
page explains why, and how the on-disk cost is kept down.
|
||||
(**IndexTTS2**, **MOSS-TTS-v1.5**, **dots.tts**, and any engine whose
|
||||
dependencies conflict with the parent's `torch`/`transformers` pins) run in a
|
||||
**dedicated sidecar venv** so their pins can't break the rest of the app. Those
|
||||
sidecars are where disk adds up — this page explains why, and how the on-disk
|
||||
cost is kept down.
|
||||
|
||||
## Why a sidecar needs its own venv
|
||||
|
||||
@@ -48,6 +49,13 @@ the parent whenever the engine allows it.** When it doesn't (IndexTTS2's
|
||||
`transformers<5` forces an older torch line), the second copy is the
|
||||
unavoidable price of isolation — not a bug.
|
||||
|
||||
The opt-in #498 engines illustrate both sides: **dots.tts** pins
|
||||
`torch==2.8.0` — the **same** build the parent constrains to — so it shares
|
||||
almost all of torch with the main venv and only its `transformers==4.57` +
|
||||
model deps are new. **MOSS-TTS-v1.5** pins `torch==2.9.1+cu128`, a **different**
|
||||
build, so it pays a full extra multi-GB torch copy on CUDA hosts (the price of
|
||||
running an 8B model whose stack pins `transformers==5.0`).
|
||||
|
||||
> On Linux, the `nvidia-*` CUDA packages are separate wheels, so even across
|
||||
> *different* torch versions any `nvidia-*` whose pinned version happens to
|
||||
> match is still shared. On Windows the CUDA DLLs live inside the one torch
|
||||
|
||||
@@ -0,0 +1,117 @@
|
||||
# OmniVoice Studio — dots.tts Engine
|
||||
|
||||
dots.tts (rednote-hilab) is a **2B** fully-continuous autoregressive TTS,
|
||||
widely cited as one of the strongest open zero-shot voice-cloning models. It
|
||||
covers **24 languages**, emits **48 kHz** audio, and is released under
|
||||
**Apache-2.0** (code + checkpoints).
|
||||
|
||||
It runs in its own subprocess **and its own Python venv** with
|
||||
`transformers==4.57.0`, isolated from the OmniVoice parent process which
|
||||
pins `transformers>=5.3` — the same isolation primitive used by
|
||||
[IndexTTS-2](indextts.md) and [MOSS-TTS-v1.5](moss-tts-v15.md).
|
||||
|
||||
> **Opt-in, and never a default.** dots.tts is selected explicitly in
|
||||
> **Settings → Engines** (or `OMNIVOICE_TTS_BACKEND=dots-tts`). It is not
|
||||
> part of the default install.
|
||||
|
||||
## Platform support
|
||||
|
||||
- **Linux / macOS only.** dots.tts's upstream package declares Linux and
|
||||
macOS classifiers and has **no Windows install path**. On Windows the
|
||||
engine reports itself unavailable in **Settings → Engines** with a clear
|
||||
reason — run OmniVoice under WSL2 or use a Linux/macOS host.
|
||||
- **No MPS.** Upstream device selection is CUDA-or-CPU with no Metal branch,
|
||||
so on Apple Silicon the official package runs on **CPU** (slow but
|
||||
correct). A faster Apple-Silicon path exists only via community MLX ports,
|
||||
which OmniVoice does not auto-wire.
|
||||
- **VRAM:** ~9 GB checkpoint; a 12–16 GB CUDA GPU is the realistic target.
|
||||
|
||||
## Install
|
||||
|
||||
dots.tts is **not** bundled (large checkpoint + conflicting `transformers`).
|
||||
|
||||
1. Clone the dots.tts repo on disk:
|
||||
|
||||
```bash
|
||||
git clone https://github.com/rednote-hilab/dots.tts.git
|
||||
```
|
||||
|
||||
2. Install the editable package into a fresh venv with the upstream
|
||||
constraints. Use `uv pip install -e . -c constraints/recommended.txt` —
|
||||
**never** `uv sync --all-extras`, which would overwrite OmniVoice's lock
|
||||
file with `transformers==4.57` and break the parent process:
|
||||
|
||||
```bash
|
||||
cd dots.tts
|
||||
uv venv .venv
|
||||
uv pip install -e . -c constraints/recommended.txt
|
||||
```
|
||||
|
||||
3. The ~9 GB checkpoint downloads from HuggingFace on first synthesize. The
|
||||
parent forwards `HF_HOME` / `HF_HUB_CACHE` to the sidecar so the cache is
|
||||
shared with the rest of OmniVoice's downloads.
|
||||
|
||||
4. Set `OMNIVOICE_DOTS_TTS_DIR` to the repo root (the directory that
|
||||
contains `pyproject.toml` and `constraints/`):
|
||||
|
||||
```bash
|
||||
# macOS / Linux
|
||||
echo 'export OMNIVOICE_DOTS_TTS_DIR=$HOME/code/dots.tts' >> ~/.zshrc
|
||||
source ~/.zshrc
|
||||
```
|
||||
|
||||
5. Restart OmniVoice. dots.tts appears in **Settings → Engines** with
|
||||
`available: true` and `isolation_mode: subprocess`.
|
||||
|
||||
## Venv resolution order
|
||||
|
||||
OmniVoice probes for a usable dots.tts Python interpreter in this priority
|
||||
order (see `backend/engines/dots_tts/bootstrap.py`):
|
||||
|
||||
1. **`${OMNIVOICE_DOTS_TTS_DIR}/.venv/`** — your existing clone's venv.
|
||||
2. **`backend/engines/dots_tts/.venv/`** — OmniVoice's own venv, created on
|
||||
demand by step 3.
|
||||
3. **Lazy bootstrap** — `uv venv` then `uv pip install -e <clone> -c
|
||||
<clone>/constraints/recommended.txt`. Requires `OMNIVOICE_DOTS_TTS_DIR`.
|
||||
|
||||
## Voice cloning
|
||||
|
||||
For best fidelity ("continuation cloning"), pass **both** a reference clip
|
||||
(`ref_audio`) and its exact transcript (`ref_text`). A reference clip alone
|
||||
does x-vector-only cloning. Keep the reference ~10 s. Upstream requires the
|
||||
reference audio whenever a transcript is given, so OmniVoice drops a stray
|
||||
`ref_text` that arrives without `ref_audio`.
|
||||
|
||||
## Optional env knobs
|
||||
|
||||
| Variable | Default | Purpose |
|
||||
|----------|---------|---------|
|
||||
| `OMNIVOICE_DOTS_TTS_DIR` | — | Path to the dots.tts clone (required). |
|
||||
| `OMNIVOICE_DOTS_TTS_MODEL` | `rednote-hilab/dots.tts-soar` | Checkpoint override (`-base`, `-soar`, `-mf`). |
|
||||
| `OMNIVOICE_DOTS_TTS_PRECISION` | `bfloat16` (CUDA) / `float32` (CPU) | Inference precision. |
|
||||
| `OMNIVOICE_DOTS_TTS_OPTIMIZE` | `0` | `1` enables `torch.compile` (slower first call, faster after). |
|
||||
|
||||
> Using the `dots.tts-mf` (MeanFlow-distilled) checkpoint? It's tuned for
|
||||
> **4** flow-matching steps — pass `num_step=4`.
|
||||
|
||||
## Common errors
|
||||
|
||||
### `dots.tts is not supported on Windows ...`
|
||||
|
||||
Upstream is Linux/macOS only. Use WSL2 or a Linux/macOS host.
|
||||
|
||||
### `dots.tts venv not found. Set OMNIVOICE_DOTS_TTS_DIR ...`
|
||||
|
||||
You haven't pointed OmniVoice at a dots.tts clone yet. Follow **Install**.
|
||||
|
||||
## License
|
||||
|
||||
Apache-2.0 (code and checkpoints). See the upstream
|
||||
[README](https://github.com/rednote-hilab/dots.tts/blob/main/README.md).
|
||||
|
||||
---
|
||||
|
||||
dots.tts runs in a dedicated sidecar venv (it pins `transformers==4.57`,
|
||||
which conflicts with the parent's `transformers>=5.3`). For why that adds
|
||||
disk and how uv keeps the cost down, see
|
||||
[Engine venvs & disk usage](disk-usage.md).
|
||||
@@ -0,0 +1,131 @@
|
||||
# OmniVoice Studio — MOSS-TTS-v1.5 Engine
|
||||
|
||||
MOSS-TTS-v1.5 (OpenMOSS) is an **8B** flagship zero-shot TTS — a Qwen3-8B
|
||||
language backbone plus a 1.6B audio codec. It covers **31 languages**, does
|
||||
zero-shot voice cloning, token-level duration control and inline
|
||||
`[pause Ns]` markers. Released under **Apache-2.0** (code + weights).
|
||||
|
||||
It runs in its own subprocess **and its own Python venv** with
|
||||
`transformers==5.0.0`, isolated from the OmniVoice parent process which
|
||||
pins `transformers>=5.3`. This is the same isolation primitive used by
|
||||
[IndexTTS-2](indextts.md): the two `transformers` pins cannot share one
|
||||
interpreter, so MOSS runs behind
|
||||
`backend/services/subprocess_backend.py::SubprocessBackend`.
|
||||
|
||||
> **Opt-in, and never a default.** MOSS-TTS-v1.5 is selected explicitly in
|
||||
> **Settings → Engines** (or `OMNIVOICE_TTS_BACKEND=moss-tts-v15`). It is
|
||||
> not part of the default install and does not change OmniVoice's
|
||||
> out-of-the-box behaviour on any platform.
|
||||
|
||||
## Hardware
|
||||
|
||||
- **VRAM/RAM:** an 8B model. The upstream llama.cpp pipeline fits the 8B on
|
||||
8 GB GPUs when quantized; the bf16 Transformers path used here is ~16 GB
|
||||
of weights, so a 16 GB+ GPU is the realistic CUDA target. It also runs on
|
||||
**CPU** (fp32) — correct but slow.
|
||||
- **Device:** CUDA when present, else CPU. **There is no MPS path** —
|
||||
upstream documents only CUDA/CPU and the custom modelling code is
|
||||
untested on Apple Silicon, so OmniVoice never routes MOSS to MPS. On a
|
||||
Mac it runs on CPU.
|
||||
|
||||
## Install
|
||||
|
||||
MOSS-TTS-v1.5 is **not** bundled (the model is large and the package pins a
|
||||
conflicting `transformers`). OmniVoice ships a sidecar runner that loads it
|
||||
into an isolated venv on demand.
|
||||
|
||||
1. Clone the MOSS-TTS repo on disk:
|
||||
|
||||
```bash
|
||||
git clone https://github.com/OpenMOSS/MOSS-TTS.git
|
||||
```
|
||||
|
||||
2. Install the editable package into a fresh venv. Use
|
||||
`uv pip install -e ".[torch-runtime]"` — **never** `uv sync --all-extras`,
|
||||
which would overwrite OmniVoice's lock file with `transformers==5.0` and
|
||||
break the parent process. The `torch-runtime` extra is CUDA (`+cu128`):
|
||||
|
||||
```bash
|
||||
cd MOSS-TTS
|
||||
uv venv .venv
|
||||
uv pip install -e ".[torch-runtime]"
|
||||
```
|
||||
|
||||
On a **non-CUDA / CPU host** (e.g. Apple Silicon), install plain
|
||||
`torch`/`torchaudio`/`transformers==5.0.0` into the venv instead of the
|
||||
`+cu128` extra (the auto-bootstrap below only targets CUDA hosts).
|
||||
|
||||
3. The ~16 GB weights download from HuggingFace on first synthesize. The
|
||||
parent forwards `HF_HOME` / `HF_HUB_CACHE` to the sidecar so the cache is
|
||||
shared with the rest of OmniVoice's downloads.
|
||||
|
||||
4. Set `OMNIVOICE_MOSS_TTS_V15_DIR` to the repo root (the directory that
|
||||
contains `pyproject.toml`):
|
||||
|
||||
```bash
|
||||
# macOS / Linux
|
||||
echo 'export OMNIVOICE_MOSS_TTS_V15_DIR=$HOME/code/MOSS-TTS' >> ~/.zshrc
|
||||
source ~/.zshrc
|
||||
```
|
||||
|
||||
```powershell
|
||||
# Windows PowerShell
|
||||
[Environment]::SetEnvironmentVariable("OMNIVOICE_MOSS_TTS_V15_DIR","$env:USERPROFILE\code\MOSS-TTS","User")
|
||||
```
|
||||
|
||||
5. Restart OmniVoice. MOSS-TTS-v1.5 appears in **Settings → Engines** with
|
||||
`available: true` and `isolation_mode: subprocess`.
|
||||
|
||||
## Venv resolution order
|
||||
|
||||
OmniVoice probes for a usable MOSS Python interpreter in this priority
|
||||
order (see `backend/engines/moss_tts_v15/bootstrap.py`):
|
||||
|
||||
1. **`${OMNIVOICE_MOSS_TTS_V15_DIR}/.venv/`** — your existing clone's venv.
|
||||
Highest priority, so a power user who already set MOSS up gets zero
|
||||
re-install.
|
||||
2. **`backend/engines/moss_tts_v15/.venv/`** — OmniVoice's own venv,
|
||||
created on demand by step 3.
|
||||
3. **Lazy bootstrap** — if neither venv exists, OmniVoice runs `uv venv`
|
||||
then `uv pip install --python <python> -e "${DIR}[torch-runtime]"`.
|
||||
Requires `OMNIVOICE_MOSS_TTS_V15_DIR`; raises a clear error otherwise.
|
||||
On a non-CUDA host the `+cu128` extra cannot resolve — set the venv up
|
||||
manually per step 2.
|
||||
|
||||
## Voice cloning
|
||||
|
||||
Pass a reference clip as `ref_audio`. MOSS's zero-shot clone mode needs only
|
||||
the audio (no transcript). Without a reference, MOSS synthesizes in its own
|
||||
default voice. `duration` (seconds) maps to MOSS's `tokens` argument at
|
||||
~12.5 tokens/second.
|
||||
|
||||
## Optional env knobs
|
||||
|
||||
| Variable | Default | Purpose |
|
||||
|----------|---------|---------|
|
||||
| `OMNIVOICE_MOSS_TTS_V15_DIR` | — | Path to the MOSS-TTS clone (required). |
|
||||
| `OMNIVOICE_MOSS_TTS_V15_MODEL` | `OpenMOSS-Team/MOSS-TTS-v1.5` | HF repo id override (mirror / air-gapped). |
|
||||
| `OMNIVOICE_MOSS_TTS_V15_ATTN` | `sdpa` | Attention impl; set `flash_attention_2` on Ampere+ CUDA with `flash-attn` installed. |
|
||||
|
||||
## Common errors
|
||||
|
||||
### `MOSS-TTS-v1.5 venv not found. Set OMNIVOICE_MOSS_TTS_V15_DIR ...`
|
||||
|
||||
You haven't pointed OmniVoice at a MOSS-TTS clone yet. Follow **Install**.
|
||||
|
||||
### `uv pip install -e failed ... '[torch-runtime]' extra (cu128) cannot resolve`
|
||||
|
||||
You're on a non-CUDA host. The upstream `torch-runtime` extra is CUDA-only;
|
||||
set up the venv manually with plain `torch`/`transformers==5.0.0` (step 2).
|
||||
|
||||
## License
|
||||
|
||||
Apache-2.0 (code and weights) — no acceptance gate. See the upstream
|
||||
[README](https://github.com/OpenMOSS/MOSS-TTS/blob/main/README.md).
|
||||
|
||||
---
|
||||
|
||||
MOSS-TTS-v1.5 runs in a dedicated sidecar venv (it pins `transformers==5.0`,
|
||||
which conflicts with the parent's `transformers>=5.3`). For why that adds
|
||||
disk and how uv keeps the cost down, see
|
||||
[Engine venvs & disk usage](disk-usage.md).
|
||||
@@ -46,6 +46,12 @@ tts_engines:
|
||||
doc: docs/engines/indextts.md
|
||||
- id: omnivoice-gguf
|
||||
- id: supertonic3
|
||||
- id: moss-tts-v15
|
||||
readme: "**MOSS-TTS-v1.5**"
|
||||
doc: docs/engines/moss-tts-v15.md
|
||||
- id: dots-tts
|
||||
readme: "**dots.tts**"
|
||||
doc: docs/engines/dots-tts.md
|
||||
|
||||
# Same contract against backend/services/asr_backend.py _REGISTRY.
|
||||
asr_engines:
|
||||
@@ -63,6 +69,8 @@ asr_engines:
|
||||
readme: Moonshine
|
||||
- id: funasr
|
||||
readme: FunASR
|
||||
- id: sherpa-onnx-asr
|
||||
readme: "**sherpa-onnx** (live dictation)"
|
||||
|
||||
# Doc files that must exist (the install path users are sent to).
|
||||
docs:
|
||||
|
||||
@@ -11,6 +11,8 @@ working OmniVoice Studio install on a Debian / Ubuntu / Fedora / Arch host.
|
||||
`sudo dnf install python3.11` on Fedora, or already installed on Arch.
|
||||
- **Bun** — `curl -fsSL https://bun.sh/install | bash`.
|
||||
- **FFmpeg** — `sudo apt install ffmpeg` (Debian/Ubuntu), `sudo dnf install ffmpeg-free` (Fedora), or `sudo pacman -S ffmpeg` (Arch).
|
||||
- **Rust / Cargo** (required for building from source only) — `curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh` or via your package manager (e.g., `sudo apt install rustc cargo`).
|
||||
If you use rustup, reopen the shell or source `"$HOME/.cargo/env"` before running `bun run desktop-prod`.
|
||||
- **GTK/WebKit deps** for the Tauri shell:
|
||||
|
||||
```bash
|
||||
|
||||
@@ -15,6 +15,8 @@ working OmniVoice Studio install on macOS (Apple Silicon or Intel).
|
||||
- **Bun** — `curl -fsSL https://bun.sh/install | bash`.
|
||||
- **Xcode Command Line Tools** — `xcode-select --install`.
|
||||
- **FFmpeg** (used by the dubbing + capture pipelines) — `brew install ffmpeg`.
|
||||
- **Rust / Cargo** (required for building from source only) — `curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh` or `brew install rust`.
|
||||
If you use rustup, reopen the terminal or source `"$HOME/.cargo/env"` before running `bun run desktop-prod`.
|
||||
|
||||
Optional but recommended:
|
||||
|
||||
|
||||
@@ -61,6 +61,29 @@ fresh install heals itself.
|
||||
**Linked issues:** [#58](https://github.com/debpalash/OmniVoice-Studio/issues/58),
|
||||
[#248](https://github.com/debpalash/OmniVoice-Studio/issues/248)
|
||||
|
||||
### 1a. Model load fails: `[Errno 2] No such file or directory: '…/transformers/…/modeling_*.py'`
|
||||
|
||||
**Symptom:** the System Check / model load fails with e.g.
|
||||
`[Errno 2] No such file or directory:
|
||||
'…/site-packages/transformers/models/qwen3/modeling_qwen3.py'`.
|
||||
|
||||
**Cause:** same class as §1 — a **corrupted/incomplete `transformers` install**.
|
||||
A model load lazily resolves a module file that's **missing from `site-packages`**
|
||||
(an interrupted `uv sync`, antivirus quarantine, or a partial update). The
|
||||
package's metadata is intact, so a plain install no-ops and never restores the
|
||||
file. Restarting does **not** help (the file is still gone).
|
||||
|
||||
**Fix:** force-reinstall transformers in the backend venv, then restart:
|
||||
|
||||
```
|
||||
uv pip install --reinstall transformers
|
||||
```
|
||||
|
||||
Or, as a quick workaround, switch ASR to **faster-whisper** in
|
||||
**Settings → Models**. If it recurs, add the backend **`.venv`** to your
|
||||
antivirus exclusions (see §1). Newer builds classify this error and show the
|
||||
reinstall hint directly instead of a bare path + "try restarting".
|
||||
|
||||
## 2. HF 401 / pyannote license not accepted
|
||||
|
||||
**Symptom:** dubbing fails with `HfHubHTTPError: 401 Client Error: Unauthorized
|
||||
@@ -192,6 +215,165 @@ for the dedicated CosyVoice path.
|
||||
|
||||
**Linked issue:** [#55](https://github.com/debpalash/OmniVoice-Studio/issues/55)
|
||||
|
||||
## 12. CUDA PyTorch wheel download fails on first run
|
||||
|
||||
**Symptom:** first-run setup stops at **Installing dependencies** with a failure
|
||||
that mentions `torch` and a `download.pytorch.org` (or `download-r2.pytorch.org`)
|
||||
URL — e.g. `Failed to download torch==2.8.0+cu128 …win_amd64.whl`. The app then
|
||||
won't launch.
|
||||
|
||||
**Cause:** on Windows/Linux NVIDIA machines, OmniVoice installs the CUDA PyTorch
|
||||
build (`torch` + `torchaudio`) from PyTorch's own index. That CUDA wheel is
|
||||
large (~2.5 GB), so a flaky or restricted network drops it partway. This is a
|
||||
download/network problem, **not** a bug in OmniVoice — but the CUDA wheels come
|
||||
from a *named, explicit* index that a PyPI mirror (`UV_DEFAULT_INDEX`) cannot
|
||||
redirect, so the generic mirror trick doesn't help here.
|
||||
|
||||
**Fix, in order:**
|
||||
|
||||
1. **Clean & Retry.** Large downloads frequently succeed on a second attempt —
|
||||
OmniVoice already retries each request 5× with long timeouts, and a fresh
|
||||
attempt restarts cleanly.
|
||||
2. **Use a VPN** if your network throttles or blocks the PyTorch CDN.
|
||||
3. **Provide the wheels manually (offline path).** Download the two wheels that
|
||||
match your machine from a source you *can* reach (the official
|
||||
[pytorch.org](https://pytorch.org/get-started/locally/) wheel index or a
|
||||
regional mirror), then drop them in the wheel folder and **Clean & Retry** —
|
||||
OmniVoice will install from your local copies instead of the network:
|
||||
- Folder: **`<env dir>/wheels`** (the exact path is printed in the error
|
||||
message and in the setup log; `<env dir>` is your chosen install/storage
|
||||
location).
|
||||
- Files: the `torch` **and** `torchaudio` wheels for your exact Python/OS/CUDA
|
||||
— e.g. `torch-2.8.0+cu128-cp311-cp311-win_amd64.whl` and the matching
|
||||
`torchaudio-2.8.0+cu128-cp311-cp311-win_amd64.whl`. They must match the
|
||||
pinned versions (shown in the failing URL).
|
||||
- On retry, OmniVoice re-resolves the install using those local wheels; the
|
||||
rest of the (small) dependencies still come from PyPI/your mirror.
|
||||
|
||||
If you don't have an NVIDIA GPU, you don't need the CUDA build at all — a CPU /
|
||||
Apple-Silicon install skips this index entirely.
|
||||
|
||||
**Linked issue:** [#569](https://github.com/debpalash/OmniVoice-Studio/issues/569)
|
||||
|
||||
## 13. Stuck on the download page / incomplete model cache ("only `refs/`")
|
||||
|
||||
**Symptom:** the setup screen never finishes the model download and you can't
|
||||
reach the main app. Looking in the HF cache, a model folder
|
||||
(`models--k2-fsa--OmniVoice`, `models--Systran--faster-whisper-large-v3`) has
|
||||
`refs/` and maybe `config.json` but **no weight files** (`blobs/` empty or tiny).
|
||||
|
||||
**Cause:** the download started but the large weight shards never finished —
|
||||
almost always the connection **dropping, throttling, or being blocked** mid-pull
|
||||
(corporate/school proxy, VPN, antivirus quarantining the multi-GB file, or a
|
||||
region where `huggingface.co` is slow/blocked). The app retries and verifies
|
||||
weights, but a connection that *trickles* rather than dies can stall for a long
|
||||
time.
|
||||
|
||||
**Fix — force a clean re-download:**
|
||||
|
||||
1. **Fully quit OmniVoice.** Check Task Manager (Windows) / Activity Monitor
|
||||
(macOS) and end any leftover `omnivoice` / `python` process — a half-running
|
||||
one keeps the cache locked.
|
||||
2. **Delete the incomplete model folder(s) entirely** from the HF cache (the
|
||||
whole `models--…` folder, not just `refs/`). Leave other models alone:
|
||||
- `models--k2-fsa--OmniVoice`
|
||||
- `models--Systran--faster-whisper-large-v3`
|
||||
3. **Relaunch** — the download page re-pulls from scratch.
|
||||
|
||||
**If it stalls again at the same spot**, the download is being blocked — try, in
|
||||
order:
|
||||
|
||||
- **Antivirus/firewall** — temporarily disable it for the download (large model
|
||||
files are a common false-positive quarantine), then re-enable.
|
||||
- **Connection** — use a stable, direct connection; pause any VPN; avoid
|
||||
corporate/school networks.
|
||||
- **Region mirror** — if `huggingface.co` is slow/blocked where you are, set a
|
||||
mirror **before** launching and relaunch:
|
||||
- macOS/Linux: `export HF_ENDPOINT=https://hf-mirror.com`
|
||||
- Windows (PowerShell): `[Environment]::SetEnvironmentVariable("HF_ENDPOINT","https://hf-mirror.com","User")`
|
||||
|
||||
**Manual fallback** (if downloads keep failing), pull the weights yourself into
|
||||
the same cache, then relaunch:
|
||||
|
||||
```bash
|
||||
pip install -U "huggingface_hub[cli]"
|
||||
huggingface-cli download k2-fsa/OmniVoice
|
||||
huggingface-cli download Systran/faster-whisper-large-v3
|
||||
```
|
||||
|
||||
(If OmniVoice uses a custom models directory, set `HF_HOME` to it first so the
|
||||
files land where the app looks.)
|
||||
|
||||
> Newer builds detect an incomplete cache and re-offer the download instead of
|
||||
> stranding you on this page — update once the fix is in your channel.
|
||||
|
||||
**Linked issue:** [#622](https://github.com/debpalash/OmniVoice-Studio/issues/622)
|
||||
|
||||
## 14. "Can't reach the local backend" *during* generation / transcription / dubbing
|
||||
|
||||
**Symptom:** the app worked at startup (you reached the main menu and the model
|
||||
loaded), but the moment you **generate audio, dub a video, transcribe, or
|
||||
dictate**, it spins for a long time and then shows **"Can't reach the local
|
||||
backend."** The backend log ends right after a line like `whisperx transcribing
|
||||
…tmpXXXX.wav` (or a generate) with nothing after it — i.e. the backend is
|
||||
**alive**, the GPU *job* is what stalled.
|
||||
|
||||
**Cause:** this is **not** a connection, download, or "network mirror" problem —
|
||||
the backend started fine. A GPU job (a **generate** on the TTS model, or an ASR
|
||||
transcribe with WhisperX/faster-whisper **large-v3**) is too heavy for the
|
||||
available compute and runs for minutes; because it wedges its GPU-pool worker,
|
||||
every *other* request — including the next generate and the health check — is
|
||||
starved, which the UI surfaces as an unreachable backend. The usual trigger is
|
||||
**VRAM starvation on NVIDIA**: models contend for memory on an 8 GB-class GPU
|
||||
(the log shows e.g. `GPU pool sized … 7.0 GB free`). CPU-only machines hit the
|
||||
same wall on long clips. This is the same root cause whether the last thing you
|
||||
did was `generate:start (audio)`, a dub, or a dictation.
|
||||
|
||||
> There is **no "Network → Restricted/Global mirror" toggle** in Settings — that
|
||||
> control (the footer/Sharing **Network** button) is for **LAN sharing**, not
|
||||
> downloads. If someone pointed you there for this error, it was the wrong knob.
|
||||
|
||||
**Fix — reduce ASR load (any one of these):**
|
||||
|
||||
1. **Pick a smaller ASR model / engine** in **Settings → Models** — e.g.
|
||||
faster-whisper **medium** or **small**, instead of large-v3. Biggest win on
|
||||
low-VRAM GPUs.
|
||||
2. **Free VRAM**: **Flush the TTS model** before dubbing so ASR isn't competing
|
||||
for memory, or
|
||||
3. **Run ASR on CPU** (slower but reliable) if your GPU is small.
|
||||
4. **Test with a 10-second clip** first — if that returns quickly, it confirms a
|
||||
compute/VRAM limit rather than a true hang.
|
||||
|
||||
Newer builds **bound** every GPU job — whole-file transcription **and** TTS
|
||||
generation: instead of hanging forever and starving the backend, a wedged job now
|
||||
fails after a timeout with this exact guidance, and the worker pool is reset so
|
||||
capacity is restored automatically (no app restart needed). Tune the bounds with
|
||||
`OMNIVOICE_ASR_TRANSCRIBE_TIMEOUT_S` (transcription) and
|
||||
`OMNIVOICE_GENERATE_TIMEOUT_S` (generation) — both in seconds, default 300.
|
||||
**Raise** them for very long single files/generations, **lower** them to fail
|
||||
faster on a small machine.
|
||||
|
||||
## Dub: "translation engine needs the optional … package"
|
||||
|
||||
**Symptom:** in the Dub tab, translating fails with e.g. *"The 'google'
|
||||
translation engine needs the optional `deep_translator` Python package, which
|
||||
isn't installed in this backend."*
|
||||
|
||||
**Cause:** the online translation engines (Google / DeepL / Microsoft / MyMemory
|
||||
via `deep_translator`, and the LLM provider via `openai`) are **optional** and
|
||||
not bundled. Only **Argos** and **NLLB** work out of the box.
|
||||
|
||||
**Fix:**
|
||||
- **From-source / Docker install:** click the highlighted **Install** button next
|
||||
to the *Engine* label in the Dub tab (or run `uv pip install deep_translator`
|
||||
in the backend venv) and restart the backend.
|
||||
- **Packaged installer build:** in-app install is disabled (read-only signed
|
||||
environment). Click the highlighted button to open the popover and **Switch to
|
||||
Argos (bundled, offline)** — or copy the command to run it in a from-source
|
||||
checkout.
|
||||
|
||||
Full guide: [dubbing/translation-engines.md](../dubbing/translation-engines.md#installing-optional-translation-engines-from-source-vs-packaged-build).
|
||||
|
||||
## First-run setup fails on a restricted network (GitHub/PyPI blocked)
|
||||
|
||||
On networks that block or can't resolve **GitHub**, the first-run bootstrap may
|
||||
|
||||
@@ -18,6 +18,8 @@ working OmniVoice Studio install on Windows 10 / 11 (x64).
|
||||
You need it for `git clone` anyway, and it includes **Git Bash**, which
|
||||
`bun run desktop-prod` uses to run its build-and-launch script. Without it,
|
||||
`desktop-prod` stops with an error telling you to install it.
|
||||
- **Rust / Cargo** (required for building from source only) — `winget install Rust.Rustup` or download `rustup-init.exe` from [rustup.rs](https://rustup.rs/).
|
||||
After installing Rustup, close and reopen PowerShell before running `bun run desktop-prod`.
|
||||
|
||||
## Install (from source)
|
||||
|
||||
|
||||
@@ -0,0 +1,93 @@
|
||||
# Maintenance Refactor — `frontend/src/pages` Modularization
|
||||
|
||||
**Status:** Plan (not yet executed) · **Drafted:** 2026-06-30 · **Type:** Pure mechanical refactor, no behavior change
|
||||
|
||||
## Why
|
||||
|
||||
`frontend/src/pages/` has grown a few files large enough that any edit reloads the
|
||||
whole thing into context and risks unrelated breakage. Editing one Settings panel
|
||||
should touch a ~150-line file, not a 1969-line one. This both improves
|
||||
maintainability and cuts token cost per edit.
|
||||
|
||||
The fix is **not** a new architecture — `components/settings/` already proves the
|
||||
target pattern (13 extracted `*Panel.jsx`, each with co-located `.css`/`.test.jsx`,
|
||||
plus a shared `primitives/` folder). This refactor **finishes a migration that
|
||||
stalled**, then locks it in so files can't silently regrow.
|
||||
|
||||
## Current state (measured 2026-06-30)
|
||||
|
||||
| File | Lines | Notes |
|
||||
|------|------:|-------|
|
||||
| `pages/Settings.jsx` | 1969 | Still inline: `ModelStoreTab` (~790L), `Settings` orchestrator (~600L), `GeneralTab`, `EnginesTab`, `HotkeyTab`, `CredentialsTab`, plus `Row`/`fmtBytes`/`orgColor` helpers |
|
||||
| `pages/DubTab.jsx` | 1592 | One mega-component + inline `DubFailureNotice`, `DubPipelineStepper`, `PrepOverlay`, `TranscribeOverlay`, `FooterBtn` |
|
||||
| `pages/CloneDesignTab.jsx` | 837 | |
|
||||
| `pages/VoiceGallery.jsx` | 768 | |
|
||||
| `pages/VoiceProfile.jsx` | 515 | |
|
||||
| `pages/AudiobookTab.jsx` | 402 | within target after Phase 3 sweep |
|
||||
| everything else | <340 | within target |
|
||||
|
||||
Already-extracted, do **not** touch (reference pattern): `components/settings/*Panel.jsx`,
|
||||
`components/settings/primitives/`.
|
||||
|
||||
## The gold standard (proposed)
|
||||
|
||||
1. **Size caps:** soft **300 lines**, hard **500 lines** per `.jsx`/`.css`. Over 500 must split.
|
||||
2. **Pages are thin orchestrators:** a page = layout + routing + state wiring that
|
||||
composes feature components. No inline sub-component over ~50 lines.
|
||||
3. **One component per file**, co-located `Foo.jsx` + `Foo.css` + `Foo.test.jsx`,
|
||||
grouped in a per-page feature folder:
|
||||
- `components/settings/` (exists)
|
||||
- `components/dub/` (new)
|
||||
- `components/clone/` (new)
|
||||
- `components/gallery/` (new)
|
||||
4. **Shared bits → `primitives/`** in the feature folder (settings already has this).
|
||||
5. **Enforce with ESLint `max-lines`** — **warn-only first** so it never breaks CI
|
||||
(respects the "keep main green" rule), upgrade to error after the backlog clears.
|
||||
|
||||
## Phases (each = one mergeable, CI-green PR)
|
||||
|
||||
### Phase 0 — Standard + guardrail
|
||||
- Add the size/structure rule to `CONTRIBUTING.md` (required by the docs-sync rule anyway).
|
||||
- Add ESLint `max-lines: ['warn', { max: 500, skipBlankLines: true, skipComments: true }]`.
|
||||
- No code moves. Smallest possible PR; establishes the contract.
|
||||
|
||||
### Phase 1 — `Settings.jsx` (biggest win: 1969 → ~300L)
|
||||
Extract into `components/settings/`, mirroring existing panel naming:
|
||||
| Extract | Current lines (approx) | New file |
|
||||
|---------|------------------------|----------|
|
||||
| `ModelStoreTab` (+ `Row`, `fmtBytes`, `orgColor`, `MODEL_ROLE_*`) | 229–1021 | `ModelStoreTab.jsx` (likely split further: table vs. matrix vs. row) |
|
||||
| `GeneralTab` | 80–201 | `GeneralTab.jsx` |
|
||||
| `EnginesTab` | 1022–1072 | `EnginesTab.jsx` |
|
||||
| `HotkeyTab` (+ `CREDENTIAL_FIELDS`, `keyEventToAccelerator`) | 1693–1870 | `HotkeyTab.jsx` |
|
||||
| `CredentialsTab` | 1871–1969 | `CredentialsTab.jsx` |
|
||||
`Settings.jsx` keeps only: imports, `TAB_DEFS`/`LOG_SOURCE_DEFS`, the `Settings`
|
||||
default export (tab router + shared state), and `askConfirm`.
|
||||
|
||||
### Phase 2 — `DubTab.jsx` (1592 → orchestrator + `components/dub/`)
|
||||
Extract `DubFailureNotice`, `DubPipelineStepper`, `PrepOverlay`,
|
||||
`TranscribeOverlay`, `FooterBtn`, and the large render sub-sections into
|
||||
`components/dub/`. `DubTab.jsx` retains the pipeline state machine + composition.
|
||||
|
||||
### Phase 3 — `CloneDesignTab`, `VoiceGallery`, `VoiceProfile`, `AudiobookTab`
|
||||
Same treatment into `components/clone/` and `components/gallery/`. Smaller, lower risk.
|
||||
|
||||
## Constraints honored
|
||||
- **No behavior change** — pure moves; diff is verifiable by "app renders
|
||||
identically + existing tests pass." Each panel that has a test keeps it.
|
||||
- **Keep main green** — ESLint rule is warn-only; each phase is independently CI-green.
|
||||
- **Docs-sync** — Phase 0 lands the `CONTRIBUTING.md` change in the same PR as the rule.
|
||||
- **No versioning impact** — frontend-only refactor; no `package.json` version bump,
|
||||
no lockfile/dep change, no Docker/Tauri/Python surface touched.
|
||||
|
||||
## Verification per phase
|
||||
1. `bun run build` (or the project's typecheck/lint) passes.
|
||||
2. Existing `components/settings/*.test.jsx` (and any new co-located tests) pass.
|
||||
3. Manual smoke: open Settings → every tab renders; open Dub → pipeline renders.
|
||||
4. `git diff --stat` shows only moves (line counts shift between files, net ~0 logic change).
|
||||
|
||||
## Out of scope (explicitly)
|
||||
- No redesign of the Settings *UI* itself (the "unorganised" look) — that's a separate
|
||||
visual-polish task; this refactor only restructures the *code*. Flag if you want
|
||||
that bundled.
|
||||
- No conversion of `.jsx` → `.tsx` (pages are currently JS; TS migration is a
|
||||
different decision).
|
||||
@@ -0,0 +1,126 @@
|
||||
# Migration — OmniVoice `ui/` Primitives → shadcn/ui (Tailwind v4)
|
||||
|
||||
**Status:** Foundation landed · P1 form primitives landed (Input/Select/Textarea/Slider backed by shadcn; Table foundation added) · **Drafted:** 2026-06-30 · **Type:** Incremental component-library adoption, no intended visual change
|
||||
**Owner stance:** wants a clean, conventional component base (shadcn) without re-skinning the app · **This plan's recommendation:** adopt shadcn *primitives* behind the existing prop APIs, themed by the OmniVoice palette via a token bridge; migrate in waves; never big-bang. See §6.
|
||||
|
||||
## Why
|
||||
|
||||
OmniVoice's `src/ui/` primitives (`Button.jsx`, `Input.jsx`, `Badge`, `Panel`, `Tabs`, …) are hand-rolled and token-faithful, but each one re-encodes variant logic, focus rings, and disabled states as long arbitrary-property Tailwind strings (see the `[transition:…]`/`[box-shadow:…]` blocks in `ui/Button.jsx`). shadcn/ui is the de-facto React primitive convention: `cva` variant maps, a `cn()` merge helper, Radix behavioural primitives (already a dependency), and a flat `components/ui/*` layout that `npx shadcn add` extends. Adopting it gives us a maintained, well-documented base and lets contributors paste canonical shadcn snippets that "just work."
|
||||
|
||||
The risk in adopting shadcn is that it ships its own **grayscale (`neutral`) palette**. Dropping stock shadcn in would repaint the app gray and break theme switching. This foundation solves that with a **token bridge** (§2) so shadcn components inherit the *existing* OmniVoice look — Gruvbox-pink default plus every `[data-theme]` — with zero per-component restyling.
|
||||
|
||||
This is **not** a redesign. The contract is the same as the CSS→Tailwind migration (`docs/css-to-tailwind-migration.md`): every step renders coherent with today's palette, and the visual-regression harness (`src/test/visual/`) is the gate that proves it.
|
||||
|
||||
## What landed in this foundation PR
|
||||
|
||||
- **shadcn init for Tailwind v4 + Vite + React 19.** `frontend/components.json` (style `new-york`, `rsc:false`, `tsx:true`, base color `neutral`, css-vars on), `src/lib/utils.ts` (`cn()` = `clsx` + `tailwind-merge`), and a `@/*` → `src/*` path alias in `vite.config.js` + `tsconfig.json` so `@/lib/utils` and future `npx shadcn add` resolve.
|
||||
- **The token bridge** in `src/index.css` (§2).
|
||||
- **Two proof components** — `src/components/ui/button.tsx`, `src/components/ui/input.tsx` (verbatim shadcn new-york, unmodified class strings) — rendered across the default / midnight / catppuccin themes in the visual harness with committed baselines.
|
||||
- **New deps:** `class-variance-authority`, `clsx`, `tailwind-merge`, `tw-animate-css`, `@radix-ui/react-slot` (root `bun.lock` regenerated; `bun install --frozen-lockfile` confirmed in sync for the Docker build).
|
||||
|
||||
No existing component was modified or replaced. The shadcn primitives are **not yet wired into the app** — they exist as the proven base for the waves below.
|
||||
|
||||
## 2. The token bridge (P0 — gates everything)
|
||||
|
||||
shadcn components reference a fixed semantic vocabulary (`bg-background`, `bg-primary`, `bg-card`, `text-muted-foreground`, `border-input`, `ring-ring`, `bg-destructive`, …). Those utilities only exist if Tailwind's theme defines `--color-background`, `--color-primary`, etc. OmniVoice's `@theme` block instead defines `--color-bg`, `--color-brand`, `--color-danger`, …. The bridge maps the former onto the latter.
|
||||
|
||||
It lives in `index.css` as a single `@theme inline` block. `inline` is load-bearing: it makes each generated utility emit `… { background-color: var(--color-bg) }` (a *live* reference) rather than baking in a static value, so runtime `[data-theme]` overrides flow through.
|
||||
|
||||
### Mapping table
|
||||
|
||||
| shadcn token (Tailwind key) | ← OmniVoice token | Notes |
|
||||
|---|---|---|
|
||||
| `--color-background` | `--color-bg` | app chrome bg |
|
||||
| `--color-foreground` | `--color-fg` | primary text |
|
||||
| `--color-card` / `--color-popover` | `--color-bg-elev-1` | raised surfaces |
|
||||
| `--color-card-foreground` / `--color-popover-foreground` | `--color-fg` | text on surfaces |
|
||||
| `--color-primary` | `--color-brand` | brand pink (theme-dependent) |
|
||||
| `--color-primary-foreground` | `--color-fg-inverse` | dark text on the brand fill (matches existing primary Button) |
|
||||
| `--color-secondary` / `--color-muted` | `--color-bg-elev-2` | subtle fills |
|
||||
| `--color-secondary-foreground` | `--color-fg` | — |
|
||||
| `--color-muted-foreground` | `--color-fg-muted` | muted/placeholder text |
|
||||
| `--color-accent` | *(reuses existing `--color-accent`)* | already in base `@theme` (amber) — `bg-accent` works as-is, not re-emitted |
|
||||
| `--color-accent-foreground` | `--color-fg-inverse` | dark text on the accent fill |
|
||||
| `--color-destructive` | `--color-danger` | error/destructive red |
|
||||
| `--color-destructive-foreground` | `--color-fg-inverse` | — |
|
||||
| `--color-border` | *(reuses existing `--color-border`)* | already in base `@theme` — `border-border` works as-is, not re-emitted |
|
||||
| `--color-input` | `--color-border` | input outline |
|
||||
| `--color-ring` | `--color-brand` | focus ring |
|
||||
| `--radius` | `--radius-lg` (6px) | shadcn base radius; the `--radius-*` scale itself is left untouched, so `rounded-md` keeps OmniVoice's 4px |
|
||||
|
||||
**Why theme switching keeps working with no themes.css changes.** Each bridged utility resolves to an OmniVoice `--color-*` token, and `ui/themes.css` already re-declares those tokens per `[data-theme]`. So switching to midnight changes `--color-brand` → purple and every shadcn `bg-primary` follows automatically. Verified in the harness: the same `button.tsx` renders brand-pink (default), purple (midnight), and lavender (catppuccin) with correct themed backgrounds and destructive reds. There is **no** separate shadcn token block to maintain per theme — a documented comment in `themes.css` records this.
|
||||
|
||||
**Why the existing tokens are safe.** `accent` and `border` already exist in the base `@theme`; re-emitting them in the bridge would be self-referential (`--color-accent: var(--color-accent)`) — a no-op at best, circular at worst — so they're intentionally omitted and reused as-is. The `--radius-*` scale is not touched, so every existing `rounded-sm/md/lg/xl` consumer is unchanged.
|
||||
|
||||
## 3. Where shadcn components live + aliases
|
||||
|
||||
| Concern | Decision |
|
||||
|---|---|
|
||||
| Location | `src/components/ui/*.tsx` (shadcn convention) — **distinct** from the existing `src/ui/*.jsx` primitives, so the two coexist during migration with no name clash |
|
||||
| `components` alias | `@/components` |
|
||||
| `ui` alias | `@/components/ui` |
|
||||
| `utils` alias | `@/lib/utils` |
|
||||
| `lib` / `hooks` | `@/lib` / `@/hooks` |
|
||||
| Path resolution | `@/*` → `src/*` in both `vite.config.js` (`resolve.alias`) and `tsconfig.json` (`paths`) |
|
||||
| Language | `.tsx` (the repo is mixed JS/JSX; new shadcn files are TS to match shadcn output and get prop typing) |
|
||||
|
||||
## 4. Primitive → shadcn mapping + prop-compatibility strategy
|
||||
|
||||
The existing `ui/*` primitives have call sites all over the app. The migration must **not** churn those call sites. Strategy: **keep the existing prop API; swap the implementation.** Each `ui/*.jsx` becomes a thin wrapper that maps its current props onto the shadcn component, so consumers (`<Button variant="subtle" size="sm">`, `<Input size="md">`) keep working unchanged.
|
||||
|
||||
| Existing `ui/` primitive | shadcn target | Prop bridge (existing → shadcn) |
|
||||
|---|---|---|
|
||||
| `Button.jsx` (`primary`/`subtle`/`ghost`/`danger`/`chip`/`preset`/`icon`, `size sm/md`, `loading`, `block`, `leading/trailing`) | `components/ui/button.tsx` | `primary→default`, `subtle→outline`, `ghost→ghost`, `danger→destructive`; `chip`/`preset`/`icon` stay as OmniVoice-only variants added to the `cva` map; `size md→default`; `loading` (spinner + disable), `block` (`w-full`), `leading/trailing` (slot children) wrapped in the JS layer |
|
||||
| `Input.jsx` `Input` | `components/ui/input.tsx` | `size sm/md/lg` → extend the shadcn `cva` (shadcn ships one size) or map to padding classes; `aria-invalid` already shared |
|
||||
| `Input.jsx` `Textarea` | `npx shadcn add textarea` | same `size` bridge |
|
||||
| `Input.jsx` `Select` | **Decided: kept NATIVE** + `ui-select` caret, wearing the shadcn shell (`inputBaseClass`). The Radix `select.tsx` was added to `components/ui/` for *new* call sites, but the primitive stays native because DubSegmentTable/CompareModal/GeneralTab depend on `onChange={(e) => …e.target.value}`, which Radix's value-only `onValueChange` would break |
|
||||
| `Input.jsx` `Field` | keep as a composition wrapper around `shadcn label` + control |
|
||||
| `Badge.jsx` | `npx shadcn add badge` | `tone` → `variant` map |
|
||||
| `Tabs.jsx` | `npx shadcn add tabs` (Radix; already a dep) | `items`/`value`/`onChange` → controlled `Tabs` |
|
||||
| `Progress.jsx` | `npx shadcn add progress` (Radix; already a dep) | `tone`/`size`/`shimmer` props preserved |
|
||||
| `Slider.jsx` | `npx shadcn add slider` (Radix; already a dep) | `value`/`onChange`/`label`/`showValue` preserved |
|
||||
| `Panel.jsx` | `npx shadcn add card` | glass variant keeps its `.css` residue (per the CSS→Tailwind plan) |
|
||||
| `Segmented.jsx` | `npx shadcn add toggle-group` (Radix; already a dep) | `items`/`value` preserved |
|
||||
|
||||
Anything shadcn doesn't cover 1:1 (the `chip`/`preset`/`icon` Button variants, the value-bubble Slider, the glass Panel) is added to the shadcn component's `cva`/markup rather than left behind — the wrapper is where OmniVoice-specific behaviour lives.
|
||||
|
||||
## 5. Phasing (staged waves)
|
||||
|
||||
Each wave = one or more independently shippable, CI-green PRs. Ordered so blast radius grows only as confidence does.
|
||||
|
||||
### P0 — Foundation (this PR)
|
||||
Init + token bridge + `cn()` + 2 proof components + baselines + this doc. No app component touched. **Done.**
|
||||
|
||||
### P1 — Primitives (swap implementation behind existing APIs)
|
||||
Convert `ui/Button.jsx` and `ui/Input.jsx` into thin wrappers over `components/ui/button.tsx` / `input.tsx`, porting the OmniVoice-only variants into the shadcn `cva`. Add a baseline-PR → conversion-PR pair per primitive (same recipe as the CSS→Tailwind plan §4). Then `Badge`, `Tabs`, `Progress`, `Slider`, `Segmented`, `Panel` one at a time.
|
||||
- **Success:** the visual harness shows the existing component specs (`Button`, `Input`, …) unchanged within tolerance after each swap; call sites untouched.
|
||||
|
||||
**Landed (form/data primitives).** `ui/Input.jsx` (`Input`/`Textarea`/`Select`/`Field`) and `ui/Slider.jsx` now wrap the shadcn components, exports + prop APIs unchanged:
|
||||
- `input.tsx` exports `inputBaseClass` (the shell, no behaviour change — `ShadcnInput` baseline byte-identical); new `textarea.tsx`, `select.tsx` (+`@radix-ui/react-select`), `slider.tsx`, `table.tsx` added to `components/ui/`.
|
||||
- `Input`/`Textarea` render the shadcn components; a small `fieldSizeVariants` `cva` (named palette utilities, tailwind-merge-clean) restores the OmniVoice padding-based `sm/md/lg` scale + filled `bg-bg-elev-2` over the shell.
|
||||
- `Select` stays native (see §4); `Slider` keeps its number-based `onChange` + label/value-bubble chrome around the shadcn `Slider`, tuned via the `data-slot` track/thumb selectors.
|
||||
- **`Table` deliberately NOT rerouted.** `ui/Table.jsx` is a flex-`<div>` chrome wrapper whose `.ui-table*`/`.segment-table` global classes (Table.css) are a SHARED CONTRACT used directly by ModelsTable / DubSegmentTable / EngineCompatibilityMatrix (virtualised react-window lists needing the div/flex layout, not a semantic `<table>`). The shadcn `table.tsx` is provided for new tabular data only; `Table.jsx` and its global classes are untouched. Its toolbar inherits the shadcn-backed `Input`/`Button` for free.
|
||||
- **Verified:** only the 3 `Input-*` baselines moved (palette-coherent across default/midnight/catppuccin); `Slider`/`Table` stayed within tolerance. `vitest` 641 green; `oxlint` 0 errors; `oxfmt --check` clean; `vite build` green; `bun install --frozen-lockfile` in sync.
|
||||
|
||||
### P2 — Usages (adopt shadcn directly where it's cleaner)
|
||||
New UI uses `@/components/ui/*` directly. High-traffic surfaces (Settings tabs, dialogs) migrate off the wrappers to native shadcn where the prop bridge adds no value. `npx shadcn add dialog/dropdown-menu/tooltip` to replace the hand-wrapped Radix usages (these need `tw-animate-css`, already imported).
|
||||
|
||||
### P3 — Delete CSS + shrink index.css
|
||||
As primitives move to shadcn, retire `Button.css`/`Input.css` residue and fold any remaining shadcn-shared tokens. Trim `index.css` globals that the shadcn components now own. Pairs naturally with the CSS→Tailwind P4.
|
||||
|
||||
## 6. Risk + effort (honest)
|
||||
|
||||
- **Biggest risk — variant fidelity.** OmniVoice's Button has 7 variants and bespoke focus/disabled treatments; shadcn ships 6 with different sizing. The wrapper approach contains this (map what maps, port the rest into `cva`), but P1 Button is the hardest single step and should ship behind a baseline diff that a human eyeballs. **Mitigation:** the visual harness already snapshots `Button`/`Input` across 3 themes; a swap that drifts fails the gate.
|
||||
- **`tw-animate-css` is unused today.** It's imported for the future Dialog/Dropdown waves; Button/Input don't need it. Low risk (additive utilities + keyframes), but it's a dep we carry before we use it. Acceptable for a foundation PR; revisit if P2 slips.
|
||||
- **knip flags the 3 new files as unused.** Expected — they're proof components only referenced by the visual harness, which knip doesn't treat as a production entry. knip is **not** a CI gate here, so this is informational; it resolves the moment P1 wires the wrappers.
|
||||
- **`@/*` alias is global.** Additive and standard; existing relative imports are unaffected. Confirmed clean against `typecheck:ci`, the Vite build, and the Docker frozen-lockfile install.
|
||||
- **Effort:** P1 ~1 day per primitive (baseline + swap + verify); ~1–1.5 weeks for the full primitive set. P2/P3 fold into the CSS→Tailwind timeline.
|
||||
|
||||
**Recommendation.** Adopt shadcn as the primitive base via wrappers, theme it through the bridge, and migrate in the wave order above — never replace en masse. The single most important guardrail is the visual baseline: do not swap a primitive's implementation without a before/after snapshot in all three harness themes.
|
||||
|
||||
## Constraints honored
|
||||
|
||||
- **Keep main green** — every wave is an independently CI-green PR. This foundation passes `vite build`, `typecheck:ci`, `oxlint` (0 errors), `oxfmt --check`, `vitest` (641), the full `test:visual` suite (48), and `bun install --frozen-lockfile`.
|
||||
- **Docs-sync** — this doc lands in the same PR as the foundation; `CONTRIBUTING.md`'s component-authoring guidance is updated when P1 makes shadcn the default primitive (no doc OmniVoice currently ships describes a *required* primitive source, so no stale doc results from P0).
|
||||
- **No versioning / Docker / Tauri / Python impact** — frontend-only; runtime deps added with the root `bun.lock` regenerated and the frozen-lockfile Docker path verified; no app-version bump (`package.json` version untouched).
|
||||
- **Local-first / cross-platform parity** — pure styling + presentational components; no behaviour, no platform divergence, no network.
|
||||
@@ -0,0 +1,176 @@
|
||||
# OmniVoice → True ElevenLabs Alternative — Spec Roadmap
|
||||
|
||||
This directory holds the implementation-ready specs that close the gap between
|
||||
OmniVoice Studio and ElevenLabs **without giving up what makes OmniVoice
|
||||
different**: fully local, no accounts, no API keys, no telemetry, 646 languages,
|
||||
cross-platform. The thesis is *counter-positioning*, not feature-cloning — we
|
||||
match the capabilities creators actually feel, and we win on "your voice never
|
||||
leaves your machine."
|
||||
|
||||
## The thesis
|
||||
|
||||
ElevenLabs' moat is **perceived voice quality + expressive control**, and its
|
||||
2025-26 expansion is **voice agents**. Everything else (library, studio editor,
|
||||
dubbing depth, API) is table-stakes polish. OmniVoice already has the hard parts
|
||||
— multi-engine TTS/ASR, cloning, design, dubbing, live dictation, an MCP server,
|
||||
a local-LLM adapter, streaming TTS, echo cancellation. The gap is mostly **the
|
||||
last mile of control and polish on top of infrastructure that already exists.**
|
||||
|
||||
## Gap analysis
|
||||
|
||||
| ElevenLabs capability | OmniVoice today | Spec that closes it |
|
||||
|---|---|---|
|
||||
| Expressive/emotional delivery (v3 audio tags) | Voice *design* attributes only; no per-utterance emotion | **01 — Expressive TTS** |
|
||||
| Pronunciation dictionaries (IPA/phoneme) | `pronunciation.py` alias/respell, not user-editable/persisted | **01 — Expressive TTS** |
|
||||
| Conversational AI / voice agents | Pieces exist (streaming STT+TTS, local LLM, AEC) but no loop | **02 — Conversational Agent** |
|
||||
| Projects / Dubbing Studio (per-line regen, edit transcript/translation, reassign speaker) | Dub already content-addresses segments; longform caches only per-chapter; no unified editor | **03 — Long-form Studio Editor** |
|
||||
| Voice Library / shareable voices / marketplace | Local profiles only; no portable/shareable format | **04 — Voice Packs (local library)** |
|
||||
| Streaming latency (Flash ~75ms) + SDKs | Streaming TTS exists; latency + API/SDK ergonomics unbenchmarked | **05 — Streaming latency + API/SDK parity** |
|
||||
| Voice Isolator (denoise), Sound Effects | Demucs is in the dub stack; not exposed as tools | **06 — Audio cleanup + Sound FX** |
|
||||
| Accounts, cloud sync, hosted marketplace, usage analytics | *(none — intentionally)* | **Won't build** (see below) |
|
||||
|
||||
## The specs
|
||||
|
||||
### Tier 1 — the moat-closers (highest leverage)
|
||||
|
||||
- **[01 — Expressive TTS](01-expressive-tts.md)** — engine-agnostic emotion/style
|
||||
intent (inline tags + controls) *lowered* onto each engine's real mechanism
|
||||
(OmniVoice `instruct`, CosyVoice NL-instruct/`[laughter]`, IndexTTS2 emotion
|
||||
vector, VoxCPM2 prefix), degrading **visibly** never silently; plus a
|
||||
user-editable, per-language, DB-persisted **pronunciation dictionary** (IPA/CMU/
|
||||
respell) applied pre-synthesis. Builds on `services/ssml_lite.py`,
|
||||
`services/pronunciation.py`, `services/longform_parser.py`. Adds
|
||||
`services/expression.py`, `api/routers/pronunciation.py`, alembic 0008.
|
||||
*Status: spec complete, 5 shippable slices.*
|
||||
|
||||
- **[02 — Conversational Agent](02-conversational-agent.md)** — fully-offline
|
||||
full-duplex voice assistant: VAD → streaming STT → local LLM → streaming TTS
|
||||
with **barge-in** (Silero VAD on AEC-cleaned mic) and a single server-side
|
||||
`/ws/converse` orchestrator. Reuses `capture_ws.py` streaming ASR, `tts_stream.py`,
|
||||
`aec.py`, `llm_backend.py` (Ollama/OpenAI-compat), `mcp_server.py`. Opt-in;
|
||||
half-duplex/push-to-talk fallback on weak hardware. *Status: spec complete, 6 slices.*
|
||||
|
||||
- **[03 — Long-form Studio Editor](03-longform-studio-editor.md)** — per-segment
|
||||
edit / **regenerate-one-line** / reassign-voice / per-segment emotion / timing,
|
||||
across dubbing, audiobooks, stories. Dubbing already content-addresses segments
|
||||
(`incremental.py`, `regen_only`, `seg_hashes`); the spec extends that **span-level
|
||||
cache to longform** (which today only caches per-chapter) and unifies the editor
|
||||
UX. *Status: spec complete, 6 slices, no alembic needed.*
|
||||
|
||||
### Tier 2 — ecosystem & developer parity (drafts — expand before implementation)
|
||||
|
||||
- **04 — Voice Packs (local Voice Library).** A portable, importable voice-pack
|
||||
format (profile + reference + design `instruct` + pronunciation overrides +
|
||||
license/attribution, signed/hashed) and an **import/export** flow, plus an
|
||||
opt-in community **GitHub index** (a JSON manifest repo, not a hosted service)
|
||||
the app can browse and pull from. Local-first replacement for the marketplace:
|
||||
creators share packs as files/links; nothing is hosted by us. *Touchpoints:*
|
||||
voice profile storage, `omnivoice_data/`, the model-store download UI pattern.
|
||||
*Open: pack schema, signing/trust, NSFW/abuse stance on the index.*
|
||||
|
||||
- **05 — Streaming latency + API/SDK parity.** Honest benchmark of streaming TTS
|
||||
**time-to-first-audio** and real-time-factor per engine/device, a latency
|
||||
budget, and a documented **OpenAI-compatible + native streaming HTTP/WS API**
|
||||
with thin Python/JS SDK wrappers so developers can drop OmniVoice in where they
|
||||
used ElevenLabs. *Touchpoints:* `tts_stream.py` (`/ws/tts`), the MCP server, the
|
||||
generate path. *Open: which engines get the low-latency "Flash-class" path; SDK
|
||||
surface; OpenAI `/v1/audio/speech` compatibility scope.*
|
||||
|
||||
- **06 — Audio cleanup + Sound FX.** Expose **Voice Isolator** (vocal/denoise via
|
||||
the Demucs already vendored in the dub stack) and a **text-to-sound-effects**
|
||||
generator as first-class tools (and MCP tools), reusing existing audio I/O.
|
||||
*Touchpoints:* the dub separation stage, `services/audio_dsp.py`, MCP. *Open:
|
||||
which local SFX model; scope vs. core TTS focus (likely lowest priority).*
|
||||
|
||||
## Dependencies & recommended sequencing
|
||||
|
||||
```
|
||||
01 Expressive TTS ──► (emotion/style field) ──► 03 Studio Editor (per-segment emotion)
|
||||
│
|
||||
└──► 02 Conversational Agent (expressive replies)
|
||||
02 reuses: 01's streaming TTS quality + the existing live-dictation STT
|
||||
05 (API/latency) underpins 02's "feels real-time" and is independently shippable
|
||||
04 / 06 are independent and can slot in anytime
|
||||
```
|
||||
|
||||
Recommended order on the v0.3.x line (each spec is already sliced so early slices
|
||||
ship value without the whole feature):
|
||||
1. **01 Expressive TTS** — biggest perceived-quality win, unblocks 03's emotion-
|
||||
per-segment and 02's expressive replies. Start with the pronunciation
|
||||
dictionary slice (fast, high-trust) + inline emotion tags.
|
||||
2. **03 Studio Editor** — the dub transcript-edit + single-line-regen slice is
|
||||
small (the cache already exists) and immediately feels "pro."
|
||||
3. **02 Conversational Agent** — the headline new *category*; ship half-duplex
|
||||
first, then barge-in. Pair with the 05 latency benchmark.
|
||||
4. **05 / 04 / 06** — as capacity allows; 05 makes OmniVoice a real developer
|
||||
drop-in, 04 builds community gravity, 06 is breadth.
|
||||
|
||||
## Cross-cutting principles (every spec obeys these)
|
||||
|
||||
- **Local-first, always.** No cloud calls, accounts, or keys on any default path.
|
||||
New capabilities run on-device; "share" means files/links the user controls.
|
||||
- **Cross-platform default parity (hard rule).** Default behavior identical on
|
||||
macOS / Windows / Linux; anything platform-specific is opt-in (Settings toggle,
|
||||
env var, CLI flag). CPU-capable baselines everywhere.
|
||||
- **Back-compat (hard rule).** Existing engines, on-disk model state, and
|
||||
`omnivoice_data/` keep working with no forced reinstall or re-render; schema
|
||||
changes go through tested alembic upgrades.
|
||||
- **Sliceable onto v0.3.x.** No big-bang merges, no v0.4 deferrals — every spec is
|
||||
decomposed into independently-shippable slices with fail-before/pass-after tests.
|
||||
- **Degrade visibly.** When an engine/host can't do something (an emotion an
|
||||
engine lacks, latency on weak hardware), tell the user — never fail silently or
|
||||
fake it.
|
||||
|
||||
## What we deliberately will NOT build
|
||||
|
||||
Accounts, login, cloud sync, a hosted voice marketplace, server-side rendering,
|
||||
and usage analytics/telemetry are ElevenLabs *features* that are **anti-features**
|
||||
for a local-first tool. We don't measure parity against them. "Fully local, no
|
||||
keys, 646 languages, free, your voice never leaves your machine" is the
|
||||
counter-position — these specs make OmniVoice match ElevenLabs on the things
|
||||
creators feel, while staying on the right side of that line.
|
||||
|
||||
## Prior art & reconciliation
|
||||
|
||||
This roadmap (00 + 01–03) is the single source of truth for the ElevenLabs-parity
|
||||
program as of **2026-06-25**. The table below classifies every pre-existing spec in
|
||||
`docs/specs/` against it: **(S) Superseded** — substantial overlap, the new specs
|
||||
are more current/grounded (a banner now points here); **(F) Folded-in** — distinct,
|
||||
still-valuable detail referenced from the new specs; **(K) Keep as-is** — distinct
|
||||
scope, no parity overlap, left untouched.
|
||||
|
||||
| Pre-existing doc | Class | Disposition |
|
||||
|---|---|---|
|
||||
| `2026-06-12-elevenlabs-parity-program.md` | **S** | The previous parity roadmap. Its waves are reorganized into Tier 1/2 here; 01 explicitly carries forward its "perceived-quality half." Banner added. |
|
||||
| `2026-06-13-stories-audiobook-maturity.md` | **S** | Stories/Audiobook convergence + maturity. Its "one shared chapterized render core" and per-line/incremental asks are absorbed by **03** (longform Studio editor + span-level cache). Banner added. |
|
||||
| `studio-v1.md` | **S** | Long-form block editor v1 (paste→split→assign→stitch). Subsumed by **03**'s unified longform editor across Dub/Audiobook/Stories. Banner added. |
|
||||
| `voice-console-10x.md` | **K** | Voice-workspace *UI polish* (pinned action bar, identity line, a11y). No parity-capability overlap; left as-is. |
|
||||
| `voice-studio-unification.md` | **K** | Clone+Design → one "Voice" workspace + data-model unification. UI/IA scope, not parity capability. Left as-is. |
|
||||
| `workspace-connectivity.md` | **K** | Navigation IA + universal "Use in ▸" handoff + transcripts-to-backend. Cross-workspace plumbing, distinct scope. Left as-is. |
|
||||
| `2026-05-29-v0.3.0-stabilization-sweep.md` | **K** | Stabilization/bug-cluster + review/security gate program. Orthogonal to parity. Left as-is. |
|
||||
| `longform/` (#21–#34) | **F** | Granular per-task implementation specs (tied to tasks #21–34). Their capabilities feed the new specs: incremental/longform render → **03**; `.ovsvoice` (#29) → **04 Voice Packs**; ACX two-pass mastering (#28), EPUB/m4b export (#24/#30), transcriptions import (#23), shared voice selector (#22) → the longform editor + Voice Library work; phone calls (#32) → the **deliberately deferred** telephony note (gated on guardrails). Retained as the detailed build specs. |
|
||||
|
||||
### Salvaged ideas now referenced (don't lose these)
|
||||
|
||||
Concrete deliverables in the pre-existing docs that the new 01–03 did **not** already
|
||||
surface, captured here so they aren't dropped:
|
||||
|
||||
- **GPU-compat preflight — "no silent CPU fallback"** (`longform/21-gpu-compat-matrix.md`). A
|
||||
canonical device-family probe + per-engine *effective device*/routing status, surfaced at
|
||||
engine-select and **every** synth entry point with an explicit warning when the active engine
|
||||
can't use the user's GPU. This is the concrete enforcement of this roadmap's "degrade visibly"
|
||||
principle and a "first-run that actually works" win — fold into the platform-robustness track.
|
||||
- **Standalone `.txt` chapter cue-sheet export** (`longform/33-cue-sheet-export.md`). A
|
||||
human-readable `HH:MM:SS<TAB>Title` cue sheet for the longform front doors (for mp3/show-notes/
|
||||
YouTube chapters), reusing existing `formatTimecode`/`buildCueSheet` helpers — no backend change.
|
||||
A small, high-value nicety for **03**'s longform editor surfaces.
|
||||
- **Consent-locked voice profiles + AudioSeal watermarking on agentic/shared output**
|
||||
(parity-program items 0.2/5.3/5.4; `longform/29-ovsvoice-format.md`, `longform/32-phone-calls.md`).
|
||||
The consent/attestation + watermark guardrail is a hard prerequisite for **04 Voice Packs**
|
||||
(sharing) and **02**'s agentic output (EU AI Act Art 50, applies 2026-08-02). The new specs assume
|
||||
local-first sharing but don't yet spell out the consent-lock/watermark gate — it must land **with**
|
||||
04 and any agentic-output path, not as a follow-up.
|
||||
- **Two-pass ACX loudness mastering + chaptered m4b/cover/metadata** (`2026-06-13-stories-audiobook-maturity.md`,
|
||||
`longform/28-two-pass-acx-mastering.md`, `#24`). The audiobook-grade mastering/packaging detail lives
|
||||
in those specs; **03** assumes the shared longform render core exists and should consume this rather
|
||||
than re-specify it.
|
||||
@@ -0,0 +1,308 @@
|
||||
# Spec 01 — Expressive TTS: emotion/style direction + pronunciation control
|
||||
|
||||
**Date:** 2026-06-25
|
||||
**Status:** Proposed
|
||||
**Target line:** v0.3.x (continuous-to-main; no RC, no deferral to v0.4)
|
||||
**Related:** `docs/specs/2026-06-12-elevenlabs-parity-program.md` (this is the "perceived-quality" half that program left open), issues #674/#679 (design-mode profile_id handling).
|
||||
|
||||
---
|
||||
|
||||
## 1. Context & Problem (gap vs ElevenLabs)
|
||||
|
||||
The #1 perceived-quality gap users report vs ElevenLabs is **expressive delivery**. ElevenLabs v3 ships two things we don't expose coherently:
|
||||
|
||||
1. **Audio tags** — inline square-bracket performance cues (`[excited]`, `[whispers]`, `[sigh]`, `[laughs]`) that steer emotion/tone *mid-line*, plus situational/reaction tags ([ElevenLabs v3 audio tags](https://elevenlabs.io/blog/eleven-v3-audio-tags-expressing-emotional-context-in-speech), [help: how audio tags work](https://help.elevenlabs.io/hc/en-us/articles/35869142561297-How-do-audio-tags-work-with-Eleven-v3)).
|
||||
2. **Pronunciation dictionaries** — per-term rules with **phoneme** (IPA/CMU via SSML `<phoneme>`) and **alias** (respelling) entries, checked start-to-end, first match wins, case-sensitive ([ElevenLabs pronunciation dictionaries](https://elevenlabs.io/docs/eleven-api/guides/how-to/text-to-speech/pronunciation-dictionaries)).
|
||||
|
||||
**What we already have (and must reuse, not reinvent):**
|
||||
|
||||
- `backend/services/ssml_lite.py` — inline `[slow]/[fast]/[emphasis]/[spell]` tags → ordered prosody segments `{text, speed, spell, emphasis}`. ReDoS-safe fixed-alternation regex. Already wired into `longform_parser.py`.
|
||||
- `backend/services/pronunciation.py` — `apply_lexicon(text, {term: respelling})`: whole-word, case-insensitive, longest-key-first, word-boundary aware, ReDoS-safe single-pass `re.sub`. **This is the alias-rule engine; it has no DB persistence and no IPA path yet.** Used today only by audiobook (`services/audiobook.py:119`, `api/routers/audiobook.py`).
|
||||
- `backend/services/longform_parser.py` — the canonical grammar: precedence `# chapter → [voice:NAME] → [pause] → SSML-lite → [spell]`. JS twin `frontend/src/utils/longformParser.js`, golden corpus `tests/fixtures/longform_parser_cases.json`. **This is where multi-voice `[voice:NAME]` story tags live — our new emotion tags must not collide with it.**
|
||||
- `backend/services/chunked_tts.py` — sentence-boundary splitter that already **refuses to cut inside `[...]` bracket tags** (`_BRACKET_TAG_RE`, line 43). New tags inherit that protection for free.
|
||||
- `omnivoice.utils.text.parse_pause_markers` — `[pause Nms]` span splitter, consumed in `generation.py:224`.
|
||||
|
||||
**What's missing / the gap:**
|
||||
|
||||
- No way to direct **emotion** at all from the Studio generate path. The only "style" the **OmniVoice base model** accepts is the validated `instruct` taxonomy (Gender/Age/Pitch/**Style=whisper only**/Accent/Dialect) — confirmed in `core/describe_voice.py` and the engine's `omnivoice/utils/voice_design.py` validator. The base model **does not** take `[happy]`/`[sad]` ([k2-fsa/OmniVoice issue #78 — Emotion/Tone](https://github.com/k2-fsa/OmniVoice/issues/78)); only a finetune does.
|
||||
- The capable engines express emotion **very differently**: CosyVoice 3 via natural-language instruct `…<|endofprompt|>` + inline `[laughter]`/`[breath]`/`<strong>` ([CosyVoice 3 paper](https://arxiv.org/html/2505.17589v1)); IndexTTS2 via an **8-dim emotion vector** `[happy,angry,sad,afraid,disgusted,melancholic,surprised,calm]` or an emotion-reference clip ([IndexTTS2](https://indextts.ai/), [arXiv 2506.21619](https://arxiv.org/html/2506.21619v2)); VoxCPM2 via a `(instruct)text` prefix (`tts_backend.py:423`).
|
||||
- The lexicon is project-local JSON only — no global, no per-language, no DB persistence, no UI, no IPA/phoneme path, no inline one-off override.
|
||||
|
||||
**Design principle:** one engine-agnostic *intent* surface (tags + sliders + dictionary) that **lowers** to whatever each engine can actually do, and **degrades visibly** (never silently) where it can't.
|
||||
|
||||
---
|
||||
|
||||
## 2. Goals / Non-goals
|
||||
|
||||
**Goals**
|
||||
|
||||
- G1. Inline emotion/style tags in the generate text — `[excited]`, `[whispers]`, `[sad]`, `[shouting]`, `[laughs]`, … — parsed into per-span *expression intent*, composing cleanly with existing `[voice:]`/`[pause]`/SSML-lite/`[spell]`.
|
||||
- G2. A non-inline alternative for users who don't want to learn tags: an **Expression** panel on the generate/Studio UI (emotion dropdown + intensity slider + optional **emotion-reference clip** picker) that sets a per-render default.
|
||||
- G3. Per-engine **capability matrix** that maps expression intent → the engine's real mechanism, surfaced in the UI so users know what their selected engine will honor before they hit generate.
|
||||
- G4. A user-editable **pronunciation dictionary** persisted in the DB (global + per-language scope), with **alias** (respelling) and **phoneme** (IPA/CMU) entry types, applied at synth time before the model, plus inline one-off overrides.
|
||||
- G5. Backward-compatible: existing engines, on-disk profiles, the project-local audiobook lexicon JSON, and plain (tag-free) text all keep working byte-identically.
|
||||
- G6. Local-first, identical default behavior on macOS/Windows/Linux.
|
||||
|
||||
**Non-goals**
|
||||
|
||||
- N1. Per-phoneme prosody curves / full SSML (`<prosody>`/`<break>` trees). SSML-lite + `[pause]` stay our prosody surface.
|
||||
- N2. Training/finetuning an emotion model. We expose what shipped engines already do; the OmniVoice base model's emotion ceiling is its instruct taxonomy, and we say so.
|
||||
- N3. A learned text→emotion classifier in the base path (IndexTTS2's own T2E module is used when *that* engine is active; we don't build a global one).
|
||||
- N4. Auto-generating IPA from spelling (no g2p engine bundled in this spec — see Open Questions Q4).
|
||||
|
||||
---
|
||||
|
||||
## 3. User Experience (UI + flows)
|
||||
|
||||
### 3.1 Inline expression tags (power path)
|
||||
|
||||
In the Studio / generate text box the user writes:
|
||||
|
||||
```
|
||||
[excited] We did it! [pause 400ms] [whispers] ...but don't tell anyone.
|
||||
```
|
||||
|
||||
- Tags are stripped from spoken text and turned into per-span intent.
|
||||
- Tags compose with multi-voice stories: `[voice:Morgan] [angry] Get out. [voice:Sam] [nervous] O-okay.` — `[voice:]` switches narrator (existing), `[angry]`/`[nervous]` set that span's emotion.
|
||||
- An **"⊕ Insert"** popover (reusing the existing clone-tab insert popover pattern, #672) lists available tags **filtered to what the active engine supports**, with a tooltip showing the lowering ("`[excited]` → CosyVoice instruct / IndexTTS2 emo-vector / OmniVoice: not supported, ignored").
|
||||
- Unsupported-on-this-engine tags render with a subtle strikethrough chip and a one-line banner: *"OmniVoice ignores emotion tags — switch to CosyVoice 3 or IndexTTS2 for emotional delivery."* (never silent; mirrors the routing-banner convention from `engine_routing.py`).
|
||||
|
||||
### 3.2 Expression panel (no-tags path)
|
||||
|
||||
A collapsible **Expression** section under the generate controls:
|
||||
|
||||
- **Emotion** dropdown: Neutral (default) / Happy / Sad / Angry / Afraid / Surprised / Calm / Whisper / Shout. Maps to the engine's mechanism (sliders for IndexTTS2; instruct phrase for CosyVoice/VoxCPM; whisper-only for OmniVoice).
|
||||
- **Intensity** slider 0–100 (default 50). Only enabled when the active engine supports graded intensity (IndexTTS2 emo-vector magnitude); otherwise greyed with a tooltip.
|
||||
- **Emotion reference** (optional): pick a short clip whose *delivery* (not timbre) is mimicked. Enabled only for engines with an emotion-ref path (IndexTTS2). This is **separate** from the voice-clone `ref_audio` — same control style, different slot.
|
||||
- A live **"This engine will: …"** line shows the resolved lowering, so the panel doubles as the capability disclosure.
|
||||
|
||||
The panel sets request-level defaults; inline tags override per span (tag wins, same precedence rule as SSML-lite speed).
|
||||
|
||||
### 3.3 Pronunciation dictionary (Settings → Voice → Pronunciation)
|
||||
|
||||
A new **PronunciationPanel** (sibling of `VoicePanel.jsx`):
|
||||
|
||||
- A table of entries: **Term** | **Scope** (Global / language) | **Type** (Respelling / IPA / CMU) | **Replacement** | **Enabled**.
|
||||
- Add/edit/delete rows; inline validation (IPA charset check; CMU ARPABET token check). Bad phoneme strings flagged before save, not at synth time.
|
||||
- A **"Test"** field: type a sentence, see the post-substitution text (and, for phoneme rows, the `[[…]]` markup that will be handed to the engine) — no model call needed.
|
||||
- Import/Export JSON (round-trips the existing audiobook lexicon shape, so a project lexicon can be promoted to global).
|
||||
|
||||
### 3.4 Inline one-off pronunciation override
|
||||
|
||||
Within text: `She lives on [[ˈnɛvʌdə]] street` (IPA in double brackets) or `[[Nuh-VAD-uh]]` (respelling). Applies once, overrides any dictionary entry for that occurrence. Chosen `[[…]]` because single `[...]` is already emotion/voice/pause tags — double brackets are unambiguous and don't collide.
|
||||
|
||||
---
|
||||
|
||||
## 4. Technical Design
|
||||
|
||||
### 4.1 Architecture overview
|
||||
|
||||
Two independent, composable layers, both **pure/CPU/stdlib** at the parse stage (model-free, unit-testable, cross-platform-identical):
|
||||
|
||||
```
|
||||
text + request expression defaults
|
||||
│
|
||||
▼
|
||||
[A] expression parse ── extends services/longform_parser.py grammar:
|
||||
# chapter → [voice:NAME] → [emotion] → [pause] → SSML-lite → [spell] → [[pronounce]]
|
||||
│ (new layer, slotted between voice and pause)
|
||||
▼
|
||||
spans: {voice_id, text, pause_ms_after, speed, expression} ← expression added
|
||||
│
|
||||
▼
|
||||
[B] pronunciation apply ── services/pronunciation.py (extended):
|
||||
apply_pronunciation(span.text, dict, language) → text with aliases substituted
|
||||
+ [[…]] inline overrides resolved to engine phoneme markup or respelling
|
||||
│
|
||||
▼
|
||||
[C] expression lowering ── services/expression.py (NEW):
|
||||
lower(expression, engine_id) → engine-specific kwargs
|
||||
(instruct phrase | emo_vector | emo_ref | whisper | <none>)
|
||||
│
|
||||
▼
|
||||
TTSBackend.generate(text, instruct=…, **expression_kwargs)
|
||||
```
|
||||
|
||||
### 4.2 Files to add / extend (real paths)
|
||||
|
||||
**Add**
|
||||
|
||||
- `backend/services/expression.py` — the expression vocabulary + lowering. Pure, model-free. Defines:
|
||||
- `EXPRESSIONS` — canonical emotion set `{neutral, happy, sad, angry, afraid, surprised, calm, whisper, shout, laugh, sigh}` (the cross-engine intersection; superset of OmniVoice's `whisper`).
|
||||
- `Expression` dataclass `{emotion: str, intensity: float, ref_audio: str|None}`.
|
||||
- `parse_expression_tags(text) -> list[(text, Expression|None)]` — splits a line on `[emotion]`/`[/emotion]` tags using the **same fixed-alternation, ReDoS-safe regex shape** as `ssml_lite.py` (`_TAG_RE`), tags drawn from `EXPRESSIONS`. Unknown bracket tokens are left **untouched** so they pass through to SSML-lite / `[voice:]` / `[pause]` — no grammar overlap.
|
||||
- `lower(expr, engine_id) -> dict` — the capability matrix in code (see 4.4). Returns kwargs to merge into `generate()`; returns `{}` + a `degraded` note for engines that can't honor it.
|
||||
- `backend/api/routers/pronunciation.py` — CRUD for dictionary entries + `/pronunciation/test` (dry-run substitution, no model). Registered in `backend/main.py` alongside the other routers.
|
||||
- `frontend/src/components/settings/PronunciationPanel.jsx` (+ `.css`, + `.test.jsx`).
|
||||
- `frontend/src/utils/expressionTags.js` — JS twin of `parse_expression_tags` (mechanically mirrored, same golden corpus, exactly like `longformParser.js`).
|
||||
- `backend/migrations/versions/0008_pronunciation_dictionary.py` — alembic migration (see §5.3).
|
||||
|
||||
**Extend**
|
||||
|
||||
- `backend/services/longform_parser.py::_parse_chapter_body` — insert the expression layer between `[voice:]` runs and the existing pause/SSML loop. Each emitted span gains an `expression` key (default `None`). The `Span` dataclass / `to_dict()` and the JS twin update in lockstep; `tests/fixtures/longform_parser_cases.json` gains expression cases.
|
||||
- `backend/services/ssml_lite.py` — **no change to its grammar**; expression parsing runs as a sibling layer above it. `[whisper]` is handled by expression (engine-level), distinct from SSML-lite's prosody — documented so they don't drift.
|
||||
- `backend/services/pronunciation.py` — add:
|
||||
- `apply_pronunciation(text, entries, language)` — alias substitution (delegates to existing `apply_lexicon` for respelling rows) **plus** phoneme rows lowered to the active engine's phoneme markup; per-language filtering (global rows always apply; language rows apply when `language` matches or is `Auto`).
|
||||
- `parse_inline_pronunciation(text, engine_id)` — resolve `[[…]]` overrides (IPA/CMU/respelling autodetected by charset) to engine markup or plain respelling, ReDoS-safe `\[\[[^\]]*\]\]`.
|
||||
- `load_dict_from_db()` / `save_dict_to_db()` — DB-backed counterpart to the existing JSON `load_lexicon`/`save_lexicon` (which stay for the audiobook project-local path).
|
||||
- `backend/api/routers/generation.py::generate_speech` — accept `expression`, `expression_intensity`, `expression_ref` (Form fields) and a `pronounce: bool` toggle (default ON). Thread them through `_run_inference` / `_run_backend_inference` so the lowering kwargs reach `model.generate()` / `backend.generate()`. Pronunciation applied **per chunk before** `split_text_into_chunks`, mirroring `services/audiobook.py:124`.
|
||||
- `backend/services/tts_backend.py` — extend the `generate(**extras)` contract with optional `emo_vector`, `emo_ref`, `emotion` kwargs. The ABC already takes `**extras`, so **no signature break**; each backend reads the kwargs it understands and ignores the rest (the existing graceful-degradation idiom, e.g. KittenTTS/MOSS at lines 510–527). Per-engine `generate()` bodies updated for CosyVoice (fold emotion into the `inference_instruct2` prompt), VoxCPM2 (`(instruct)` prefix), IndexTTS2 (`emo_vector`/`emo_ref` — its module in `engines/indextts/`).
|
||||
- A new class attribute on `TTSBackend`: `expression_caps: dict` (e.g. `{"mode": "instruct"|"emo_vector"|"emo_ref"|"whisper_only"|"none", "emotions": [...]}`), defaulting to `{"mode":"none"}`. `list_backends()` surfaces it next to `gpu_compat` so the UI can filter tags/sliders per engine without a model load.
|
||||
|
||||
### 4.3 Data flow & composition (no collisions)
|
||||
|
||||
Grammar precedence (extends `longform_parser.py:10`):
|
||||
|
||||
```
|
||||
# chapter → [voice:NAME] → [emotion] → [pause] → SSML-lite → [spell] → [[pronounce]]
|
||||
```
|
||||
|
||||
- `[voice:NAME]` (existing, `_VOICE_RE`) is matched first → switches narrator. Emotion tags live **inside** a voice run, so `[voice:Sam]` and `[angry]` never compete for the same token.
|
||||
- `[emotion]` tags are a **closed set** drawn from `EXPRESSIONS`; the regex only matches those literals, so a stray `[whatever]` is not consumed and flows to the lower layers (or out as literal text). This is the same closed-alternation safety `ssml_lite.py` relies on.
|
||||
- `[pause Nms]`, SSML-lite, `[spell]` are unchanged and parse *within* an emotion span — emotion is an outer attribute, prosody/speed inner, exactly like the current voice→pause→ssml nesting.
|
||||
- `[[pronounce]]` (double bracket) is resolved last, after tag stripping, so it can't be confused with single-bracket tags. `chunked_tts._BRACKET_TAG_RE` already protects single `[...]`; we widen it (or add a sibling) to also never split inside `[[...]]`.
|
||||
|
||||
**Tag-vs-text disambiguation rule (the load-bearing invariant):** a `[token]` is consumed by a layer **iff** `token` (case-insensitive) is in that layer's closed vocabulary (`EXPRESSIONS`, SSML-lite `_TAGS`, `voice:`-prefixed, or `pause …`). Everything else is literal. This is asserted by the shared golden corpus against both Python and JS parsers.
|
||||
|
||||
### 4.4 Per-engine capability matrix (`expression.lower`)
|
||||
|
||||
| Engine (`id`) | Mechanism | emotion | intensity | emo-ref | How `lower()` maps it |
|
||||
|---|---|---|---|---|---|
|
||||
| `omnivoice` (base) | `instruct` taxonomy, **Style=whisper only** | whisper only | no | no | `[whispers]` → append `whisper` to instruct (validator-safe via existing `heal_design_instruct`). All other emotions → `degraded`, banner shown. |
|
||||
| `cosyvoice` | NL instruct `…<\|endofprompt\|>` + inline `[laughter]`/`[breath]`/`<strong>` | full set | coarse (phrasing) | no | emotion → instruct phrase ("speak in an excited tone") merged into `inference_instruct2`'s instruct arg (`tts_backend.py:846`). `[laughs]`/`[sigh]` → CosyVoice's own `[laughter]`/`[breath]` literals injected into text. |
|
||||
| `indextts2` | 8-dim **emo_vector** + emo-ref + text-infer | full set | **yes (0–1)** | **yes** | emotion+intensity → one-hot-ish 8-vector scaled by intensity; emo-ref → `emo_audio` path. (`engines/indextts/`.) |
|
||||
| `voxcpm2` | `(instruct)text` prefix | full set | coarse | no | emotion → `(speak excitedly)` prefix (`tts_backend.py:423`). |
|
||||
| `kittentts`, `sherpa-onnx`, `gpt-sovits`, `mlx-audio`(varies), `moss-tts-nano`, `supertonic3` | none / preset | — | — | — | `lower()` → `{}` + `degraded`; tags stripped, text spoken neutrally. Banner: *"This engine has no emotion control."* |
|
||||
|
||||
`lower()` is the single source of truth for this table; `list_backends()['expression_caps']` is derived from the same constants so UI and synth never disagree (the `describe_voice.py` import-time-validation discipline — a missing/renamed capability fails loudly in a test, not at synth).
|
||||
|
||||
### 4.5 Pronunciation lowering
|
||||
|
||||
- **Respelling (alias) rows** → existing `apply_lexicon` (already correct: longest-first, boundary-aware, ReDoS-safe). Works on **every** engine — it's just text substitution.
|
||||
- **Phoneme rows (IPA/CMU)** → engine phoneme markup where supported (e.g. CosyVoice/sherpa-onnx models with a phoneme front-end), else **graceful fallback to the respelling** if the row also has one, else passed through and flagged "phoneme not honored on this engine" (parity-rule: visible degradation). No engine is *broken* by a phoneme row; worst case it's spoken as the literal grapheme.
|
||||
- Per-language: global rows always apply; language-tagged rows apply when request `language` matches (or is `Auto`). Matching is case-insensitive on the 2-letter prefix, consistent with `CosyVoiceBackend.LANG_TAGS` handling.
|
||||
|
||||
---
|
||||
|
||||
## 5. API / Schema / Data-model changes
|
||||
|
||||
### 5.1 Endpoints
|
||||
|
||||
- `POST /generate` (extend, `generation.py`): new optional Form fields
|
||||
- `expression: str = Form("")` — emotion name or `""`/`neutral`.
|
||||
- `expression_intensity: float = Form(50, ge=0, le=100)`.
|
||||
- `expression_ref: UploadFile = File(None)` — emotion reference clip (engines that support it).
|
||||
- `pronounce: bool = Form(True)` — apply the pronunciation dictionary.
|
||||
Omitting all of them = byte-identical legacy behavior.
|
||||
- `GET /pronunciation` → `[{id, term, scope, type, replacement, enabled, language}]`.
|
||||
- `POST /pronunciation` / `PUT /pronunciation/{id}` / `DELETE /pronunciation/{id}` — CRUD with server-side IPA/CMU validation.
|
||||
- `POST /pronunciation/test` → `{input} → {substituted, phoneme_markup}` (no model).
|
||||
- `GET /engines/tts` (existing `list_backends`) gains `expression_caps` per entry. No new endpoint.
|
||||
|
||||
### 5.2 Prefs / settings
|
||||
|
||||
- New pref key `pronunciation_enabled` (default `true`) via `core/prefs.py` (`prefs.resolve`, env `OMNIVOICE_PRONUNCIATION` for power-users) — same pattern as `tts_backend`.
|
||||
- Expression defaults are per-request, not persisted globally (a render-time choice, like `effect_preset`).
|
||||
|
||||
### 5.3 Alembic migration (additive, idempotent — sketch)
|
||||
|
||||
`backend/migrations/versions/0008_pronunciation_dictionary.py`, `down_revision="0007_rebuild_poisoned_design_instruct"`. Follows the 0004 idempotent-guard pattern exactly.
|
||||
|
||||
```python
|
||||
revision = "0008_pronunciation_dictionary"
|
||||
down_revision = "0007_rebuild_poisoned_design_instruct"
|
||||
|
||||
def _has_table(name): # same helper as 0004
|
||||
bind = op.get_bind()
|
||||
return bind.execute(sa.text(
|
||||
"SELECT name FROM sqlite_master WHERE type='table' AND name=:n"),
|
||||
{"n": name}).fetchone() is not None
|
||||
|
||||
def upgrade():
|
||||
if _has_table("pronunciation_entries"):
|
||||
return
|
||||
op.create_table(
|
||||
"pronunciation_entries",
|
||||
sa.Column("id", sa.Text(), primary_key=True),
|
||||
sa.Column("term", sa.Text(), nullable=False),
|
||||
sa.Column("replacement", sa.Text(), nullable=False, server_default=""),
|
||||
sa.Column("type", sa.Text(), nullable=False, server_default="respelling"), # respelling|ipa|cmu
|
||||
sa.Column("language", sa.Text(), nullable=False, server_default="*"), # '*' = global
|
||||
sa.Column("enabled", sa.Integer(), nullable=False, server_default="1"),
|
||||
sa.Column("created_at", sa.Float(), nullable=True),
|
||||
)
|
||||
op.create_index("idx_pron_lang", "pronunciation_entries", ["language"])
|
||||
|
||||
def downgrade():
|
||||
if _has_table("pronunciation_entries"):
|
||||
op.drop_table("pronunciation_entries")
|
||||
```
|
||||
|
||||
**Critically:** the same table must also be added to `core/db.py::_BASE_SCHEMA` as `CREATE TABLE IF NOT EXISTS pronunciation_entries (...)` so fresh installs and the `_reconcile_additive_columns` safety net converge on the identical end-state (the dual-path discipline already documented in `db.py:140`). The audiobook project-local JSON lexicon (`load_lexicon`/`save_lexicon`) is **untouched** — it remains the per-project override; DB entries are the global/default layer, merged dict-style (project JSON wins on key conflict, longest-first preserved).
|
||||
|
||||
---
|
||||
|
||||
## 6. Local-first & cross-platform compliance
|
||||
|
||||
- **No cloud.** Parsing, lexicon, and lowering are pure Python/stdlib + JS — zero network, zero model for the control plane. Emotion is realized entirely by the **already-on-device** engine the user selected; IPA/CMU validation is charset/table-based (no g2p service).
|
||||
- **Default-parity (strict 2026-05-20 rule).** The *default* path — emotion tags, the dictionary, `[[…]]` overrides — behaves identically on macOS/Windows/Linux because it's pure text transformation guarded by the shared golden corpus run on both the Python and JS parsers. `longform_parser._normalize` already CRLF-normalizes, so Windows-authored scripts parse identically. **Engine-specific** emotion fidelity differs by engine (a property of the engine, not the platform) and is disclosed in the capability matrix UI — this is allowed because the *user-visible default behavior of the feature* (tags parse, dictionary applies, degradation is shown) is identical everywhere; only the opt-in *engine* changes what's honored.
|
||||
- No platform-only tag or shortcut. The "⊕ Insert" popover and panels are the same component on every OS.
|
||||
- Emotion-reference clip is processed on-device by the same engine path as voice refs; privacy unchanged (never uploaded, scrubbed from any bug report per CLAUDE.md capture rules).
|
||||
|
||||
---
|
||||
|
||||
## 7. Phasing (each slice lands independently on v0.3.x)
|
||||
|
||||
- **Phase 1 — Pronunciation dictionary (DB + UI).** Migration 0008 + `_BASE_SCHEMA` row + `pronunciation.py` extensions (`apply_pronunciation`, DB load/save, per-language) + `/pronunciation` router + `PronunciationPanel`. Wires into `/generate` (`pronounce` toggle) and reuses the existing audiobook apply-site. **Respelling rows only** in this phase (alias rules work on every engine). Ships value immediately, zero engine risk.
|
||||
- **Phase 2 — Inline `[[pronounce]]` overrides + IPA/CMU rows.** Phoneme validation + engine-markup lowering for the engines that have a phoneme front-end; respelling fallback elsewhere. Widen `chunked_tts` bracket guard for `[[…]]`.
|
||||
- **Phase 3 — Expression engine + lowering.** `services/expression.py` + `expression_caps` on backends + `lower()` for CosyVoice/VoxCPM2/IndexTTS2/OmniVoice-whisper. `/generate` Form fields + `_run_*_inference` threading. No UI yet (API + tags usable headless/MCP).
|
||||
- **Phase 4 — Inline emotion tags in the grammar.** Extend `longform_parser` + JS twin + golden corpus; `[emotion]` spans flow through audiobook/story/Studio. This is the collision-sensitive change, landed only after Phase 3's vocabulary is stable.
|
||||
- **Phase 5 — Expression panel UI** (dropdown + intensity + emo-ref picker + "this engine will…" line + per-engine tag filtering in the Insert popover).
|
||||
|
||||
Phases 1–2 (pronunciation) and 3–5 (expression) are independent tracks; either can lead. Each phase = one PR through the review+security gate with its tests, bisectable, docs-synced.
|
||||
|
||||
---
|
||||
|
||||
## 8. Testing strategy (fail-before / pass-after)
|
||||
|
||||
**Pure-parser (no model, run in CI everywhere):**
|
||||
|
||||
- `tests/test_expression_parse.py` — `parse_expression_tags`: plain text → one neutral span (fail-before: function doesn't exist); nested `[voice:][excited]…[pause]…` precedence; unknown `[token]` passes through untouched (the anti-collision invariant); unclosed tag → applies to EOL; ReDoS corpus (long adversarial bracket runs) completes < 50 ms.
|
||||
- Extend `tests/fixtures/longform_parser_cases.json` with expression cases; assert **byte-identical** output from `longform_parser.py` and `frontend/src/utils/expressionTags.js`/`longformParser.js` (the existing twin-parity gate). Fail-before: JS twin missing emotion key.
|
||||
- `tests/test_pronunciation.py` — `apply_pronunciation`: per-language filtering (global applies, mismatched-language skipped, `Auto` applies all); respelling delegation matches legacy `apply_lexicon`; `[[ipa]]` inline override resolves; phoneme-on-unsupported-engine falls back to respelling and sets `degraded`. Idempotency (apply twice == once).
|
||||
- `tests/test_expression_lowering.py` — `lower(expr, engine)` for every registered engine returns the matrix's expected kwargs; `expression_caps` on each backend matches what `lower()` actually consumes (the describe_voice-style import-time consistency assert, so a renamed cap fails a test, not synth).
|
||||
|
||||
**API:**
|
||||
|
||||
- `tests/test_pronunciation_api.py` — CRUD round-trip; IPA/CMU validation rejects garbage with 400; `/pronunciation/test` returns substituted text with no model loaded.
|
||||
- `tests/test_generate_expression.py` — `/generate` with `expression=excited` on OmniVoice returns 200 + an `X-OmniVoice-Expression: degraded` header (visible degradation, never silent); on a mock CosyVoice backend, asserts the instruct phrase was injected.
|
||||
|
||||
**Migration:**
|
||||
|
||||
- `tests/test_migration_0008.py` — upgrade on a 0007-stamped DB creates the table; re-running is a no-op (idempotent guard); a fresh `_BASE_SCHEMA` DB and a migrated DB have **identical** `PRAGMA table_info(pronunciation_entries)` (the dual-path convergence assert, like the existing `_reconcile_additive_columns` tests). Backward-compat: an existing `omnivoice_data/` DB with no table upgrades cleanly and old rows untouched.
|
||||
|
||||
**Cross-platform / CI-green:**
|
||||
|
||||
- Twin-parity test covers Win/mac/Linux line endings via `_normalize`.
|
||||
- No new Python runtime dep (Phase 1–4 are stdlib); no `frontend/package.json` change unless the panel pulls a new lib (it shouldn't) — if it does, regenerate root `bun.lock` and assert `bun install --frozen-lockfile` per the keep-main-green rule.
|
||||
|
||||
---
|
||||
|
||||
## 9. Risks & mitigations
|
||||
|
||||
- **R1 — tag/grammar collision** (emotion tag eats a `[voice:]` or a literal `[bracketed]` word). *Mitigation:* closed-vocabulary matching + the shared golden corpus asserting passthrough of unknown tokens against both parsers. This is the single highest-risk change → isolated to Phase 4, after the vocab is frozen.
|
||||
- **R2 — silent degradation** (user picks `[excited]` on OmniVoice, hears neutral, blames us). *Mitigation:* strikethrough chips, banner, and an `X-OmniVoice-Expression: degraded` response header — parity with the no-silent-CPU-fallback rule. Capability matrix shown *before* generate.
|
||||
- **R3 — IPA/CMU garbage → engine crash.** *Mitigation:* validate on save (400), fall back to respelling/literal at synth, never pass unvalidated phoneme strings to a model. Worst case = spoken as written.
|
||||
- **R4 — IndexTTS2 emo-vector API drift** (it's subprocess-isolated, own venv). *Mitigation:* lowering for IndexTTS2 lives behind its `engines/indextts/` adapter; a contract test pins the kwarg names; if the kwarg is absent the adapter degrades to neutral (existing `**extras` ignore idiom).
|
||||
- **R5 — lexicon/dictionary double-apply** (project JSON + DB). *Mitigation:* single merge point, project-wins ordering, idempotency test; respelling pass is already idempotent by construction (`pronunciation.py` single-pass `re.sub`).
|
||||
- **R6 — DB migration on a preview-stamped DB** (alembic_version at a removed rev). *Mitigation:* the `_BASE_SCHEMA` + `_reconcile_additive_columns` belt already covers this class (`db.py`); the convergence test asserts it.
|
||||
|
||||
---
|
||||
|
||||
## 10. Open questions / decisions for the owner
|
||||
|
||||
- Q1. **Tag vocabulary:** adopt ElevenLabs' exact tag names (`[excited]`, `[whispers]`, `[sighs]`) for muscle-memory parity, or a neutral set? (Recommendation: alias the common ElevenLabs names to our canonical set so pasted ElevenLabs scripts "just work.")
|
||||
- Q2. **Reaction tags** (`[laughs]`, `[sigh]`, `[gasp]`): treat as emotion spans, or as literal text injected for engines that support them (CosyVoice `[laughter]`/`[breath]`)? (Recommendation: a third tag class "sound" lowered per-engine; out of scope for Phase 3, note for later.)
|
||||
- Q3. **OmniVoice base-model emotion:** ship as whisper-only with honest degradation (this spec), or also wire the `ModelsLab/omnivoice-singing` finetune as an opt-in engine variant that *does* take `[happy]`/`[sad]`? (Recommendation: ship honest degradation now; the finetune is a separate engine-registry entry, a clean follow-up.)
|
||||
- Q4. **g2p for IPA generation:** bundle a small grapheme→IPA helper (e.g. `g2p-en` for English, ARPABET) so users get a suggested phoneme string, or keep this spec validation-only (user supplies IPA/CMU)? (Recommendation: validation-only now; g2p is a per-language native-dep rabbit hole — defer, document.)
|
||||
- Q5. **Docs-sync:** this adds inline-tag and pronunciation surfaces → `docs/voice-design.md`, `docs/generation-parameters.md`, and `docs/features.yaml` must update in the same PRs (hard rule). Confirm whether a dedicated `docs/expressive-tts.md` page is wanted.
|
||||
@@ -0,0 +1,587 @@
|
||||
# Local Conversational Voice Agent — Implementation Spec
|
||||
|
||||
**Date:** 2026-06-25
|
||||
**Status:** Proposed
|
||||
**Owner:** debpalash
|
||||
**Spec #:** 02
|
||||
|
||||
A fully-offline, low-latency full-duplex voice assistant for OmniVoice Studio:
|
||||
**VAD → streaming STT → local LLM (streaming tokens) → streaming TTS**, with
|
||||
barge-in / turn-taking and echo cancellation so the agent never hears itself.
|
||||
Opt-in, heavier "Conversation" mode. Composes components OmniVoice already ships
|
||||
(sub-second streaming ASR, sentence-chunked streaming TTS, an NLMS echo
|
||||
canceller, an OpenAI-compatible local LLM adapter, far-end audio bus) rather
|
||||
than introducing a parallel stack.
|
||||
|
||||
---
|
||||
|
||||
## Context & Problem
|
||||
|
||||
### The category gap
|
||||
|
||||
Voice **agents** are the product ElevenLabs (Conversational AI), OpenAI (Realtime
|
||||
API), and the open-source frameworks (LiveKit Agents, Pipecat, Ten) are all
|
||||
racing on. The defining UX is *full-duplex conversation*: you talk, it answers in
|
||||
~250–450 ms, and you can **interrupt it mid-sentence** and it stops and listens.
|
||||
Every production stack today is **cloud-tethered** — the STT, the LLM, and often
|
||||
the TTS are remote API calls, which means an account, an API key, per-minute
|
||||
billing, and your microphone audio leaving the machine.
|
||||
|
||||
### Why OmniVoice is uniquely positioned
|
||||
|
||||
OmniVoice already has **every pipeline stage** of a voice agent, running locally,
|
||||
and they were each built (and hardened) for the live-dictation feature that just
|
||||
landed:
|
||||
|
||||
| Agent stage | Already in the codebase | Path |
|
||||
|---|---|---|
|
||||
| Streaming STT with endpointing | sherpa-onnx `OnlineRecognizer`, frame-by-frame decode, `is_endpoint()` turn detection, <300 ms perceived latency on CPU | `backend/api/routers/capture_ws.py:414-533` (`_run_sherpa_streaming`), `backend/services/sherpa_dictation.py:275-316` |
|
||||
| Echo cancellation (anti-self-trigger) | `NlmsEchoCanceller` with Geigel double-talk detector, server-side so it's platform-identical | `backend/services/aec.py:75-282` |
|
||||
| Far-end reference plumbing | publish/subscribe far-end bus + playback tap worklet feeding the AEC reference frame | `frontend/src/utils/aec/farEndBus.js`, `playbackTap.js`, `public/aec-worklet.js` |
|
||||
| Streaming TTS | `/ws/tts` sentence-chunked synthesis, <100 ms TTFA target, conversational keep-open socket | `backend/api/routers/tts_stream.py:54-272` |
|
||||
| Local LLM "brain" | OpenAI-compat adapter (Ollama / LM Studio / llama.cpp server), structured chat-messages surface | `backend/services/llm_backend.py:66-141` |
|
||||
| Tool surface | FastMCP server (`generate_speech`, `list_voices`, `transcribe`, …) | `backend/mcp_server.py:101-246` |
|
||||
|
||||
No competitor can offer **"voice agent, zero cloud, your voice never leaves the
|
||||
box, runs on a CPU laptop."** OmniVoice can, because the parts are already here
|
||||
and already cross-platform. This spec wires them into one full-duplex loop.
|
||||
|
||||
### The problem this solves for users
|
||||
|
||||
Today a user can *dictate* to OmniVoice and *generate speech* from OmniVoice, but
|
||||
the two are disconnected. They cannot **talk to** it. The asks already arriving in
|
||||
Issues/Discord — "local Alexa", "offline ChatGPT voice mode", "talk to my docs
|
||||
without an API key" — all reduce to the same missing primitive: a turn-taking
|
||||
voice loop. That primitive is also the substrate for later agentic features
|
||||
(voice-driven dubbing direction, hands-free batch control via the MCP tools).
|
||||
|
||||
---
|
||||
|
||||
## Goals / Non-goals
|
||||
|
||||
### Goals
|
||||
|
||||
1. **Full-duplex conversation mode** — speak, get a spoken answer in a natural
|
||||
turn gap (target **median end-of-speech → first-audio ≤ 700 ms** on Apple
|
||||
Silicon / discrete GPU; degrade gracefully on CPU, see Phasing).
|
||||
2. **Barge-in** — the user can interrupt the agent mid-utterance; TTS playback
|
||||
stops within **≤ 200 ms** and the loop returns to listening.
|
||||
3. **No self-trigger** — the agent's own TTS playback, leaking into the mic, must
|
||||
not be transcribed as user speech. Reuse the existing AEC + far-end bus.
|
||||
4. **Fully local & opt-in** — no cloud, no keys, no accounts. Off by default;
|
||||
one Settings toggle turns it on. Functions with reporting/telemetry disabled.
|
||||
5. **Cross-platform parity** — identical default behavior on macOS / Windows /
|
||||
Linux. Any platform-only optimization is opt-in.
|
||||
6. **CPU-capable** — usable (if slower) on a CPU-only machine with a small quant
|
||||
LLM and a CPU-realtime TTS engine; never a hard GPU requirement.
|
||||
7. **Conversation persistence** — turn history kept across a session and
|
||||
resumable, with an additive alembic migration and no migration of existing
|
||||
`omnivoice_data/`.
|
||||
8. **Engine back-compat** — no change to on-disk engine/model state; existing
|
||||
IndexTTS/CosyVoice/etc. installs are untouched.
|
||||
|
||||
### Non-goals
|
||||
|
||||
- **Bundling an LLM in the installer.** We *standardize on* a local runtime and
|
||||
guide the user to install a model (one-click where possible), but the ~1–4 GB
|
||||
weights are a first-use download, not installer payload (mirrors the existing
|
||||
TTS model-on-first-use pattern).
|
||||
- **A new TTS engine.** Real-time uses the engines already present (KittenTTS /
|
||||
MOSS-TTS-Nano / Kokoro-via-MLX); no sample-level streaming engine is added.
|
||||
- **Telephony / SIP / multi-party.** Single local user, one mic, one speaker.
|
||||
- **Sample-level (sub-sentence) TTS streaming.** Sentence-chunked streaming is
|
||||
the latency mechanism; sub-sentence is an open question, not a v0.3.x goal.
|
||||
- **Cloud LLM as a default.** Cloud OpenAI-compat endpoints remain *possible*
|
||||
(the adapter already supports them) but stay opt-in and never the default.
|
||||
- **Wake-word / always-listening.** Mode is explicitly entered; no background
|
||||
hot-mic.
|
||||
|
||||
---
|
||||
|
||||
## User Experience
|
||||
|
||||
### Entering the mode
|
||||
|
||||
- **Opt-in gate.** Settings → *Conversation (beta)* toggle (`prefs` key
|
||||
`conversation.enabled`, default `false`). While off, nothing in the loop loads
|
||||
and no new socket opens — zero footprint, identical to today on every platform.
|
||||
- A new left-nav entry **"Talk"** appears only when the toggle is on. First entry
|
||||
runs a **readiness check**: is a local LLM reachable (`llm_backend.is_available()`),
|
||||
is a streaming sherpa ASR model installed, is a real-time-capable TTS engine
|
||||
selected? Any miss shows an inline, actionable card (the project's house error
|
||||
style) — e.g. *"No local LLM detected. Install Ollama and pull `llama3.2:3b`,
|
||||
then click Recheck"* with a copy-paste command per OS. Nothing auto-installs.
|
||||
|
||||
### The conversation screen
|
||||
|
||||
A single focused view:
|
||||
|
||||
- **Big mic orb** at center with four visible states: *Idle* → *Listening*
|
||||
(waveform reacts to mic) → *Thinking* (LLM streaming) → *Speaking* (orb pulses
|
||||
with TTS playback). State transitions are the user's mental model of whose turn
|
||||
it is.
|
||||
- **Live transcript rail** — the user's partial ASR text appears as they speak
|
||||
(greyed, italic), commits on endpoint, then the agent's reply streams in token
|
||||
by token as it's generated, with a speaker label and the **voice profile** the
|
||||
agent is using (any saved clone/design voice — reuse the profile picker).
|
||||
- **Barge-in affordance** — while *Speaking*, a subtle "interrupt anytime" hint;
|
||||
starting to talk visibly cuts the agent off (orb snaps Listening, the agent's
|
||||
half-spoken line is marked *(interrupted)* in the rail).
|
||||
- **Controls** — push-to-talk vs. open-mic toggle (open-mic is VAD-gated;
|
||||
push-to-talk is the CPU-friendly / noisy-room fallback), voice picker, LLM
|
||||
model indicator, *End conversation* (persists + closes the session), mute.
|
||||
- **System prompt / persona** — a small "Agent persona" field (persisted per
|
||||
conversation) so the user can set behavior ("You are a terse coding helper").
|
||||
Defaults to a neutral, concise assistant prompt.
|
||||
|
||||
### Core flow (happy path)
|
||||
|
||||
1. User clicks **Talk**, mode initializes (warm the ASR recognizer, TTS model,
|
||||
and confirm LLM reachable — show a one-time spinner).
|
||||
2. User speaks. Partial transcript streams (existing `partial` frames). On
|
||||
`is_endpoint()` (trailing-silence turn detection) the utterance commits.
|
||||
3. The committed user turn (+ short rolling history + persona system prompt) is
|
||||
sent to the local LLM, which **streams tokens**.
|
||||
4. Tokens feed the existing `SentenceChunker`; each completed sentence is handed
|
||||
to streaming TTS the moment it's ready (first sentence starts speaking while
|
||||
the LLM is still generating the rest — the core latency trick).
|
||||
5. TTS audio plays; **every playback frame is published to the far-end bus** and
|
||||
sent to the ASR socket as an AEC reference (tag `0x01`) so the agent doesn't
|
||||
transcribe itself.
|
||||
6. The mic stays open (open-mic mode): if the user starts talking (VAD speech +
|
||||
AEC-cleaned energy over threshold for the barge-in window) → **barge-in**:
|
||||
cancel the LLM stream, flush the TTS queue, stop playback, return to step 2.
|
||||
7. On *End conversation*, the turn history is persisted and the sockets close.
|
||||
|
||||
### Degraded / edge flows
|
||||
|
||||
- **Weak hardware:** if warm-up profiling predicts response latency over a
|
||||
threshold, the UI suggests **push-to-talk + half-duplex** (no barge-in) and a
|
||||
smaller LLM/TTS, but still works.
|
||||
- **No LLM:** mode is unavailable with the actionable install card; the rest of
|
||||
the app is untouched.
|
||||
- **Noisy room / open-mic false triggers:** a sensitivity slider and a
|
||||
push-to-talk escape hatch; barge-in defaults conservative to avoid the agent
|
||||
interrupting itself on its own echo tail.
|
||||
|
||||
---
|
||||
|
||||
## Technical Design
|
||||
|
||||
### The full-duplex pipeline
|
||||
|
||||
```
|
||||
mic ──worklet──► PCM16 frames ──┐
|
||||
│ (tag 0x00 near-end)
|
||||
TTS playback ──playbackTap──► far-end bus ──► PCM16 (tag 0x01 far-end)
|
||||
│
|
||||
▼
|
||||
┌────────────── /ws/converse (NEW orchestration socket) ──────────────┐
|
||||
│ │
|
||||
│ NlmsEchoCanceller.process_near_end() ── clean mic ──► OnlineRecognizer│
|
||||
│ (services/aec.py) (sherpa streaming) │
|
||||
│ │ partial/final│
|
||||
│ is_endpoint() → TURN │
|
||||
│ ▼ │
|
||||
│ ConversationSession (NEW) ── history + persona ──► │
|
||||
│ │ │
|
||||
│ ▼ streaming chat │
|
||||
│ llm_backend.chat_messages_stream() (NEW streaming surface) │
|
||||
│ │ tokens │
|
||||
│ ▼ │
|
||||
│ SentenceChunker.push() ── sentence ──► TTS generate │
|
||||
│ │ (services/tts_backend) │
|
||||
│ ▼ PCM16 chunks │
|
||||
│ ◄── audio frames back to client ──► (barge-in cancels) │
|
||||
└─────────────────────────────────────────────────────────────────────────┘
|
||||
```
|
||||
|
||||
The whole loop is one **server-side orchestrator** so turn-state, barge-in
|
||||
cancellation, and history live in one place rather than being coordinated across
|
||||
three independent client sockets. The client streams mic+reference PCM up and
|
||||
receives transcript/state/audio frames down — one connection.
|
||||
|
||||
### Files/services to add or extend
|
||||
|
||||
**New — backend**
|
||||
|
||||
- `backend/api/routers/converse_ws.py` — the `/ws/converse` orchestration
|
||||
endpoint. Models on `capture_ws.py`'s loopback guard + `ws_remote_authorized`
|
||||
pattern (`capture_ws.py:139-142`), the AEC-tagged PCM transport
|
||||
(`_demux_aec_frame`, `_recv_pcm_frame` at `capture_ws.py:62-73, 383-411`), and
|
||||
the sherpa streaming decode loop (`capture_ws.py:457-497`). Owns the
|
||||
per-session `ConversationSession` and the cancellation token.
|
||||
- `backend/services/conversation.py` — `ConversationSession`: holds the rolling
|
||||
message list (system persona + last *N* turns, token-budgeted), drives one
|
||||
turn (ASR-final → LLM stream → sentence-chunk → TTS), and exposes an
|
||||
`asyncio.Event`-based **interrupt** that barge-in trips to cancel the in-flight
|
||||
LLM generation + drain the TTS queue. Persists turns via the new store.
|
||||
- `backend/services/conversation_store.py` — CRUD over the new `conversations`
|
||||
and `conversation_turns` tables (below). Thin, mirrors `mcp_bindings.py`.
|
||||
- `backend/services/vad.py` — Silero-VAD wrapper (ONNX, CPU, ~1 MB) for
|
||||
**barge-in detection** specifically: scores AEC-*cleaned* mic frames while the
|
||||
agent is speaking, so the agent's own echo tail can't trip it. Endpointing of
|
||||
the *user's* turn stays with sherpa's `is_endpoint()` (already tuned, rule1
|
||||
2.4 s / rule2 1.2 s trailing silence — `sherpa_dictation.py:298-301`); VAD is a
|
||||
fast speech-onset gate, not a replacement for endpointing.
|
||||
|
||||
**Extend — backend**
|
||||
|
||||
- `backend/services/llm_backend.py` — add `chat_messages_stream(messages, …)`
|
||||
yielding token deltas. The `openai` client already supports `stream=True`;
|
||||
this is an additive surface alongside the existing one-shot `chat_messages`
|
||||
(`llm_backend.py:123-141`). `OffBackend` raises the same clear error.
|
||||
- `backend/services/tts_backend.py` — reuse `get_active_tts_backend` /
|
||||
`generate` unchanged; add a thin per-sentence helper that the session calls so
|
||||
TTS runs in the GPU/CPU pool exactly as `tts_stream.py:188-209` does today.
|
||||
- `backend/core/prefs.py` — new `conversation.*` keys (enabled, llm_model,
|
||||
tts_engine, mode `open-mic|push-to-talk`, vad_sensitivity, persona_default).
|
||||
Mirrors the existing `dictation.*` namespace and rebuild-on-change pattern
|
||||
(`api/routers/dictation.py:99-127`).
|
||||
|
||||
**New — frontend**
|
||||
|
||||
- `frontend/src/components/Conversation/ConversationView.jsx` — the screen.
|
||||
Reuses `startMicCapture` (`utils/aec/micCapture.js`), `frameFromFloat` +
|
||||
`AEC_NEAR`/`AEC_FAR` tags (`utils/aec/pcm.js`), `subscribeFarEnd`
|
||||
(`utils/aec/farEndBus.js`), and the voice/profile picker.
|
||||
- `frontend/src/utils/conversationSocket.js` — opens `/ws/converse?aec=1&sr=16000
|
||||
&model=<sherpa>`, multiplexes: uploads tagged mic + far-end PCM, receives
|
||||
`partial`/`final`/`token`/`state`/audio-bytes/`done` frames.
|
||||
- `frontend/src/utils/conversationPlayer.js` — a **gapless PCM16 queue player**
|
||||
(Web Audio `AudioBufferSourceNode` scheduling) that (a) plays streamed agent
|
||||
audio with minimal gaps between sentences and (b) **publishes each played frame
|
||||
to `publishFarEnd()`** so it becomes the AEC reference — closing the
|
||||
anti-self-trigger loop. Exposes `flush()` for instant barge-in stop. This is
|
||||
the one genuinely new client primitive (today's TTS path buffers a whole WAV
|
||||
then plays via `playBlobAudio`; a conversation needs incremental, interruptible
|
||||
playback).
|
||||
|
||||
### Per-stage latency budget
|
||||
|
||||
Production voice agents target a **200–450 ms** end-of-user-speech → first-audio
|
||||
gap (human turn-taking rhythm), and **< 200 ms** barge-in stop
|
||||
([LiveKit](https://livekit.com/blog/turn-detection-voice-agents-vad-endpointing-model-based-detection),
|
||||
[FutureAGI](https://futureagi.com/blog/voice-ai-barge-in-turn-taking-2026/)).
|
||||
We split the gap as follows. Two budgets: a **GPU/Apple-Silicon** target and an
|
||||
honest **CPU-only** reality.
|
||||
|
||||
| Stage | What | GPU/MPS target | CPU-only realistic |
|
||||
|---|---|---|---|
|
||||
| Endpoint detection | sherpa `is_endpoint()` trailing-silence commit | ~150–250 ms (silence rule, inherent) | same |
|
||||
| ASR finalize | drain stream for committed text (already decoded incrementally) | < 30 ms | < 80 ms |
|
||||
| LLM TTFT | first token from local model | 80–250 ms (3B 4-bit) | 300–600 ms (1–3B) |
|
||||
| First sentence ready | enough tokens for `SentenceChunker` first emit (aggressive first-clause flush, `sentence_chunker.py:475-555`) | +50–150 ms | +150–400 ms |
|
||||
| TTS TTFA | synth first sentence, first PCM chunk out | 80–200 ms (Kokoro/Kitten) | 150–400 ms (Kitten/MOSS-Nano) |
|
||||
| Playback startup | queue player schedules first buffer | < 30 ms | < 30 ms |
|
||||
| **Perceived gap** | end-of-speech → first audio | **≈ 450–750 ms** | **≈ 1.0–1.9 s** |
|
||||
| **Barge-in stop** | VAD onset → playback flush + LLM cancel | **< 200 ms** | < 250 ms |
|
||||
|
||||
Notes grounding the numbers:
|
||||
- Local LLM TTFT for 1–3B models is the long pole on CPU; small-model streaming
|
||||
runs ~9–14 ms/token with sub-500 ms TTFT on modern hardware, slower on old CPUs
|
||||
([daily.dev](https://daily.dev/blog/running-llms-locally-ollama-llama-cpp-self-hosted-ai-developers/),
|
||||
[quantizelab](https://www.quantizelab.dev/articles/vllm-vs-llama-cpp-vs-ollama-benchmark-guide)).
|
||||
The CPU path is *usable*, not snappy — hence push-to-talk + half-duplex on weak
|
||||
hardware, set honestly by warm-up profiling rather than hidden.
|
||||
- The **sentence-chunk overlap is the core trick**: the first sentence speaks
|
||||
while the LLM finishes the rest, so perceived latency is *first-sentence*
|
||||
latency, not whole-response latency. `SentenceChunker`'s aggressive-first-flush
|
||||
(emit first clause at ≥ 40 chars on a comma/dash) already exists to shave
|
||||
200–500 ms off TTFA.
|
||||
- The **endpoint silence rule is itself ~1.2–2.4 s** in the current dictation
|
||||
tuning, which is too slow for snappy conversation. The session will run a
|
||||
**conversation-tuned endpoint profile** (shorter `rule2` trailing silence, e.g.
|
||||
~0.6–0.8 s) configured at recognizer build time — a new spec on
|
||||
`sherpa_dictation.py`'s online builder, *not* a change to the dictation
|
||||
defaults (back-compat).
|
||||
|
||||
### Barge-in + AEC handling
|
||||
|
||||
This is the make-or-break of full-duplex, and the existing AEC plumbing is what
|
||||
makes it tractable locally and identically cross-platform.
|
||||
|
||||
1. **Reference path.** Every agent-audio frame the `conversationPlayer` schedules
|
||||
is also `publishFarEnd()`-ed; `conversationSocket` subscribes and sends it up
|
||||
tagged `0x01`. Server-side, `converse_ws` feeds it to
|
||||
`NlmsEchoCanceller.push_far_end()` and cleans the mic with
|
||||
`process_near_end()` before *either* ASR or VAD sees it
|
||||
(`aec.py:153-207`). The canceller already passes-through when the far-end is
|
||||
stale (`aec.py:_FAR_STALE_S`), so it won't buzz once the agent stops talking.
|
||||
2. **Onset detection.** While state == *Speaking*, the Silero VAD scores the
|
||||
**cleaned** mic frames. Sustained speech for a short window (e.g. ≥ 120–200 ms,
|
||||
`vad_sensitivity`-tunable) = barge-in. Using cleaned audio + a sustain window
|
||||
is what prevents the agent's residual echo from self-interrupting (the classic
|
||||
"agent talks over itself" bug).
|
||||
3. **Cancellation.** On barge-in the session: trips the interrupt `Event` →
|
||||
the LLM stream generator is cancelled (stop pulling tokens, the `openai`
|
||||
stream is closed), the pending-sentence TTS queue is dropped, a `state:
|
||||
listening` + `interrupted` frame is sent, and the client `conversationPlayer.flush()`
|
||||
stops playback **immediately** (Web Audio `stop()` on scheduled sources). The
|
||||
committed-so-far agent text is saved as a partial turn.
|
||||
4. **Turn handoff.** The recognizer stream is `reset()` (as in
|
||||
`capture_ws.py:493`) and the user's new utterance is decoded fresh.
|
||||
|
||||
Server-side AEC was a deliberate cross-platform-parity choice for dictation
|
||||
(`aec.py:1-27`: browser `echoCancellation` quality/availability varies per
|
||||
webview, which would make a *default* behave differently per OS). The same
|
||||
reasoning applies — and is now load-bearing — for the agent.
|
||||
|
||||
### LLM runtime choice (with alternatives)
|
||||
|
||||
**Standardize on the existing OpenAI-compatible adapter pointed at a local
|
||||
server — recommend Ollama as the default local runtime, llama.cpp's
|
||||
`llama-server` as the power-user equal.** Rationale:
|
||||
|
||||
- **Zero new code path.** `llm_backend.OpenAICompatBackend` already speaks this
|
||||
shape and already names Ollama (`http://localhost:11434/v1`) and LM Studio as
|
||||
first-class (`llm_backend.py:66-90`). Adding streaming is one additive method.
|
||||
- **Local-first & cross-platform.** Ollama ships for macOS/Windows/Linux, runs
|
||||
CPU or GPU, auto-detects, streams over SSE with consistent inter-token latency
|
||||
— same default behavior everywhere, which the parity rule demands.
|
||||
- **No weights in our installer.** Model is a guided first-use pull (e.g.
|
||||
`ollama pull llama3.2:3b`), matching how OmniVoice already does models.
|
||||
- **Recommended default model:** a small instruct model (~3B, 4-bit) for the GPU
|
||||
path; a ~1–1.5B for the CPU path. Selectable in Settings; we ship *guidance*,
|
||||
not weights.
|
||||
|
||||
Alternatives considered:
|
||||
|
||||
| Option | When to prefer | Why not the default |
|
||||
|---|---|---|
|
||||
| **llama.cpp `llama-server`** (OpenAI-compat) | Power users wanting GGUF control / no Ollama daemon; lowest TTFT single-user | Same OpenAI-compat surface — *fully supported* via base-url, just less turn-key to install than Ollama. Documented as the equal alternative. |
|
||||
| **In-process `llama-cpp-python`** | Eliminate the localhost hop, bundle-friendlier | Adds a native build dep per platform (the very cross-platform fragility we avoid elsewhere); the localhost hop costs < 5 ms. Revisit only if a "no external daemon" install becomes a top ask. |
|
||||
| **`transformers` in-process** | Reuse the Python env | Heavy load, weak streaming ergonomics, GPU-memory contention with TTS in the single-worker `_gpu_pool` (`model_manager.py:71-104`). Wrong tool for low-latency chat. |
|
||||
| **Cloud OpenAI-compat** | User explicitly opts in | Violates the local-first default; allowed but never default, never required. |
|
||||
|
||||
### Concurrency reality
|
||||
|
||||
`model_manager._gpu_pool` is **1 worker on MPS/CPU and budget-limited on CUDA**
|
||||
(`model_manager.py:71-104`) — ASR, TTS, and a `transformers` LLM would *serialize*
|
||||
on one GPU. Standardizing the LLM on a **separate local server process** (Ollama /
|
||||
llama-server) sidesteps this entirely: the LLM runs in its own process/accelerator
|
||||
context, ASR runs on its sherpa CPU/ONNX path, and TTS uses the existing pool —
|
||||
three independent lanes, which is exactly what overlapping the pipeline stages
|
||||
requires. For CPU-only, choosing a **CPU-realtime TTS** (KittenTTS English /
|
||||
MOSS-TTS-Nano multilingual / Kokoro-via-MLX on Apple) keeps TTS off the LLM's
|
||||
cores enough to stay usable.
|
||||
|
||||
---
|
||||
|
||||
## API / Schema / Data-model changes
|
||||
|
||||
### WebSocket protocol — `/ws/converse`
|
||||
|
||||
Loopback-guarded (or `OMNIVOICE_API_KEY` bearer for the thin-client case), exactly
|
||||
like `/ws/transcribe`. Query: `?aec=1&sr=16000&model=<sherpa_id>&conversation=<id?>`.
|
||||
|
||||
**Client → server**
|
||||
- Binary frames: tagged PCM16 mono, `0x00` near-end (mic), `0x01` far-end
|
||||
(agent-playback reference) — identical framing to the AEC dictation transport.
|
||||
- JSON control frames:
|
||||
- `{"type":"start","persona":"...","voice":"<profile_id>","llm_model":"...","mode":"open-mic|push-to-talk"}`
|
||||
- `{"type":"barge_in"}` — explicit interrupt (push-to-talk re-key / UI button); server also detects barge-in via VAD autonomously.
|
||||
- `{"type":"end"}` — persist + close.
|
||||
|
||||
**Server → client**
|
||||
- `{"type":"state","value":"idle|listening|thinking|speaking"}`
|
||||
- `{"type":"partial","text":"..."}` — user ASR interim (reused frame shape).
|
||||
- `{"type":"final","text":"...","role":"user"}` — committed user turn.
|
||||
- `{"type":"token","text":"...","role":"assistant"}` — streamed LLM delta.
|
||||
- `{"type":"start_audio","sample_rate":N,"format":"pcm16","engine":"..."}` then
|
||||
binary PCM16 chunks (mirrors `tts_stream.py`'s `start` + bytes contract).
|
||||
- `{"type":"interrupted","spoken_text":"..."}` — barge-in fired; partial agent turn.
|
||||
- `{"type":"turn_done","turn_id":N}` / `{"type":"error","detail":"..."}`.
|
||||
|
||||
### REST endpoints (loopback-gated, Settings UI)
|
||||
|
||||
- `GET/POST /conversation/prefs` — read/write the `conversation.*` prefs +
|
||||
readiness status (LLM reachable, ASR model installed, TTS engine real-time).
|
||||
Mirrors `dictation.py` prefs router.
|
||||
- `GET /conversations` — list saved conversations (id, title, started_at, turn_count).
|
||||
- `GET /conversations/{id}` — full turn history.
|
||||
- `DELETE /conversations/{id}` — delete one.
|
||||
|
||||
### Persistence — additive alembic migration
|
||||
|
||||
New migration `0008_conversations.py` (next after `0007_*`), additive only, no
|
||||
backfill, existing `omnivoice_data/` untouched:
|
||||
|
||||
```sql
|
||||
CREATE TABLE conversations (
|
||||
id TEXT PRIMARY KEY, -- uuid
|
||||
title TEXT, -- first user turn, truncated
|
||||
persona TEXT, -- system prompt for the session
|
||||
voice_profile TEXT, -- profile_id the agent speaks with
|
||||
llm_model TEXT, -- model id used
|
||||
created_at REAL NOT NULL,
|
||||
updated_at REAL NOT NULL
|
||||
);
|
||||
CREATE TABLE conversation_turns (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
conversation_id TEXT NOT NULL REFERENCES conversations(id) ON DELETE CASCADE,
|
||||
role TEXT NOT NULL, -- 'user' | 'assistant'
|
||||
text TEXT NOT NULL,
|
||||
interrupted INTEGER NOT NULL DEFAULT 0,
|
||||
created_at REAL NOT NULL
|
||||
);
|
||||
CREATE INDEX ix_turns_conversation ON conversation_turns(conversation_id, id);
|
||||
```
|
||||
|
||||
**Privacy:** transcripts are stored locally only (same trust boundary as
|
||||
existing history). **No audio is persisted** — only text turns. The auto
|
||||
bug-reporter must never attach conversation transcripts (extend its scrub
|
||||
allow/deny exactly as it strips reference audio today).
|
||||
|
||||
### Prefs (`core/prefs.py`, JSON store)
|
||||
|
||||
`conversation.enabled` (bool, default false), `conversation.mode`
|
||||
(`open-mic|push-to-talk`, default `push-to-talk` for the safe first run),
|
||||
`conversation.llm_model`, `conversation.tts_engine`,
|
||||
`conversation.vad_sensitivity` (float), `conversation.endpoint_profile`
|
||||
(`conversation|dictation`), `conversation.persona_default` (str). Env overrides
|
||||
follow the existing `prefs.resolve` precedence.
|
||||
|
||||
---
|
||||
|
||||
## Local-first & Cross-platform compliance
|
||||
|
||||
- **No cloud, no keys, no accounts.** STT (sherpa ONNX), VAD (Silero ONNX), TTS
|
||||
(local engines), AEC (server NLMS) all run on-device. The LLM runs on a
|
||||
**local** server (Ollama/llama-server) by default. A cloud OpenAI-compat
|
||||
endpoint is only reachable if the user explicitly configures one — never the
|
||||
default, never required.
|
||||
- **Opt-in by definition.** Off until the Settings toggle; while off, no socket,
|
||||
no model load, no nav entry — the app is byte-for-byte today's behavior on
|
||||
every platform. This satisfies the strict opt-in rule for a heavy new mode.
|
||||
- **Identical default behavior on mac/win/linux.** Server-side AEC was *chosen*
|
||||
over browser `echoCancellation` precisely so the default behaves the same on
|
||||
every webview (`aec.py:1-27`); the agent inherits that. The mic worklet, PCM
|
||||
framing, sherpa decode, sentence chunker, and queue player are platform-neutral
|
||||
JS/Python. The one platform-specific *implementation* allowance is the
|
||||
Apple-only Kokoro-via-MLX TTS fast path — and it's behind the engine picker
|
||||
(opt-in), with a cross-platform default (KittenTTS/MOSS-Nano) so the
|
||||
*user-visible default* never diverges. No P0 platform gap.
|
||||
- **CPU-capable.** A 1–1.5B quant LLM + a CPU-realtime TTS + the CPU sherpa ASR +
|
||||
the numpy NLMS AEC is the documented CPU path. Slower turns, push-to-talk +
|
||||
half-duplex by default on weak hardware (set by warm-up profiling), but
|
||||
functional — no hard GPU requirement.
|
||||
- **Engine back-compat.** Reuses `get_active_tts_backend` and the LLM adapter as-is;
|
||||
no on-disk engine/model state changes; existing installs untouched.
|
||||
- **Docs-sync.** Lands with a `docs/` page (setup: install Ollama, pull a model,
|
||||
pick a voice; the CPU vs GPU expectation table) **in the same PR** as the
|
||||
feature, per the docs-sync rule. README feature grid updated.
|
||||
|
||||
---
|
||||
|
||||
## Phasing (sliceable on the v0.3.x line)
|
||||
|
||||
Each phase is an independently-mergeable, bisectable PR (or small cluster) with
|
||||
tests in the same PR, continuous-to-main — no RC, no version bump beyond the
|
||||
standing patch. Value lands incrementally.
|
||||
|
||||
- **P0 — Streaming LLM surface.** Add `chat_messages_stream()` to
|
||||
`llm_backend.py` (+ `OffBackend` parity, + tests). Independently useful
|
||||
(dictation refinement could stream later). *No UI.*
|
||||
- **P1 — Half-duplex conversation (the spine).** `/ws/converse` +
|
||||
`ConversationSession` wiring ASR-final → LLM-stream → sentence-chunk → TTS →
|
||||
client queue player. **Push-to-talk, no barge-in, no AEC reference loop yet**
|
||||
(user holds to talk, releases, listens to the full answer). `ConversationView`
|
||||
with the orb + transcript rail. This alone is a shippable "talk to your local
|
||||
LLM, get a spoken answer" feature.
|
||||
- **P2 — Persistence.** `0008` migration + `conversation_store` + history list /
|
||||
resume + the `GET/DELETE /conversations` endpoints. Bug-reporter scrub guard.
|
||||
- **P3 — AEC reference loop.** `conversationPlayer` publishes far-end frames;
|
||||
`converse_ws` cancels echo via the existing `NlmsEchoCanceller`. Enables
|
||||
**open-mic** safely (agent stops self-transcribing). Still no interruption.
|
||||
- **P4 — Barge-in.** `services/vad.py` (Silero) on cleaned mic during *Speaking*,
|
||||
interrupt `Event`, LLM-stream cancel, TTS-queue flush, client `flush()`. This
|
||||
is the full-duplex payoff. Conversation-tuned endpoint profile lands here.
|
||||
- **P5 — Graceful degradation + polish.** Warm-up latency profiling →
|
||||
auto-suggest push-to-talk/half-duplex + smaller models on weak hardware;
|
||||
sensitivity tuning; persona presets; readiness-card install guidance per OS;
|
||||
docs page + README.
|
||||
|
||||
---
|
||||
|
||||
## Testing strategy
|
||||
|
||||
- **Unit (backend):**
|
||||
- `chat_messages_stream` yields deltas, cancels cleanly on interrupt, `OffBackend`
|
||||
raises (mock the `openai` stream).
|
||||
- `ConversationSession` turn lifecycle: final→tokens→sentences→tts calls in
|
||||
order; interrupt `Event` cancels mid-stream and saves the partial turn.
|
||||
- `conversation_store` CRUD; `0008` migration **up-then-down** on a copy of a
|
||||
real `omnivoice_data/` DB (back-compat: existing tables untouched).
|
||||
- VAD barge-in gate: synthetic cleaned-mic frames with/without speech onset →
|
||||
fires only on sustained speech, **never** on a far-end echo fixture (the
|
||||
self-interrupt regression — a fail-before/pass-after test per the fix-quality
|
||||
rule).
|
||||
- Conversation-tuned endpoint profile builds without touching dictation defaults.
|
||||
- **Integration (backend):** a fake LLM (deterministic token stream) + a fake
|
||||
fast TTS through real `/ws/converse`; assert frame ordering
|
||||
(`state`/`partial`/`final`/`token`/audio/`turn_done`) and that a `barge_in`
|
||||
frame mid-speech produces `interrupted` + returns to `listening`.
|
||||
- **Frontend (vitest):** `conversationPlayer` gapless scheduling + instant
|
||||
`flush()`; far-end publish on each played frame; `conversationSocket` framing
|
||||
(tags `0x00`/`0x01`, control frames). Reuse the existing AEC PCM test patterns
|
||||
(`frontend/src/test/aecPcm.test.js`, `aecFarEndBus.test.js`).
|
||||
- **Latency harness (non-gating, like the eval tier):** a scripted turn measures
|
||||
endpoint→TTFT→TTFA→first-audio on the CI box and on a CPU-only profile, logging
|
||||
the budget table so regressions are visible. Not a hard gate (hardware-variable)
|
||||
but tracked.
|
||||
- **Cross-platform / full-matrix green:** no `frontend/package.json` dep churn
|
||||
expected beyond a tiny Silero ONNX asset (verify root `bun.lock` regen +
|
||||
`bun install --frozen-lockfile` for Docker), `uv tree` clean after adding the
|
||||
Silero/onnxruntime path (onnxruntime already transitively present), Tauri cargo
|
||||
build unaffected (no Rust change). CodeQL/security re-run on the new socket.
|
||||
- **Off-by-default proof:** a test that with `conversation.enabled=false`, no
|
||||
conversation route/socket is reachable and no model loads — the opt-in guarantee.
|
||||
|
||||
## Risks & mitigations
|
||||
|
||||
| Risk | Likelihood | Mitigation |
|
||||
|---|---|---|
|
||||
| **CPU latency feels sluggish** (LLM TTFT is the long pole). | High on old CPUs | Honest warm-up profiling → default to push-to-talk + half-duplex + smaller model; sentence-overlap so perceived latency is first-sentence; never advertise sub-second on CPU. |
|
||||
| **Self-trigger / feedback loop** (agent transcribes itself, or interrupts itself). | High without care | Server AEC cleans mic *before* ASR+VAD; barge-in scores **cleaned** audio with a sustain window; far-end-stale pass-through already handled (`aec.py`). The dedicated VAD-vs-echo regression test gates this. |
|
||||
| **NLMS AEC is "good-enough," not WebRTC AES3** (`aec.py:14-18`) — residual echo on loud speakers. | Medium | Conservative barge-in sensitivity default; recommend headphones in the readiness card; sustain window; optional future upgrade to a stronger canceller is isolated behind the `aec.py` interface. |
|
||||
| **GPU contention** (LLM + TTS on one accelerator). | Medium | Standardize LLM on a **separate process** (Ollama/llama-server), keeping it off the single-worker `_gpu_pool`; CPU-realtime TTS option. |
|
||||
| **Endpoint silence too slow → laggy turns** (dictation tuning is 1.2–2.4 s). | Medium | Conversation-specific endpoint profile (shorter trailing silence) built at recognizer init; *dictation defaults unchanged* (back-compat). |
|
||||
| **User has no local LLM installed.** | High at launch | Readiness card with copy-paste per-OS install (Ollama) + Recheck; mode simply unavailable until satisfied; rest of app untouched. |
|
||||
| **Open-mic false triggers in noisy rooms.** | Medium | Push-to-talk is the default first-run mode; sensitivity slider; VAD sustain window. |
|
||||
| **Privacy regression via bug reporter.** | Low but serious | No audio persisted; transcripts excluded from auto bug reports by scrub rule + a test asserting it. |
|
||||
|
||||
## Open questions / decisions for the owner
|
||||
|
||||
1. **Default local LLM runtime + model.** Recommend **Ollama + `llama3.2:3b`
|
||||
(GPU) / a ~1–1.5B (CPU)** as the documented default, llama-server as the
|
||||
equal power-user path. Approve, or prefer llama-server-first / a different
|
||||
default model?
|
||||
2. **Default first-run mode.** Spec proposes **push-to-talk** (safe, CPU-kind,
|
||||
no false barge-in) with open-mic as opt-in once P3/P4 land. Agree, or
|
||||
open-mic-first on capable hardware?
|
||||
3. **Ship half-duplex (P1) standalone?** It's a real, useful "talk to your local
|
||||
LLM" feature before barge-in exists. Ship it as soon as it's green, or hold
|
||||
the whole mode until P4?
|
||||
4. **Bundle the Silero VAD ONNX asset** (~1–2 MB) in-repo/installer vs.
|
||||
first-use download? It's tiny and load-bearing for barge-in — leaning bundle,
|
||||
but it's a (small) installer-size decision.
|
||||
5. **Conversation persistence default.** On (resumable history) or off
|
||||
(ephemeral, nothing written) by default? Privacy-conservative would be
|
||||
**ephemeral by default, opt-in to save**.
|
||||
6. **Persona / system-prompt library.** Ship a small preset set (concise
|
||||
assistant, coding helper, tutor) or just a free-text field for v1?
|
||||
7. **MCP tool-calling in v1?** The agent *could* call the FastMCP tools
|
||||
(`generate_speech`, `list_voices`, …) to act on the app by voice. Powerful but
|
||||
adds tool-call orchestration + latency. Recommend **deferring tool-calling to
|
||||
a follow-up spec** and shipping a pure conversational loop first — confirm.
|
||||
|
||||
---
|
||||
|
||||
**Sources (turn-taking, barge-in, local-LLM latency):**
|
||||
[LiveKit — Turn Detection: VAD, Endpointing, Model-Based](https://livekit.com/blog/turn-detection-voice-agents-vad-endpointing-model-based-detection) ·
|
||||
[FutureAGI — Voice AI Barge-In & Turn-Taking 2026](https://futureagi.com/blog/voice-ai-barge-in-turn-taking-2026/) ·
|
||||
[Sparkco — Optimizing Barge-in Detection 2025](https://sparkco.ai/blog/optimizing-voice-agent-barge-in-detection-for-2025) ·
|
||||
[Softcery — Real-Time vs Turn-Based Voice Agents](https://softcery.com/lab/ai-voice-agents-real-time-vs-turn-based-tts-stt-architecture) ·
|
||||
[daily.dev — Running LLMs Locally 2026 (Ollama/llama.cpp)](https://daily.dev/blog/running-llms-locally-ollama-llama-cpp-self-hosted-ai-developers/) ·
|
||||
[QuantizeLab — vLLM vs llama.cpp vs Ollama Benchmarks](https://www.quantizelab.dev/articles/vllm-vs-llama-cpp-vs-ollama-benchmark-guide)
|
||||
@@ -0,0 +1,265 @@
|
||||
# Implementation Spec — 03: Long-form "Studio" Editor (per-segment edit · regenerate-one-line · reassign-voice · emotion/timing)
|
||||
|
||||
> Status: draft for owner review · Target line: **v0.3.x** (continuous-to-main, no RC) · Surfaces: Dub, Audiobook, Stories
|
||||
> Foundation: this is **90% a great editor UX + gap-filling on top of machinery that already exists**. Dubbing already content-addresses segments and regenerates exactly one line; the work is (a) lifting that pattern to the longform (Audiobook/Stories) renderer, which today only caches at *chapter* granularity, and (b) unifying the editor affordances to the ElevenLabs "Studio" bar.
|
||||
|
||||
---
|
||||
|
||||
## Context & Problem
|
||||
|
||||
### The gap vs ElevenLabs Studio / Dubbing Studio
|
||||
|
||||
ElevenLabs has converged its "pro polish" loop into two editors that OmniVoice partially matches and partially does not:
|
||||
|
||||
- **Studio (Projects / Audiobooks)** — paste a manuscript, see chapters laid out, assign different voices per character/paragraph, and make **surgical edits without regenerating everything**: a *Replace voice* pop-up tells you *how many paragraphs* will be re-rendered, and editing one fragment re-renders only that fragment ([Studio overview](https://elevenlabs.io/docs/eleven-creative/products/studio), [change voice across paragraphs](https://help.elevenlabs.io/hc/en-us/articles/23370957112721-How-can-I-change-the-voice-and-settings-across-multiple-paragraphs-in-Studio), [Audiobooks](https://elevenlabs.io/docs/eleven-creative/products/audiobooks)).
|
||||
- **Dubbing Studio** — transcript **and** translation are edited inline in speaker cards; a clip carries a **"stale" badge** when its text/settings/length change; you **regenerate one clip** (refresh icon) or *Generate Stale Audio* in bulk; you **reassign a clip to another speaker** by dragging it to that track; and you adjust **timing** by dragging clip handles / Split / Merge ([Dubbing Studio](https://elevenlabs.io/docs/eleven-creative/products/dubbing/dubbing-studio)).
|
||||
|
||||
OmniVoice today is **asymmetric** across its three longform surfaces:
|
||||
|
||||
| Capability | Dub | Stories | Audiobook |
|
||||
|---|---|---|---|
|
||||
| Per-line text edit | ✅ `DubSegmentRow.jsx` (text 232–252, restore 253–273) | ✅ per-line `<textarea>` (`StoriesEditor.jsx:712–720`) | ❌ one script `<textarea>` only (`AudiobookTab.jsx:227–233`) |
|
||||
| Per-line voice/speaker reassign | ✅ profile `<select>` (288–313) + speaker_id (213–224) | ✅ per-line voice override (734–742) + cast panel | ⚠️ only via inline `[voice:NAME]` markup typed in the blob |
|
||||
| Per-line emotion/direction | ✅ `direction` (split/direction menu 335–383) | ✅ `emotion` tone tags (tune drawer 773–795) | ❌ none |
|
||||
| Per-line timing / fit | ✅ fit strategies + fit badges (`DubSegmentRow.jsx:74–105`) | n/a (longform has no slots) | n/a |
|
||||
| **Stale badge + regenerate ONE line** | ✅ `plan_incremental` + `regen_only` + "Regen changed (N)" (`DubTab.jsx:781–786`) | ❌ no incremental — server **re-streams the whole plan** on every export | ❌ chapter-cache only; editing one span re-keys the **whole chapter** |
|
||||
| Re-stitch / re-mux after a partial edit | ✅ `dub_generate` always rebuilds `dubbed_{lang}.wav`; mux is lazy in `dub_export` | ❌ full re-render | ⚠️ resume reuses *unchanged chapters* but never sub-chapter |
|
||||
|
||||
### What already exists in-repo (the foundation we build on)
|
||||
|
||||
The dub pipeline **already** content-addresses segments so that editing one line re-synthesizes only that line:
|
||||
|
||||
- **`backend/services/incremental.py`** — `segment_fingerprint(seg)` (52–67) is a sha1 over the **generation inputs that actually affect TTS output**: `_GEN_INPUT_FIELDS = ("text","target_lang","profile_id","instruct","speed","direction","effect_preset")` (line 23). `_canon_value` (32–49) normalizes None/""/missing and int↔float so the **server-parsed view** and the **client-raw view** of the same logical segment hash identically — the root-cause fix for #281 ("1 edit re-dubs all N lines"). `fit_fingerprint(params)` (101–116) hashes the **fit configuration separately and on purpose** (70–76): a fit-knob change must trigger a **re-mix** of already-rendered natural-rate WAVs (`regen_only=[]`), never a re-TTS. `plan_incremental(segments, *, stored_hashes)` (119–157) returns `{stale, fresh, total, fingerprints}`.
|
||||
- **`backend/api/routers/dub_generate.py`** — honors `regen_only` (113): for a segment **not** in `regen_only` it reloads the cached `dub_seg_path(job_id, seg_id)` (160–197, with a legacy index-name fallback at 162–166) instead of re-running TTS; for stale segments it runs `_gen(...)`. After the loop it **always re-stitches the full `dubbed_{lang}.wav`** (766–770) and persists `job["seg_hashes"]` (498–512) + `seg_order` (127). The `done` SSE ships `seg_hashes`/`seg_num_step` back (840). Strategy-transition guard (122–123) and `seg_wav_kind` (830) keep smart_fit reuse correct.
|
||||
- **`tests/test_redub_incremental.py`** — already asserts the contract end-to-end with a mocked TTS engine: `test_edited_line_produces_different_cached_output` (228–281) proves an edited line's cached WAV changes, the **untouched line's cached WAV is reused byte-for-byte** (272–273), TTS ran exactly once (266), and the final track was rebuilt (281). `test_one_edit_marks_exactly_one_segment_stale` (101–118) proves the planner.
|
||||
- **Per-segment audio + metadata keying.** Audio lives on disk as `{DUB_DIR}/{job_id}/seg_{seg_id}.wav` via `dub_seg_path(job_id, seg_id)` (`backend/core/config.py:54–71`). Metadata lives in the job's `job_data` JSON blob (`dub_history` table, `backend/core/db.py:74–85`), holding `segments`, `seg_order`, `seg_hashes`, `seg_num_step`, `seg_wav_kind`, `dubbed_tracks`, `fit_plans`, `video_stretch_plans`. Stable ids are minted at transcribe time (`s{NNNNN:05x}`, `dub_core.py:600`). Job persistence is `dub_pipeline.get_job`/`save_job`/`put_job` (`backend/services/dub_pipeline.py:158–214`).
|
||||
- **Longform (Audiobook + Stories) share one renderer** `_render_longform_sse` (`backend/api/routers/audiobook.py:403–595`) and one **chapter-level** content-addressed key `chapter_cache_key` (`backend/services/longform_render.py:110–136`), cached under `OUTPUTS_DIR/longform_cache` (`_render_chapter_cached`, `audiobook.py:314–355`). The lexicon is folded into the key (338–341). Resume (`audiobook.py:714–759`) reuses already-rendered chapters because the key is content-based. **But the granularity is the whole chapter** (longform_render.py:117–125) — that is the gap.
|
||||
|
||||
**Conclusion:** Dub is the reference implementation. Stories already has the *editor UI* but no incremental backend. Audiobook has neither a per-line editor nor sub-chapter incrementality. This spec makes all three behave like a "Studio" by (1) standardizing the editor affordances and (2) pushing the dub-proven `fingerprint → stale → regen_only → re-stitch` loop down into the longform renderer at **span granularity**.
|
||||
|
||||
---
|
||||
|
||||
## Goals / Non-goals
|
||||
|
||||
### Goals
|
||||
1. **Per-segment edit across all three surfaces**: edit a line's **text** (and, for dub, its **translation** independently of the source text), change its **assigned voice/speaker**, set per-segment **emotion/style** (composes with the emotion/style field — see "Composition with emotion/style"), and adjust **timing** (dub only: fit/slot).
|
||||
2. **A visible "stale" badge + "regenerate this one line"** affordance on every surface, mirroring Dubbing Studio's refresh-icon + *Generate Stale Audio*. Editing a line invalidates **exactly one fingerprint** and triggers a **single-segment regen + re-stitch/re-mux**; unchanged lines are cache-hits (reused byte-for-byte).
|
||||
3. **Reuse the existing machinery, don't reinvent it.** Dub keeps `plan_incremental`/`regen_only`. Longform gets a **span-level** twin of `chapter_cache_key` so editing one span re-synthesizes one span, not the chapter.
|
||||
4. **A unified editor *contract*** (stale model, regen verb, voice-reassign affordance, the "N lines will regenerate" preview) shared in concept across surfaces, while respecting each surface's distinct timing model.
|
||||
5. **Zero forced re-render of existing projects.** Existing `omnivoice_data/` dub jobs, story projects, and audiobook renders keep working untouched; new keying degrades gracefully to "treat as stale once" rather than corrupting cached audio.
|
||||
6. **Local-first, cross-platform parity preserved**, docs-sync in the same PR.
|
||||
|
||||
### Non-goals (explicitly deferred)
|
||||
- **A timeline/waveform DAW view.** ElevenLabs reassigns a speaker by dragging a clip between tracks on a timeline; OmniVoice reassigns via the existing per-row `<select>`. No timeline canvas, no clip-drag-to-track, no multi-track audio lanes in this spec. (Split/Merge already exist in `DubSegmentRow.jsx`.)
|
||||
- **New TTS engines or new emotion *models*.** This spec consumes whatever emotion/style field exists; it does not add a new style engine.
|
||||
- **Sub-chapter resume of an *interrupted* render.** Resume stays chapter-granular (`audiobook.py:714–759`); span-level incrementality is for *edits*, not for crash recovery, in this milestone.
|
||||
- **Collaborative / multi-user editing.** Local-first, single-user.
|
||||
- **Audiobook/Stories *slot* timing.** Longform has no per-line time slots (it's narration, not lip-sync); per-segment *timing* is a **dub-only** capability here. Longform "timing" is limited to the existing `[pause]` markers and per-line `speed`.
|
||||
- **Changing the dub fingerprint contract.** `segment_fingerprint`'s field set is stable (back-compat tested in `test_redub_incremental.py:87–98`); emotion/style integration extends it **only if** the field genuinely affects TTS output (see Open Questions Q1).
|
||||
|
||||
---
|
||||
|
||||
## User Experience
|
||||
|
||||
The three surfaces converge on **one mental model — "edit a line, see it go stale, regenerate just that line"** — but keep surface-appropriate controls.
|
||||
|
||||
### Shared "Studio" affordances (all surfaces)
|
||||
- **Stale badge.** A line whose generation inputs changed since its last successful render shows an amber **"changed"** chip (mirrors Dubbing Studio's stale state). Derived from `fingerprint(line) ≠ stored_hash[line.id]` — the dub planner today.
|
||||
- **Regenerate-this-line.** A per-row **↻** button regenerates only that line, then re-stitches. Disabled (no-op) when the line isn't stale.
|
||||
- **Regenerate-changed (bulk).** A header button **"Regenerate changed (N)"** counts stale lines and regenerates exactly those (dub already has this — `DubTab.jsx:781–786`; Stories/Audiobook gain it).
|
||||
- **"This will regenerate N lines" preview.** When a change has fan-out (e.g. reassigning a *cast* voice that M lines inherit, or editing the pronunciation lexicon that affects K chapters), a confirm step states the count before clearing that audio — ElevenLabs' *Replace voice* pop-up pattern.
|
||||
- **Instant preview vs final export.** A line-level **▶ preview** renders that one line quickly (low step count, no watermark/mux) and is *not* persisted; the **final** render/export re-stitches and re-muxes the full artifact. Dub already splits these (`preview_segment`, `dub_generate.py:867–951`; preview is `preview: true`, no disk write, 8 steps).
|
||||
|
||||
### Dub surface (extend, don't rebuild)
|
||||
`DubTab.jsx` + `DubSegmentRow.jsx` already implement nearly all of this. Remaining UX work is **polish + parity**:
|
||||
- **Independent translation edit.** Today text edit + restore-original exists (`DubSegmentRow.jsx:232–273`); make the **transcript (`text_original`)** and the **translation (`text`)** independently editable in the row, with a **↻ re-translate this line** affordance routing to `dub_translate` (`dub_translate.py:222`) for one id, then marking the line stale. (ElevenLabs: edit transcript *or* translation freely.)
|
||||
- **Reassign speaker** keeps the per-row speaker `<select>` (213–224) and the per-row voice override (288–313); changing either flips the fingerprint via `profile_id` and goes stale.
|
||||
- **Timing** keeps the existing fit strategy + fit badges (`smart_fit`/`concise`/`stretch_video`/`strict_slot`, `DubSegmentRow.jsx:74–105`). A **fit-knob** change re-mixes (not re-TTS) via the existing `fit_fingerprint` path. Split/Merge/Direction stay (335–383).
|
||||
- Emotion/style → the per-segment **direction** field (already a fingerprint input).
|
||||
|
||||
### Stories surface
|
||||
`StoriesEditor.jsx` already has the richest line editor (per-line text, per-line voice override, emotion tone tags, per-line speed, per-line preview). The **only** missing piece is the *incremental backend*: today `generateAll` (374–409) POSTs the whole plan to `/longform/render` and re-streams everything. New UX:
|
||||
- Each track card gains the shared **stale chip** + **↻ regenerate this line** + header **"Regenerate changed (N)"**.
|
||||
- Single-line preview stays client-side (`previewTrack`, 301–352) — unchanged.
|
||||
- Full export switches from "always full render" to "render with `regen_only`": only stale spans re-synthesize; fresh spans reuse cached span WAVs. Reassigning a **cast** voice shows the "N lines inherit this voice and will regenerate" confirm.
|
||||
|
||||
### Audiobook surface
|
||||
`AudiobookTab.jsx` is the least mature — a single script `<textarea>` (227–233) with only per-*chapter* audition (372–397). It gets the biggest UX uplift:
|
||||
- **A segment/transcript view** for the parsed plan: render `AudiobookPlan.chapters[].spans[]` (the parser already produces spans — `audiobook.py:80–94`, `Span` at `services/audiobook.py:28–43`) as an editable list **under each chapter heading**, each span row carrying: editable text, a per-span **voice `<select>`** (writes back as inline `[voice:NAME]` so the script blob stays the single source of truth — see Technical Design), an emotion/style affordance, and the shared **stale chip + ↻**.
|
||||
- Editing a span edits the underlying script blob region; the plan re-parses; **only the edited span goes stale**, not the chapter.
|
||||
- Per-chapter audition stays; per-*span* preview is added (reuses the longform single-span render path).
|
||||
- The single-blob textarea remains available as a "raw script" toggle for power users; the structured view is the default. (Both bind to the same `script` string per the `LongformProject` store, spec 31.)
|
||||
|
||||
---
|
||||
|
||||
## Technical Design
|
||||
|
||||
### Principle: one cache contract, two implementations
|
||||
|
||||
Dub and longform both reduce to: **`fingerprint(unit) → diff vs stored → regen the stale units → reuse cached unit audio for the rest → re-stitch/re-mux the whole artifact`**. Dub's `unit` = dub segment (already shipped). Longform's `unit` becomes the **span** (new). We do **not** force them onto one code path — their timing/mux differ — but they share `incremental.py`'s primitives and the same persisted-hash discipline.
|
||||
|
||||
### Part A — Dub (extend existing, minimal change)
|
||||
|
||||
The edit → single-segment-regen → re-stitch flow already exists and is tested. Deltas:
|
||||
|
||||
1. **Independent transcript/translation edit.** `DubSegment` already has `text` (translation) and the job carries `text_original` (set in `dub_core.py:601`, preserved by `_sync_job_segments`, `dub_generate.py:51–88`). Expose `text_original` as an editable field on the row; **only `text` is a fingerprint input** (incremental.py:23), so editing the *transcript* alone does **not** force a re-TTS unless the user re-translates. A **per-line re-translate** calls `POST /dub/translate` (`dub_translate.py:222`) for that single id; the returned `text` flips the fingerprint → the line goes stale → user clicks ↻.
|
||||
2. **No fingerprint change needed** for voice/speaker/direction/effect — all already in `_GEN_INPUT_FIELDS`. Timing/fit already routes through `fit_fingerprint` (re-mix, not re-TTS).
|
||||
3. **Re-mux** stays lazy in `dub_export.py` (download/preview endpoints, `dub_download` 366–740, `dub_preview_video` 789–1034), driven by the persisted `fit_plans`/`video_stretch_plans`. A single-segment regen only rebuilds `dubbed_{lang}.wav`; the video re-mux happens on next download/preview, gated by plan-staleness helpers (`_video_retime_plan_for` 260–277).
|
||||
|
||||
> Net dub change is small: a transcript field + a single-id re-translate call. The heavy lifting (`regen_only`, `seg_hashes`, re-stitch) is untouched.
|
||||
|
||||
### Part B — Longform (the real new machinery): span-level incremental
|
||||
|
||||
Today `_render_chapter_cached` (`audiobook.py:314–355`) keys the **whole chapter** with `chapter_cache_key` (`longform_render.py:110–136`). We add a **span-level** key and a span cache, then make `_render_longform_sse` reuse fresh span WAVs and re-synthesize only stale ones.
|
||||
|
||||
**New: `span_fingerprint` + span cache** (in `backend/services/longform_render.py`, alongside `chapter_cache_key`):
|
||||
- `span_fingerprint(span, *, engine_id, sample_rate, voice_sig, lexicon) -> str` — sha1 over the inputs that affect a span's rendered audio: `(voice_id, text, pause_ms_after, speed, emotion/style, lexicon-respelling-of-text, engine_id, sample_rate, voice_sig)`. This is the **longform twin of `segment_fingerprint`**; it deliberately mirrors the dub field discipline (same #281 canonicalization rules — reuse `incremental._canon_value` so int↔float / None↔"" parity holds).
|
||||
- Span audio cached at `OUTPUTS_DIR/longform_cache/span_{key}.wav` (sibling to the existing chapter WAVs). The chapter WAV becomes a **stitch of its span WAVs** (crossfade + inter-span silence already done by `synthesize_chapter`, `services/audiobook.py:97–141`) rather than a single monolithic render. `chapter_cache_key` is retained for the **final stitched chapter** (so resume still hits at chapter granularity), but the chapter render now internally reuses fresh span WAVs.
|
||||
|
||||
**Refactor `synthesize_chapter`** (`services/audiobook.py:97–141`) so each span renders through a `render_span(span) -> tensor` that first checks the span cache by `span_fingerprint`; on hit it loads the WAV, on miss it runs the injected `synth` and writes the WAV. The function already iterates spans and stitches (121–139) — we wrap the per-span synth call (125) in the cache check. **This is the exact analog of dub's per-segment cache-load-or-`_gen` branch** (`dub_generate.py:160–199`).
|
||||
|
||||
**New: `regen_only` for longform.** `_render_longform_sse` (`audiobook.py:403`) and the `POST /audiobook` / `POST /longform/render` request bodies accept an optional `regen_only: list[str]` (span ids) + `stored_span_hashes: dict[str,str]`. When present:
|
||||
- Spans **not** in `regen_only` and whose stored hash matches → **cache-hit**, reuse WAV.
|
||||
- Spans in `regen_only` (or all spans, when omitted = today's behavior) → re-synthesize.
|
||||
- The chapter is re-stitched from the (mostly cached) span WAVs; the book is re-muxed (the existing `build_ffmetadata` + concat-demux mux, `audiobook.py:536` / `services/longform_render.py`).
|
||||
- Emit `seg_hashes`-equivalent (`span_hashes`) in the `done` SSE so the client persists them, exactly like dub's `done` payload (`dub_generate.py:840`).
|
||||
|
||||
**Span identity.** Longform spans need **stable ids** the way dub segments do (`s{NNNNN:05x}`). The longform parser (`services/longform_parser.parse_script_to_spans`, wrapped at `audiobook.py:80–94`) currently emits positional spans with no id. We add a **deterministic span id** derived from `(chapter_index, span_index)` *plus a content-stable suffix*, OR — preferred — mint stable ids in the parser and thread them through `Span` (`services/audiobook.py:28–43`, add `id: Optional[str]`). For Stories, the track card already has a stable `id` (`makeTrack`, `StoriesEditor.jsx:91–94`) → `storyToSpans` (`utils/storyToSpans.js:27–60`) threads it onto the span. The id is what `regen_only` addresses.
|
||||
|
||||
> **Why not just keep chapter keys?** Because a 30-page chapter re-synthesizing on a one-word fix is exactly the wall this spec closes. Span keying makes "fix one sentence" cost one sentence — the ElevenLabs bar.
|
||||
|
||||
### Composition with emotion/style (per-segment)
|
||||
|
||||
Assume a per-segment emotion/style field exists (dub: `direction`; stories: `emotion` track field, `StoriesEditor.jsx:91–94`; audiobook: to be added per-span). Integration rule, derived from the existing `fit_fingerprint` precedent:
|
||||
|
||||
- **If the field changes the TTS *output*** (e.g. an instruct/direction string fed to the engine) → it is a **fingerprint input**. Dub already includes `direction` (incremental.py:23; `test_direction_change_flips_fingerprint`, `test_redub_incremental.py:131–134`). Longform `span_fingerprint` includes the emotion/style field symmetrically.
|
||||
- **If the field is post-processing only** (a DSP/effect knob that re-mixes already-rendered audio) → it belongs in a **separate** fit-style fingerprint that triggers a re-mix, not a re-TTS — exactly how `fit_fingerprint` (incremental.py:70–116) is kept *out* of `segment_fingerprint`.
|
||||
|
||||
This composes cleanly with a future emotion/style spec: whichever bucket the field falls into, the fingerprint discipline already has a slot for it. (Owner decision Q1.)
|
||||
|
||||
### Edit → single-segment-regen → re-stitch/re-mux (end-to-end)
|
||||
|
||||
```
|
||||
User edits line L's text / voice / emotion (any surface)
|
||||
│
|
||||
├─ client recomputes fingerprint(L) → ≠ stored_hash[L.id] → L shows "stale" chip
|
||||
│ (dub: segment_fingerprint; longform: span_fingerprint — same canonicalization)
|
||||
│
|
||||
User clicks ↻ (one line) or "Regenerate changed (N)"
|
||||
│
|
||||
├─ DUB: POST /dub/generate/{job} { segments, segment_ids, regen_only:[L.id], preview:false }
|
||||
│ → dub_generate reuses cached seg_{id}.wav for all but L (dub_generate.py:160–197)
|
||||
│ → re-TTS L only (_gen) → re-stitch dubbed_{lang}.wav (766–770)
|
||||
│ → persist seg_hashes[L.id] (498–512) → done SSE returns new hashes (840)
|
||||
│ → re-mux is lazy on next /dub/download (dub_export.py:366–740)
|
||||
│
|
||||
└─ LONGFORM: POST /audiobook (or /longform/render) { chapters, regen_only:[L.id], stored_span_hashes }
|
||||
→ _render_longform_sse reuses span_{key}.wav for fresh spans
|
||||
→ re-synthesize L only → re-stitch L's chapter → re-mux m4b/mp3 (audiobook.py:536)
|
||||
→ done SSE returns span_hashes → client persists them
|
||||
```
|
||||
|
||||
Both paths **always rebuild the full final artifact** from a set that is mostly cache-hits — the dub invariant proven by `test_redub_incremental.py:280–281` (final track rebuilt) generalizes to longform's m4b/mp3.
|
||||
|
||||
### Files to extend (with paths)
|
||||
|
||||
| File | Change |
|
||||
|---|---|
|
||||
| `backend/services/incremental.py` | Add `span_fingerprint(...)` (or factor a shared `_fingerprint(fields, canon)` core that both `segment_fingerprint` and `span_fingerprint` call). Reuse `_canon_value`. |
|
||||
| `backend/services/longform_render.py` | Add span-level key + `span_{key}.wav` cache load/write helper; keep `chapter_cache_key` for the stitched chapter. |
|
||||
| `backend/services/audiobook.py` | `Span` gains stable `id`; `synthesize_chapter` (97–141) wraps per-span synth in span-cache load-or-render. |
|
||||
| `backend/api/routers/audiobook.py` | `_render_longform_sse` (403–595) honors `regen_only` + `stored_span_hashes`; `POST /audiobook` (598–610) + `POST /longform/render` (639–667) accept them; `done` SSE emits `span_hashes`. |
|
||||
| `backend/services/longform_parser.py` (+ `frontend/src/utils/longformParser.js` twin, byte-for-byte per #27) | Mint stable span ids. **Both must change together** (the JS twin is golden-corpus-verified). |
|
||||
| `backend/api/routers/dub_translate.py` | Allow a **single-id** re-translate (already id-keyed, `dub_translate.py:222`); ensure one-segment requests are cheap. |
|
||||
| `frontend/src/components/DubSegmentRow.jsx` | Expose editable `text_original` (transcript) distinct from `text` (translation); add per-line re-translate. |
|
||||
| `frontend/src/pages/AudiobookTab.jsx` | New structured span/transcript view over `AudiobookPlan`; per-span text/voice/emotion edit + stale chip + ↻; raw-script toggle retained. |
|
||||
| `frontend/src/components/StoriesEditor.jsx` | Add stale chip + ↻ + "Regenerate changed (N)"; `generateAll` (374–409) sends `regen_only`/`stored_span_hashes`. |
|
||||
| `frontend/src/utils/storyToSpans.js` | Thread track `id` → span `id`. |
|
||||
| `frontend/src/store/longformSlice.ts` (per spec 31) | Persist `span_hashes` alongside the project (the localStorage analog of `job_data.seg_hashes`). |
|
||||
|
||||
---
|
||||
|
||||
## API / Schema / Data-model changes
|
||||
|
||||
### New / extended request fields (additive, all optional → back-compat)
|
||||
- **`DubSegment`** (`backend/schemas/requests.py:19–43`): unchanged field set; `text_original` is carried in the job, not the segment fingerprint. (No schema change required for dub — the transcript edit reuses existing job state.) If exposed in the request, add `text_original: Optional[str] = None` (additive, defaulted → old payloads parse unchanged).
|
||||
- **Longform render bodies** (`LongformChapter`/`LongformSpan` Pydantic models, `audiobook.py:615–636`; `AudiobookSynthesizeRequest`): add `regen_only: Optional[list[str]] = None` and `stored_span_hashes: Optional[dict[str,str]] = None`. `LongformSpan` gains `id: Optional[str] = None`. All defaulted → existing clients unaffected.
|
||||
|
||||
### Endpoints
|
||||
- **Reuse** `POST /dub/generate/{job_id}` (`dub_generate.py:93`) — already takes `regen_only`. No new dub endpoint.
|
||||
- **Reuse** `POST /dub/translate` (`dub_translate.py:222`) for single-id re-translate (already id-keyed). No new endpoint.
|
||||
- **Extend** `POST /audiobook` (`audiobook.py:598`) and `POST /longform/render` (`audiobook.py:639`) to honor `regen_only`/`stored_span_hashes`; `done` SSE adds a `span_hashes` field (additive — old clients ignore unknown keys, matching dub's `seg_hashes`/`seg_num_step` additive precedent at `dub_generate.py:840`).
|
||||
- **New (optional, thin)** `POST /audiobook/preview-span/{job_id}` mirroring `dub_generate.py:867` `preview_segment` — fast single-span audition (low steps, no mux, no persist). Only if Stories' client-side preview (`previewTrack`) doesn't already cover the audiobook need; otherwise skip.
|
||||
|
||||
### Persistence
|
||||
- **Dub**: no change. `seg_hashes`/`seg_order`/`seg_num_step` already live in `job_data` (`dub_history.job_data`, `db.py:74–85`), written by `_save_job` (`dub_pipeline.py:187–214`).
|
||||
- **Longform**: span audio cached on disk under `OUTPUTS_DIR/longform_cache/span_{key}.wav` (content-addressed → self-cleaning, pruned by the existing `prune_cache_dir`, `audiobook.py:494`). **`span_hashes` persist client-side** in the `LongformProject` zustand store (spec 31, `longformSlice.ts`) — the localStorage analog of `job_data.seg_hashes`. **No new SQLite table, no alembic migration** for longform (it's filesystem cache + browser state). If the owner later wants server-side longform job rows to carry `span_hashes`, *that* would go through alembic — flagged as a non-goal here.
|
||||
- **Migration / back-compat**: existing dub jobs already carry `seg_hashes` (or get them on next generate). Existing longform renders have **no** `span_hashes` → first edit treats all spans as stale **once** (the planner's "missing stored hash → stale" default, `incremental.plan_incremental` doc 134–136), which is safe (re-renders correctly, just not incrementally that one time). **No forced re-render** of any existing project on upgrade. No DB schema change → **no alembic migration needed**; the localStorage versioned `migrate` fn (spec 31) tolerates the absent `span_hashes` key.
|
||||
|
||||
### Preferences
|
||||
- Per-surface toggle (Settings): "Default to structured editor view" (Audiobook), defaulting **on**. No network prefs. Stored in existing settings store.
|
||||
|
||||
---
|
||||
|
||||
## Local-first & Cross-platform compliance
|
||||
|
||||
- **Local-first preserved.** Every path is local: TTS/regen runs on-device (MPS/CUDA/ROCm/CPU auto-detect, unchanged); span/segment WAVs are local files; fingerprints are local sha1; `span_hashes` persist locally (job_data / localStorage). **No cloud call, no account, no API key, no telemetry** is added. The optional `POST /dub/translate` offline providers (nllb, argos) keep translation local; cloud LLM translation stays the user's existing opt-in.
|
||||
- **Cross-platform parity (strict rule, 2026-05-20).** The editor + per-segment regen is **default behavior** and must be identical on macOS / Windows / Linux. The implementation uses only cross-platform pieces: `dub_seg_path`/cache paths use `os.path.join` + realpath containment (`config.py:54–71`, already cross-platform), ffmpeg mux is the existing cross-platform invocation (`build_render_cmd`), fingerprints are pure Python, and the UI is the existing Tauri/React stack. **No platform-only affordance** is introduced — no macOS-only shortcut, no Windows-only picker. The keyboard shortcuts already in `DubSegmentRow.jsx` (⌘D split / ⌘M merge, 335–383) must map to Ctrl on Windows/Linux (verify they already do; if not, that's a P0 parity fix in the same PR).
|
||||
- **Existing-engine + existing-`omnivoice_data/` back-compat.** No engine code touched in a way that requires reinstall; on-disk model state untouched. Existing dub jobs, story projects, audiobook renders open and play unchanged; the only first-edit cost is one non-incremental regen (correct output, just not cached yet). Span cache is purely additive on disk.
|
||||
|
||||
---
|
||||
|
||||
## Phasing (sliceable milestones on the v0.3.x line)
|
||||
|
||||
Each slice is independently shippable, continuous-to-main, with its own regression test. Ordering puts the **lowest-risk, highest-leverage** work first (dub is already 90% there).
|
||||
|
||||
- **03a — Dub transcript/translation split + single-id re-translate.** Expose editable transcript (`text_original`) vs translation (`text`) in `DubSegmentRow.jsx`; per-line re-translate via `POST /dub/translate` for one id. Reuses existing `regen_only`. *Smallest, proves the "edit translation → one line stale → ↻" loop end-to-end on the surface that already supports it.*
|
||||
- **03b — Longform span fingerprint + span cache (backend only, no UI).** `span_fingerprint` in `incremental.py`; span-cache load/write in `longform_render.py`; `synthesize_chapter` reuses fresh span WAVs; stable span ids in the parser (both twins). Gated by a test asserting **one edited span re-synthesizes; siblings cache-hit** (the dub test, lifted to longform).
|
||||
- **03c — Longform `regen_only` wiring + `span_hashes` in the store.** `_render_longform_sse` + the two POST bodies honor `regen_only`/`stored_span_hashes`; `done` SSE emits `span_hashes`; `longformSlice` persists them.
|
||||
- **03d — Stories editor: stale chip + ↻ + "Regenerate changed (N)".** Stories already has the row editor; just add the stale UX and switch `generateAll` to send `regen_only`. *First user-visible longform incrementality.*
|
||||
- **03e — Audiobook structured span editor.** The big UX uplift: structured per-span view over `AudiobookPlan`, per-span text/voice/emotion edit, stale chip + ↻, raw-script toggle. Rides 03b/03c.
|
||||
- **03f — Emotion/style-per-segment composition.** Wire the emotion/style field into `span_fingerprint` (re-TTS bucket) or a longform fit-style fingerprint (re-mix bucket) per the owner's Q1 decision; symmetric with dub's existing `direction`.
|
||||
|
||||
(03a, 03b can land in parallel; 03d depends on 03c; 03e depends on 03b+03c; 03f is last.)
|
||||
|
||||
---
|
||||
|
||||
## Testing strategy
|
||||
|
||||
**The load-bearing assertion (every surface): a single-line edit regenerates exactly one segment; all others are cache-hits.** This is already proven for dub — the new tests **lift that exact contract** to longform.
|
||||
|
||||
- **Reuse + extend `tests/test_redub_incremental.py`.** It already asserts the dub contract: `test_one_edit_marks_exactly_one_segment_stale` (101–118), `test_edited_line_produces_different_cached_output` (228–281: edited line's WAV changes, **untouched line's WAV reused byte-for-byte** 272–273, TTS ran exactly once 266, final track rebuilt 281). 03a adds a test that editing **only `text_original`** does *not* flip the fingerprint, but a re-translate that changes `text` does (composes with `test_direction_change_flips_fingerprint`, 131–134).
|
||||
- **New `tests/test_longform_incremental.py`** (the centerpiece, mirrors the dub test with a stub `synth`):
|
||||
- `span_fingerprint` parity: server-parsed span vs client-raw span hash identically (the #281 class, reusing `_canon_value`).
|
||||
- **One-edit-one-span**: a 3-span chapter renders all 3; edit span 1's text; assert the planner marks **exactly** span 1 stale; regen reuses spans 0 and 2's cached WAVs **byte-for-byte**, re-synthesizes span 1 only (stub `synth` call-count == 1), and the re-stitched chapter WAV changed.
|
||||
- **Voice reassign**: changing a span's `voice_id` flips its fingerprint (and only its); a *cast*-level reassign that M spans inherit marks exactly M stale.
|
||||
- **Cross-chapter isolation**: editing a span in chapter 2 leaves chapter 1's cached chapter WAV untouched (chapter-key still hits).
|
||||
- **Lexicon edit fan-out**: editing a lexicon entry marks stale exactly the spans whose respelled text changed (composes with the lexicon-in-key behavior, `audiobook.py:338–341`).
|
||||
- **Back-compat tests**: existing dub job with stored `seg_hashes` from an older build still matches (`test_backcompat_with_hashes_stored_by_previous_builds`, 87–98); an existing longform render with **no** `span_hashes` is treated as all-stale exactly once, then incremental thereafter — and produces byte-identical audio to a full render.
|
||||
- **Cross-platform**: the cache-path/realpath tests run on all three OSes in CI; assert `span_{key}.wav` path construction and containment hold on Windows path separators.
|
||||
- **Frontend**: a Playwright/unit test that editing one line shows the stale chip and that "Regenerate changed (N)" sends `regen_only` with exactly the stale ids.
|
||||
- **Keep main green**: any `frontend/package.json` touch regenerates root `bun.lock` and passes `bun install --frozen-lockfile` (Docker parity); parser twin change re-runs the golden-corpus equality test (#27).
|
||||
|
||||
---
|
||||
|
||||
## Risks & mitigations
|
||||
|
||||
| Risk | Mitigation |
|
||||
|---|---|
|
||||
| **Span-cache explosion** (one WAV per span per fingerprint → thousands of files for a long book). | Content-addressed names self-dedupe; reuse the existing `prune_cache_dir` (`audiobook.py:494`) with an LRU/size cap; only the *current* project's fresh spans are kept hot. |
|
||||
| **Fingerprint parity drift** (server vs client hash differently → every span looks stale → degrades to full render, the #281 regression class). | Reuse `incremental._canon_value` verbatim; add the server-vs-client parity test first (TDD), exactly as dub did. |
|
||||
| **Parser-twin divergence** (Python `longform_parser` vs JS `longformParser.js` mint different span ids). | Stable ids derived from `(chapter_index, span_index)` are computed identically in both; the existing golden-corpus byte-for-byte test (#27) gates it. |
|
||||
| **Stitch seams** (per-span WAVs stitched may click vs a monolithic chapter render). | `synthesize_chapter` already crossfades spans (50 ms) and hard-concats silences (`services/audiobook.py:121–139`) — the seam behavior is identical whether a span was freshly rendered or cache-loaded (same WAV bytes). |
|
||||
| **smart_fit-style double-processing on dub** (reusing a slotted WAV under smart_fit). | Already solved: the strategy-transition guard (`dub_generate.py:122–123`) + `seg_wav_kind` (830) force a full regen when the cached WAVs are the wrong kind. No new exposure. |
|
||||
| **Existing projects forced to re-render.** | First edit treats unknown-hash spans as stale **once** (correct output), never corrupts cache; no upgrade-time mass re-render. |
|
||||
| **Emotion/style field lands in the wrong fingerprint bucket** (re-TTS when it should re-mix, or vice-versa, wasting compute or shipping stale audio). | Q1 decision pins the bucket per field; the `fit_fingerprint` precedent (incremental.py:70–116) gives both buckets a tested home; 03f ships last, after the field's semantics are known. |
|
||||
|
||||
---
|
||||
|
||||
## Open questions / decisions for the owner
|
||||
|
||||
1. **Emotion/style field → which fingerprint bucket?** If the per-segment emotion/style field is fed to the engine as an instruct/direction (changes TTS output), it's a **`span_fingerprint`/`segment_fingerprint` input** (re-TTS on change). If it's a post-render DSP/style transfer (re-mix), it belongs in a **fit-style fingerprint** (re-mix, no re-TTS). Dub's `direction` is already the former. Please confirm the bucket per the emotion/style spec when it lands (drives 03f).
|
||||
2. **Audiobook span ids vs the raw-script-blob single-source-of-truth.** Editing a span writes back inline `[voice:NAME]`/text into the `script` string (so the blob stays authoritative, per spec 31). Stable span ids derived from `(chapter_index, span_index)` shift when the user inserts a paragraph above. Acceptable to treat an inserted span as "new → stale" and shift downstream ids (cheap, correct), or do we want content-anchored ids? Recommendation: positional ids + "shifted = stale once"; revisit only if users report churn.
|
||||
3. **Single-id re-translate cost.** `POST /dub/translate` (`dub_translate.py:222`) loads a translation model; a per-line re-translate pays that once per call. Cache the loaded model module-level (nllb already is, 232–306) so single-id calls are cheap — confirm acceptable, or batch re-translate the stale set instead of per-line.
|
||||
4. **`POST /audiobook/preview-span` — build it or reuse client-side preview?** Stories previews client-side (`previewTrack`). If audiobook structured view needs server-side single-span audition, add the thin endpoint; otherwise reuse the client path. Owner call on whether the thin endpoint is worth it for 03e.
|
||||
5. **Scope of "timing" for longform.** Confirmed non-goal: longform has no per-line slots, so per-segment *timing* stays dub-only; longform "timing" = existing `[pause]` + per-line `speed`. Flag if the owner wants longform pause/speed surfaced as first-class per-span timing controls (small add to the span row).
|
||||
@@ -1,3 +1,5 @@
|
||||
> **Superseded (2026-06-25)** by [00-roadmap-elevenlabs-parity.md](00-roadmap-elevenlabs-parity.md) and specs 01–03. Retained for historical context.
|
||||
|
||||
# ElevenLabs-Parity Program — Implementation Spec
|
||||
|
||||
**Date:** 2026-06-12
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
> **Superseded (2026-06-25)** by [00-roadmap-elevenlabs-parity.md](00-roadmap-elevenlabs-parity.md) and specs 01–03. Retained for historical context.
|
||||
|
||||
# Stories Editor & Audiobook — Maturity Spec
|
||||
|
||||
> Compiled 2026-06-13. Targets the v0.3.x continuous-to-main line. Grounds every
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
> **Superseded (2026-06-25)** by [00-roadmap-elevenlabs-parity.md](00-roadmap-elevenlabs-parity.md) and specs 01–03. Retained for historical context.
|
||||
|
||||
# Studio / Projects — v1 spec
|
||||
|
||||
**Goal:** ElevenLabs-Studio parity for long-form narration. A user pastes (or drags in) a 10-page script, the app splits it into blocks, they assign a voice per block, preview inline, then hit Generate to get one stitched WAV.
|
||||
|
||||
+15
-9
@@ -25,18 +25,24 @@ manifest:
|
||||
Both manifests are signed with the same minisign key, so a tampered build is
|
||||
rejected regardless of channel.
|
||||
|
||||
## For maintainers — cutting a preview build
|
||||
## For maintainers — how previews are built
|
||||
|
||||
Preview builds are **manual** (no scheduled spend, nothing auto-published):
|
||||
Preview builds come from **`main`**, two ways:
|
||||
|
||||
1. Go to **Actions → Desktop Release → Run workflow**.
|
||||
2. Pick the branch to build (usually `main`).
|
||||
3. Set **publish_preview = true** and run.
|
||||
- **Nightly (automatic).** A scheduled job (07:00 UTC) rebuilds the rolling
|
||||
`preview` prerelease from `main` — but only when `main` actually moved in the
|
||||
last day, so idle days cost nothing. Preview is never more than ~24h behind
|
||||
`main`.
|
||||
- **On demand.** **Actions → Desktop Release → Run workflow**, pick a branch
|
||||
(usually `main`), set **publish_preview = true**. Useful to cut a preview off
|
||||
a feature branch, or to refresh immediately without waiting for the nightly.
|
||||
|
||||
This builds the matrix and publishes/updates a single rolling `preview`
|
||||
**prerelease** with its own signed `latest.json`. The tagged `latest` stable
|
||||
release is never affected. Preview users get the new build on their next check;
|
||||
stable users see nothing.
|
||||
Either way it builds the matrix and publishes/updates a single rolling
|
||||
`preview` **prerelease** — always flagged prerelease, and carrying the same
|
||||
platform set as stable (both verified in CI after each preview publish) — with
|
||||
its own signed `latest.json`. The tagged `latest` stable release is never
|
||||
affected. Preview users get the new build on their next check; stable users see
|
||||
nothing.
|
||||
|
||||
To stop offering previews, delete the `preview` release/tag on GitHub — the
|
||||
Preview channel then falls back to stable.
|
||||
|
||||
@@ -0,0 +1,13 @@
|
||||
{
|
||||
"$schema": "./node_modules/oxfmt/configuration_schema.json",
|
||||
"singleQuote": true,
|
||||
"jsxSingleQuote": false,
|
||||
"ignorePatterns": [
|
||||
"**/*.css",
|
||||
"**/*.json",
|
||||
"**/*.toml",
|
||||
"**/*.html",
|
||||
"**/*.md",
|
||||
"src-tauri/**"
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,38 @@
|
||||
{
|
||||
"$schema": "./node_modules/oxlint/configuration_schema.json",
|
||||
"env": {
|
||||
"browser": true,
|
||||
"es2024": true
|
||||
},
|
||||
"globals": {
|
||||
"__APP_VERSION__": "readonly",
|
||||
"AudioWorklet": "readonly",
|
||||
"AudioWorkletNode": "readonly",
|
||||
"AudioWorkletProcessor": "readonly",
|
||||
"registerProcessor": "readonly"
|
||||
},
|
||||
"plugins": ["react", "import", "unicorn"],
|
||||
"ignorePatterns": ["dist/", "src-tauri/", "node_modules/"],
|
||||
"categories": {
|
||||
"correctness": "error"
|
||||
},
|
||||
"rules": {
|
||||
"no-unused-vars": ["error", { "varsIgnorePattern": "^[A-Z_]", "argsIgnorePattern": "^_", "caughtErrors": "none" }],
|
||||
"no-empty": ["warn", { "allowEmptyCatch": true }],
|
||||
"no-unused-expressions": ["error", { "allowShortCircuit": true, "allowTernary": true }],
|
||||
"react/exhaustive-deps": "warn",
|
||||
"react/rules-of-hooks": "error",
|
||||
"react/only-export-components": ["warn", { "allowConstantExport": true }],
|
||||
"max-lines": ["warn", { "max": 500, "skipBlankLines": true, "skipComments": true }]
|
||||
},
|
||||
"overrides": [
|
||||
{
|
||||
"files": ["vite.config.js", "*.config.js", "eslint.config.js"],
|
||||
"env": { "node": true }
|
||||
},
|
||||
{
|
||||
"files": ["**/*.test.{js,jsx}", "**/*.spec.{js,jsx}", "src/test/**"],
|
||||
"env": { "node": true, "vitest": true }
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
{
|
||||
"$schema": "https://ui.shadcn.com/schema.json",
|
||||
"style": "new-york",
|
||||
"rsc": false,
|
||||
"tsx": true,
|
||||
"tailwind": {
|
||||
"config": "",
|
||||
"css": "src/index.css",
|
||||
"baseColor": "neutral",
|
||||
"cssVariables": true,
|
||||
"prefix": ""
|
||||
},
|
||||
"iconLibrary": "lucide",
|
||||
"aliases": {
|
||||
"components": "@/components",
|
||||
"ui": "@/components/ui",
|
||||
"utils": "@/lib/utils",
|
||||
"lib": "@/lib",
|
||||
"hooks": "@/hooks"
|
||||
}
|
||||
}
|
||||
@@ -2,8 +2,18 @@ import type { Page } from '@playwright/test';
|
||||
|
||||
/** Every routable view (the `mode` values in App.jsx). */
|
||||
export const MODES = [
|
||||
'launchpad', 'clone', 'design', 'gallery', 'dub', 'stories',
|
||||
'projects', 'queue', 'tools', 'transcriptions', 'settings', 'donate',
|
||||
'launchpad',
|
||||
'clone',
|
||||
'design',
|
||||
'gallery',
|
||||
'dub',
|
||||
'stories',
|
||||
'projects',
|
||||
'queue',
|
||||
'tools',
|
||||
'transcriptions',
|
||||
'settings',
|
||||
'donate',
|
||||
] as const;
|
||||
|
||||
/**
|
||||
|
||||
@@ -18,7 +18,9 @@ test.describe('OmniVoice Gallery', () => {
|
||||
expect(bg).toBe('rgba(255, 255, 255, 0.04)');
|
||||
});
|
||||
|
||||
test('opening an archetype in the Designer mounts the design view (no chunk-load failure)', async ({ page }) => {
|
||||
test('opening an archetype in the Designer mounts the design view (no chunk-load failure)', async ({
|
||||
page,
|
||||
}) => {
|
||||
const errors = collectErrors(page);
|
||||
await gotoMode(page, 'gallery');
|
||||
|
||||
@@ -29,9 +31,9 @@ test.describe('OmniVoice Gallery', () => {
|
||||
|
||||
// The design view (CloneDesignTab — the lazy chunk that failed when Vite
|
||||
// was down) must mount. Its prompt/personality UI is the tell.
|
||||
await expect(
|
||||
page.getByText(/personality|prompt|steps/i).first()
|
||||
).toBeVisible({ timeout: 15_000 });
|
||||
await expect(page.getByText(/personality|prompt|steps/i).first()).toBeVisible({
|
||||
timeout: 15_000,
|
||||
});
|
||||
|
||||
await expect(page.getByText(/this tab hit a snag/i)).toHaveCount(0);
|
||||
expect(errors.fatal, errors.fatal.join('\n')).toEqual([]);
|
||||
|
||||
+18
-14
@@ -1,29 +1,33 @@
|
||||
import js from '@eslint/js'
|
||||
import globals from 'globals'
|
||||
import reactHooks from 'eslint-plugin-react-hooks'
|
||||
import reactRefresh from 'eslint-plugin-react-refresh'
|
||||
import { defineConfig, globalIgnores } from 'eslint/config'
|
||||
// Advisory-only lint for the React Compiler rule family that oxlint does not yet
|
||||
// implement natively (e.g. react-hooks/set-state-in-effect, immutability,
|
||||
// preserve-manual-memoization). Run locally via `bun run lint:hooks`.
|
||||
//
|
||||
// This is NOT a CI gate. oxlint (`bun run lint`, see .oxlintrc.json) is the gate.
|
||||
// Everything oxlint already covers (no-unused-vars, exhaustive-deps,
|
||||
// rules-of-hooks, no-undef, max-lines, react-refresh) is turned OFF here so the
|
||||
// two tools don't double-report. Drop this file once oxlint's JS-plugin support
|
||||
// graduates from alpha and can run eslint-plugin-react-hooks directly.
|
||||
import reactHooks from 'eslint-plugin-react-hooks';
|
||||
import globals from 'globals';
|
||||
import { defineConfig, globalIgnores } from 'eslint/config';
|
||||
|
||||
export default defineConfig([
|
||||
globalIgnores(['dist']),
|
||||
{
|
||||
files: ['**/*.{js,jsx}'],
|
||||
extends: [
|
||||
js.configs.recommended,
|
||||
reactHooks.configs.flat.recommended,
|
||||
reactRefresh.configs.vite,
|
||||
],
|
||||
extends: [reactHooks.configs.flat.recommended],
|
||||
languageOptions: {
|
||||
ecmaVersion: 2020,
|
||||
ecmaVersion: 'latest',
|
||||
globals: globals.browser,
|
||||
parserOptions: {
|
||||
ecmaVersion: 'latest',
|
||||
ecmaFeatures: { jsx: true },
|
||||
sourceType: 'module',
|
||||
},
|
||||
},
|
||||
rules: {
|
||||
'no-unused-vars': ['error', { varsIgnorePattern: '^[A-Z_]' }],
|
||||
// Owned by oxlint — disabled here to avoid duplicate diagnostics.
|
||||
'react-hooks/exhaustive-deps': 'off',
|
||||
'react-hooks/rules-of-hooks': 'off',
|
||||
},
|
||||
},
|
||||
])
|
||||
]);
|
||||
|
||||
@@ -0,0 +1,15 @@
|
||||
{
|
||||
"$schema": "https://unpkg.com/knip@6/schema.json",
|
||||
"ignore": [
|
||||
"public/aec-worklet.js"
|
||||
],
|
||||
"ignoreUnresolved": [
|
||||
"/@react-refresh"
|
||||
],
|
||||
"ignoreDependencies": [
|
||||
"tailwindcss",
|
||||
"@tauri-apps/plugin-updater",
|
||||
"@tauri-apps/plugin-window-state",
|
||||
"playwright-core"
|
||||
]
|
||||
}
|
||||
+19
-6
@@ -1,14 +1,19 @@
|
||||
{
|
||||
"name": "omnivoice-studio",
|
||||
"version": "0.3.8",
|
||||
"private": true,
|
||||
"version": "0.3.6",
|
||||
"license": "AGPL-3.0-only",
|
||||
"type": "module",
|
||||
"scripts": {
|
||||
"dev": "vite",
|
||||
"desktop": "tauri dev",
|
||||
"build": "vite build",
|
||||
"lint": "eslint .",
|
||||
"lint": "oxlint",
|
||||
"lint:fix": "oxlint --fix",
|
||||
"lint:hooks": "eslint .",
|
||||
"format": "oxfmt --write",
|
||||
"format:check": "oxfmt --check",
|
||||
"knip": "knip",
|
||||
"typecheck": "tsc --noEmit",
|
||||
"typecheck:ci": "tsc --noEmit --checkJs false",
|
||||
"test": "vitest run",
|
||||
@@ -16,7 +21,9 @@
|
||||
"test:legacy": "node --experimental-strip-types --no-warnings --test ../tests/frontend/*.test.mjs",
|
||||
"preview": "vite preview",
|
||||
"tauri": "tauri",
|
||||
"e2e": "playwright test"
|
||||
"e2e": "playwright test",
|
||||
"test:visual": "playwright test --config=playwright.visual.config.ts",
|
||||
"test:visual:update": "playwright test --config=playwright.visual.config.ts --update-snapshots"
|
||||
},
|
||||
"dependencies": {
|
||||
"@fontsource-variable/inter": "^5.2.8",
|
||||
@@ -24,11 +31,12 @@
|
||||
"@fontsource/ibm-plex-mono": "^5.2.7",
|
||||
"@radix-ui/react-dialog": "^1.1.17",
|
||||
"@radix-ui/react-dropdown-menu": "^2.1.18",
|
||||
"@radix-ui/react-popover": "^1.1.17",
|
||||
"@radix-ui/react-progress": "^1.1.10",
|
||||
"@radix-ui/react-select": "^2.3.1",
|
||||
"@radix-ui/react-slider": "^1.4.1",
|
||||
"@radix-ui/react-slot": "^1.3.0",
|
||||
"@radix-ui/react-tabs": "^1.1.15",
|
||||
"@radix-ui/react-toggle": "^1.1.12",
|
||||
"@radix-ui/react-toggle-group": "^1.1.13",
|
||||
"@radix-ui/react-tooltip": "^1.2.10",
|
||||
"@tailwindcss/vite": "^4.3.1",
|
||||
@@ -40,6 +48,8 @@
|
||||
"@tauri-apps/plugin-process": "^2.3.1",
|
||||
"@tauri-apps/plugin-updater": "^2.10.1",
|
||||
"@tauri-apps/plugin-window-state": "^2.4.1",
|
||||
"class-variance-authority": "^0.7.1",
|
||||
"clsx": "^2.1.1",
|
||||
"country-flag-icons": "^1.6.17",
|
||||
"i18next": "^26.3.1",
|
||||
"i18next-browser-languagedetector": "^8.2.1",
|
||||
@@ -50,12 +60,13 @@
|
||||
"react-hot-toast": "^2.6.0",
|
||||
"react-i18next": "^17.0.8",
|
||||
"react-window": "^2.2.7",
|
||||
"tailwind-merge": "^3.6.0",
|
||||
"tailwindcss": "^4.3.1",
|
||||
"tw-animate-css": "^1.4.0",
|
||||
"wavesurfer.js": "^7.12.8",
|
||||
"zustand": "^5.0.14"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@eslint/js": "^10.0.1",
|
||||
"@playwright/test": "^1.61.0",
|
||||
"@tauri-apps/api": "^2.11.0",
|
||||
"@tauri-apps/cli": "^2.11.2",
|
||||
@@ -66,9 +77,11 @@
|
||||
"@vitejs/plugin-react": "^6.0.2",
|
||||
"eslint": "^10.5.0",
|
||||
"eslint-plugin-react-hooks": "^7.1.1",
|
||||
"eslint-plugin-react-refresh": "^0.5.3",
|
||||
"globals": "^17.6.0",
|
||||
"jsdom": "^29.1.1",
|
||||
"knip": "^6.23.0",
|
||||
"oxfmt": "^0.57.0",
|
||||
"oxlint": "^1.71.0",
|
||||
"playwright-core": "1.61.0",
|
||||
"typescript": "^6.0.3",
|
||||
"vite": "^8.0.16",
|
||||
|
||||
@@ -0,0 +1,51 @@
|
||||
import { defineConfig, devices } from '@playwright/test';
|
||||
|
||||
// Visual-regression config — SEPARATE from playwright.config.ts (e2e).
|
||||
//
|
||||
// The e2e suite drives the full app against a Python backend; this one renders
|
||||
// individual presentational components in isolation via the Vite harness
|
||||
// (src/test/visual/harness.html), so NO backend is required. It runs its own
|
||||
// Vite dev server on a dedicated port and snapshots leaf components across
|
||||
// themes. Local/manual only (`bun run test:visual`) — see
|
||||
// src/test/visual/README.md for why it is not yet a CI gate.
|
||||
const PORT = Number(process.env.VISUAL_PORT || 3902);
|
||||
|
||||
export default defineConfig({
|
||||
testDir: './src/test/visual',
|
||||
testMatch: /.*\.visual\.spec\.ts$/,
|
||||
timeout: 30_000,
|
||||
fullyParallel: true,
|
||||
retries: 0,
|
||||
reporter: [['list']],
|
||||
// Flat, platform-agnostic baseline filenames (Badge-midnight.png …). These
|
||||
// are committed; they are correct for the machine that generated them.
|
||||
snapshotPathTemplate: '{testDir}/__screenshots__/{arg}{ext}',
|
||||
expect: {
|
||||
timeout: 10_000,
|
||||
toHaveScreenshot: {
|
||||
animations: 'disabled',
|
||||
caret: 'hide',
|
||||
// Small tolerance absorbs sub-pixel font anti-aliasing jitter on the
|
||||
// same OS without masking real layout/color regressions.
|
||||
maxDiffPixelRatio: 0.01,
|
||||
},
|
||||
},
|
||||
use: {
|
||||
baseURL: `http://localhost:${PORT}`,
|
||||
headless: true,
|
||||
deviceScaleFactor: 1,
|
||||
// Default to Playwright's managed chromium; set PLAYWRIGHT_CHROMIUM to
|
||||
// pin a specific binary (e.g. the system chromium used by e2e in CI).
|
||||
...(process.env.PLAYWRIGHT_CHROMIUM
|
||||
? { launchOptions: { executablePath: process.env.PLAYWRIGHT_CHROMIUM } }
|
||||
: {}),
|
||||
},
|
||||
projects: [{ name: 'chromium', use: { ...devices['Desktop Chrome'], deviceScaleFactor: 1 } }],
|
||||
webServer: {
|
||||
command: 'bun run dev',
|
||||
url: `http://localhost:${PORT}`,
|
||||
reuseExistingServer: !process.env.CI,
|
||||
timeout: 60_000,
|
||||
env: { OMNIVOICE_UI_PORT: String(PORT) },
|
||||
},
|
||||
});
|
||||
@@ -10,8 +10,9 @@
|
||||
class AecFrameEmitter extends AudioWorkletProcessor {
|
||||
constructor(options) {
|
||||
super();
|
||||
const frame = (options && options.processorOptions && options.processorOptions.frameSize) || 320;
|
||||
this._frameSize = frame; // 320 samples = 20 ms @ 16 kHz
|
||||
const frame =
|
||||
(options && options.processorOptions && options.processorOptions.frameSize) || 320;
|
||||
this._frameSize = frame; // 320 samples = 20 ms @ 16 kHz
|
||||
this._buf = new Float32Array(frame);
|
||||
this._n = 0;
|
||||
}
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"raised": 137.5,
|
||||
"raised": 10,
|
||||
"goal": 200,
|
||||
"currency": "USD",
|
||||
"sponsorCount": 23,
|
||||
"updated": "2026-06-16"
|
||||
"sponsorCount": 1,
|
||||
"updated": "2026-06-17"
|
||||
}
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user