Files
VoiceStudio/pyproject.toml
T
debpalashandClaude Opus 4.7 994c6cf065 feat(backend): setup wizard router, translation engines, export options, client-disconnect handling
- Add setup router (backend/api/routers/setup.py) for first-run wizard:
  system checks, engine probes, model downloads with progress
- Add translation engines service with pluggable backends
- Add utils/hf_progress for HuggingFace download progress streaming
- Add PyInstaller runtime hooks (numpy compat, torch compiler disable)
- Global exception handler short-circuits h11 LocalProtocolError and
  Starlette ClientDisconnect with HTTP 499 to silence noisy stack traces
  when users scrub or cancel video mid-stream
- /dub/download-mp3 accepts bitrate query param (clamped 64–320kbps)
- Refactor ASR/TTS backends, dub pipeline, engine management
- Update backend.spec for PyInstaller packaging
- Bump pyproject version to 0.2.0; refresh uv.lock

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-04-22 17:49:04 +05:30

157 lines
5.3 KiB
TOML
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
[build-system]
requires = ["hatchling"]
build-backend = "hatchling.build"
[project]
name = "omnivoice"
version = "0.2.0"
description = "OmniVoice: Towards Omnilingual Zero-Shot Text-to-Speech with Diffusion Language Models"
readme = "README.md"
license = "Apache-2.0"
requires-python = ">=3.11"
authors = [{name = "Han Zhu"}]
keywords = [
"tts",
"text-to-speech",
"speech-synthesis",
"zero-shot",
"multilingual",
"diffusion",
"voice-cloning",
]
classifiers = [
"Intended Audience :: Science/Research",
"Intended Audience :: Developers",
"Topic :: Scientific/Engineering :: Artificial Intelligence",
"Topic :: Multimedia :: Sound/Audio :: Speech",
"Operating System :: OS Independent",
"Programming Language :: Python :: 3",
]
dependencies = [
"torch>=2.4",
"torchaudio>=2.4",
"transformers>=5.3.0",
"accelerate",
"pydub",
"gradio",
"tensorboardX",
"webdataset",
"numpy",
"soundfile",
"psutil>=7.2.2",
# Pinned to 3.x — pyannote 4.x removed `use_auth_token` from `Inference`
# which whisperx 3.4.2 still passes, blowing up `whisperx.load_model()`
# with TypeError. whisperx tests against pyannote 3.3.2+, so we track
# that range and revisit when whisperx releases a 4-compatible build.
"pyannote-audio>=3.3.2,<4.0",
"pyinstaller>=6.19.0",
"imageio-ffmpeg>=0.6.0",
"pedalboard>=0.9.14",
# Primary ASR — cross-platform, CTranslate2-based under the hood. WhisperX
# adds wav2vec2 forced alignment (±10-30 ms word timing vs Whisper's own
# ±100-300 ms) which directly improves lip-sync on the dub pipeline.
# Pulls `faster-whisper` transitively, so a WhisperX install also provides
# the plain faster-whisper backend as a fallback for rare-language audio
# where no wav2vec2 alignment model exists.
"whisperx>=3.1.0",
"faster-whisper>=1.0.0",
# Apple Silicon-only speedup; skipped everywhere else so `uv sync` can
# succeed on Linux/Windows/mac-Intel (no mlx wheels exist for those).
"mlx-whisper>=0.2.1 ; sys_platform == 'darwin' and platform_machine == 'arm64'",
# Apple Silicon-only rich TTS library — 14+ engines (Kokoro, CSM, Dia,
# Qwen3-TTS, Chatterbox, MeloTTS, OuteTTS, Spark, Higgs-Audio, Voxtral,
# …). Gives mac-ARM users a broad engine picker in Settings. Also gated
# by platform markers because it depends on mlx (Apple Silicon only).
"mlx-audio>=0.3.0 ; sys_platform == 'darwin' and platform_machine == 'arm64'",
"demucs>=4.0.1",
"yt-dlp>=2024.12.13",
"alembic>=1.13",
# Lightweight English TTS "Turbo" tier — 25-80 MB ONNX model, 8 preset
# voices (Bella, Jasper, Luna, Bruno, Rosie, Hugo, Kiki, Leo), CPU
# realtime on any platform. Complements OmniVoice's 2.4 GB multilingual
# zero-shot clone: when the caller just needs fast English narration with
# no reference sample, this is ~100× smaller + ~10× faster. Pinned to
# the exact wheel because the project is in developer preview and the
# 0.x API is explicitly unstable.
"kittentts @ https://github.com/KittenML/KittenTTS/releases/download/0.8.1/kittentts-0.8.1-py3-none-any.whl",
]
[project.optional-dependencies]
eval = [
"jiwer==3.1.0", # WER
"librosa", # Audio processing
"s3prl", # Speech representation (HuBERT etc.)
"funasr", # ASR models
"zhconv", # Chinese character normalization
"zhon", # Chinese punctuation
"unidecode", # Unicode normalization
]
api = [
"fastapi",
"uvicorn",
"python-multipart"
]
ui = [
"gradio",
"gradio_client",
"requests",
]
[project.scripts]
omnivoice-infer = "omnivoice.cli.infer:main"
omnivoice-infer-batch = "omnivoice.cli.infer_batch:main"
omnivoice-demo = "omnivoice.cli.demo:main"
omnivoice-dub = "omnivoice.cli.dub:main"
[project.urls]
Homepage = "https://github.com/k2-fsa/OmniVoice"
Repository = "https://github.com/k2-fsa/OmniVoice"
"Bug Tracker" = "https://github.com/k2-fsa/OmniVoice/issues"
[tool.uv.sources]
# Install PyTorch with CUDA support on Linux/Windows (CUDA doesn't exist for Mac).
# NOTE: We must explicitly request them as `dependencies` above. These improved
# versions will not be selected if they're only third-party dependencies.
torch = [
{ index = "pytorch-cuda", marker = "sys_platform == 'linux' or sys_platform == 'win32'" },
]
torchaudio = [
{ index = "pytorch-cuda", marker = "sys_platform == 'linux' or sys_platform == 'win32'" },
]
[[tool.uv.index]]
name = "pytorch-cuda"
# Use PyTorch built for NVIDIA Toolkit version 12.8.
# Available versions: https://pytorch.org/get-started/locally/
url = "https://download.pytorch.org/whl/cu128"
# Only use this index when explicitly requested by `tool.uv.sources`.
explicit = true
[tool.uv]
constraint-dependencies = [
"torch==2.8.0",
"torchaudio==2.8.0",
]
[tool.hatch.metadata]
# Needed so the KittenTTS wheel-URL dep in `project.dependencies` is accepted
# by hatchling's metadata validator. KittenTTS isn't on PyPI (dev preview),
# so pulling it via GH Releases URL is the only option today.
allow-direct-references = true
[tool.hatch.build.targets.sdist]
include = ["omnivoice"]
[tool.hatch.build.targets.wheel]
packages = ["omnivoice"]
[dependency-groups]
dev = [
"httpx>=0.28.1",
"pytest>=9.0.3",
"pytest-asyncio>=1.3.0",
]