Backend:
- Split monolithic main.py into backend/{api/routers,core,schemas,services}
- core/db.py: allowlist-gated migrations, db_conn context manager (kills SQL injection on ALTER)
- core/tasks.py: lock-guarded listener add/remove/push, snapshot-before-iterate
- services/ffmpeg_utils.py: run_ffmpeg helper with concurrency semaphore, EAGAIN retry, guaranteed reap
- services/segmentation.py: Bengali/CJK/Arabic punctuation, ultra-short tier, stitch_adjacent_shorts,
bounded-loop merge; public clean_up_segments API
- services/model_manager.py: robust lock.locked() handling
- api/routers/dub_core.py: job_id traversal guard, thread-safe _active_procs, timeouts on ffmpeg/demucs,
POST /dub/cleanup-segments endpoint
- api/routers/dub_export.py: guarded SSE listener remove, ffmpeg timeouts via run_ffmpeg
- api/routers/exports.py: destination_path validation, safe source resolver, subprocess list-form
- api/routers/generation.py: contextlib.suppress on tempfile cleanup, db_conn usage, safe output-path helper
- api/routers/system.py: try/finally tmp cleanup, subprocess timeouts
- schemas/requests.py: TranslateSegment.id int->str to match hex segment IDs
- main.py: threading.Lock around crash log writes
Frontend:
- components/SearchableSelect.jsx: popover combobox with search, keyboard nav, popular+recent pins, 200-item cap
- App.jsx: wire SearchableSelect for dub language / ISO code / voice-gen language; Clean Up segments button;
fix blob URL leak (object-shaped prev in setter, unmount cleanup via ref)
- components/WaveformTimeline.jsx: explicit <video> detach instead of innerHTML='' to release decoder
- index.css: ss-* combobox styles matching Gruvbox theme
Tests:
- tests/test_segmentation.py (26 cases), test_dub_transcribe.py, test_dub_export_unique.py, conftest.py
Chore:
- .gitignore: exclude omnivoice.zip, /research/ reference clones
- Remove tracked stray root test scripts + crash_log.txt
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
121 lines
3.0 KiB
TOML
121 lines
3.0 KiB
TOML
[build-system]
|
|
requires = ["hatchling"]
|
|
build-backend = "hatchling.build"
|
|
|
|
[project]
|
|
name = "omnivoice"
|
|
version = "0.2.0"
|
|
description = "OmniVoice: Towards Omnilingual Zero-Shot Text-to-Speech with Diffusion Language Models"
|
|
readme = "README.md"
|
|
license = "Apache-2.0"
|
|
requires-python = ">=3.10"
|
|
authors = [{name = "Han Zhu"}]
|
|
keywords = [
|
|
"tts",
|
|
"text-to-speech",
|
|
"speech-synthesis",
|
|
"zero-shot",
|
|
"multilingual",
|
|
"diffusion",
|
|
"voice-cloning",
|
|
]
|
|
classifiers = [
|
|
"Intended Audience :: Science/Research",
|
|
"Intended Audience :: Developers",
|
|
|
|
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
"Topic :: Multimedia :: Sound/Audio :: Speech",
|
|
|
|
"Operating System :: OS Independent",
|
|
"Programming Language :: Python :: 3",
|
|
]
|
|
dependencies = [
|
|
"torch>=2.4",
|
|
"torchaudio>=2.4",
|
|
"transformers>=5.3.0",
|
|
"accelerate",
|
|
"pydub",
|
|
"gradio",
|
|
"tensorboardX",
|
|
"webdataset",
|
|
"numpy",
|
|
"soundfile",
|
|
"psutil>=7.2.2",
|
|
"pyannote-audio>=4.0.4",
|
|
"pyinstaller>=6.19.0",
|
|
"imageio-ffmpeg>=0.6.0",
|
|
"pedalboard>=0.9.14",
|
|
"mlx-whisper>=0.2.1",
|
|
"demucs>=4.0.1",
|
|
]
|
|
|
|
[project.optional-dependencies]
|
|
|
|
eval = [
|
|
"jiwer==3.1.0", # WER
|
|
"librosa", # Audio processing
|
|
"s3prl", # Speech representation (HuBERT etc.)
|
|
"funasr", # ASR models
|
|
"zhconv", # Chinese character normalization
|
|
"zhon", # Chinese punctuation
|
|
"unidecode", # Unicode normalization
|
|
]
|
|
api = [
|
|
"fastapi",
|
|
"uvicorn",
|
|
"python-multipart"
|
|
]
|
|
ui = [
|
|
"gradio",
|
|
"gradio_client",
|
|
"requests",
|
|
]
|
|
|
|
[project.scripts]
|
|
omnivoice-infer = "omnivoice.cli.infer:main"
|
|
omnivoice-infer-batch = "omnivoice.cli.infer_batch:main"
|
|
omnivoice-demo = "omnivoice.cli.demo:main"
|
|
|
|
[project.urls]
|
|
Homepage = "https://github.com/k2-fsa/OmniVoice"
|
|
Repository = "https://github.com/k2-fsa/OmniVoice"
|
|
"Bug Tracker" = "https://github.com/k2-fsa/OmniVoice/issues"
|
|
|
|
[tool.uv.sources]
|
|
# Install PyTorch with CUDA support on Linux/Windows (CUDA doesn't exist for Mac).
|
|
# NOTE: We must explicitly request them as `dependencies` above. These improved
|
|
# versions will not be selected if they're only third-party dependencies.
|
|
torch = [
|
|
{ index = "pytorch-cuda", marker = "sys_platform == 'linux' or sys_platform == 'win32'" },
|
|
]
|
|
torchaudio = [
|
|
{ index = "pytorch-cuda", marker = "sys_platform == 'linux' or sys_platform == 'win32'" },
|
|
]
|
|
|
|
[[tool.uv.index]]
|
|
name = "pytorch-cuda"
|
|
# Use PyTorch built for NVIDIA Toolkit version 12.8.
|
|
# Available versions: https://pytorch.org/get-started/locally/
|
|
url = "https://download.pytorch.org/whl/cu128"
|
|
# Only use this index when explicitly requested by `tool.uv.sources`.
|
|
explicit = true
|
|
|
|
[tool.uv]
|
|
constraint-dependencies = [
|
|
"torch==2.8.0",
|
|
"torchaudio==2.8.0",
|
|
]
|
|
|
|
[tool.hatch.build.targets.sdist]
|
|
include = ["omnivoice"]
|
|
|
|
[tool.hatch.build.targets.wheel]
|
|
packages = ["omnivoice"]
|
|
|
|
[dependency-groups]
|
|
dev = [
|
|
"httpx>=0.28.1",
|
|
"pytest>=9.0.3",
|
|
"pytest-asyncio>=1.3.0",
|
|
]
|