Files
VoiceStudio/pyproject.toml
T
Palash Debnath ca8a2e8eb8 feat(audiobook): PDF ingest for /audiobook/import (ebook-in core value) (#459)
The audiobook importer accepted .txt/.md/.epub but not PDF — the single most
common "ebook in" format. Add a pure `pdf_to_chapter_script(data)` that
extracts the text layer page-by-page and runs it through the existing
chapterizer, so PDFs land in the same `# Heading` + body grammar EPUB and
plaintext already produce (one front door onto the unchanged render pipeline).

- Dep: `pypdf>=4.0` — pure-Python, MIT, zero native deps, so PDF import behaves
  identically on macOS/Windows/Linux (default-feature cross-platform rule).
  EPUB + plaintext stay stdlib-only; only PDF needs a real parser.
- Robustness, surfaced as actionable 400s rather than silent empty imports:
  corrupt file, password-protected (empty-password decrypt attempted first),
  scanned/image-only (no text layer → clear "scanned PDF" message), and a
  page-count ceiling. A single unparseable page is skipped, not fatal.
- Route: `.pdf` branch in audiobook_import; frontend accept filter +
  api-client doc updated to `.txt,.md,.epub,.pdf`.

tests/test_longform_import.py: 5 PDF cases (extract+chapterize, no-marker
single chapter, corrupt, image-only, page-cap) using a hand-built in-memory
PDF — no PDF-authoring test dep, mirroring the in-memory-EPUB approach.
16 passed; frontend suite 401; CJK guard green.
2026-06-14 17:14:28 +05:30

238 lines
9.6 KiB
TOML
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
[build-system]
requires = ["hatchling"]
build-backend = "hatchling.build"
[project]
name = "omnivoice"
version = "0.3.6"
description = "OmniVoice: Towards Omnilingual Zero-Shot Text-to-Speech with Diffusion Language Models"
readme = "README.md"
# Free and open-source under the GNU Affero General Public License v3 (see
# LICENSE). A commercial license is available for proprietary/closed-source use
# without AGPL obligations — contact OmniVoice@palash.dev. The bundled omnivoice/
# TTS model by Han Zhu remains Apache-2.0 upstream (Apache-2.0 is AGPL-compatible).
license = "AGPL-3.0-only"
requires-python = ">=3.11"
authors = [{name = "Han Zhu"}]
keywords = [
"tts",
"text-to-speech",
"speech-synthesis",
"zero-shot",
"multilingual",
"diffusion",
"voice-cloning",
]
classifiers = [
"Intended Audience :: Science/Research",
"Intended Audience :: Developers",
"Topic :: Scientific/Engineering :: Artificial Intelligence",
"Topic :: Multimedia :: Sound/Audio :: Speech",
"Operating System :: OS Independent",
"Programming Language :: Python :: 3",
]
dependencies = [
"torch>=2.4",
"torchaudio>=2.4",
"transformers>=5.3.0",
"accelerate",
"pydub",
"gradio",
"tensorboardX",
"webdataset",
"numpy",
"soundfile",
# whisperx / faster-whisper import `pkg_resources` at runtime. setuptools
# 80+ DROPPED the bundled pkg_resources, so an unpinned ">=75" now resolves
# to a version WITHOUT it → "No module named 'pkg_resources'", which both
# breaks WhisperX transcription and makes its is_available() report false
# ("No ASR backend is ready"). Cap below 80 so pkg_resources stays present
# (issue #224; also #58). Revisit when whisperx/faster-whisper drop the
# pkg_resources dependency.
"setuptools>=75,<80",
"psutil>=7.2.2",
# Pinned to 3.x — pyannote 4.x removed `use_auth_token` from `Inference`
# which whisperx 3.4.2 still passes, blowing up `whisperx.load_model()`
# with TypeError. whisperx tests against pyannote 3.3.2+, so we track
# that range and revisit when whisperx releases a 4-compatible build.
"pyannote-audio>=3.3.2,<4.0",
"pyinstaller>=6.19.0",
"imageio-ffmpeg>=0.6.0",
"pedalboard>=0.9.14",
# Primary ASR — cross-platform, CTranslate2-based under the hood. WhisperX
# adds wav2vec2 forced alignment (±10-30 ms word timing vs Whisper's own
# ±100-300 ms) which directly improves lip-sync on the dub pipeline.
# Pulls `faster-whisper` transitively, so a WhisperX install also provides
# the plain faster-whisper backend as a fallback for rare-language audio
# where no wav2vec2 alignment model exists.
"whisperx>=3.1.0",
"faster-whisper>=1.0.0",
# Apple Silicon-only speedup; skipped everywhere else so `uv sync` can
# succeed on Linux/Windows/mac-Intel (no mlx wheels exist for those).
"mlx-whisper>=0.2.1 ; sys_platform == 'darwin' and platform_machine == 'arm64'",
# Apple Silicon-only rich TTS library — 14+ engines (Kokoro, CSM, Dia,
# Qwen3-TTS, Chatterbox, MeloTTS, OuteTTS, Spark, Higgs-Audio, Voxtral,
# …). Gives mac-ARM users a broad engine picker in Settings. Also gated
# by platform markers because it depends on mlx (Apple Silicon only).
"mlx-audio>=0.3.0 ; sys_platform == 'darwin' and platform_machine == 'arm64'",
"demucs>=4.0.1",
"yt-dlp>=2024.12.13",
"alembic>=1.13",
# Lightweight English TTS "Turbo" tier — 25-80 MB ONNX model, 8 preset
# voices (Bella, Jasper, Luna, Bruno, Rosie, Hugo, Kiki, Leo), CPU
# realtime on any platform. Complements OmniVoice's 2.4 GB multilingual
# zero-shot clone: when the caller just needs fast English narration with
# no reference sample, this is ~100× smaller + ~10× faster. Pinned to
# the exact wheel because the project is in developer preview and the
# 0.x API is explicitly unstable.
"kittentts @ https://github.com/KittenML/KittenTTS/releases/download/0.8.1/kittentts-0.8.1-py3-none-any.whl",
# Invisible audio watermarking — embeds imperceptible neural watermarks
# in AI-generated speech for provenance detection (SynthID-like).
# MIT license, ~5ms per segment on CPU, 16-bit message payload.
"audioseal>=0.1.3",
# API server — always needed for the Studio UI.
"fastapi",
"scalar-fastapi",
"uvicorn",
"python-multipart",
# Required by uvicorn for WebSocket support (real-time sidebar events).
"websockets",
# Offline translation — listed as builtin in the engine registry so the
# "Argos (Local, Fast)" option in the Dub tab works out-of-the-box.
"argostranslate>=1.9.0",
# Phase 1 AUTH-02: Fernet symmetric encryption + scrypt KDF for the
# at-rest HF token in the SQLite settings store. Pulled in directly so
# we don't depend on a transitive arrival via pyannote/huggingface_hub
# (Assumption A1 in RESEARCH.md was checked at execute-time and proved
# false — `cryptography` is not on the install path today).
"cryptography>=41",
"mcp>=1.2",
# Fast model downloads (FDL plan). huggingface_hub arrives transitively via
# transformers, but we pin it directly so the Xet fast-download path can't
# silently disappear on a resolve, and we pull `hf-xet` explicitly: it is
# the content-defined-chunking, parallel byte-range downloader that gives
# IDM/uGet-style speed for Xet-backed repos (the entire current model
# catalog — see .planning/quick/260613-fdl-fast-model-downloads/SPIKE).
# hf-xet is 64-bit only (fine for every OmniVoice target). Do NOT add
# `hf_transfer` — it is deprecated in favour of Xet and breaks progress
# callbacks (the accurate-progress work in this same plan depends on tqdm).
"huggingface_hub>=1.7",
"hf-xet>=1.1",
# Audiobook PDF ingest (/audiobook/import). Pure-Python, MIT, zero native
# deps → identical behaviour on macOS/Windows/Linux (default-parity rule).
# EPUB + plaintext stay stdlib-only; only PDF needs a real parser, and
# pypdf is the lightest one that ships no C extensions.
"pypdf>=4.0",
]
[project.optional-dependencies]
eval = [
"jiwer==3.1.0", # WER
"librosa", # Audio processing
"s3prl", # Speech representation (HuBERT etc.)
"funasr", # ASR models
"zhconv", # Chinese character normalization
"zhon", # Chinese punctuation
"unidecode", # Unicode normalization
]
ui = [
"gradio",
"gradio_client",
"requests",
]
# Phase 3 Plan 03-01 — Supertonic-3 opt-in engine. CPU-only ONNX TTS,
# 31 languages, ~99M params, ~400 MB model on first use. Default
# `uv sync --no-dev` does NOT install this; users opt in with
# `uv sync --extra supertonic` after accepting the OpenRAIL-M model
# license in Settings → Engines.
#
# Publisher verified per Plan 03-01 Task 1 (Package Legitimacy Audit):
# • PyPI maintainers = Yu Yechan / Juheon Lee / Hyeongju Kim (Supertone Inc.)
# • Repository = github.com/supertone-inc/supertonic-py
# • Same publisher ships supertonic-js on npm (same maintainer email)
# • Wheel inspected: pure-Python, no postinstall scripts, no subprocess/exec
# at module top level. ``supertonic.config.MODEL_CONFIGS["supertonic-3"]``
# itself pins the HF model revision by SHA — we re-pin to the same
# SHA in backend/engines/supertonic3/constants.py for TTS-03.
supertonic = [
"supertonic==1.3.1",
]
[project.scripts]
omnivoice-infer = "omnivoice.cli.infer:main"
omnivoice-infer-batch = "omnivoice.cli.infer_batch:main"
omnivoice-demo = "omnivoice.cli.demo:main"
omnivoice-dub = "omnivoice.cli.dub:main"
[project.urls]
Homepage = "https://github.com/k2-fsa/OmniVoice"
Repository = "https://github.com/k2-fsa/OmniVoice"
"Bug Tracker" = "https://github.com/k2-fsa/OmniVoice/issues"
[tool.uv.sources]
# Install PyTorch with CUDA support on Linux/Windows (CUDA doesn't exist for Mac).
# NOTE: We must explicitly request them as `dependencies` above. These improved
# versions will not be selected if they're only third-party dependencies.
torch = [
{ index = "pytorch-cuda", marker = "sys_platform == 'linux' or sys_platform == 'win32'" },
]
torchaudio = [
{ index = "pytorch-cuda", marker = "sys_platform == 'linux' or sys_platform == 'win32'" },
]
[[tool.uv.index]]
name = "pytorch-cuda"
# Use PyTorch built for NVIDIA Toolkit version 12.8.
# Available versions: https://pytorch.org/get-started/locally/
url = "https://download.pytorch.org/whl/cu128"
# Only use this index when explicitly requested by `tool.uv.sources`.
explicit = true
[tool.uv]
constraint-dependencies = [
"torch==2.8.0",
"torchaudio==2.8.0",
]
[tool.hatch.metadata]
# Needed so the KittenTTS wheel-URL dep in `project.dependencies` is accepted
# by hatchling's metadata validator. KittenTTS isn't on PyPI (dev preview),
# so pulling it via GH Releases URL is the only option today.
allow-direct-references = true
[tool.hatch.build.targets.sdist]
include = ["omnivoice"]
[tool.hatch.build.targets.wheel]
packages = ["omnivoice"]
[dependency-groups]
dev = [
"httpx>=0.28.1",
"pytest>=9.0.3",
"pytest-asyncio>=1.3.0",
"pytest-cov>=6.0",
]
[tool.pytest.ini_options]
# Bare `pytest` would otherwise walk into `research/` (1.2 GB of vendored
# upstream projects, each with its own test_*.py that calls sys.exit at
# module level) and INTERNALERROR. `backend/tests/` runs separately because
# it stubs `core.config` in sys.modules, which pollutes import state for
# other modules in the same session.
testpaths = ["tests"]
norecursedirs = [
"research",
"omnivoice/training",
"omnivoice/eval",
"frontend",
"deploy",
".venv",
"node_modules",
"omnivoice_data",
"backend/omnivoice_data",
]