Files
VoiceStudio/tests/test_longform_render.py
T
Palash DebnathandClaude Opus 4.8 7af5143fac feat(audiobook): per-chapter preview + resume + chapter fault-isolation (PR 3/8) (#411)
* feat(audiobook): per-chapter preview + resume + chapter fault-isolation (PR 3/8)

Builds on the shared core (#408) and metadata UI (#409). Chapter-level control,
the spec's PR 3.

Shared core:
- chapter_cache_key(spans, sr, engine_id, voice_sig) — deterministic content
  hash of a chapter's audio inputs. Same inputs → reuse; any change (text,
  voice, order, pauses, sr, engine, resolved-voice signature) → re-render.

Backend (audiobook router):
- Chapter WAVs are now content-addressed in OUTPUTS_DIR/audiobook_cache. A
  re-run after a failure/interruption reuses already-rendered chapters and only
  synthesizes the missing/changed ones (resume). Job emits `cached` per chapter
  and `cached_chapters`/`failed_chapters` on done.
- Per-chapter fault isolation: a chapter that throws emits `chapter_error` and
  the job continues; the m4b assembles from the successful chapters. Re-running
  retries only the failed (un-cached) chapters.
- POST /audiobook/preview — render a single chapter to audition it; shares the
  same cache so a preview warms the full run and a re-preview is instant.
- _build_synth now exposes resolve + engine_id; _prepare_synth unifies the
  omnivoice/generic paths for both the job and preview.

Frontend:
- Plan view: a ▶ preview button per chapter with inline playback.
- Done panel: "reused N chapters" + "N failed — click Create to retry" notes.

Tests: chapter_cache_key determinism + sensitivity (8); preview validation +
cache-hit-skips-synth (3). 55 backend + 326 frontend green; build clean.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

* fix(audiobook): mark cache-key SHA1 usedforsecurity=False (bandit B324)

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

---------

Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-13 14:19:48 +05:30

215 lines
8.5 KiB
Python

"""Shared long-form render core (Stories + Audiobook convergence).
Pure builders for the chapterized mux: FFMETADATA (global tags + chapters),
concat list, loudness filter, cover validation, and the ffmpeg render argv.
All unit-testable without ffmpeg/torch/GPU.
"""
from __future__ import annotations
import pytest
from services.longform_render import (
LOUDNESS_PRESETS,
build_concat_list,
build_ffmetadata,
build_loudnorm_filter,
build_render_cmd,
chapter_cache_key,
validate_cover_image,
)
# ── loudness ────────────────────────────────────────────────────────────────
def test_loudnorm_acx_filter():
f = build_loudnorm_filter("acx")
assert f == "loudnorm=I=-19.0:TP=-3.0:LRA=11.0"
def test_loudnorm_podcast_filter():
assert build_loudnorm_filter("podcast") == "loudnorm=I=-16.0:TP=-1.5:LRA=11.0"
def test_loudnorm_case_insensitive():
assert build_loudnorm_filter("ACX") == build_loudnorm_filter("acx")
@pytest.mark.parametrize("val", [None, "", "off", "none", "bogus"])
def test_loudnorm_off_or_unknown_is_none(val):
assert build_loudnorm_filter(val) is None
def test_loudness_presets_within_acx_window():
# ACX wants integrated near -19 LUFS and a -3 dB peak ceiling.
acx = LOUDNESS_PRESETS["acx"]
assert -23.0 <= acx.i <= -18.0
assert acx.tp == -3.0
# ── FFMETADATA ──────────────────────────────────────────────────────────────
def test_ffmetadata_chapters_only_matches_legacy_shape():
doc = build_ffmetadata([("One", 1000), ("Two", 500)])
assert doc.startswith(";FFMETADATA1\n")
assert "[CHAPTER]\nTIMEBASE=1/1000\nSTART=0\nEND=1000\ntitle=One" in doc
assert "START=1000\nEND=1500\ntitle=Two" in doc
# No global tags when none supplied.
assert "artist=" not in doc
def test_ffmetadata_global_tags_mapped_and_ordered():
doc = build_ffmetadata(
[("Ch", 1000)],
global_meta={
"title": "My Book", "author": "Ada", "narrator": "Grace",
"year": "2026", "genre": "Sci-Fi", "description": "A tale",
},
)
# field → tag mapping (author→artist, narrator→composer, year→date,
# description→comment) and stable order (title before artist).
head = doc.split("[CHAPTER]")[0]
assert head.index("title=My Book") < head.index("artist=Ada")
assert "composer=Grace" in head
assert "date=2026" in head
assert "genre=Sci-Fi" in head
assert "comment=A tale" in head
def test_ffmetadata_skips_empty_global_values():
doc = build_ffmetadata([("Ch", 1)], global_meta={"title": "T", "author": " ", "genre": None})
head = doc.split("[CHAPTER]")[0]
assert "title=T" in head
assert "artist=" not in head # whitespace-only dropped
assert "genre=" not in head # None dropped
def test_ffmetadata_escapes_special_chars():
doc = build_ffmetadata([("a=b;c#d", 100)], global_meta={"title": "x=y"})
assert r"title=x\=y" in doc
assert r"title=a\=b\;c\#d" in doc
# ── concat list ─────────────────────────────────────────────────────────────
def test_concat_list_quotes_and_escapes():
out = build_concat_list(["/a/one.wav", "/weird/it's here.wav"])
assert "file '/a/one.wav'" in out
assert "file '/weird/it'\\''s here.wav'" in out
# ── cover validation ────────────────────────────────────────────────────────
def test_cover_valid(tmp_path):
p = tmp_path / "cover.jpg"
p.write_bytes(b"\xff\xd8\xff" + b"x" * 100)
assert validate_cover_image(str(p)) is True
def test_cover_rejects_missing_and_bad_type(tmp_path):
assert validate_cover_image(None) is False
assert validate_cover_image(str(tmp_path / "nope.jpg")) is False
txt = tmp_path / "c.txt"
txt.write_bytes(b"hi")
assert validate_cover_image(str(txt)) is False
def test_cover_rejects_oversize(tmp_path):
big = tmp_path / "big.png"
big.write_bytes(b"\x89PNG" + b"0" * (8 * 1024 * 1024 + 1))
assert validate_cover_image(str(big)) is False
def test_cover_rejects_empty(tmp_path):
empty = tmp_path / "empty.jpg"
empty.write_bytes(b"")
assert validate_cover_image(str(empty)) is False
# ── render command ──────────────────────────────────────────────────────────
def test_render_cmd_m4b_default():
cmd = build_render_cmd("ffmpeg", "concat.txt", "ch.ffmeta", "out.m4b")
assert cmd[0] == "ffmpeg"
assert "-f" in cmd and "concat" in cmd
assert cmd[-3:] == ["-f", "mp4", "out.m4b"]
assert "-c:a" in cmd and "aac" in cmd
assert "+faststart" in cmd
assert "-map_metadata" in cmd
# no cover, no loudnorm by default
assert "attached_pic" not in cmd
assert "-af" not in cmd
def test_render_cmd_mp3_format():
cmd = build_render_cmd("ffmpeg", "c.txt", "m.ffmeta", "out.mp3", fmt="mp3")
assert "libmp3lame" in cmd
assert cmd[-3:] == ["-f", "mp3", "out.mp3"]
assert "+faststart" not in cmd
def test_render_cmd_bitrate_validation():
ok = build_render_cmd("ffmpeg", "c", "m", "o", bitrate="192k")
assert "192k" in ok
bad = build_render_cmd("ffmpeg", "c", "m", "o", bitrate="; rm -rf /")
assert "128k" in bad # rejected → default
assert "; rm -rf /" not in bad
def test_render_cmd_loudnorm_adds_af():
cmd = build_render_cmd("ffmpeg", "c", "m", "o", loudness="acx")
assert "-af" in cmd
assert any(a.startswith("loudnorm=") for a in cmd)
def test_render_cmd_with_cover(tmp_path):
cover = tmp_path / "cover.jpg"
cover.write_bytes(b"\xff\xd8\xff" + b"x" * 50)
cmd = build_render_cmd("ffmpeg", "c", "m", "o.m4b", cover_path=str(cover))
# cover becomes input 2, mapped as attached_pic, copied
assert str(cover) in cmd
assert "-map" in cmd and "2:v" in cmd
assert "attached_pic" in cmd
assert "-c:v" in cmd and "copy" in cmd
def test_render_cmd_drops_invalid_cover(tmp_path):
cmd = build_render_cmd("ffmpeg", "c", "m", "o.m4b", cover_path=str(tmp_path / "missing.jpg"))
assert "attached_pic" not in cmd # silently dropped, render still proceeds
# ── chapter cache key (resume) ──────────────────────────────────────────────
_SPANS = [(None, "Once upon a time.", 350), ("narrator", "The end.", 0)]
def test_cache_key_deterministic():
a = chapter_cache_key(_SPANS, sample_rate=24000, engine_id="omnivoice")
b = chapter_cache_key(list(_SPANS), sample_rate=24000, engine_id="omnivoice")
assert a == b and len(a) == 20
@pytest.mark.parametrize("mutate", [
lambda: chapter_cache_key([(None, "Different.", 350), ("narrator", "The end.", 0)],
sample_rate=24000, engine_id="omnivoice"), # text
lambda: chapter_cache_key([("x", "Once upon a time.", 350), ("narrator", "The end.", 0)],
sample_rate=24000, engine_id="omnivoice"), # voice
lambda: chapter_cache_key([(None, "Once upon a time.", 500), ("narrator", "The end.", 0)],
sample_rate=24000, engine_id="omnivoice"), # pause
lambda: chapter_cache_key(list(reversed(_SPANS)), sample_rate=24000, engine_id="omnivoice"), # order
lambda: chapter_cache_key(_SPANS, sample_rate=44100, engine_id="omnivoice"), # sr
lambda: chapter_cache_key(_SPANS, sample_rate=24000, engine_id="kokoro"), # engine
lambda: chapter_cache_key(_SPANS, sample_rate=24000, engine_id="omnivoice",
voice_sig={"narrator": "ref.wav|warm|7"}), # voice sig
])
def test_cache_key_changes_on_any_input(mutate):
base = chapter_cache_key(_SPANS, sample_rate=24000, engine_id="omnivoice")
assert mutate() != base
def test_cache_key_voice_sig_order_irrelevant():
a = chapter_cache_key(_SPANS, sample_rate=24000, engine_id="omnivoice",
voice_sig={"a": "1", "b": "2"})
b = chapter_cache_key(_SPANS, sample_rate=24000, engine_id="omnivoice",
voice_sig={"b": "2", "a": "1"})
assert a == b