Files
VoiceStudio/tests/test_dub_qc.py
T
Palash DebnathandClaude Fable 5 a12492af07 feat(dub): second-pass ASR QC — flag lines whose dub drifts from target (Wave 3.3) (#370)
After a dub is generated, re-recognize the synthetic audio and compare what
the ASR heard against what we asked the TTS to say. Lines that drift are
flagged for the user to re-listen / re-dub — turning subtitle timing and
pronunciation from trusted math into measured truth, and doubling as an
automatic dub-quality check.

Design delta from pyvideotrans (which lets recognized text REPLACE the
subtitles wholesale): we keep the generated text authoritative and use the
second pass only for MEASUREMENT — a per-line drift score + measured
start/end that feed the incremental re-dub loop, never silently overwriting
the translation.

- services/dub_qc.py (pure, tested): word_error_rate (normalized token edit
  distance, case/punct-insensitive, script-agnostic) + score_dub (matches
  recognized segments to dub segments by time overlap, concatenates the
  hypothesis, scores drift, derives measured bounds).
- POST /dub/qc/{job_id}: runs the active ASR backend on the dubbed track in
  the GPU pool, annotates each segment with qc_drift/qc_flagged/
  qc_recognized/qc_measured_start-end (non-destructive — content untouched),
  persists, emits a qc_done job event. Opt-in, never fatal.
- Frontend: dubQc() API fn + a red 'Verify' badge on flagged segment rows
  (en.json keys; other locales fall back).

12 pure scoring tests (identical/substitution/empty/no-overlap/multi-segment
matching/measured-timing); endpoint validated in CI.

Spec 5 / parity program Wave 3.3.

Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
2026-06-12 12:50:39 +05:30

91 lines
3.1 KiB
Python

"""Second-pass ASR QC scoring (Wave 3.3 / Spec 5) — pure, no ASR/main import."""
import pytest
from services.dub_qc import score_dub, word_error_rate
# ── word_error_rate ──────────────────────────────────────────────────────────
def test_wer_identical_is_zero():
assert word_error_rate("hello there world", "hello there world") == 0.0
def test_wer_case_and_punctuation_insensitive():
assert word_error_rate("Hello, world!", "hello world") == 0.0
def test_wer_one_substitution():
# 1 edit over 3 reference tokens.
assert word_error_rate("the cat sat", "the dog sat") == pytest.approx(1 / 3)
def test_wer_empty_reference_with_hypothesis_is_one():
assert word_error_rate("", "something") == 1.0
def test_wer_both_empty_is_zero():
assert word_error_rate("", "") == 0.0
def test_wer_completely_different():
assert word_error_rate("alpha beta", "x y z") >= 1.0
# ── score_dub ────────────────────────────────────────────────────────────────
def _seg(start, end, text, sid=None):
d = {"start": start, "end": end, "text": text}
if sid is not None:
d["id"] = sid
return d
def test_clean_dub_no_flags():
dub = [_seg(0, 3, "hello world", "a"), _seg(3, 6, "good morning", "b")]
recog = [_seg(0.1, 2.9, "hello world"), _seg(3.0, 5.8, "good morning")]
out = score_dub(dub, recog)
assert [q.flagged for q in out] == [False, False]
assert out[0].seg_id == "a"
assert all(q.drift == 0.0 for q in out)
def test_drifted_segment_is_flagged():
dub = [_seg(0, 3, "the quarterly report is ready", "a")]
# ASR heard something quite different (mispronunciation / bad clone).
recog = [_seg(0, 3, "the quarter lee deport is read")]
out = score_dub(dub, recog, drift_threshold=0.5)
assert out[0].flagged is True
assert out[0].recognized_text == "the quarter lee deport is read"
def test_measured_timing_from_recognition():
dub = [_seg(0.0, 5.0, "hello", "a")]
recog = [_seg(0.4, 1.2, "hello")]
out = score_dub(dub, recog)
assert out[0].new_start == pytest.approx(0.4)
assert out[0].new_end == pytest.approx(1.2)
def test_segment_with_no_overlap_scores_full_drift():
dub = [_seg(0, 3, "spoken line", "a")]
recog = [_seg(10, 12, "elsewhere")] # no time overlap
out = score_dub(dub, recog)
assert out[0].recognized_text == ""
assert out[0].drift == 1.0 and out[0].flagged
assert out[0].new_start is None
def test_multiple_recognized_segments_concatenate():
dub = [_seg(0, 6, "one two three four", "a")]
recog = [_seg(0, 3, "one two"), _seg(3, 6, "three four")]
out = score_dub(dub, recog)
assert out[0].recognized_text == "one two three four"
assert out[0].drift == 0.0
def test_seg_ids_override():
dub = [_seg(0, 3, "x")]
out = score_dub(dub, [_seg(0, 3, "x")], seg_ids=["custom"])
assert out[0].seg_id == "custom"