diff --git a/CHANGELOG.md b/CHANGELOG.md index 2bf42b70..f806c846 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -10,6 +10,8 @@ the frozen-backend fallback mirror it for their toolchains. **Highlights** +- Number ranges are read as ranges — "20~30초" no longer comes out as "23" (#1821) + ### Changed ### Added @@ -18,6 +20,8 @@ the frozen-backend fallback mirror it for their toolchains. ### Fixed +- Digit ranges written with a tilde are now spoken: `20~30초` read aloud as "23" because the separator never reached the engine, and the two numbers ran together (#1821) + ## [0.5.2] — 2026-09-02 diff --git a/backend/services/text_normalization.py b/backend/services/text_normalization.py index 274124d8..20e60ade 100644 --- a/backend/services/text_normalization.py +++ b/backend/services/text_normalization.py @@ -112,6 +112,13 @@ _FULL_NAME_TO_CODE = { "vietnamese": "vi", "kazakh": "kz", "standard arabic": "ar", + # Below: inert for num2words (absent from _NUM2WORDS_LANGS, which reads + # digits natively for these scripts), present so _plain_lang_code can + # resolve them for the digit-range rule. + "korean": "ko", + "japanese": "ja", + "chinese": "zh", + "mandarin chinese": "zh", } # ISO codes whose num2words locale name differs. @@ -178,6 +185,68 @@ def _num2words_lang(language: Optional[str]) -> Optional[str]: return None +def _plain_lang_code(language: Optional[str]) -> Optional[str]: + """Resolve a request language to a bare ISO code, with no num2words gate. + + :func:`_num2words_lang` answers "may I call num2words for this?" and so + returns ``None`` for ko/ja/zh/th/vi. Rules that are not num2words-backed + need the code itself, which is what this returns. + """ + if not language: + return None + s = str(language).strip().lower() + if not s or s == "auto": + return None + code = _FULL_NAME_TO_CODE.get(s) + if code: + return code + m = _ISO_CODE_RE.match(s) + if m: + return _ISO_ALIASES.get(m.group(1), m.group(1)) + return None + + +# ── Digit ranges ───────────────────────────────────────────────────────────── +# "20~30" loses its separator at the engine and reads as ONE number: OmniVoice +# says "이십삼" (23) for "20~30초". Speak the separator instead. Verified by +# rendering each form and transcribing it back (ko, OmniVoice): +# "20~30초" → heard "23초" ✗ +# "20에서 30초" → heard "20에서 30초" ✓ +# Only the tilde family is rewritten — those are unambiguously range marks +# between digits. An ASCII hyphen is left alone on purpose: it also spells +# dates, phone numbers and product codes, where "to" would be wrong. +#: Spacing is part of the form, not decoration: a Korean postposition binds to +#: the numeral ("20에서 30"), Japanese and Chinese set no spaces at all, and +#: English needs them on both sides. +_RANGE_FORM = { + "ko": "{a}에서 {b}", + "ja": "{a}から{b}", + "zh": "{a}到{b}", + "en": "{a} to {b}", +} + +#: ASCII tilde, wave dash, fullwidth tilde — Japanese and Korean IMEs emit the +#: latter two, so all three have to match. +#: +#: The neighbour guards block digits/decimal marks (so a longer number is never +#: split) and ASCII letters (so a product code like "AB20~30CD" is left alone), +#: but deliberately allow everything else: CJK writes its unit right against +#: the digits — "20~30초", "20〜30分", "20~30秒" — and a \w guard would reject +#: exactly the cases this rule exists for. +_NUM_RANGE_RE = re.compile( + r"(? str: + """``20~30`` → ``20에서 30``. No-op where the spoken form isn't verified.""" + form = _RANGE_FORM.get(lang) + if not form: + return text + return _NUM_RANGE_RE.sub(lambda m: form.format(a=m.group(1), b=m.group(2)), text) + + # ── Universal safety filters (all languages) ───────────────────────────────── # Zero-width & bidi controls, C0/C1 controls (except \t \n \r), BOM, U+FFFD. @@ -487,6 +556,11 @@ def normalize_text(text: str, language: Optional[str] = None) -> str: if not text: return text or "" out = _safety_filters(text) + # Runs outside the num2words gate below: ko/ja/zh keep their digits (that + # gate returns None for them) but still need the range mark spoken. + plain = _plain_lang_code(language) + if plain: + out = _outside_brackets(out, lambda t: _speak_number_ranges(t, plain)) lang = _num2words_lang(language) if lang: if lang in _ABBREV_COMPILED: diff --git a/tests/test_no_hardcoded_cjk.py b/tests/test_no_hardcoded_cjk.py index b247fbf0..6e0dc19b 100644 --- a/tests/test_no_hardcoded_cjk.py +++ b/tests/test_no_hardcoded_cjk.py @@ -57,6 +57,7 @@ _ALLOWED_FILES = { "backend/services/segmentation.py", "backend/services/sentence_chunker.py", # streaming-TTS terminator tables (Patter port, Wave 1.4) "backend/services/subtitle_segmenter.py", + "backend/services/text_normalization.py", # spoken range word per language ("20~30" must not read as one number) "backend/core/http_headers.py", # docstring quotes the CJK filename that 500'd the header (#1262) "frontend/src/components/DubSegmentRow.jsx", "frontend/src/components/StoriesEditor.jsx", diff --git a/tests/test_text_normalization.py b/tests/test_text_normalization.py index 3e85e89e..9f3e749e 100644 --- a/tests/test_text_normalization.py +++ b/tests/test_text_normalization.py @@ -65,6 +65,18 @@ _CHANGE_CASES = [ ("French", "il a 42 chats", "il a quarante-deux chats"), ("French", "Mme Dupont arrive", "Madame Dupont arrive"), ("Russian", "у меня 42 кота", "у меня сорок два кота"), + # Digit ranges: the tilde has to be SPOKEN or the engine mashes the two + # numbers into one ("20~30초" was read as "이십삼"). Spacing is part of the + # per-language form — a Korean postposition binds to its numeral, Japanese + # and Chinese set no spaces, English needs them. + ("Korean", "대략 20~30초짜리", "대략 20에서 30초짜리"), + ("Korean", "가격은 20~30만원", "가격은 20에서 30만원"), + ("ko", "20~30초", "20에서 30초"), + ("Japanese", "20〜30分ぐらい", "20から30分ぐらい"), # wave dash U+301C + ("Japanese", "20~30分ぐらい", "20から30分ぐらい"), # fullwidth U+FF5E + ("Chinese", "大约需要20~30秒", "大约需要20到30秒"), + # EN runs the range through num2words afterwards, as it does any digit + ("English", "It takes 20~30 seconds", "It takes twenty to thirty seconds"), # Universal safety filters (language-independent) (None, "hello​ ‍world", "hello world"), (None, "too many\t spaces", "too many spaces"), @@ -124,6 +136,16 @@ _UNCHANGED_CASES = [ ("English", "I said no. Fine."), # the word "no.", not "number" ("English", "down main st. Anyway"), # lowercase "st." is not Saint ("German", "es kostet 3,5 Euro"), # decimal comma: ambiguous + # Digit ranges: only the tilde family is a range mark. Everything else that + # sits between digits means something other than "to". + ("Korean", "대략 20-30초"), # ASCII hyphen: also dates/phones + ("Korean", "2026-09-05 회의"), # date + ("Korean", "010-1234-5678"), # phone number + ("Korean", "AB20~30CD"), # product code, not a range + ("Korean", "1.20~30.5"), # decimals either side + ("Japanese", "そうですね〜"), # tilde not between digits + ("Vietnamese", "20~30 giây"), # no verified spoken form + (None, "20~30초"), # no language given # Unsupported languages keep every digit (num2words unmapped) ("Japanese", "42 cats and 3.5 stars at 3:30"), ("Thai", "42 cats"),