From 033f8f2f64b30e264077ec04e955767f368ca1bc Mon Sep 17 00:00:00 2001 From: Palash Debnath <4178343+debpalash@users.noreply.github.com> Date: Thu, 10 Sep 2026 15:06:45 -0700 Subject: [PATCH 1/3] fix(asr): PyTorch Whisper budgets the VRAM its model needs The CUDA preflight demanded 5.0 GB of free VRAM for every model, sized for full large-v3. The default model is large-v3-turbo, which the pipeline loads in fp16: about 1.6 GB of weights, not the 3.2 GiB fp32 figure in the old comment. So a 6 GB card with nothing else resident reported 5.0 GB free and was sent to CPU every time, although CUDA ran the same audio in 37 s against minutes on CPU. The budget is now fp16 weights + 1.5 GB workspace (batch 16) + 0.5 GB headroom, per model and capped at the old 5.0 GB. That is 3.6 GB for turbo and 5.0 GB for full large-v3 and any unrecognised model. Both CPU-fallback warnings name the model and the OMNIVOICE_ASR_VRAM_PREFLIGHT=0 opt-out. The engine doc lists the budgets. Fixes #2041 --- backend/services/asr_backend.py | 63 ++++++++++++++----- docs/engines/pytorch-whisper.md | 24 +++++-- tests/test_pytorch_whisper_fallback.py | 2 +- .../test_pytorch_whisper_vram_budget_2041.py | 50 +++++++++++++++ 4 files changed, 118 insertions(+), 21 deletions(-) create mode 100644 tests/test_pytorch_whisper_vram_budget_2041.py diff --git a/backend/services/asr_backend.py b/backend/services/asr_backend.py index acc3ae25..a4e1c193 100644 --- a/backend/services/asr_backend.py +++ b/backend/services/asr_backend.py @@ -648,7 +648,8 @@ class WhisperXBackend(ASRBackend): logger.warning( "whisperx VRAM preflight: %.1f GB free is too little for %s on CUDA " "(needs ≥%.1f GB even at int8) — using CPU int8 instead. Free VRAM " - "(flush the TTS model, or close other GPU apps) for GPU-speed ASR. (#723)", + "(flush the TTS model, or close other GPU apps) for GPU-speed ASR, or " + "set OMNIVOICE_ASR_VRAM_PREFLIGHT=0 to skip this check. (#723)", free, self._model_name, self._CUDA_VRAM_BUDGET_GB["int8"] * scale, ) @@ -1292,10 +1293,42 @@ class PyTorchWhisperBackend(ASRBackend): # Reuses the `_asr_pipe` attached to the TTS model when available. self._pipe = asr_pipe - # whisper-large-v3-turbo occupies roughly 3.2 GiB before generation adds - # its encoder/decoder workspace. Loading it onto a nearly full card works, - # then the first transcribe fails with a CUDA OOM and yields zero segments. - _CUDA_VRAM_BUDGET_GB = 5.0 + # Free VRAM needed on CUDA, where _ensure_pipe loads fp16 weights: the + # weights (parameters x 2 bytes), the batch-16 decode workspace, and + # headroom. Loading onto a nearly full card works, then the first + # transcribe fails with a CUDA OOM and yields zero segments, so the device + # pick checks this against actually-free VRAM first. + # + # #2041: this was a flat 5.0 GB for every model, sized for full large-v3. + # The default is large-v3-turbo (0.81B parameters, about 1.6 GB in fp16), + # so a 6 GB card with nothing else resident reported 5.0 GB free and was + # sent to CPU every time, although CUDA transcribed the same audio in 37 s. + _CUDA_VRAM_BUDGET_GB = 5.0 # full large-v3, and any model not listed below + _FP16_WEIGHTS_GB = ( + ("turbo", 1.6), ("large", 3.1), ("medium", 1.5), + ("small", 0.5), ("base", 0.15), ("tiny", 0.08), + ) + _CUDA_WORKSPACE_GB = 1.5 # batch 16 x 15 s chunks + _CUDA_HEADROOM_GB = 0.5 + + @classmethod + def _cuda_budget_gb(cls, model_name: str) -> float: + """Free VRAM (GB) this model needs on CUDA; never above the 5.0 GB + that full large-v3 was measured to need.""" + name = (model_name or "").lower() + for key, weights_gb in cls._FP16_WEIGHTS_GB: + if key in name: + return min( + cls._CUDA_VRAM_BUDGET_GB, + weights_gb + cls._CUDA_WORKSPACE_GB + cls._CUDA_HEADROOM_GB, + ) + return cls._CUDA_VRAM_BUDGET_GB + + @staticmethod + def _model_name() -> str: + return os.environ.get( + "OMNIVOICE_PYTORCH_ASR_MODEL", "openai/whisper-large-v3-turbo" + ) @classmethod def is_available(cls) -> tuple[bool, str]: @@ -1306,7 +1339,7 @@ class PyTorchWhisperBackend(ASRBackend): return False, f"transformers not installed: {e}" @classmethod - def _pick_device(cls) -> str: + def _pick_device(cls, model_name: str | None = None) -> str: from services.model_manager import get_best_device device = str(get_best_device()) @@ -1321,14 +1354,18 @@ class PyTorchWhisperBackend(ASRBackend): free_gb = free / 1024**3 except Exception: # noqa: BLE001 — an unavailable probe must not block ASR return device - if free_gb >= cls._CUDA_VRAM_BUDGET_GB: + model_name = model_name or cls._model_name() + budget_gb = cls._cuda_budget_gb(model_name) + if free_gb >= budget_gb: return device logger.warning( "PyTorch Whisper VRAM preflight: %.1f GB free < %.1f GB needed " - "for reliable CUDA transcription — using CPU instead. Close other " - "GPU apps or Flush models to restore GPU-speed ASR.", + "for %s on CUDA — using CPU instead. Close other GPU apps or Flush " + "models to restore GPU-speed ASR, or set " + "OMNIVOICE_ASR_VRAM_PREFLIGHT=0 to skip this check.", free_gb, - cls._CUDA_VRAM_BUDGET_GB, + budget_gb, + model_name, ) return "cpu" @@ -1351,10 +1388,8 @@ class PyTorchWhisperBackend(ASRBackend): # constructor and this path is skipped. import torch from transformers import pipeline as hf_pipeline - model_name = os.environ.get( - "OMNIVOICE_PYTORCH_ASR_MODEL", "openai/whisper-large-v3-turbo" - ) - device = self._pick_device() + model_name = self._model_name() + device = self._pick_device(model_name) asr_dtype = torch.float16 if str(device).startswith("cuda") else torch.float32 logger.info( "PyTorchWhisperBackend: loading standalone ASR pipeline %s on %s", diff --git a/docs/engines/pytorch-whisper.md b/docs/engines/pytorch-whisper.md index 815a327e..15ed1c56 100644 --- a/docs/engines/pytorch-whisper.md +++ b/docs/engines/pytorch-whisper.md @@ -41,12 +41,24 @@ transformers-format Whisper repo works. Weights download on first load — see ## VRAM preflight -whisper-large-v3-turbo needs roughly 3.2 GiB before generation adds its -workspace; loading it onto a nearly-full card "succeeds" and then the first -transcribe OOMs with zero segments. So on CUDA the engine checks free VRAM -against a 5 GB budget before loading and uses the CPU instead when the card -is too full (flush the TTS model to restore GPU-speed ASR). Disable with -`OMNIVOICE_ASR_VRAM_PREFLIGHT=0`. +Loading a model onto a nearly-full card "succeeds", and then the first +transcribe runs out of memory with zero segments. So on CUDA the engine +checks free VRAM before loading and uses the CPU instead when the card is +too full (flush the TTS model to restore GPU-speed ASR). + +The budget follows the model it loads in fp16: the weights, plus about +1.5 GB of working memory and 0.5 GB of headroom. + +| Model | Free VRAM needed | +|---|---| +| whisper-large-v3-turbo (default) | 3.6 GB | +| whisper-large-v3 | 5 GB | +| medium / small / base / tiny | 3.5 / 2.5 / 2.2 / 2.1 GB | +| any other model | 5 GB | + +A 6 GB card with nothing else loaded runs the default model on the GPU +([#2041](https://github.com/debpalash/VoiceStudio/issues/2041)). Disable +the check with `OMNIVOICE_ASR_VRAM_PREFLIGHT=0`. ## Quirks diff --git a/tests/test_pytorch_whisper_fallback.py b/tests/test_pytorch_whisper_fallback.py index 6a23e9b7..395b8e98 100644 --- a/tests/test_pytorch_whisper_fallback.py +++ b/tests/test_pytorch_whisper_fallback.py @@ -84,7 +84,7 @@ def test_low_free_vram_routes_pytorch_whisper_to_cpu(monkeypatch): import torch monkeypatch.setattr("services.model_manager.get_best_device", lambda: "cuda:0") - monkeypatch.setattr(torch.cuda, "mem_get_info", lambda: (4 * 1024**3, 24 * 1024**3)) + monkeypatch.setattr(torch.cuda, "mem_get_info", lambda: (3 * 1024**3, 24 * 1024**3)) assert ab.PyTorchWhisperBackend._pick_device() == "cpu" diff --git a/tests/test_pytorch_whisper_vram_budget_2041.py b/tests/test_pytorch_whisper_vram_budget_2041.py new file mode 100644 index 00000000..f606e15d --- /dev/null +++ b/tests/test_pytorch_whisper_vram_budget_2041.py @@ -0,0 +1,50 @@ +"""#2041 — PyTorch Whisper demanded 5.0 GB of free VRAM for every model, sized +for full large-v3, so a 6 GB card was sent to CPU for the default +large-v3-turbo although CUDA ran the same audio in 37 s.""" +import pytest + +GiB = 1024**3 + + +@pytest.fixture +def pick(monkeypatch): + torch = pytest.importorskip("torch") + from services import asr_backend as ab + + for name in ("OMNIVOICE_PYTORCH_ASR_MODEL", "OMNIVOICE_ASR_VRAM_PREFLIGHT"): + monkeypatch.setenv(name, "x") # recorded, so the teardown restores it + monkeypatch.delenv(name) + monkeypatch.setattr("services.model_manager.get_best_device", lambda: "cuda:0") + warnings = [] + monkeypatch.setattr(ab.logger, "warning", lambda msg, *args: warnings.append(msg % args)) + + def with_free(free_gb, model=None): + monkeypatch.setattr(torch.cuda, "mem_get_info", lambda: (int(free_gb * GiB), 6 * GiB)) + return ab.PyTorchWhisperBackend._pick_device(model) + + with_free.warnings = warnings + return with_free + + +def test_the_reported_6gb_card_keeps_turbo_on_cuda(pick): + # The report: a 6 GB Quadro RTX 3000 with nothing else resident, 5.0 GB free. + assert pick(5.0) == "cuda:0" + + +def test_turbo_still_falls_back_on_a_genuinely_full_card(pick): + assert pick(2.0) == "cpu" + + +def test_full_large_v3_keeps_its_five_gigabyte_budget(pick): + assert pick(4.5, "openai/whisper-large-v3") == "cpu" + assert pick(5.2, "openai/whisper-large-v3") == "cuda:0" + + +def test_an_unknown_model_gets_the_conservative_budget(pick): + assert pick(4.5, "someone/custom-asr") == "cpu" + + +def test_the_fallback_warning_names_the_model_and_the_opt_out(pick): + pick(1.0) + assert "OMNIVOICE_ASR_VRAM_PREFLIGHT=0" in pick.warnings[-1] + assert "whisper-large-v3-turbo" in pick.warnings[-1] From e501a596d3004b63cf2df4c12b3bef8fb8148d17 Mon Sep 17 00:00:00 2001 From: Palash Debnath <4178343+debpalash@users.noreply.github.com> Date: Thu, 10 Sep 2026 15:06:52 -0700 Subject: [PATCH 2/3] docs(changelog): note the PyTorch Whisper VRAM budget fix (#2044) --- CHANGELOG.md | 1 + 1 file changed, 1 insertion(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 595e949a..624484f7 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -26,6 +26,7 @@ the frozen-backend fallback mirror it for their toolchains. - Stopping a process on macOS no longer fails with "Operation not permitted" when it was already exiting (#2032) - A YouTube link blocked by its "not a bot" check now says how to attach signed-in cookies in Dub, instead of quoting yt-dlp's command-line flags (#2036, #2034) - An engine that fails to start now says whether it timed out, crashed (with its exit code and last output) or answered wrongly, instead of "did not signal ready: None" (#2037, #2026) +- PyTorch Whisper runs on 6 GB NVIDIA cards instead of falling back to CPU, because its memory check now fits the model it loads (#2044, #2041) ### CI From e7911e3e97940bfeec1c97b0413c961bfc59cd81 Mon Sep 17 00:00:00 2001 From: Palash Debnath <4178343+debpalash@users.noreply.github.com> Date: Thu, 10 Sep 2026 15:33:11 -0700 Subject: [PATCH 3/3] fix(asr): reduced VRAM budgets only for exact OpenAI Whisper ids Review follow-up. The budget matched model names by substring, so a custom repo whose name contains "turbo", "small" or "base" (or a word such as "database") got a reduced budget and could be admitted to CUDA without enough memory. Reduced budgets now apply only to the exact OpenAI checkpoint ids, .en variants included. Any other repository, fine-tunes included, keeps the conservative 5.0 GB, as the engine doc says. --- backend/services/asr_backend.py | 36 ++++++++++++------- docs/engines/pytorch-whisper.md | 13 +++---- .../test_pytorch_whisper_vram_budget_2041.py | 20 +++++++++++ 3 files changed, 51 insertions(+), 18 deletions(-) diff --git a/backend/services/asr_backend.py b/backend/services/asr_backend.py index a4e1c193..e947a9e0 100644 --- a/backend/services/asr_backend.py +++ b/backend/services/asr_backend.py @@ -1304,10 +1304,23 @@ class PyTorchWhisperBackend(ASRBackend): # so a 6 GB card with nothing else resident reported 5.0 GB free and was # sent to CPU every time, although CUDA transcribed the same audio in 37 s. _CUDA_VRAM_BUDGET_GB = 5.0 # full large-v3, and any model not listed below - _FP16_WEIGHTS_GB = ( - ("turbo", 1.6), ("large", 3.1), ("medium", 1.5), - ("small", 0.5), ("base", 0.15), ("tiny", 0.08), - ) + # fp16 weights (GB) of the OpenAI Whisper checkpoints, by exact repo id. + # Anything else, including a fine-tune or a custom repo whose name happens + # to contain "small" or "turbo", keeps the conservative 5.0 GB budget. + _FP16_WEIGHTS_GB = { + "openai/whisper-large-v3-turbo": 1.6, + "openai/whisper-large-v3": 3.1, + "openai/whisper-large-v2": 3.1, + "openai/whisper-large": 3.1, + "openai/whisper-medium": 1.5, + "openai/whisper-medium.en": 1.5, + "openai/whisper-small": 0.5, + "openai/whisper-small.en": 0.5, + "openai/whisper-base": 0.15, + "openai/whisper-base.en": 0.15, + "openai/whisper-tiny": 0.08, + "openai/whisper-tiny.en": 0.08, + } _CUDA_WORKSPACE_GB = 1.5 # batch 16 x 15 s chunks _CUDA_HEADROOM_GB = 0.5 @@ -1315,14 +1328,13 @@ class PyTorchWhisperBackend(ASRBackend): def _cuda_budget_gb(cls, model_name: str) -> float: """Free VRAM (GB) this model needs on CUDA; never above the 5.0 GB that full large-v3 was measured to need.""" - name = (model_name or "").lower() - for key, weights_gb in cls._FP16_WEIGHTS_GB: - if key in name: - return min( - cls._CUDA_VRAM_BUDGET_GB, - weights_gb + cls._CUDA_WORKSPACE_GB + cls._CUDA_HEADROOM_GB, - ) - return cls._CUDA_VRAM_BUDGET_GB + weights_gb = cls._FP16_WEIGHTS_GB.get((model_name or "").strip().lower()) + if weights_gb is None: + return cls._CUDA_VRAM_BUDGET_GB + return min( + cls._CUDA_VRAM_BUDGET_GB, + weights_gb + cls._CUDA_WORKSPACE_GB + cls._CUDA_HEADROOM_GB, + ) @staticmethod def _model_name() -> str: diff --git a/docs/engines/pytorch-whisper.md b/docs/engines/pytorch-whisper.md index 15ed1c56..a177dd4d 100644 --- a/docs/engines/pytorch-whisper.md +++ b/docs/engines/pytorch-whisper.md @@ -46,15 +46,16 @@ transcribe runs out of memory with zero segments. So on CUDA the engine checks free VRAM before loading and uses the CPU instead when the card is too full (flush the TTS model to restore GPU-speed ASR). -The budget follows the model it loads in fp16: the weights, plus about -1.5 GB of working memory and 0.5 GB of headroom. +For the OpenAI Whisper checkpoints, the budget follows the model it loads in +fp16: the weights, plus about 1.5 GB of working memory and 0.5 GB of +headroom. Any other repository, including a fine-tune, keeps 5 GB. | Model | Free VRAM needed | |---|---| -| whisper-large-v3-turbo (default) | 3.6 GB | -| whisper-large-v3 | 5 GB | -| medium / small / base / tiny | 3.5 / 2.5 / 2.2 / 2.1 GB | -| any other model | 5 GB | +| `openai/whisper-large-v3-turbo` (default) | 3.6 GB | +| `openai/whisper-large`, `-large-v2`, `-large-v3` | 5 GB | +| `openai/whisper-medium` / `-small` / `-base` / `-tiny` (and `.en`) | 3.5 / 2.5 / 2.2 / 2.1 GB | +| any other repository | 5 GB | A 6 GB card with nothing else loaded runs the default model on the GPU ([#2041](https://github.com/debpalash/VoiceStudio/issues/2041)). Disable diff --git a/tests/test_pytorch_whisper_vram_budget_2041.py b/tests/test_pytorch_whisper_vram_budget_2041.py index f606e15d..36ed5ce1 100644 --- a/tests/test_pytorch_whisper_vram_budget_2041.py +++ b/tests/test_pytorch_whisper_vram_budget_2041.py @@ -48,3 +48,23 @@ def test_the_fallback_warning_names_the_model_and_the_opt_out(pick): pick(1.0) assert "OMNIVOICE_ASR_VRAM_PREFLIGHT=0" in pick.warnings[-1] assert "whisper-large-v3-turbo" in pick.warnings[-1] + + +@pytest.mark.parametrize( + "custom", + [ + "someone/turbo-whisper-xl", + "acme/small-talk-asr", + "org/database-asr", + "openai/whisper-small-finetuned", + ], +) +def test_a_custom_repo_with_a_size_word_keeps_the_conservative_budget(pick, custom): + # Substring matching gave these a reduced budget and could admit a model + # that then ran out of VRAM (#2044 review). + assert pick(4.5, custom) == "cpu" + + +def test_english_only_checkpoints_and_stray_case_are_recognised(pick): + assert pick(3.0, "openai/whisper-small.en") == "cuda:0" + assert pick(3.0, " OpenAI/Whisper-Small ") == "cuda:0"