Merge branch 'fix/review-2083' into fix/community-integration
# Conflicts: # CHANGELOG.md # docs/install/troubleshooting.md
This commit is contained in:
@@ -22,6 +22,8 @@ the frozen-backend fallback mirror it for their toolchains.
|
|||||||
|
|
||||||
- Include cuDNN 8 compatibility libraries for CTranslate2 in CUDA containers (#2072) — thanks @basil-k-aji-dev!
|
- Include cuDNN 8 compatibility libraries for CTranslate2 in CUDA containers (#2072) — thanks @basil-k-aji-dev!
|
||||||
|
|
||||||
|
- Preserve audio reads, writes, and reference amplitude without TorchCodec (#2083) — thanks @Moep90!
|
||||||
|
|
||||||
## [0.5.3] — 2026-09-17
|
## [0.5.3] — 2026-09-17
|
||||||
|
|
||||||
**Highlights**
|
**Highlights**
|
||||||
|
|||||||
@@ -204,6 +204,42 @@ def _safe_torchaudio_save(
|
|||||||
fmt, e,
|
fmt, e,
|
||||||
)
|
)
|
||||||
torchaudio.save(path_or_buf, tensor, sample_rate, format=fmt)
|
torchaudio.save(path_or_buf, tensor, sample_rate, format=fmt)
|
||||||
|
except (ImportError, RuntimeError) as e:
|
||||||
|
if isinstance(e, RuntimeError) and "could not load libtorchcodec" not in str(e).lower():
|
||||||
|
raise # Unrelated encoder failures must remain visible.
|
||||||
|
# torchaudio >= 2.9 routes save() through TorchCodec, which needs
|
||||||
|
# FFmpeg *shared libraries* on the system. Where those are absent the
|
||||||
|
# write raises ImportError and every generation fails. #1931 guarded
|
||||||
|
# set_audio_backend() against that torchaudio but left save() itself
|
||||||
|
# unprotected; arm64 CUDA hosts reach it unavoidably, since torch
|
||||||
|
# 2.8.0 publishes no aarch64 wheel. soundfile is already a locked
|
||||||
|
# dependency and the tensor is normalized by this point, so hand it to
|
||||||
|
# the audited sibling helper rather than failing the request.
|
||||||
|
logger.warning(
|
||||||
|
"torchaudio.save needs TorchCodec (%s); writing via soundfile", e
|
||||||
|
)
|
||||||
|
if hasattr(path_or_buf, "seek") and hasattr(path_or_buf, "truncate"):
|
||||||
|
try:
|
||||||
|
path_or_buf.seek(0)
|
||||||
|
path_or_buf.truncate(0)
|
||||||
|
except (OSError, io.UnsupportedOperation):
|
||||||
|
pass # Non-seekable streams cannot be rewound; preserve fallback behavior.
|
||||||
|
_subtype = {
|
||||||
|
"wav": "FLOAT" if bits_per_sample == 32 else "PCM_16",
|
||||||
|
"flac": "PCM_16",
|
||||||
|
"ogg": "VORBIS",
|
||||||
|
"mp3": "MPEG_LAYER_III",
|
||||||
|
}.get(fmt, "PCM_16")
|
||||||
|
try:
|
||||||
|
_safe_soundfile_write(
|
||||||
|
path_or_buf,
|
||||||
|
tensor.transpose(0, 1).contiguous().numpy(),
|
||||||
|
sample_rate,
|
||||||
|
subtype=_subtype,
|
||||||
|
format=fmt.upper(),
|
||||||
|
)
|
||||||
|
except Exception as e2:
|
||||||
|
raise _describe_write_failure(e2, path_or_buf) from e2
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
# #1221: libsndfile reports OS-level write failures as a bare
|
# #1221: libsndfile reports OS-level write failures as a bare
|
||||||
# "LibsndfileError: System error." — no path, no errno, nothing the
|
# "LibsndfileError: System error." — no path, no errno, nothing the
|
||||||
@@ -267,6 +303,7 @@ def _safe_soundfile_write(
|
|||||||
sample_rate: int,
|
sample_rate: int,
|
||||||
*,
|
*,
|
||||||
subtype: str = "PCM_16",
|
subtype: str = "PCM_16",
|
||||||
|
format: str | None = None,
|
||||||
) -> None:
|
) -> None:
|
||||||
"""Sibling helper for the one in-tree ``sf.write`` site.
|
"""Sibling helper for the one in-tree ``sf.write`` site.
|
||||||
|
|
||||||
@@ -284,6 +321,10 @@ def _safe_soundfile_write(
|
|||||||
subtype: Soundfile subtype string. ``"PCM_16"`` (default) for
|
subtype: Soundfile subtype string. ``"PCM_16"`` (default) for
|
||||||
standard 16-bit PCM WAV; ``"PCM_24"``, ``"FLOAT"`` etc.
|
standard 16-bit PCM WAV; ``"PCM_24"``, ``"FLOAT"`` etc.
|
||||||
also work.
|
also work.
|
||||||
|
format: Container format (``"WAV"``, ``"FLAC"``, ``"OGG"``,
|
||||||
|
``"MP3"``). ``None`` lets soundfile infer it from the path's
|
||||||
|
extension — which it cannot do for a file-like object, so
|
||||||
|
callers passing a buffer must name it.
|
||||||
|
|
||||||
Raises:
|
Raises:
|
||||||
ValueError: if the array is empty.
|
ValueError: if the array is empty.
|
||||||
@@ -321,7 +362,7 @@ def _safe_soundfile_write(
|
|||||||
samples = np.ascontiguousarray(samples)
|
samples = np.ascontiguousarray(samples)
|
||||||
|
|
||||||
_ensure_audio_parent(path)
|
_ensure_audio_parent(path)
|
||||||
sf.write(path, samples, sample_rate, subtype=subtype)
|
sf.write(path, samples, sample_rate, subtype=subtype, format=format)
|
||||||
|
|
||||||
|
|
||||||
def atomic_save_wav(
|
def atomic_save_wav(
|
||||||
|
|||||||
@@ -1152,3 +1152,7 @@ remove the app binary itself are in
|
|||||||
### Pedalboard illegal-instruction crashes
|
### Pedalboard illegal-instruction crashes
|
||||||
|
|
||||||
VoiceStudio pins pedalboard to `>=0.9.14,<0.9.21` while [upstream portable-wheel repair #466](https://github.com/spotify/pedalboard/pull/466) remains open. Re-sync the locked environment after updating from a build with newer affected wheels. This keeps the existing effects API floor while avoiding the reported Linux CPU import crash.
|
VoiceStudio pins pedalboard to `>=0.9.14,<0.9.21` while [upstream portable-wheel repair #466](https://github.com/spotify/pedalboard/pull/466) remains open. Re-sync the locked environment after updating from a build with newer affected wheels. This keeps the existing effects API floor while avoiding the reported Linux CPU import crash.
|
||||||
|
|
||||||
|
### TorchCodec unavailable
|
||||||
|
|
||||||
|
When torchaudio requires an unavailable TorchCodec installation, VoiceStudio writes through soundfile and reads reference audio through its FFmpeg fallback. Reference amplitude is normalized using the decoded sample representation, including 8-, 24-, and 32-bit PCM.
|
||||||
|
|||||||
@@ -58,10 +58,21 @@ def load_audio(audio_path: str, sampling_rate: int):
|
|||||||
waveform, prompt_sampling_rate = torchaudio.load(
|
waveform, prompt_sampling_rate = torchaudio.load(
|
||||||
audio_path, backend="soundfile"
|
audio_path, backend="soundfile"
|
||||||
)
|
)
|
||||||
except (RuntimeError, OSError):
|
except (ImportError, RuntimeError, OSError):
|
||||||
# Fallback via pydub+ffmpeg for formats torchaudio can't handle
|
# Fallback via pydub+ffmpeg for formats torchaudio can't handle.
|
||||||
|
# ImportError belongs here too: torchaudio >= 2.9 routes load()
|
||||||
|
# through TorchCodec and raises ImportError when its FFmpeg shared
|
||||||
|
# libraries are absent. The ``backend="soundfile"`` argument above
|
||||||
|
# does NOT avoid that — 2.9 accepts and ignores it.
|
||||||
aseg = AudioSegment.from_file(audio_path)
|
aseg = AudioSegment.from_file(audio_path)
|
||||||
audio_data = np.array(aseg.get_array_of_samples()).astype(np.float32) / 32768.0
|
# Scale by the decoded sample width instead of a hardcoded 16-bit
|
||||||
|
# divisor. pydub reports 8-bit as sample_width 1 and widens 24-bit to
|
||||||
|
# a full-range int32 (sample_width 4), so /32768 makes 24- and 32-bit
|
||||||
|
# references 32768x too loud and 8-bit ones 256x too quiet.
|
||||||
|
audio_data = (
|
||||||
|
np.array(aseg.get_array_of_samples()).astype(np.float32)
|
||||||
|
/ aseg.max_possible_amplitude
|
||||||
|
)
|
||||||
if aseg.channels == 1:
|
if aseg.channels == 1:
|
||||||
waveform = torch.from_numpy(audio_data).unsqueeze(0)
|
waveform = torch.from_numpy(audio_data).unsqueeze(0)
|
||||||
else:
|
else:
|
||||||
|
|||||||
@@ -372,3 +372,89 @@ def test_atomic_save_wav_delegates_to_safe_helper(tmp_path):
|
|||||||
decoded, _ = sf.read(str(target))
|
decoded, _ = sf.read(str(target))
|
||||||
assert abs(decoded).max() <= 1.0 + 1e-3
|
assert abs(decoded).max() <= 1.0 + 1e-3
|
||||||
assert abs(decoded).max() > 0.5
|
assert abs(decoded).max() > 0.5
|
||||||
|
|
||||||
|
|
||||||
|
# ── torchaudio 2.9 + no TorchCodec (#1931 follow-up) ───────────────────────
|
||||||
|
#
|
||||||
|
# torchaudio >= 2.9 routes save() through TorchCodec, which needs FFmpeg
|
||||||
|
# *shared libraries* on the system. Where those are absent every write raises
|
||||||
|
# ImportError. #1931 guarded set_audio_backend() against that torchaudio but
|
||||||
|
# left save() unprotected; arm64 CUDA hosts reach it unavoidably, since torch
|
||||||
|
# 2.8.0 publishes no aarch64 wheel. Without the soundfile fallback these three
|
||||||
|
# tests raise instead of producing a file.
|
||||||
|
|
||||||
|
|
||||||
|
def _torchcodec_missing(*_a, **_kw):
|
||||||
|
raise ImportError(
|
||||||
|
"TorchCodec is required for save_with_torchcodec. "
|
||||||
|
"Please install torchcodec to use this function."
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_safe_save_falls_back_to_soundfile_when_torchcodec_missing(
|
||||||
|
tmp_path, monkeypatch
|
||||||
|
):
|
||||||
|
import torchaudio
|
||||||
|
|
||||||
|
monkeypatch.setattr(torchaudio, "save", _torchcodec_missing)
|
||||||
|
|
||||||
|
target = tmp_path / "fallback.wav"
|
||||||
|
_safe_torchaudio_save(str(target), _sine_tensor(), 24000)
|
||||||
|
|
||||||
|
info = sf.info(str(target))
|
||||||
|
assert info.subtype == "PCM_16"
|
||||||
|
assert info.frames == 24000
|
||||||
|
decoded, sr = sf.read(str(target))
|
||||||
|
assert sr == 24000
|
||||||
|
assert abs(decoded).max() > 0.1
|
||||||
|
|
||||||
|
|
||||||
|
def test_safe_save_buffer_falls_back_when_torchcodec_missing(monkeypatch):
|
||||||
|
"""The OpenAI-compatible /v1/audio/speech path writes to a BytesIO."""
|
||||||
|
import torchaudio
|
||||||
|
|
||||||
|
monkeypatch.setattr(torchaudio, "save", _torchcodec_missing)
|
||||||
|
|
||||||
|
buf = io.BytesIO()
|
||||||
|
_safe_torchaudio_save(buf, _sine_tensor(), 24000, format="wav")
|
||||||
|
|
||||||
|
assert buf.getvalue()[:4] == b"RIFF"
|
||||||
|
buf.seek(0)
|
||||||
|
decoded, sr = sf.read(buf)
|
||||||
|
assert sr == 24000
|
||||||
|
assert len(decoded) == 24000
|
||||||
|
|
||||||
|
|
||||||
|
def test_safe_save_flac_buffer_fallback_names_the_format(monkeypatch):
|
||||||
|
"""soundfile cannot infer a container from a file-like object, so the
|
||||||
|
fallback must pass ``format=`` explicitly — otherwise this raises."""
|
||||||
|
import torchaudio
|
||||||
|
|
||||||
|
monkeypatch.setattr(torchaudio, "save", _torchcodec_missing)
|
||||||
|
|
||||||
|
buf = io.BytesIO()
|
||||||
|
_safe_torchaudio_save(buf, _sine_tensor(), 24000, format="flac")
|
||||||
|
|
||||||
|
assert buf.getvalue()[:4] == b"fLaC"
|
||||||
|
buf.seek(0)
|
||||||
|
decoded, sr = sf.read(buf)
|
||||||
|
assert sr == 24000
|
||||||
|
assert len(decoded) == 24000
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("message, fallback", [
|
||||||
|
("Could not load libtorchcodec. Missing FFmpeg shared libraries", True),
|
||||||
|
("unrelated encoder failure", False),
|
||||||
|
])
|
||||||
|
def test_save_native_loader_failure_only_uses_fallback(tmp_path, monkeypatch, message, fallback):
|
||||||
|
import torchaudio
|
||||||
|
def fail(*args, **kwargs):
|
||||||
|
raise RuntimeError(message)
|
||||||
|
monkeypatch.setattr(torchaudio, "save", fail)
|
||||||
|
target = tmp_path / "codec.wav"
|
||||||
|
if fallback:
|
||||||
|
_safe_torchaudio_save(str(target), _sine_tensor(), 24000)
|
||||||
|
assert sf.info(target).frames == 24000
|
||||||
|
else:
|
||||||
|
with pytest.raises(RuntimeError, match="unrelated encoder failure"):
|
||||||
|
_safe_torchaudio_save(str(target), _sine_tensor(), 24000)
|
||||||
|
|||||||
@@ -0,0 +1,108 @@
|
|||||||
|
"""``load_audio`` must survive torchaudio 2.9 without TorchCodec (#1931).
|
||||||
|
|
||||||
|
torchaudio >= 2.9 routes ``load()`` through TorchCodec, which needs FFmpeg
|
||||||
|
*shared libraries* on the system. Where those are absent the call raises
|
||||||
|
``ImportError`` — but ``load_audio`` caught only ``(RuntimeError, OSError)``,
|
||||||
|
so the pydub fallback written for precisely this situation never ran. Every
|
||||||
|
voice-clone reference read, watermark check and dub segment load then failed
|
||||||
|
with "TorchCodec is required for load_with_torchcodec".
|
||||||
|
|
||||||
|
This is the read-side twin of the ``_safe_torchaudio_save`` regression in
|
||||||
|
``tests/backend/services/test_audio_io.py``, and it reaches the same users:
|
||||||
|
#1931 guarded ``set_audio_backend()`` against torchaudio 2.9 but left
|
||||||
|
``load()`` unprotected. arm64 CUDA hosts reach it unavoidably, since torch
|
||||||
|
2.8.0 publishes no aarch64 wheel.
|
||||||
|
|
||||||
|
Note that ``backend="soundfile"`` does not avoid this — torchaudio 2.9
|
||||||
|
accepts that argument and ignores it.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
import pytest
|
||||||
|
import soundfile as sf
|
||||||
|
import torch
|
||||||
|
|
||||||
|
from omnivoice.utils.audio import load_audio
|
||||||
|
|
||||||
|
|
||||||
|
def _write_sine_wav(path, *, seconds: float = 0.5, sample_rate: int = 24000):
|
||||||
|
n = int(seconds * sample_rate)
|
||||||
|
t = np.arange(n, dtype=np.float32) / sample_rate
|
||||||
|
sf.write(str(path), 0.5 * np.sin(2 * np.pi * 440.0 * t), sample_rate,
|
||||||
|
subtype="PCM_16")
|
||||||
|
return sample_rate
|
||||||
|
|
||||||
|
|
||||||
|
def _torchcodec_missing(*_a, **_kw):
|
||||||
|
raise ImportError(
|
||||||
|
"TorchCodec is required for load_with_torchcodec. "
|
||||||
|
"Please install torchcodec to use this function."
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_load_audio_falls_back_when_torchcodec_missing(tmp_path, monkeypatch):
|
||||||
|
"""Without the ImportError catch this raises instead of returning audio."""
|
||||||
|
import torchaudio
|
||||||
|
|
||||||
|
ref = tmp_path / "ref.wav"
|
||||||
|
sample_rate = _write_sine_wav(ref)
|
||||||
|
monkeypatch.setattr(torchaudio, "load", _torchcodec_missing)
|
||||||
|
|
||||||
|
waveform = load_audio(str(ref), sample_rate)
|
||||||
|
|
||||||
|
assert isinstance(waveform, torch.Tensor)
|
||||||
|
assert waveform.ndim == 2, f"expected (1, T), got {tuple(waveform.shape)}"
|
||||||
|
assert waveform.shape[0] == 1, "load_audio must return mono"
|
||||||
|
assert waveform.shape[-1] > 0
|
||||||
|
assert waveform.abs().max() > 0.05, "fallback produced silence"
|
||||||
|
|
||||||
|
|
||||||
|
def test_load_audio_fallback_resamples_to_target(tmp_path, monkeypatch):
|
||||||
|
"""The fallback path must still honour the requested sampling rate."""
|
||||||
|
import torchaudio
|
||||||
|
|
||||||
|
ref = tmp_path / "ref_16k.wav"
|
||||||
|
_write_sine_wav(ref, seconds=0.5, sample_rate=16000)
|
||||||
|
monkeypatch.setattr(torchaudio, "load", _torchcodec_missing)
|
||||||
|
|
||||||
|
waveform = load_audio(str(ref), 24000)
|
||||||
|
|
||||||
|
# 0.5 s resampled 16k -> 24k is ~12000 samples; allow resampler edge slack.
|
||||||
|
assert abs(waveform.shape[-1] - 12000) <= 64, (
|
||||||
|
f"expected ~12000 samples at 24 kHz, got {waveform.shape[-1]}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("subtype", ["PCM_U8", "PCM_16", "PCM_24", "PCM_32", "FLOAT"])
|
||||||
|
def test_load_audio_fallback_amplitude_matches_bit_depth(
|
||||||
|
tmp_path, monkeypatch, subtype
|
||||||
|
):
|
||||||
|
"""The fallback must scale by the decoded width, not a fixed 32768.
|
||||||
|
|
||||||
|
pydub reports 8-bit as ``sample_width`` 1 and widens 24-bit to a
|
||||||
|
full-range int32 (``sample_width`` 4, contrary to the stale comment in
|
||||||
|
its own source). Dividing every decode by 32768 therefore returned 24-
|
||||||
|
and 32-bit references 32768x too loud and 8-bit ones 256x too quiet.
|
||||||
|
Nothing downstream clamps, so a clone reference silently became noise.
|
||||||
|
|
||||||
|
Before the ImportError catch this path was rare; on torchaudio >= 2.9
|
||||||
|
without TorchCodec it is the only path, which is what makes it a bug
|
||||||
|
worth fixing here.
|
||||||
|
"""
|
||||||
|
import torchaudio
|
||||||
|
|
||||||
|
sample_rate = 24000
|
||||||
|
peak = 0.5
|
||||||
|
t = np.arange(sample_rate // 2, dtype=np.float64) / sample_rate
|
||||||
|
ref = tmp_path / f"ref_{subtype.lower()}.wav"
|
||||||
|
sf.write(str(ref), peak * np.sin(2 * np.pi * 440.0 * t), sample_rate,
|
||||||
|
subtype=subtype)
|
||||||
|
monkeypatch.setattr(torchaudio, "load", _torchcodec_missing)
|
||||||
|
|
||||||
|
waveform = load_audio(str(ref), sample_rate)
|
||||||
|
|
||||||
|
assert waveform.abs().max().item() == pytest.approx(peak, abs=0.02), (
|
||||||
|
f"{subtype} decoded at the wrong scale: peak "
|
||||||
|
f"{waveform.abs().max().item():.6f}, expected ~{peak}"
|
||||||
|
)
|
||||||
Reference in New Issue
Block a user