* feat(gallery): save gallery voices as profiles, with validated audio references Work-in-progress lifted from the concurrent gallery session at the owner's request (its uncommitted working tree, preserved verbatim from base 92b1ee5d; safety snapshot remains at rescue/gallery-wip): - gallery voices can be saved as local profiles: audio is copied into the profile store with content-addressed filenames, existing profiles are detected and refreshed only when the source clip changed - backend/core/audio_validation.py: symlink-rejecting, root-contained resolution for persisted profile WAV references, with tests - archetype/community routers and the Voice Gallery UI updated for the save-as-profile handoff (spec: docs/specs/longform/26-gallery-use-handoff.md) - locale updates for the new gallery strings across all 21 files Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * chore: drop a stray local screenshot script that rode in with the tree copy * fix(community): explain the tolerated Content-Length parse failure; drop an unused import CodeQL on #1542: the empty except now says why it is safe (the streamed byte counter enforces the same cap regardless), and the test file loses an unused Path import. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * fix(gallery): review findings — copy outside the write lock, no stale completions CodeRabbit on #1542, all findings addressed: - the profile-audio copy stages to a .part temp BEFORE BEGIN IMMEDIATE and publishes via atomic os.replace inside it — other backend writers no longer block for the duration of an audio copy; a mid-copy failure leaves no temp droppings and no profile row (both pinned by tests) - VoiceGallery async ops carry per-operation generation tokens: a preview or save-as-profile that resolves after unmount (or after a newer operation) can no longer play audio, redirect into a workspace, or touch state — three fail-before regression tests - VoiceGalleryActions imports the page at test runtime; the e2e locator uses a stable data-testid instead of a translated string; symlink tests skip cleanly where the OS can't create symlinks; the changelog line carries its PR ref Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * ci: static ffmpeg fallback when the chocolatey feed is down Third feed outage to break a PR run (2026-07-20, 2026-07-28, today — three attempts, three 'installed 0/1'). Chocolatey is a distribution channel, not the dependency: after the retry loop exhausts, fetch the static gyan.dev build from its GitHub release mirror and put it on PATH — same binary, no feed in the path. URL verified live (HTTP 200). Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> --------- Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
107 lines
4.2 KiB
Python
107 lines
4.2 KiB
Python
"""Lightweight validation for persisted profile WAV references.
|
|
|
|
This module deliberately uses only the standard library. Gallery routers import
|
|
it during startup, so pulling in torch/torchaudio merely to validate a cached
|
|
file would make every Gallery open pay the model stack's import cost.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
import wave
|
|
from pathlib import Path
|
|
from typing import Optional
|
|
|
|
from core.path_security import UnsafePath, resolve_within, safe_filename
|
|
|
|
_READ_CHUNK_BYTES = 1 << 20
|
|
_MAX_CHANNELS = 64
|
|
_MAX_SAMPLE_RATE = 768_000
|
|
_MAX_SAMPLE_WIDTH = 8
|
|
|
|
|
|
def resolve_regular_file(root: os.PathLike[str] | str, value: object) -> Optional[Path]:
|
|
"""Resolve a portable bare filename inside *root*, rejecting symlinks."""
|
|
try:
|
|
name = safe_filename(value)
|
|
unresolved = Path(root).resolve(strict=False) / name
|
|
if unresolved.is_symlink():
|
|
return None
|
|
return resolve_within(root, name)
|
|
except (OSError, UnsafePath):
|
|
return None
|
|
|
|
|
|
def is_playable_wav(path: Optional[Path]) -> bool:
|
|
"""Return true only for a regular, decodable WAV with audio frames."""
|
|
if path is None:
|
|
return False
|
|
try:
|
|
if not path.is_file() or path.is_symlink():
|
|
return False
|
|
file_size = path.stat().st_size
|
|
with wave.open(str(path), "rb") as wav:
|
|
channels = wav.getnchannels()
|
|
sample_rate = wav.getframerate()
|
|
sample_width = wav.getsampwidth()
|
|
frame_count = wav.getnframes()
|
|
if (
|
|
not 0 < channels <= _MAX_CHANNELS
|
|
or not 0 < sample_rate <= _MAX_SAMPLE_RATE
|
|
or not 0 < sample_width <= _MAX_SAMPLE_WIDTH
|
|
or frame_count <= 0
|
|
):
|
|
return False
|
|
# ``wave.getnframes`` trusts the header. Read through the declared
|
|
# payload so an interrupted write with a complete header but a
|
|
# truncated data chunk cannot masquerade as playable audio.
|
|
frame_size = channels * sample_width
|
|
expected_bytes = frame_count * frame_size
|
|
# A PCM payload cannot be larger than the containing file. Check
|
|
# before calling ``readframes`` so hostile header values cannot
|
|
# turn a tiny file into a multi-gigabyte allocation request.
|
|
if expected_bytes > file_size:
|
|
return False
|
|
read_bytes = 0
|
|
chunk_frames = max(1, min(frame_count, _READ_CHUNK_BYTES // frame_size))
|
|
while read_bytes < expected_bytes:
|
|
chunk = wav.readframes(chunk_frames)
|
|
if not chunk or len(chunk) % frame_size:
|
|
return False
|
|
read_bytes += len(chunk)
|
|
return read_bytes == expected_bytes
|
|
except (MemoryError, OSError, EOFError, OverflowError, wave.Error):
|
|
# Python 3.11's wave module rejects valid IEEE-float/WAVE_EXTENSIBLE
|
|
# files. SoundFile is already a runtime dependency and recognizes those
|
|
# containers; import it only on the uncommon fallback path.
|
|
try:
|
|
import soundfile as sf
|
|
|
|
with sf.SoundFile(str(path)) as audio:
|
|
if (
|
|
audio.format != "WAV"
|
|
or not 0 < audio.channels <= _MAX_CHANNELS
|
|
or not 0 < audio.samplerate <= _MAX_SAMPLE_RATE
|
|
or len(audio) <= 0
|
|
):
|
|
return False
|
|
remaining = len(audio)
|
|
# Decode through the declared payload in byte-bounded chunks;
|
|
# ``sf.info`` alone also trusts a truncated file's header.
|
|
chunk_frames = max(
|
|
1, _READ_CHUNK_BYTES // (audio.channels * 4),
|
|
)
|
|
while remaining:
|
|
frames = audio.read(
|
|
min(remaining, chunk_frames), dtype="float32", always_2d=True,
|
|
)
|
|
count = len(frames)
|
|
if count <= 0:
|
|
return False
|
|
remaining -= count
|
|
return True
|
|
except Exception:
|
|
return False
|
|
|
|
|
|
__all__ = ["is_playable_wav", "resolve_regular_file"]
|