import asyncio import io import json import logging import ntpath import os import re import time import uuid from pathlib import Path, PureWindowsPath from typing import Optional from core.config import DUB_DIR from core.http_headers import content_disposition from core.logging_utils import log_safe from core.path_security import UnsafePath, resolve_within from core.tasks import task_manager from fastapi import APIRouter, Header, HTTPException, Query, Response from fastapi.responses import FileResponse, StreamingResponse from services.ffmpeg_utils import ( bed_mix_filter, explain_ffmpeg_failure, find_ffmpeg, run_ffmpeg, ) from services.karaoke_ass import build_ass, scale_words from services.video_retime import ( DRIFT_TOLERANCE_S, RetimeError, build_chunk_filter_graph, expand_retime_chunks, prepare_smart_fit_video, ) from api.routers.dub_core import _get_job router = APIRouter() logger = logging.getLogger("omnivoice.api") def _unique_stamp() -> str: """Return a short unique suffix like '20260415T142301-ab12cd34' for export files.""" return f"{time.strftime('%Y%m%dT%H%M%S')}-{uuid.uuid4().hex[:8]}" _SAFE_LANG = re.compile(r"^[A-Za-z0-9_-]{1,32}$") def _job_dir_or_400(job_id: str) -> str: if not re.fullmatch(r"[A-Za-z0-9_-]{1,64}", job_id or ""): raise HTTPException(status_code=400, detail="Invalid job id") try: return str(resolve_within(DUB_DIR, job_id)) except UnsafePath as exc: raise HTTPException(status_code=400, detail="Invalid job id") from exc def _existing_job_dir_or_404(job_id: str) -> str: """Discover a real job directory without passing request data to a path sink.""" _job_dir_or_400(job_id) try: for entry in os.scandir(DUB_DIR): if entry.name == job_id and not entry.is_symlink() and entry.is_dir(follow_symlinks=False): return entry.path except OSError as exc: raise HTTPException(status_code=404, detail="Job directory not found") from exc raise HTTPException(status_code=404, detail="Job directory not found") def _resolve_dub_artifact(value: object, job_id: str) -> Path: """Resolve current or safely rebased pre-relocation dub artifact paths.""" raw = str(value or "") try: resolved = resolve_within(DUB_DIR, raw) relative = resolved.relative_to(Path(DUB_DIR).resolve()) if not relative.parts or relative.parts[0] != job_id: raise UnsafePath("Artifact does not belong to the requested job") return resolved except UnsafePath: # Older job rows store absolute paths. After the user relocates the # data directory, preserve only the suffix rooted at the exact # ``dub_jobs`` boundary; never touch the old host path itself. if ntpath.isabs(raw): parts = PureWindowsPath(raw).parts elif os.path.isabs(raw): parts = Path(raw).parts else: raise anchor = Path(DUB_DIR).name positions = [index for index, part in enumerate(parts) if part == anchor] if not positions: raise relative_parts = parts[positions[-1] + 1:] if ( not relative_parts or relative_parts[0] != job_id or any( part in {"", ".", ".."} or "/" in part or "\\" in part or ":" in part for part in relative_parts ) ): raise return resolve_within(DUB_DIR, Path(*relative_parts)) def _discover_job_artifact(path: Path, job_id: str) -> Path | None: """Return an existing artifact by walking the validated job directory. Persisted paths select names but never reach a filesystem sink. Each returned path comes from ``os.scandir`` beneath the validated job root, and symlinks are rejected so a post-validation swap cannot escape. """ job_root = Path(_existing_job_dir_or_404(job_id)).resolve() try: parts = path.relative_to(job_root).parts except ValueError: return None if not parts: return None current = job_root for index, requested in enumerate(parts): if os.path.basename(requested) != requested or requested in {"", ".", ".."}: return None try: entry = next( ( item for item in os.scandir(current) if item.name == requested and not item.is_symlink() ), None, ) except OSError: return None if entry is None: return None if index < len(parts) - 1 and not entry.is_dir(follow_symlinks=False): return None current = Path(entry.path) return current if current.is_file() else None def _dub_artifact(value: object, job_id: str, *, missing_detail: str = "File not found") -> str: """Resolve a persisted job artifact inside the global dub-data boundary.""" try: resolved = _resolve_dub_artifact(value, job_id) except UnsafePath as exc: raise HTTPException(status_code=400, detail="Invalid job artifact path") from exc path = _discover_job_artifact(resolved, job_id) if path is None: raise HTTPException(status_code=404, detail=missing_detail) return str(path) def _optional_dub_artifact(value: object, job_id: str) -> str | None: if not value: return None try: resolved = _resolve_dub_artifact(value, job_id) except UnsafePath as exc: raise HTTPException(status_code=400, detail="Invalid job artifact path") from exc path = _discover_job_artifact(resolved, job_id) return str(path) if path is not None else None def _safe_lang_or_400(lang: str | None) -> str | None: if lang is not None and not _SAFE_LANG.fullmatch(lang): raise HTTPException(status_code=400, detail="Invalid language code") return lang def _consume_native_save(authorization: str) -> str | None: if not authorization: return None from core.path_authorization import PathAuthorizationError, consume try: return consume(authorization, "dub_export") except PathAuthorizationError as exc: raise HTTPException(status_code=403, detail=str(exc)) from exc def _native_save(source: str, destination: str, display_name: str, media_type: str): """Copy a generated export file to a user-chosen destination and return JSON.""" import shutil dest = os.path.expanduser(destination) # Reject traversal against the user's home dir — Tauri save dialog returns abs path. if not os.path.isabs(dest): raise HTTPException(status_code=400, detail="save_path must be absolute") try: os.makedirs(os.path.dirname(dest) or ".", exist_ok=True) shutil.copy2(source, dest) except PermissionError as e: raise HTTPException(status_code=403, detail=f"Permission denied: {e}") except OSError as e: raise HTTPException(status_code=500, detail=f"Copy failed: {e}") if not os.path.exists(dest) or os.path.getsize(dest) == 0: raise HTTPException(status_code=500, detail="Copy produced empty file at destination") logger.info("Native save completed (%d bytes)", os.path.getsize(dest)) return { "saved": True, "path": dest, "size": os.path.getsize(dest), "media_type": media_type, "display_name": display_name, } @router.get("/tasks/stream/{task_id}") async def stream_task(task_id: str, after_seq: int = 0): """Universal Server-Sent Event stream for background tasks. `?after_seq=N` enables resumption: on reconnect, the client replays persisted events with seq > N, then (if the job is still live) attaches to the in-memory listener for live updates. After a server restart the in-memory task is gone but the persisted tail + final `jobs.status` are still readable, so a mid-stream reload still sees the final state. """ from core import job_store job_row = job_store.get(task_id) live = task_manager.active_tasks.get(task_id) if not live and not job_row: raise HTTPException( status_code=404, detail="No such task. It may have been cleaned up or was never created.", ) async def _reader(): # 1) Replay any persisted events after the client's last-seen seq. try: persisted = job_store.events_since(task_id, after_seq=after_seq) except Exception: persisted = [] for evt in persisted: yield evt["payload"] # 2) If the job has finished (whether in-memory or persisted-only), done. if not live: return if live["status"] in ("done", "failed", "cancelled"): return # 3) Attach to the in-memory listener for live updates. q = asyncio.Queue() await task_manager.add_listener(task_id, q) try: while True: evt = await q.get() if evt is None: break yield evt finally: await task_manager.remove_listener(task_id, q) return StreamingResponse(_reader(), media_type="text/event-stream") @router.get("/jobs") async def list_jobs(status: str | None = None, project_id: str | None = None, limit: int = 100): """List persisted jobs, newest first. `status=active` → running + pending (what the batch-queue UI wants). `status=failed|done|cancelled|pending|running` → exact match. `project_id=...` → scope to one project. """ from core import job_store limit = max(1, min(500, int(limit))) return job_store.list_jobs(status=status, project_id=project_id, limit=limit) @router.get("/jobs/{job_id}") async def get_job(job_id: str): from core import job_store row = job_store.get(job_id) if not row: raise HTTPException( status_code=404, detail="No such job. It may have been cleaned up or never created.", ) return row @router.get("/jobs/{job_id}/events") async def list_job_events(job_id: str, after_seq: int = 0, limit: int = 500): """Persisted SSE tail. Strict ascending seq so the client can stitch it onto a live feed (which starts above the last returned seq). """ from core import job_store row = job_store.get(job_id) if not row: raise HTTPException( status_code=404, detail="No job with that id. It may have expired, been deleted, or the server restarted before it was persisted — check the dub history in the sidebar.", ) limit = max(1, min(2000, int(limit))) return { "job": row, "events": job_store.events_since(job_id, after_seq=after_seq, limit=limit), } @router.post("/tasks/cancel/{task_id}") async def cancel_task(task_id: str): """Cancel a running background task (e.g. dub generation).""" ok = task_manager.cancel_task(task_id) if not ok: raise HTTPException(status_code=404, detail="Task not found") return {"cancelled": True, "task_id": task_id} @router.get("/dub/tracks/{job_id}") async def dub_list_tracks(job_id: str): job = _get_job(job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") return {"tracks": job.get("dubbed_tracks", {})} @router.get("/dub/segments-text/{job_id}") async def dub_segments_text(job_id: str, lang: str = Query(...)): """Per-segment texts for one generated track: ``{"texts": {segKey: text}}``. Backing store is ``job["segments_i18n"]`` (P1.2) — the authoritative per-language map every generate rebuilds. The Export preview tabs use it to hydrate segments whose in-browser ``translations[lang]`` entry is missing (tracks generated before per-language persistence, partial regens), so switching the preview language can't leave a mixed-language transcript. Empty map when the job predates segments_i18n or the track was never generated — the client keeps whatever it has. """ job = _get_job(job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") i18n = job.get("segments_i18n") or {} return {"texts": i18n.get(lang) or {}} def _segments_for_lang(job: dict, lang: "str | None") -> list: """Job segments with `text` overlaid from ``job["segments_i18n"][lang]``. P1.2 — ``job["segments"]`` is single-slot: it holds whichever language was generated LAST, so exporting subtitles for track A after generating track B emitted B's text under A's language label (the "N identical subtitle files" class). ``segments_i18n`` ({lang: {segKey: text}}, written by ``dub_generate._sync_job_segments``) preserves each generated track's text; this overlays it non-destructively when present. Back-compat: no lang requested, no ``segments_i18n`` on the job (predates the field), no entry for this lang, or no text for a given segment — each falls back to the segment as-is, i.e. exactly today's behaviour. Segment keys are the stable id (str) with the list index (str) as the legacy fallback, mirroring how the map is written. """ segments = job.get("segments", []) if not lang: return segments i18n = job.get("segments_i18n") lang_texts = i18n.get(lang) if isinstance(i18n, dict) else None if not isinstance(lang_texts, dict) or not lang_texts: return segments out = [] for i, seg in enumerate(segments): key = str(seg.get("id")) if seg.get("id") is not None else str(i) txt = lang_texts.get(key) if txt is None: txt = lang_texts.get(str(i)) out.append(dict(seg, text=txt) if isinstance(txt, str) and txt.strip() else seg) return out def _write_burn_srt(job: dict, exports_dir: str, stamp: str, dual: bool, fitted_segments: "list[dict] | None" = None, lang: "str | None" = None) -> str | None: """Build a temp SRT from job segments for use with ffmpeg's subtitles filter. Returned path is already ffmpeg-filter-safe (plain ASCII basename under exports_dir). Returns None if there are no segments to render. ``fitted_segments`` (Smart Fit): {id, start, end} cue records on the fitted timeline — when provided, cue times come from there instead of the original ``job["segments"]`` timings, so burned subs track the retimed video / fitted audio rather than the source timeline. ``lang`` (P1.2): burn the named track's text (see ``_segments_for_lang``) instead of whatever language generated last. """ segments = _segments_for_lang(job, lang) if not segments: return None if fitted_segments: segments = _apply_fitted_times(segments, fitted_segments) lines = [] for i, seg in enumerate(segments): lines.append(str(i + 1)) lines.append(f"{_format_srt_time(seg['start'])} --> {_format_srt_time(seg['end'])}") lines.append(_pick_subtitle_text(seg, dual)) lines.append("") sub_path = os.path.join(exports_dir, f"burn_subs_{stamp}.srt") with open(sub_path, "w", encoding="utf-8") as f: f.write("\n".join(lines)) return sub_path def _write_burn_ass(job: dict, exports_dir: str, stamp: str, fitted_segments: "list[dict] | None" = None, lang: "str | None" = None) -> str | None: """Karaoke variant of ``_write_burn_srt``: word-timed ASS via ``build_ass``. Same text/timing resolution (``_segments_for_lang`` + fitted-cue overlay, which also scales per-word times onto the fitted timeline); the basename is plain ASCII under exports_dir so it is ffmpeg-filter-safe. Returns None if there are no segments to render. """ segments = _segments_for_lang(job, lang) if not segments: return None if fitted_segments: segments = _apply_fitted_times(segments, fitted_segments) sub_path = os.path.join(exports_dir, f"burn_subs_{stamp}.ass") with open(sub_path, "w", encoding="utf-8") as f: f.write(build_ass(segments)) return sub_path def _ffmpeg_filter_escape(path: str) -> str: """Escape a path for use inside an ffmpeg filter value (subtitles=...). ffmpeg's filter parser treats `:` as an option separator and `\\`, `'` specially. Backslashes first, then colons, then single quotes. """ return path.replace("\\", "\\\\").replace(":", "\\:").replace("'", "\\'") def _build_video_stretch_filter_graph( plan: list[dict], orig_dur: float, video_input_idx: int = 0, in_label: str | None = None, ) -> tuple[str, str]: """Build an ffmpeg filter_complex graph that stretches the source video per-segment so each segment's visual duration matches a dub audio layout. `plan` is a list of {orig_start, orig_end, new_start, new_end, stretch_ratio} in original-time order (as persisted by dub_generate for stretch_video mode). Gaps between plan entries — and the pre-roll / tail — are passed through at 1.0× rate so silence and B-roll don't get squashed. `in_label` overrides the input stream reference; e.g. pass "[vsub]" when a subtitles filter has already written to that label. Defaults to `[{video_input_idx}:v]` for direct source-stream consumption. Returns (filter_graph, output_label). Output label is "[vstretched]" when chunks were emitted, or the original input label when the plan was empty (caller should fall back to stream-copy in that case). """ # Empty plan = no stretch — caller should stream-copy the video. Return # early so we don't synthesise a degenerate "stretch whole video at 1.0×" # graph that would force a needless re-encode. if not plan: return "", in_label or f"[{video_input_idx}:v]" # Chunk expansion + graph emission live in services.video_retime now so # the Smart Fit batched pipeline shares the exact same boundary math. # With default options the emitted graph is byte-identical to the # original inline implementation. chunks = expand_retime_chunks(plan, orig_dur) if not chunks: return "", in_label or f"[{video_input_idx}:v]" return build_chunk_filter_graph(chunks, in_label or f"[{video_input_idx}:v]") def _video_stretch_plan_for(job: dict, lang_code: str) -> dict | None: """Return the persisted stretch plan + total durations for `lang_code`, or None if this job didn't use stretch_video mode (or no plan exists). """ if (job.get("timing_strategy") or "").lower() != "stretch_video": return None plans = job.get("video_stretch_plans") or {} entry = plans.get(lang_code) if not entry or not entry.get("plan"): return None return entry def _video_retime_plan_for(job: dict, lang_code: str) -> "tuple[str, dict] | None": """Resolve the video retime plan for ``lang_code`` across both keyspaces. Returns ``(kind, entry)`` where kind is ``"stretch_video"`` (legacy Mode B plans — resolution byte-identical to ``_video_stretch_plan_for``) or ``"smart_fit"`` (Phase A ``job["fit_plans"]`` entries, gated on the track actually having been generated under smart_fit so a stale plan from an earlier run can't retime a track re-generated under another strategy). ``None`` when neither applies. """ legacy = _video_stretch_plan_for(job, lang_code) if legacy is not None: return "stretch_video", legacy entry = (job.get("fit_plans") or {}).get(lang_code) track = (job.get("dubbed_tracks") or {}).get(lang_code) or {} if entry and entry.get("plan") and track.get("timing_strategy") == "smart_fit": return "smart_fit", entry return None def _fitted_segments_for(job: dict, lang_code: "str | None") -> "list[dict] | None": """Fitted-timeline subtitle cues ({id, start, end}) for a Smart Fit track, or None. Same staleness gate as ``_video_retime_plan_for``.""" if not lang_code: return None entry = (job.get("fit_plans") or {}).get(lang_code) track = (job.get("dubbed_tracks") or {}).get(lang_code) or {} if not entry or track.get("timing_strategy") != "smart_fit": return None fitted = entry.get("fitted_segments") return fitted or None def _apply_fitted_times(segments: list[dict], fitted: list[dict]) -> list[dict]: """Overlay fitted cue times onto subtitle segments (copies; non-destructive). Matches by segment ``id``; when the fitted record carries no ids at all (defensive), falls back to positional pairing. Segments without a match keep their original timings. """ by_id = {str(f["id"]): f for f in fitted if f.get("id") is not None} out: list[dict] = [] for i, seg in enumerate(segments): cue = None if seg.get("id") is not None: cue = by_id.get(str(seg["id"])) if cue is None and not by_id and i < len(fitted): cue = fitted[i] if cue is None: out.append(seg) continue patched = dict(seg) patched["start"] = float(cue["start"]) patched["end"] = float(cue["end"]) # Karaoke burn-in: persisted word times live on the original timeline; # scale them linearly onto the fitted cue span so the highlight sweep # follows the retimed audio. Degenerate spans drop the words — export # then falls back to an even split over the fitted span. Inert for # SRT/VTT, which never read ``words``. if isinstance(seg.get("words"), list) and seg.get("words"): scaled = scale_words( seg["words"], seg.get("start", 0.0), seg.get("end", 0.0), patched["start"], patched["end"], ) if scaled is not None: patched["words"] = scaled else: patched.pop("words", None) out.append(patched) return out def _burn_subs_allowed(retime_kind: "str | None") -> bool: """Subtitle burn-in combined with video retime. Allowed for ``smart_fit`` (fitted cue records exist, and the burn pass runs AFTER the retime graph so cues land on the retimed timeline) and for plain exports. Still rejected for legacy ``stretch_video``, which has no fitted-cue record — cues would burn at original timestamps onto a re-timed video and drift. """ return retime_kind != "stretch_video" #: Audio export formats → ffmpeg codec args. Unknown formats fall back to #: AAC/m4a so a bad request can never produce a broken command. _AUDIO_FORMAT_CODECS: dict[str, list[str]] = { "wav": ["-c:a", "pcm_s16le"], "m4a": ["-c:a", "aac", "-b:a", "192k"], "mp3": ["-c:a", "libmp3lame", "-q:a", "2"], "flac": ["-c:a", "flac"], } def _build_audio_export_cmd( ffmpeg: str, track_path: str, bg_path: Optional[str], out_path: str, fmt: str, ) -> list[str]: """Build the ffmpeg command for an audio-only dub export (#119). No video input, stream-map, or codec — just the dubbed track, optionally mixed with the separated background (``no_vocals``), written to ``out_path`` in the requested ``fmt``. Unknown formats fall back to AAC. """ codec = _AUDIO_FORMAT_CODECS.get((fmt or "").lower(), _AUDIO_FORMAT_CODECS["m4a"]) cmd = [ffmpeg, "-y", "-i", track_path] if bg_path: # Mix the dubbed voice over the original background bed (same weights # as the video mux path) so ambience/music is preserved. cmd += ["-i", bg_path, "-filter_complex", bed_mix_filter("1:a", "0:a"), "-map", "[aout]"] cmd += codec cmd.append(out_path) return cmd @router.get("/dub/download/{job_id}") @router.get("/dub/download/{job_id}/{filename}") async def dub_download( job_id: str, preserve_bg: bool = Query(True, description="Mix background noise into dubbed tracks"), default_track: str = Query("", description="Default audio track; omitted selects the first dubbed track"), include_tracks: str = Query("", description="Comma-separated list of tracks to include (e.g. 'original,de,es'). Empty = include all."), save_authorization: str = Header("", alias="X-VoiceStudio-Path-Authorization"), burn_subs: bool = Query(False, description="Burn subtitles into the video stream (forces re-encode). Uses dual-subtitle layout when dual=1."), dual: bool = Query(False, description="When burn_subs=1, render translated on top of italicised original."), karaoke: bool = Query(False, description="When burn_subs=1, burn a word-timed karaoke highlight (ASS) instead of line subtitles. Ignored when dual=1 (dual karaoke is unsupported — the line burn renders instead)."), out_format: str = Query("m4a", description="Audio-only jobs (#119): output container — wav, m4a, mp3, or flac. Ignored for video jobs."), ): # Strict allowlist on the path param BEFORE it reaches any filesystem # path or ffmpeg argv (export dir, retime work path, slice paths). Real # job ids are short uuid slices — alnum/hyphen/underscore only. job_dir = _job_dir_or_400(job_id) job = _get_job(job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") tracks = job.get("dubbed_tracks", {}) if not tracks: raise HTTPException(status_code=400, detail="No dubbed tracks generated yet") include_set = set(t.strip() for t in include_tracks.split(",") if t.strip()) if include_tracks else None include_original = include_set is None or "original" in include_set if include_set: filtered_tracks = {k: v for k, v in tracks.items() if k in include_set} else: filtered_tracks = dict(tracks) filtered_tracks = { key: { **value, "path": _dub_artifact(value.get("path"), job_id, missing_detail="Dubbed track not found"), } for key, value in filtered_tracks.items() } # A dub export should play the dub without requiring player-specific track # selection. Keep ``original`` as an explicit opt-in, but when callers omit # the preference choose the first generated dub consistently (#1575). if ( filtered_tracks and not (default_track == "original" and include_original) and default_track not in filtered_tracks ): default_track = next(iter(filtered_tracks)) elif not filtered_tracks and include_original: default_track = "original" if not filtered_tracks and not include_original: raise HTTPException(status_code=400, detail="No tracks selected for export") video_path = _dub_artifact(job["video_path"], job_id, missing_detail="Source video not found") stamp = _unique_stamp() exports_dir = os.path.join(job_dir, "exports") os.makedirs(exports_dir, exist_ok=True) output_path = os.path.join(exports_dir, f"dubbed_video_{stamp}.mp4") ffmpeg = find_ffmpeg() # ── Audio-only dubbing (#119) ───────────────────────────────────────── # No source video to mux into — export the dubbed track (optionally mixed # with the separated background) straight to an audio container. if (job.get("input_type") or "video").lower() == "audio": if default_track and default_track != "original" and default_track in filtered_tracks: lang_code, track_info = default_track, filtered_tracks[default_track] elif filtered_tracks: lang_code, track_info = next(iter(filtered_tracks.items())) else: raise HTTPException(status_code=400, detail="No dubbed track selected for audio export") fmt = (out_format or "m4a").lower() if fmt not in _AUDIO_FORMAT_CODECS: fmt = "m4a" # Keep route/job data out of the filesystem and logging trust boundary. # The selected format reaches the path only through literal branches. if fmt == "wav": output_name = f"dubbed_audio_{stamp}.wav" elif fmt == "mp3": output_name = f"dubbed_audio_{stamp}.mp3" elif fmt == "flac": output_name = f"dubbed_audio_{stamp}.flac" else: output_name = f"dubbed_audio_{stamp}.m4a" out_path = os.path.join(exports_dir, output_name) bg = _optional_dub_artifact(job.get("no_vocals_path"), job_id) if preserve_bg else None cmd = _build_audio_export_cmd(ffmpeg, track_info["path"], bg, out_path, fmt) try: rc, _, stderr = await run_ffmpeg(cmd, timeout=1800.0) if rc != 0: raise Exception(stderr.decode(errors="replace") if stderr else "ffmpeg audio export non-zero") except asyncio.TimeoutError: raise HTTPException(status_code=504, detail="ffmpeg audio export timed out") except HTTPException: raise except Exception as e: raise HTTPException( status_code=500, detail=explain_ffmpeg_failure(e, "export dubbed audio", cmd=cmd), ) if not os.path.exists(out_path) or os.path.getsize(out_path) == 0: raise HTTPException(status_code=500, detail="ffmpeg audio export produced no output file") logger.info("Dub audio export completed (%d bytes)", os.path.getsize(out_path)) # Response metadata must not become a second path-like sink for job or # request data. Keep the user-selected format through explicit literal # branches; source names and language keys never enter the label. if fmt == "wav": dl_name = f"dubbed_audio_{stamp}.wav" elif fmt == "mp3": dl_name = f"dubbed_audio_{stamp}.mp3" elif fmt == "flac": dl_name = f"dubbed_audio_{stamp}.flac" else: dl_name = f"dubbed_audio_{stamp}.m4a" media_type = _MEDIA_TYPES.get(f".{fmt}", "audio/mp4") save_path = _consume_native_save(save_authorization) if save_path: # Keep the request-derived download label out of the filesystem # trust boundary. It is response metadata, not a source or # destination path (CodeQL, #1575). result = _native_save(out_path, save_path, "dubbed_audio", media_type=media_type) result["display_name"] = dl_name return result return FileResponse( out_path, media_type=media_type, headers={"Content-Disposition": content_disposition(dl_name)}, ) # Determine whether this export should drive video through a per-segment # retime (legacy stretch_video Mode B, or Smart Fit). Retime is keyed off # the default_track's plan because the video can only physically follow # one timeline at a time. If multiple dub tracks are included, only the # default_track is visually in sync — other tracks share the same # (retimed) video. Single-track export is the supported common case. retime_kind: "str | None" = None retime_entry: "dict | None" = None if default_track and default_track != "original": _retime = _video_retime_plan_for(job, default_track) if _retime: retime_kind, retime_entry = _retime stretch_entry = retime_entry if retime_kind == "stretch_video" else None # Subtitle burn under legacy stretch_video would render cues at the # original timestamps onto a re-timed video — they'd drift (no fitted-cue # record exists for that mode). Skip the burn pass in that combo and log; # the user can still export the SRT/VTT separately. Smart Fit DOES carry # fitted cues, so burn+retime is allowed there (burn runs post-retime). if not _burn_subs_allowed(retime_kind) and burn_subs: logger.warning( "stretch_video + burn_subs is not supported in one pass; " "skipping subtitle burn for job %s. Export the SRT/VTT separately.", log_safe(job_id), ) burn_subs = False # Smart Fit: cue times come from the fitted timeline — that's where the # dubbed audio actually sits, whether or not the video retime succeeds. fitted_segments = _fitted_segments_for(job, default_track) if default_track and default_track != "original" else None # Burn the DEFAULT track's text (P1.2) — it's the audio the viewer hears. _burn_lang = default_track if default_track and default_track != "original" else None # Karaoke (word-highlight) burn writes an ASS instead of the line SRT. # Dual layout keeps the line burn — dual karaoke is out of scope, matching # the disabled control in the Export drawer. The default (karaoke off) # takes exactly the legacy SRT path. sub_path = None sub_is_ass = False if burn_subs: if karaoke and not dual: sub_path = _write_burn_ass(job, exports_dir, stamp, fitted_segments=fitted_segments, lang=_burn_lang) sub_is_ass = sub_path is not None if sub_path is None: sub_path = _write_burn_srt(job, exports_dir, stamp, dual, fitted_segments=fitted_segments, lang=_burn_lang) # ── Smart Fit video retime (two-tier) ───────────────────────────────── # Tier 1 (≤48 chunks): single filter_complex graph inlined into the mux # command below. Tier 2: batched slice renders joined by the concat # demuxer into an intermediate file, muxed as an extra input. Failures # fall back to an un-retimed export with a structured warning rather # than failing the whole download. retime_decision = None retime_warning: "dict | None" = None smart_track_dur = 0.0 if retime_kind == "smart_fit" and retime_entry: smart_orig_dur = float(retime_entry.get("orig_duration") or job.get("duration") or 0.0) smart_track_dur = float( retime_entry.get("total_duration") or (filtered_tracks.get(default_track) or {}).get("duration") or 0.0 ) # A fresh export is a fresh user intent — clear any sticky abort flag # from a previous /dub/abort so it can't kill this run's first batch. job.pop("aborted", None) # realpath-normalised + containment-checked inline at the sink (the # file's established pattern — CodeQL does not track the guard # through a helper's return value). _base = os.path.realpath(DUB_DIR) retime_work_path = os.path.realpath( os.path.join(exports_dir, f"retimed_{stamp}.mp4") ) if retime_work_path != _base and not retime_work_path.startswith(_base + os.sep): raise HTTPException(status_code=400, detail="Invalid export path") try: retime_decision = await prepare_smart_fit_video( job_id=job_id, ffmpeg=ffmpeg, video_path=video_path, plan=retime_entry["plan"], orig_dur=smart_orig_dur, track_dur=smart_track_dur, work_path=retime_work_path, abort_check=lambda: bool(job.get("aborted")), ) except Exception as e: if (isinstance(e, RetimeError) and e.stage == "aborted") or job.get("aborted"): raise HTTPException(status_code=409, detail="Export aborted") from core.failure import build_failure retime_warning = build_failure(e, stage="video-retime", include_diagnostic=False) job["last_export_warning"] = {"type": "video_retime_fallback", **retime_warning} logger.exception( "Smart Fit video retime failed for job %s — exporting " "without per-segment retime", log_safe(job_id), ) cmd = [ffmpeg, "-i", video_path] input_idx = 1 retimed_idx = None if retime_decision is not None and retime_decision.mode == "file": cmd += ["-i", retime_decision.file_path] retimed_idx = input_idx input_idx += 1 bg_audio = _optional_dub_artifact(job.get("no_vocals_path"), job_id) if preserve_bg else None bg_idx = None if bg_audio and filtered_tracks: cmd += ["-i", bg_audio] bg_idx = input_idx input_idx += 1 tracks_to_process = [] for lang_code, track_info in filtered_tracks.items(): cmd += ["-i", track_info["path"]] tracks_to_process.append({"lang_code": lang_code, "idx": input_idx, "info": track_info}) input_idx += 1 filter_parts: list[str] = [] video_map = "0:v:0" video_reencode = False if retime_decision is not None: if retime_decision.mode == "filter": filter_parts.append(retime_decision.graph) video_map = retime_decision.label video_reencode = True else: video_map = f"{retimed_idx}:v:0" # Residual drift after the batched render (fps rounding): video # shorter than the fitted track → freeze the last frame out to # the track length. Rare — the predicted tail pad inside the # render usually lands within tolerance. residual = smart_track_dur - retime_decision.video_dur if smart_track_dur and residual > DRIFT_TOLERANCE_S: filter_parts.append( f"[{retimed_idx}:v]tpad=stop_mode=clone:stop_duration={residual:.4f}[vtpad]" ) video_map = "[vtpad]" video_reencode = True if sub_path: esc = _ffmpeg_filter_escape(sub_path) # Burn AFTER any retime so cues (already on the fitted timeline for # Smart Fit) land on the retimed video. Without retime this reduces # to the legacy `[0:v]subtitles=…[vsub]` graph. Karaoke burns the # word-timed ASS through the ass filter at the same graph position. if video_map.startswith("["): sub_src = video_map elif retimed_idx is not None: sub_src = f"[{retimed_idx}:v]" else: sub_src = "[0:v]" _sub_filter = "ass" if sub_is_ass else "subtitles" filter_parts.append(f"{sub_src}{_sub_filter}='{esc}'[vsub]") video_map = "[vsub]" if stretch_entry: orig_dur = float(stretch_entry.get("orig_duration") or job.get("duration") or 0.0) graph, vlabel = _build_video_stretch_filter_graph( stretch_entry["plan"], orig_dur, video_input_idx=0, in_label=video_map if video_map != "0:v:0" else None, ) if graph: filter_parts.append(graph) video_map = vlabel cmd += ["-map", video_map] if include_original: cmd += ["-map", "0:a:0"] # Smart Fit drift absorption, audio side: when the retimed video runs # longer than the fitted track (its tail passes through at 1.0× beyond # the last cue, or encoder rounding), pad the dub-track chain with # silence out to the video length so players don't end audio early. apad_dur = 0.0 if ( retime_decision is not None and smart_track_dur and retime_decision.video_dur - smart_track_dur > DRIFT_TOLERANCE_S ): apad_dur = retime_decision.video_dur if bg_idx is not None: for i, t in enumerate(tracks_to_process): tail = f",apad=whole_dur={apad_dur:.4f}" if apad_dur else "" filter_parts.append(bed_mix_filter( f"{bg_idx}:a", f"{t['idx']}:a", out=f"aout{i}", tail=tail, uniq=str(i), )) t["out_label"] = f"[aout{i}]" for t in tracks_to_process: cmd += ["-map", t["out_label"]] elif apad_dur: for i, t in enumerate(tracks_to_process): out_label = f"[aout{i}]" filter_parts.append(f"[{t['idx']}:a]apad=whole_dur={apad_dur:.4f}{out_label}") t["out_label"] = out_label for t in tracks_to_process: cmd += ["-map", t["out_label"]] else: for t in tracks_to_process: cmd += ["-map", f"{t['idx']}:a:0"] if filter_parts: cmd += ["-filter_complex", ";".join(filter_parts)] # Burning subs or per-segment video retime both force a real video # re-encode; stream-copy is viable when nothing touches the video # filter chain — including the batched Smart Fit path, whose retimed # intermediate is already encoded with these exact settings. if sub_path or stretch_entry or video_reencode: cmd += ["-c:v", "libx264", "-preset", "medium", "-crf", "20", "-pix_fmt", "yuv420p"] else: cmd += ["-c:v", "copy"] cmd += ["-c:a", "aac", "-b:a", "192k"] audio_stream_idx = 0 if include_original: cmd += [f"-metadata:s:a:{audio_stream_idx}", "language=und", f"-metadata:s:a:{audio_stream_idx}", "title=Original"] audio_stream_idx += 1 for t in tracks_to_process: cmd += [ f"-metadata:s:a:{audio_stream_idx}", f"language={t['lang_code']}", f"-metadata:s:a:{audio_stream_idx}", f"title={t['info']['language']}" ] t["stream_idx"] = audio_stream_idx audio_stream_idx += 1 total_audio = (1 if include_original else 0) + len(tracks_to_process) for i in range(total_audio): cmd += [f"-disposition:a:{i}", "0"] if default_track == "original" and include_original: cmd += ["-disposition:a:0", "default"] else: # A stale/missing language preference still means "play a dub", not # "silently fall back to the source". The first processed dub is the # deterministic fallback; ``original`` above remains explicit. target_idx = tracks_to_process[0]["stream_idx"] if tracks_to_process else 0 for t in tracks_to_process: if t['lang_code'] == default_track: target_idx = t["stream_idx"] break cmd += [f"-disposition:a:{target_idx}", "default"] # When retiming (legacy stretch_video or Smart Fit) the video and audio # durations should match within sub-frame precision, but `-shortest` # can still cut off the trailing frame; let ffmpeg keep both streams. # Otherwise keep the legacy `-shortest` so a slightly-overrunning track # doesn't extend the mux past the video. if not stretch_entry and retime_decision is None: cmd += ["-shortest"] cmd += [output_path, "-y"] try: rc, _, stderr = await run_ffmpeg(cmd, timeout=1800.0, job_id=job_id) if rc != 0: raise Exception(stderr.decode(errors="replace") if stderr else "ffmpeg mux non-zero") except asyncio.TimeoutError: raise HTTPException(status_code=504, detail="ffmpeg mux timed out") except HTTPException: raise except Exception as e: raise HTTPException( status_code=500, detail=explain_ffmpeg_failure(e, "combine video + dubbed audio", cmd=cmd), ) finally: # The batched retime intermediate is a full re-encoded video — never # leave it behind (success or failure; it's stamp-unique, no reuse). if retime_decision is not None and retime_decision.mode == "file": try: os.remove(retime_decision.file_path) except OSError as e: logger.debug("cleanup remove failed: %s", e) if not os.path.exists(output_path) or os.path.getsize(output_path) == 0: raise HTTPException(status_code=500, detail="ffmpeg mux produced no output file") logger.info("Dub mux completed (%d bytes)", os.path.getsize(output_path)) base_name = os.path.splitext(job.get('filename', 'output'))[0] safe_name = ''.join(c for c in base_name if c.isalnum() or c in '-_ ').strip() or 'output' dl_name = f"dubbed_{safe_name}_{stamp}.mp4" # Structured warning surface for the Smart Fit fallback ladder: header is # a fixed ASCII token (FileResponse headers must be latin-1 safe); the # full build_failure payload is persisted on the job for the UI to read. extra_headers = {} if retime_warning is not None: extra_headers["X-Dub-Export-Warning"] = "video-retime-fallback" save_path = _consume_native_save(save_authorization) if save_path: result = _native_save(output_path, save_path, dl_name, media_type="video/mp4") if retime_warning is not None: result["warning"] = {"type": "video_retime_fallback", **retime_warning} return result return FileResponse( output_path, media_type="video/mp4", headers={"Content-Disposition": content_disposition(dl_name), **extra_headers}, ) _MEDIA_TYPES = { ".mp4": "video/mp4", ".m4v": "video/mp4", ".mov": "video/quicktime", ".webm": "video/webm", ".mkv": "video/x-matroska", ".m4a": "audio/mp4", ".mp3": "audio/mpeg", ".wav": "audio/wav", ".flac": "audio/flac", ".ogg": "audio/ogg", } @router.get("/dub/media/{job_id}") async def dub_get_media(job_id: str): _job_dir_or_400(job_id) job = _get_job(job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") video_path = _dub_artifact(job["video_path"], job_id, missing_detail="Media file not found") # Pass an explicit media_type. Without this Starlette falls back to # mimetypes.guess_type, which on some platforms returns the wrong # MIME (e.g. "application/octet-stream" for .mkv), and the Tauri # WebView then refuses to render the