fix(dub): actionable transcription errors for missing ffmpeg and broken pipe (#2404, #2405)

Classify private per-chunk ASR exceptions into fixed public replies, raise a
typed MediaToolUnavailableError from the validated decoder, and expose the
resolved ffmpeg under the bare name so dependencies that exec 'ffmpeg' find
imageio-ffmpeg's renamed binary on every platform.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
Palash Debnath
2026-10-01 18:52:07 +05:30
co-authored by Claude Sonnet 5.5
parent e240580ccc
commit 1a30c3abb0
6 changed files with 291 additions and 16 deletions
+9 -8
View File
@@ -1574,17 +1574,18 @@ async def dub_transcribe_stream(
return {"chunks": shifted, "language": r.get("language"), "speaker_turns": turns}
except Exception as exc:
# Keep diagnostics local and fixed-shape. In particular,
# CUDA OOM is a distinct, actionable recovery class rather
# than the generic "no segments" dead end.
# CUDA OOM, a missing ffmpeg and a closed stdio pipe are
# distinct, actionable recovery classes rather than the
# generic "no segments" dead end.
is_memory = isinstance(exc, torch.OutOfMemoryError)
logger.error(
"Chunk transcription failed (backend=%s; class=%s; details withheld)",
_asr_backend.id,
type(exc).__name__,
)
from core.public_errors import stream_failure
from core.public_errors import stream_failure, transcription_failure_code
failure = stream_failure(
"transcription_memory" if is_memory else "transcription_failed"
"transcription_memory" if is_memory else transcription_failure_code(exc)
)
return {
"chunks": [],
@@ -2251,10 +2252,10 @@ async def dub_transcribe_stream(
try:
async for ev in _gen_body():
yield ev
except Exception: # noqa: BLE001 — last-resort stream finalizer
logger.error("Transcription stream failed unexpectedly")
from core.public_errors import stream_failure
yield _sse_event("error", stream_failure("transcription_failed"))
except Exception as exc: # noqa: BLE001 — last-resort stream finalizer
logger.error("Transcription stream failed unexpectedly (class=%s)", type(exc).__name__)
from core.public_errors import stream_failure, transcription_failure_code
yield _sse_event("error", stream_failure(transcription_failure_code(exc)))
yield _sse_event("done", {})
finally:
_asr_work.stop()
+68
View File
@@ -2,6 +2,7 @@
from __future__ import annotations
import logging
import os
from typing import Any
_PROVIDER_DETAILS = {
@@ -70,10 +71,77 @@ def stream_failure(code: str) -> dict[str, object]:
),
"retryable": True,
},
"transcription_media_tool": {
"code": "transcription_media_tool",
"detail": (
"Transcription needs ffmpeg, which VoiceStudio could not find or "
"run. Open Settings → Audio tools and use Download/Repair, or "
"install ffmpeg (macOS: brew install ffmpeg; Windows: winget "
"install Gyan.FFmpeg; Linux: your package manager) and restart "
"VoiceStudio. You can also point FFMPEG_PATH at an ffmpeg binary."
),
"retryable": True,
},
"transcription_pipe_lost": {
"code": "transcription_pipe_lost",
"detail": (
"The backend lost its output pipe to the app while transcribing "
"(broken pipe). Restart VoiceStudio and try again."
),
"retryable": True,
},
}
return dict(failures.get(code, failures["generation_failed"]))
def _exception_chain(error: object, limit: int = 8):
"""The exception, then its ``__cause__``/``__context__`` links (bounded)."""
seen: set[int] = set()
while isinstance(error, BaseException) and id(error) not in seen and len(seen) < limit:
seen.add(id(error))
yield error
error = error.__cause__ or error.__context__
def transcription_failure_code(error: object) -> str:
"""Stable ``stream_failure`` code for a private ASR chunk exception.
Only the exception's type, errno and filename are inspected locally; the
caller ships a VoiceStudio-owned message for the returned code, never the
exception text. A missing/unrunnable ffmpeg (``[Errno 2] ... 'ffmpeg'``,
``[WinError 193]``, :class:`MediaToolUnavailableError`) and a closed stdio
pipe (``[Errno 32] Broken pipe``) each have a different remedy, so they must
not collapse into the generic "check the selected ASR engine" reply (#2404,
#2405).
"""
import errno
from services.ffmpeg_utils import MediaToolUnavailableError
pipe = False
for exc in _exception_chain(error):
if isinstance(exc, MediaToolUnavailableError):
return "transcription_media_tool"
if isinstance(exc, OSError):
name = os.path.basename(str(exc.filename or "")).lower()
if name.startswith(("ffmpeg", "ffprobe")) and (
exc.errno in (errno.ENOENT, errno.ENOEXEC, errno.EACCES)
or getattr(exc, "winerror", None) in (2, 193)
):
return "transcription_media_tool"
low = str(exc).lower()
try:
from core.failure import _is_missing_media_tool
if _is_missing_media_tool(low):
return "transcription_media_tool"
except Exception: # noqa: BLE001 — classification must never raise
pass
if isinstance(exc, BrokenPipeError) or "broken pipe" in low or "[errno 32]" in low:
pipe = True
return "transcription_pipe_lost" if pipe else "transcription_failed"
def stream_generation_failure(error: BaseException | object) -> dict[str, object]:
"""``generation_failed`` stream metadata, enriched with the actual cause.
+3 -3
View File
@@ -333,11 +333,11 @@ def _decode_audio_16k_mono(audio_path: str):
import numpy as np
from services.ffmpeg_utils import find_ffmpeg
from services.ffmpeg_utils import MediaToolUnavailableError, find_ffmpeg
ffmpeg = find_ffmpeg()
if not ffmpeg:
raise RuntimeError(
raise MediaToolUnavailableError(
"Cannot transcribe: ffmpeg is missing or not runnable. Install "
"ffmpeg (or let VoiceStudio's bundled binary download), then retry. "
"On Windows a '[WinError 193]' here means the ffmpeg binary is "
@@ -354,7 +354,7 @@ def _decode_audio_16k_mono(audio_path: str):
# Belt-and-suspenders: find_ffmpeg() already -version-validated this
# binary, so a WinError 193 here is unexpected — surface it clearly
# rather than letting it become "no segments".
raise RuntimeError(
raise MediaToolUnavailableError(
f"ffmpeg at {ffmpeg!r} could not be executed ({e}). Reinstall "
"ffmpeg or clear the imageio-ffmpeg cache."
) from e
+57 -4
View File
@@ -190,6 +190,15 @@ def _binary_runs(path: str) -> bool:
return ok
class MediaToolUnavailableError(RuntimeError):
"""ffmpeg/ffprobe is missing, or present but not runnable.
A ``RuntimeError`` so existing handlers keep working; the dedicated type
lets the transcription path pick the actionable "repair the media engine"
reply by class instead of by sniffing message text.
"""
def find_ffmpeg():
"""Locate an ffmpeg binary.
@@ -320,6 +329,49 @@ def find_ffprobe():
return None
def _bare_name_shim(real: str, tool: str) -> "str | None":
"""Directory exposing *real* under the bare name ``<tool>[.exe]``, or None.
imageio-ffmpeg ships its binary as ``ffmpeg-<platform>-vN[.exe]``, so
publishing its directory on ``PATH`` never satisfies a dependency's literal
``ffmpeg`` lookup (parakeet-mlx, openai-whisper, pydub, ...): they still die
with ``[Errno 2] No such file or directory: 'ffmpeg'``. A symlink (hardlink
or copy where Windows refuses symlinks) under the bare name closes that gap
on every platform without asking the user to install anything. Best-effort:
returns None when the name is already bare or the shim cannot be written.
"""
exe = f"{tool}.exe" if os.name == "nt" else tool
if os.path.basename(real).lower() == exe:
return None
try:
from core.config import DATA_DIR
directory = os.path.join(DATA_DIR, "media_tools", "shims")
link = os.path.join(directory, exe)
os.makedirs(directory, exist_ok=True)
if os.path.lexists(link):
try:
if os.path.samefile(link, real) or (
not os.path.islink(link)
and os.path.getsize(link) == os.path.getsize(real)
):
return directory
except OSError:
pass
os.unlink(link)
try:
os.symlink(real, link)
except (OSError, NotImplementedError):
try:
os.link(real, link)
except OSError:
shutil.copy2(real, link)
return directory
except OSError as e:
logger.debug("bare-name %s shim unavailable: %s", tool, e)
return None
def ensure_media_tools_on_path() -> list[str]:
"""Put the resolved ffmpeg/ffprobe on ``PATH`` for third-party code (#1256).
@@ -341,16 +393,17 @@ def ensure_media_tools_on_path() -> list[str]:
added: list[str] = []
try:
directories: list[str] = []
for resolve in (find_ffmpeg, find_ffprobe):
for tool, resolve in (("ffmpeg", find_ffmpeg), ("ffprobe", find_ffprobe)):
try:
path = resolve()
except Exception:
continue
if not path:
continue
directory = os.path.dirname(os.path.abspath(path))
if directory and directory not in directories:
directories.append(directory)
shim = _bare_name_shim(path, tool)
for directory in (shim, os.path.dirname(os.path.abspath(path))):
if directory and directory not in directories:
directories.append(directory)
current = os.environ.get("PATH", "")
entries = current.split(os.pathsep) if current else []
+9 -1
View File
@@ -481,6 +481,14 @@ If a running install ever reports "Media engine unavailable":
`sudo apt install ffmpeg`, `winget install ffmpeg`) also works — press
**Use system copy** afterwards.
Transcription tells these apart from other failures. If a dub transcription
reports that it needs ffmpeg, follow the steps above; VoiceStudio also exposes
its resolved FFmpeg under the bare name `ffmpeg` to the engines it launches, so
a library that runs plain `ffmpeg` no longer fails with `[Errno 2] No such file
or directory: 'ffmpeg'`. A "lost its output pipe" (`[Errno 32] Broken pipe`)
reply means the app that launched the backend closed or relaunched; restart
VoiceStudio.
The same panel updates **yt-dlp** (video imports): site support changes
faster than app releases, so when video-URL imports start failing, press
**Update** there — the new version survives app updates, and **Restore tested
@@ -1247,7 +1255,7 @@ VoiceStudio pins pedalboard to `>=0.9.14,<0.9.21` while [upstream portable-wheel
### TorchCodec unavailable
When torchaudio requires an unavailable TorchCodec installation, VoiceStudio writes through soundfile and reads reference audio through its FFmpeg fallback. Reference amplitude is normalized using the decoded sample representation, including 8-, 24-, and 32-bit PCM.
When torchaudio requires an unavailable TorchCodec installation, VoiceStudio writes through soundfile and reads audio (reference clips, dub segments, cached-segment headers) through soundfile or its FFmpeg fallback, so dub assembly works on torchaudio 2.9 without TorchCodec. Reference amplitude is normalized using the decoded sample representation, including 8-, 24-, and 32-bit PCM.
### Isolated engine timeouts
+145
View File
@@ -0,0 +1,145 @@
"""A missing ffmpeg / closed stdio pipe must reach the user as its own fix (#2404, #2405).
Per-chunk ASR failures are deliberately reduced to fixed public messages (no
raw exception text may leave the process). That floor used to be the generic
"Check the selected ASR engine" reply for *every* class, so a Mac without a
resolvable ffmpeg (``[Errno 2] ... 'ffmpeg'``) or an orphaned backend whose
stdio pipe closed (``[Errno 32] Broken pipe``) got no next step. The exception
is now classified privately and mapped to a VoiceStudio-owned, actionable
message per class.
"""
from __future__ import annotations
import errno
import struct
import wave
import pytest
pytestmark = pytest.mark.usefixtures("asr_model_installed")
def _cases():
from services.ffmpeg_utils import MediaToolUnavailableError
return [
(FileNotFoundError(errno.ENOENT, "No such file or directory", "ffmpeg"),
"transcription_media_tool"),
(RuntimeError("[Errno 2] No such file or directory: 'ffmpeg'"),
"transcription_media_tool"),
(MediaToolUnavailableError("Cannot transcribe: ffmpeg is missing"),
"transcription_media_tool"),
(BrokenPipeError(errno.EPIPE, "Broken pipe"), "transcription_pipe_lost"),
(RuntimeError("TOKEN=chunk-secret /home/alice/a.wav"), "transcription_failed"),
# A missing INPUT file is not a missing media engine.
(FileNotFoundError(errno.ENOENT, "No such file or directory", "/tmp/in.wav"),
"transcription_failed"),
]
def test_failure_code_follows_the_exception_chain():
from core.public_errors import transcription_failure_code
try:
try:
raise BrokenPipeError(errno.EPIPE, "Broken pipe")
except OSError as inner:
raise RuntimeError("decode failed") from inner
except RuntimeError as outer:
assert transcription_failure_code(outer) == "transcription_pipe_lost"
for exc, code in _cases():
assert transcription_failure_code(exc) == code, repr(exc)
@pytest.mark.parametrize("index", range(6))
def test_chunk_failure_class_reaches_the_user(tmp_path, monkeypatch, index):
import asyncio
from api.routers import dub_core as dc
from core.public_errors import stream_failure
exc, code = _cases()[index]
job_id = f"t_chunk_class_{index}"
audio = tmp_path / "a.wav"
with wave.open(str(audio), "wb") as wf:
wf.setnchannels(1)
wf.setsampwidth(2)
wf.setframerate(16000)
wf.writeframes(struct.pack("<16000h", *([0] * 16000)))
dc._dub_jobs[job_id] = {
"audio_path": str(audio), "vocals_path": None, "scene_cuts": [],
}
class _ASR:
id = "fake"
def ensure_loaded(self):
pass
def transcribe(self, path, *, word_timestamps=True):
raise exc
def unload(self):
pass
monkeypatch.setattr(dc, "_CHUNK_TRANSCRIBE_ATTEMPTS", 1)
monkeypatch.setattr(dc, "offload_tts_for_asr", lambda *a, **k: None)
monkeypatch.setattr(
"services.asr_backend.get_active_asr_backend", lambda *a, **k: _ASR()
)
async def _collect():
resp = await dc.dub_transcribe_stream(job_id)
parts = []
async for chunk in resp.body_iterator:
parts.append(chunk.decode() if isinstance(chunk, (bytes, bytearray)) else str(chunk))
return "".join(parts)
try:
body = asyncio.run(_collect())
finally:
dc._dub_jobs.pop(job_id, None)
assert f'"code": "{code}"' in body, body
assert stream_failure(code)["detail"] in body
assert "chunk-secret" not in body and "/home/alice" not in body
def test_imageio_style_binary_is_reachable_by_bare_name(tmp_path, monkeypatch):
"""Dependencies that exec a literal ``ffmpeg`` must find a renamed binary.
imageio-ffmpeg names its build ``ffmpeg-<platform>-vN``; publishing only
that directory on PATH never satisfied ``subprocess.run(["ffmpeg", ...])``
(parakeet-mlx, openai-whisper, pydub), the ``[Errno 2] ... 'ffmpeg'`` of
#2404.
"""
import os
import shutil
import stat
import sys
import core.config as cfg
from services import ffmpeg_utils
if sys.platform == "win32":
pytest.skip("POSIX shell stub; the shim logic is identical on Windows")
real_dir = tmp_path / "imageio"
real_dir.mkdir()
real = real_dir / "ffmpeg-linux-x86_64-v7.1"
real.write_text("#!/bin/sh\nexit 0\n")
real.chmod(real.stat().st_mode | stat.S_IEXEC)
monkeypatch.setattr(cfg, "DATA_DIR", str(tmp_path / "data"))
monkeypatch.setattr(ffmpeg_utils, "find_ffmpeg", lambda: str(real))
monkeypatch.setattr(ffmpeg_utils, "find_ffprobe", lambda: None)
monkeypatch.setenv("PATH", "/nonexistent")
added = ffmpeg_utils.ensure_media_tools_on_path()
assert added
found = shutil.which("ffmpeg")
assert found and os.path.samefile(found, real)
# Idempotent: a second call must not break or duplicate the shim.
monkeypatch.setenv("PATH", "/nonexistent")
ffmpeg_utils.ensure_media_tools_on_path()
assert os.path.samefile(shutil.which("ffmpeg"), real)