From 1a30c3abb07cae4ad13f3ea51139856a51ee2589 Mon Sep 17 00:00:00 2001 From: Palash Debnath <4178343+debpalash@users.noreply.github.com> Date: Thu, 1 Oct 2026 18:52:07 +0530 Subject: [PATCH] fix(dub): actionable transcription errors for missing ffmpeg and broken pipe (#2404, #2405) Classify private per-chunk ASR exceptions into fixed public replies, raise a typed MediaToolUnavailableError from the validated decoder, and expose the resolved ffmpeg under the bare name so dependencies that exec 'ffmpeg' find imageio-ffmpeg's renamed binary on every platform. Co-Authored-By: Claude Sonnet 5.5 --- backend/api/routers/dub_core.py | 17 +-- backend/core/public_errors.py | 68 +++++++++ backend/services/asr_backend.py | 6 +- backend/services/ffmpeg_utils.py | 61 +++++++- docs/install/troubleshooting.md | 10 +- tests/test_transcribe_media_tool_failure.py | 145 ++++++++++++++++++++ 6 files changed, 291 insertions(+), 16 deletions(-) create mode 100644 tests/test_transcribe_media_tool_failure.py diff --git a/backend/api/routers/dub_core.py b/backend/api/routers/dub_core.py index ec646028d..14fc150a0 100644 --- a/backend/api/routers/dub_core.py +++ b/backend/api/routers/dub_core.py @@ -1574,17 +1574,18 @@ async def dub_transcribe_stream( return {"chunks": shifted, "language": r.get("language"), "speaker_turns": turns} except Exception as exc: # Keep diagnostics local and fixed-shape. In particular, - # CUDA OOM is a distinct, actionable recovery class rather - # than the generic "no segments" dead end. + # CUDA OOM, a missing ffmpeg and a closed stdio pipe are + # distinct, actionable recovery classes rather than the + # generic "no segments" dead end. is_memory = isinstance(exc, torch.OutOfMemoryError) logger.error( "Chunk transcription failed (backend=%s; class=%s; details withheld)", _asr_backend.id, type(exc).__name__, ) - from core.public_errors import stream_failure + from core.public_errors import stream_failure, transcription_failure_code failure = stream_failure( - "transcription_memory" if is_memory else "transcription_failed" + "transcription_memory" if is_memory else transcription_failure_code(exc) ) return { "chunks": [], @@ -2251,10 +2252,10 @@ async def dub_transcribe_stream( try: async for ev in _gen_body(): yield ev - except Exception: # noqa: BLE001 — last-resort stream finalizer - logger.error("Transcription stream failed unexpectedly") - from core.public_errors import stream_failure - yield _sse_event("error", stream_failure("transcription_failed")) + except Exception as exc: # noqa: BLE001 — last-resort stream finalizer + logger.error("Transcription stream failed unexpectedly (class=%s)", type(exc).__name__) + from core.public_errors import stream_failure, transcription_failure_code + yield _sse_event("error", stream_failure(transcription_failure_code(exc))) yield _sse_event("done", {}) finally: _asr_work.stop() diff --git a/backend/core/public_errors.py b/backend/core/public_errors.py index b987d01ca..1d608a0d4 100644 --- a/backend/core/public_errors.py +++ b/backend/core/public_errors.py @@ -2,6 +2,7 @@ from __future__ import annotations import logging +import os from typing import Any _PROVIDER_DETAILS = { @@ -70,10 +71,77 @@ def stream_failure(code: str) -> dict[str, object]: ), "retryable": True, }, + "transcription_media_tool": { + "code": "transcription_media_tool", + "detail": ( + "Transcription needs ffmpeg, which VoiceStudio could not find or " + "run. Open Settings → Audio tools and use Download/Repair, or " + "install ffmpeg (macOS: brew install ffmpeg; Windows: winget " + "install Gyan.FFmpeg; Linux: your package manager) and restart " + "VoiceStudio. You can also point FFMPEG_PATH at an ffmpeg binary." + ), + "retryable": True, + }, + "transcription_pipe_lost": { + "code": "transcription_pipe_lost", + "detail": ( + "The backend lost its output pipe to the app while transcribing " + "(broken pipe). Restart VoiceStudio and try again." + ), + "retryable": True, + }, } return dict(failures.get(code, failures["generation_failed"])) +def _exception_chain(error: object, limit: int = 8): + """The exception, then its ``__cause__``/``__context__`` links (bounded).""" + seen: set[int] = set() + while isinstance(error, BaseException) and id(error) not in seen and len(seen) < limit: + seen.add(id(error)) + yield error + error = error.__cause__ or error.__context__ + + +def transcription_failure_code(error: object) -> str: + """Stable ``stream_failure`` code for a private ASR chunk exception. + + Only the exception's type, errno and filename are inspected locally; the + caller ships a VoiceStudio-owned message for the returned code, never the + exception text. A missing/unrunnable ffmpeg (``[Errno 2] ... 'ffmpeg'``, + ``[WinError 193]``, :class:`MediaToolUnavailableError`) and a closed stdio + pipe (``[Errno 32] Broken pipe``) each have a different remedy, so they must + not collapse into the generic "check the selected ASR engine" reply (#2404, + #2405). + """ + import errno + + from services.ffmpeg_utils import MediaToolUnavailableError + + pipe = False + for exc in _exception_chain(error): + if isinstance(exc, MediaToolUnavailableError): + return "transcription_media_tool" + if isinstance(exc, OSError): + name = os.path.basename(str(exc.filename or "")).lower() + if name.startswith(("ffmpeg", "ffprobe")) and ( + exc.errno in (errno.ENOENT, errno.ENOEXEC, errno.EACCES) + or getattr(exc, "winerror", None) in (2, 193) + ): + return "transcription_media_tool" + low = str(exc).lower() + try: + from core.failure import _is_missing_media_tool + + if _is_missing_media_tool(low): + return "transcription_media_tool" + except Exception: # noqa: BLE001 — classification must never raise + pass + if isinstance(exc, BrokenPipeError) or "broken pipe" in low or "[errno 32]" in low: + pipe = True + return "transcription_pipe_lost" if pipe else "transcription_failed" + + def stream_generation_failure(error: BaseException | object) -> dict[str, object]: """``generation_failed`` stream metadata, enriched with the actual cause. diff --git a/backend/services/asr_backend.py b/backend/services/asr_backend.py index 29ca1dfab..d4ed858ad 100644 --- a/backend/services/asr_backend.py +++ b/backend/services/asr_backend.py @@ -333,11 +333,11 @@ def _decode_audio_16k_mono(audio_path: str): import numpy as np - from services.ffmpeg_utils import find_ffmpeg + from services.ffmpeg_utils import MediaToolUnavailableError, find_ffmpeg ffmpeg = find_ffmpeg() if not ffmpeg: - raise RuntimeError( + raise MediaToolUnavailableError( "Cannot transcribe: ffmpeg is missing or not runnable. Install " "ffmpeg (or let VoiceStudio's bundled binary download), then retry. " "On Windows a '[WinError 193]' here means the ffmpeg binary is " @@ -354,7 +354,7 @@ def _decode_audio_16k_mono(audio_path: str): # Belt-and-suspenders: find_ffmpeg() already -version-validated this # binary, so a WinError 193 here is unexpected — surface it clearly # rather than letting it become "no segments". - raise RuntimeError( + raise MediaToolUnavailableError( f"ffmpeg at {ffmpeg!r} could not be executed ({e}). Reinstall " "ffmpeg or clear the imageio-ffmpeg cache." ) from e diff --git a/backend/services/ffmpeg_utils.py b/backend/services/ffmpeg_utils.py index 9c8764af9..b500aafc7 100644 --- a/backend/services/ffmpeg_utils.py +++ b/backend/services/ffmpeg_utils.py @@ -190,6 +190,15 @@ def _binary_runs(path: str) -> bool: return ok +class MediaToolUnavailableError(RuntimeError): + """ffmpeg/ffprobe is missing, or present but not runnable. + + A ``RuntimeError`` so existing handlers keep working; the dedicated type + lets the transcription path pick the actionable "repair the media engine" + reply by class instead of by sniffing message text. + """ + + def find_ffmpeg(): """Locate an ffmpeg binary. @@ -320,6 +329,49 @@ def find_ffprobe(): return None +def _bare_name_shim(real: str, tool: str) -> "str | None": + """Directory exposing *real* under the bare name ``[.exe]``, or None. + + imageio-ffmpeg ships its binary as ``ffmpeg--vN[.exe]``, so + publishing its directory on ``PATH`` never satisfies a dependency's literal + ``ffmpeg`` lookup (parakeet-mlx, openai-whisper, pydub, ...): they still die + with ``[Errno 2] No such file or directory: 'ffmpeg'``. A symlink (hardlink + or copy where Windows refuses symlinks) under the bare name closes that gap + on every platform without asking the user to install anything. Best-effort: + returns None when the name is already bare or the shim cannot be written. + """ + exe = f"{tool}.exe" if os.name == "nt" else tool + if os.path.basename(real).lower() == exe: + return None + try: + from core.config import DATA_DIR + + directory = os.path.join(DATA_DIR, "media_tools", "shims") + link = os.path.join(directory, exe) + os.makedirs(directory, exist_ok=True) + if os.path.lexists(link): + try: + if os.path.samefile(link, real) or ( + not os.path.islink(link) + and os.path.getsize(link) == os.path.getsize(real) + ): + return directory + except OSError: + pass + os.unlink(link) + try: + os.symlink(real, link) + except (OSError, NotImplementedError): + try: + os.link(real, link) + except OSError: + shutil.copy2(real, link) + return directory + except OSError as e: + logger.debug("bare-name %s shim unavailable: %s", tool, e) + return None + + def ensure_media_tools_on_path() -> list[str]: """Put the resolved ffmpeg/ffprobe on ``PATH`` for third-party code (#1256). @@ -341,16 +393,17 @@ def ensure_media_tools_on_path() -> list[str]: added: list[str] = [] try: directories: list[str] = [] - for resolve in (find_ffmpeg, find_ffprobe): + for tool, resolve in (("ffmpeg", find_ffmpeg), ("ffprobe", find_ffprobe)): try: path = resolve() except Exception: continue if not path: continue - directory = os.path.dirname(os.path.abspath(path)) - if directory and directory not in directories: - directories.append(directory) + shim = _bare_name_shim(path, tool) + for directory in (shim, os.path.dirname(os.path.abspath(path))): + if directory and directory not in directories: + directories.append(directory) current = os.environ.get("PATH", "") entries = current.split(os.pathsep) if current else [] diff --git a/docs/install/troubleshooting.md b/docs/install/troubleshooting.md index 1bc1ef012..cb4477291 100644 --- a/docs/install/troubleshooting.md +++ b/docs/install/troubleshooting.md @@ -481,6 +481,14 @@ If a running install ever reports "Media engine unavailable": `sudo apt install ffmpeg`, `winget install ffmpeg`) also works — press **Use system copy** afterwards. +Transcription tells these apart from other failures. If a dub transcription +reports that it needs ffmpeg, follow the steps above; VoiceStudio also exposes +its resolved FFmpeg under the bare name `ffmpeg` to the engines it launches, so +a library that runs plain `ffmpeg` no longer fails with `[Errno 2] No such file +or directory: 'ffmpeg'`. A "lost its output pipe" (`[Errno 32] Broken pipe`) +reply means the app that launched the backend closed or relaunched; restart +VoiceStudio. + The same panel updates **yt-dlp** (video imports): site support changes faster than app releases, so when video-URL imports start failing, press **Update** there — the new version survives app updates, and **Restore tested @@ -1247,7 +1255,7 @@ VoiceStudio pins pedalboard to `>=0.9.14,<0.9.21` while [upstream portable-wheel ### TorchCodec unavailable -When torchaudio requires an unavailable TorchCodec installation, VoiceStudio writes through soundfile and reads reference audio through its FFmpeg fallback. Reference amplitude is normalized using the decoded sample representation, including 8-, 24-, and 32-bit PCM. +When torchaudio requires an unavailable TorchCodec installation, VoiceStudio writes through soundfile and reads audio (reference clips, dub segments, cached-segment headers) through soundfile or its FFmpeg fallback, so dub assembly works on torchaudio 2.9 without TorchCodec. Reference amplitude is normalized using the decoded sample representation, including 8-, 24-, and 32-bit PCM. ### Isolated engine timeouts diff --git a/tests/test_transcribe_media_tool_failure.py b/tests/test_transcribe_media_tool_failure.py new file mode 100644 index 000000000..2b4153edf --- /dev/null +++ b/tests/test_transcribe_media_tool_failure.py @@ -0,0 +1,145 @@ +"""A missing ffmpeg / closed stdio pipe must reach the user as its own fix (#2404, #2405). + +Per-chunk ASR failures are deliberately reduced to fixed public messages (no +raw exception text may leave the process). That floor used to be the generic +"Check the selected ASR engine" reply for *every* class, so a Mac without a +resolvable ffmpeg (``[Errno 2] ... 'ffmpeg'``) or an orphaned backend whose +stdio pipe closed (``[Errno 32] Broken pipe``) got no next step. The exception +is now classified privately and mapped to a VoiceStudio-owned, actionable +message per class. +""" +from __future__ import annotations + +import errno +import struct +import wave + +import pytest + +pytestmark = pytest.mark.usefixtures("asr_model_installed") + + +def _cases(): + from services.ffmpeg_utils import MediaToolUnavailableError + + return [ + (FileNotFoundError(errno.ENOENT, "No such file or directory", "ffmpeg"), + "transcription_media_tool"), + (RuntimeError("[Errno 2] No such file or directory: 'ffmpeg'"), + "transcription_media_tool"), + (MediaToolUnavailableError("Cannot transcribe: ffmpeg is missing"), + "transcription_media_tool"), + (BrokenPipeError(errno.EPIPE, "Broken pipe"), "transcription_pipe_lost"), + (RuntimeError("TOKEN=chunk-secret /home/alice/a.wav"), "transcription_failed"), + # A missing INPUT file is not a missing media engine. + (FileNotFoundError(errno.ENOENT, "No such file or directory", "/tmp/in.wav"), + "transcription_failed"), + ] + + +def test_failure_code_follows_the_exception_chain(): + from core.public_errors import transcription_failure_code + + try: + try: + raise BrokenPipeError(errno.EPIPE, "Broken pipe") + except OSError as inner: + raise RuntimeError("decode failed") from inner + except RuntimeError as outer: + assert transcription_failure_code(outer) == "transcription_pipe_lost" + for exc, code in _cases(): + assert transcription_failure_code(exc) == code, repr(exc) + + +@pytest.mark.parametrize("index", range(6)) +def test_chunk_failure_class_reaches_the_user(tmp_path, monkeypatch, index): + import asyncio + + from api.routers import dub_core as dc + from core.public_errors import stream_failure + + exc, code = _cases()[index] + job_id = f"t_chunk_class_{index}" + audio = tmp_path / "a.wav" + with wave.open(str(audio), "wb") as wf: + wf.setnchannels(1) + wf.setsampwidth(2) + wf.setframerate(16000) + wf.writeframes(struct.pack("<16000h", *([0] * 16000))) + dc._dub_jobs[job_id] = { + "audio_path": str(audio), "vocals_path": None, "scene_cuts": [], + } + + class _ASR: + id = "fake" + + def ensure_loaded(self): + pass + + def transcribe(self, path, *, word_timestamps=True): + raise exc + + def unload(self): + pass + + monkeypatch.setattr(dc, "_CHUNK_TRANSCRIBE_ATTEMPTS", 1) + monkeypatch.setattr(dc, "offload_tts_for_asr", lambda *a, **k: None) + monkeypatch.setattr( + "services.asr_backend.get_active_asr_backend", lambda *a, **k: _ASR() + ) + + async def _collect(): + resp = await dc.dub_transcribe_stream(job_id) + parts = [] + async for chunk in resp.body_iterator: + parts.append(chunk.decode() if isinstance(chunk, (bytes, bytearray)) else str(chunk)) + return "".join(parts) + + try: + body = asyncio.run(_collect()) + finally: + dc._dub_jobs.pop(job_id, None) + + assert f'"code": "{code}"' in body, body + assert stream_failure(code)["detail"] in body + assert "chunk-secret" not in body and "/home/alice" not in body + + +def test_imageio_style_binary_is_reachable_by_bare_name(tmp_path, monkeypatch): + """Dependencies that exec a literal ``ffmpeg`` must find a renamed binary. + + imageio-ffmpeg names its build ``ffmpeg--vN``; publishing only + that directory on PATH never satisfied ``subprocess.run(["ffmpeg", ...])`` + (parakeet-mlx, openai-whisper, pydub), the ``[Errno 2] ... 'ffmpeg'`` of + #2404. + """ + import os + import shutil + import stat + import sys + + import core.config as cfg + from services import ffmpeg_utils + + if sys.platform == "win32": + pytest.skip("POSIX shell stub; the shim logic is identical on Windows") + real_dir = tmp_path / "imageio" + real_dir.mkdir() + real = real_dir / "ffmpeg-linux-x86_64-v7.1" + real.write_text("#!/bin/sh\nexit 0\n") + real.chmod(real.stat().st_mode | stat.S_IEXEC) + + monkeypatch.setattr(cfg, "DATA_DIR", str(tmp_path / "data")) + monkeypatch.setattr(ffmpeg_utils, "find_ffmpeg", lambda: str(real)) + monkeypatch.setattr(ffmpeg_utils, "find_ffprobe", lambda: None) + monkeypatch.setenv("PATH", "/nonexistent") + + added = ffmpeg_utils.ensure_media_tools_on_path() + + assert added + found = shutil.which("ffmpeg") + assert found and os.path.samefile(found, real) + # Idempotent: a second call must not break or duplicate the shim. + monkeypatch.setenv("PATH", "/nonexistent") + ffmpeg_utils.ensure_media_tools_on_path() + assert os.path.samefile(shutil.which("ffmpeg"), real)