mirror of
https://github.com/debpalash/VoiceStudio.git
synced 2026-10-02 01:26:35 +08:00
Classify private per-chunk ASR exceptions into fixed public replies, raise a typed MediaToolUnavailableError from the validated decoder, and expose the resolved ffmpeg under the bare name so dependencies that exec 'ffmpeg' find imageio-ffmpeg's renamed binary on every platform. Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 5.5
parent
e240580ccc
commit
1a30c3abb0
@@ -1574,17 +1574,18 @@ async def dub_transcribe_stream(
|
||||
return {"chunks": shifted, "language": r.get("language"), "speaker_turns": turns}
|
||||
except Exception as exc:
|
||||
# Keep diagnostics local and fixed-shape. In particular,
|
||||
# CUDA OOM is a distinct, actionable recovery class rather
|
||||
# than the generic "no segments" dead end.
|
||||
# CUDA OOM, a missing ffmpeg and a closed stdio pipe are
|
||||
# distinct, actionable recovery classes rather than the
|
||||
# generic "no segments" dead end.
|
||||
is_memory = isinstance(exc, torch.OutOfMemoryError)
|
||||
logger.error(
|
||||
"Chunk transcription failed (backend=%s; class=%s; details withheld)",
|
||||
_asr_backend.id,
|
||||
type(exc).__name__,
|
||||
)
|
||||
from core.public_errors import stream_failure
|
||||
from core.public_errors import stream_failure, transcription_failure_code
|
||||
failure = stream_failure(
|
||||
"transcription_memory" if is_memory else "transcription_failed"
|
||||
"transcription_memory" if is_memory else transcription_failure_code(exc)
|
||||
)
|
||||
return {
|
||||
"chunks": [],
|
||||
@@ -2251,10 +2252,10 @@ async def dub_transcribe_stream(
|
||||
try:
|
||||
async for ev in _gen_body():
|
||||
yield ev
|
||||
except Exception: # noqa: BLE001 — last-resort stream finalizer
|
||||
logger.error("Transcription stream failed unexpectedly")
|
||||
from core.public_errors import stream_failure
|
||||
yield _sse_event("error", stream_failure("transcription_failed"))
|
||||
except Exception as exc: # noqa: BLE001 — last-resort stream finalizer
|
||||
logger.error("Transcription stream failed unexpectedly (class=%s)", type(exc).__name__)
|
||||
from core.public_errors import stream_failure, transcription_failure_code
|
||||
yield _sse_event("error", stream_failure(transcription_failure_code(exc)))
|
||||
yield _sse_event("done", {})
|
||||
finally:
|
||||
_asr_work.stop()
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
from typing import Any
|
||||
|
||||
_PROVIDER_DETAILS = {
|
||||
@@ -70,10 +71,77 @@ def stream_failure(code: str) -> dict[str, object]:
|
||||
),
|
||||
"retryable": True,
|
||||
},
|
||||
"transcription_media_tool": {
|
||||
"code": "transcription_media_tool",
|
||||
"detail": (
|
||||
"Transcription needs ffmpeg, which VoiceStudio could not find or "
|
||||
"run. Open Settings → Audio tools and use Download/Repair, or "
|
||||
"install ffmpeg (macOS: brew install ffmpeg; Windows: winget "
|
||||
"install Gyan.FFmpeg; Linux: your package manager) and restart "
|
||||
"VoiceStudio. You can also point FFMPEG_PATH at an ffmpeg binary."
|
||||
),
|
||||
"retryable": True,
|
||||
},
|
||||
"transcription_pipe_lost": {
|
||||
"code": "transcription_pipe_lost",
|
||||
"detail": (
|
||||
"The backend lost its output pipe to the app while transcribing "
|
||||
"(broken pipe). Restart VoiceStudio and try again."
|
||||
),
|
||||
"retryable": True,
|
||||
},
|
||||
}
|
||||
return dict(failures.get(code, failures["generation_failed"]))
|
||||
|
||||
|
||||
def _exception_chain(error: object, limit: int = 8):
|
||||
"""The exception, then its ``__cause__``/``__context__`` links (bounded)."""
|
||||
seen: set[int] = set()
|
||||
while isinstance(error, BaseException) and id(error) not in seen and len(seen) < limit:
|
||||
seen.add(id(error))
|
||||
yield error
|
||||
error = error.__cause__ or error.__context__
|
||||
|
||||
|
||||
def transcription_failure_code(error: object) -> str:
|
||||
"""Stable ``stream_failure`` code for a private ASR chunk exception.
|
||||
|
||||
Only the exception's type, errno and filename are inspected locally; the
|
||||
caller ships a VoiceStudio-owned message for the returned code, never the
|
||||
exception text. A missing/unrunnable ffmpeg (``[Errno 2] ... 'ffmpeg'``,
|
||||
``[WinError 193]``, :class:`MediaToolUnavailableError`) and a closed stdio
|
||||
pipe (``[Errno 32] Broken pipe``) each have a different remedy, so they must
|
||||
not collapse into the generic "check the selected ASR engine" reply (#2404,
|
||||
#2405).
|
||||
"""
|
||||
import errno
|
||||
|
||||
from services.ffmpeg_utils import MediaToolUnavailableError
|
||||
|
||||
pipe = False
|
||||
for exc in _exception_chain(error):
|
||||
if isinstance(exc, MediaToolUnavailableError):
|
||||
return "transcription_media_tool"
|
||||
if isinstance(exc, OSError):
|
||||
name = os.path.basename(str(exc.filename or "")).lower()
|
||||
if name.startswith(("ffmpeg", "ffprobe")) and (
|
||||
exc.errno in (errno.ENOENT, errno.ENOEXEC, errno.EACCES)
|
||||
or getattr(exc, "winerror", None) in (2, 193)
|
||||
):
|
||||
return "transcription_media_tool"
|
||||
low = str(exc).lower()
|
||||
try:
|
||||
from core.failure import _is_missing_media_tool
|
||||
|
||||
if _is_missing_media_tool(low):
|
||||
return "transcription_media_tool"
|
||||
except Exception: # noqa: BLE001 — classification must never raise
|
||||
pass
|
||||
if isinstance(exc, BrokenPipeError) or "broken pipe" in low or "[errno 32]" in low:
|
||||
pipe = True
|
||||
return "transcription_pipe_lost" if pipe else "transcription_failed"
|
||||
|
||||
|
||||
def stream_generation_failure(error: BaseException | object) -> dict[str, object]:
|
||||
"""``generation_failed`` stream metadata, enriched with the actual cause.
|
||||
|
||||
|
||||
@@ -333,11 +333,11 @@ def _decode_audio_16k_mono(audio_path: str):
|
||||
|
||||
import numpy as np
|
||||
|
||||
from services.ffmpeg_utils import find_ffmpeg
|
||||
from services.ffmpeg_utils import MediaToolUnavailableError, find_ffmpeg
|
||||
|
||||
ffmpeg = find_ffmpeg()
|
||||
if not ffmpeg:
|
||||
raise RuntimeError(
|
||||
raise MediaToolUnavailableError(
|
||||
"Cannot transcribe: ffmpeg is missing or not runnable. Install "
|
||||
"ffmpeg (or let VoiceStudio's bundled binary download), then retry. "
|
||||
"On Windows a '[WinError 193]' here means the ffmpeg binary is "
|
||||
@@ -354,7 +354,7 @@ def _decode_audio_16k_mono(audio_path: str):
|
||||
# Belt-and-suspenders: find_ffmpeg() already -version-validated this
|
||||
# binary, so a WinError 193 here is unexpected — surface it clearly
|
||||
# rather than letting it become "no segments".
|
||||
raise RuntimeError(
|
||||
raise MediaToolUnavailableError(
|
||||
f"ffmpeg at {ffmpeg!r} could not be executed ({e}). Reinstall "
|
||||
"ffmpeg or clear the imageio-ffmpeg cache."
|
||||
) from e
|
||||
|
||||
@@ -190,6 +190,15 @@ def _binary_runs(path: str) -> bool:
|
||||
return ok
|
||||
|
||||
|
||||
class MediaToolUnavailableError(RuntimeError):
|
||||
"""ffmpeg/ffprobe is missing, or present but not runnable.
|
||||
|
||||
A ``RuntimeError`` so existing handlers keep working; the dedicated type
|
||||
lets the transcription path pick the actionable "repair the media engine"
|
||||
reply by class instead of by sniffing message text.
|
||||
"""
|
||||
|
||||
|
||||
def find_ffmpeg():
|
||||
"""Locate an ffmpeg binary.
|
||||
|
||||
@@ -320,6 +329,49 @@ def find_ffprobe():
|
||||
return None
|
||||
|
||||
|
||||
def _bare_name_shim(real: str, tool: str) -> "str | None":
|
||||
"""Directory exposing *real* under the bare name ``<tool>[.exe]``, or None.
|
||||
|
||||
imageio-ffmpeg ships its binary as ``ffmpeg-<platform>-vN[.exe]``, so
|
||||
publishing its directory on ``PATH`` never satisfies a dependency's literal
|
||||
``ffmpeg`` lookup (parakeet-mlx, openai-whisper, pydub, ...): they still die
|
||||
with ``[Errno 2] No such file or directory: 'ffmpeg'``. A symlink (hardlink
|
||||
or copy where Windows refuses symlinks) under the bare name closes that gap
|
||||
on every platform without asking the user to install anything. Best-effort:
|
||||
returns None when the name is already bare or the shim cannot be written.
|
||||
"""
|
||||
exe = f"{tool}.exe" if os.name == "nt" else tool
|
||||
if os.path.basename(real).lower() == exe:
|
||||
return None
|
||||
try:
|
||||
from core.config import DATA_DIR
|
||||
|
||||
directory = os.path.join(DATA_DIR, "media_tools", "shims")
|
||||
link = os.path.join(directory, exe)
|
||||
os.makedirs(directory, exist_ok=True)
|
||||
if os.path.lexists(link):
|
||||
try:
|
||||
if os.path.samefile(link, real) or (
|
||||
not os.path.islink(link)
|
||||
and os.path.getsize(link) == os.path.getsize(real)
|
||||
):
|
||||
return directory
|
||||
except OSError:
|
||||
pass
|
||||
os.unlink(link)
|
||||
try:
|
||||
os.symlink(real, link)
|
||||
except (OSError, NotImplementedError):
|
||||
try:
|
||||
os.link(real, link)
|
||||
except OSError:
|
||||
shutil.copy2(real, link)
|
||||
return directory
|
||||
except OSError as e:
|
||||
logger.debug("bare-name %s shim unavailable: %s", tool, e)
|
||||
return None
|
||||
|
||||
|
||||
def ensure_media_tools_on_path() -> list[str]:
|
||||
"""Put the resolved ffmpeg/ffprobe on ``PATH`` for third-party code (#1256).
|
||||
|
||||
@@ -341,16 +393,17 @@ def ensure_media_tools_on_path() -> list[str]:
|
||||
added: list[str] = []
|
||||
try:
|
||||
directories: list[str] = []
|
||||
for resolve in (find_ffmpeg, find_ffprobe):
|
||||
for tool, resolve in (("ffmpeg", find_ffmpeg), ("ffprobe", find_ffprobe)):
|
||||
try:
|
||||
path = resolve()
|
||||
except Exception:
|
||||
continue
|
||||
if not path:
|
||||
continue
|
||||
directory = os.path.dirname(os.path.abspath(path))
|
||||
if directory and directory not in directories:
|
||||
directories.append(directory)
|
||||
shim = _bare_name_shim(path, tool)
|
||||
for directory in (shim, os.path.dirname(os.path.abspath(path))):
|
||||
if directory and directory not in directories:
|
||||
directories.append(directory)
|
||||
|
||||
current = os.environ.get("PATH", "")
|
||||
entries = current.split(os.pathsep) if current else []
|
||||
|
||||
@@ -481,6 +481,14 @@ If a running install ever reports "Media engine unavailable":
|
||||
`sudo apt install ffmpeg`, `winget install ffmpeg`) also works — press
|
||||
**Use system copy** afterwards.
|
||||
|
||||
Transcription tells these apart from other failures. If a dub transcription
|
||||
reports that it needs ffmpeg, follow the steps above; VoiceStudio also exposes
|
||||
its resolved FFmpeg under the bare name `ffmpeg` to the engines it launches, so
|
||||
a library that runs plain `ffmpeg` no longer fails with `[Errno 2] No such file
|
||||
or directory: 'ffmpeg'`. A "lost its output pipe" (`[Errno 32] Broken pipe`)
|
||||
reply means the app that launched the backend closed or relaunched; restart
|
||||
VoiceStudio.
|
||||
|
||||
The same panel updates **yt-dlp** (video imports): site support changes
|
||||
faster than app releases, so when video-URL imports start failing, press
|
||||
**Update** there — the new version survives app updates, and **Restore tested
|
||||
@@ -1247,7 +1255,7 @@ VoiceStudio pins pedalboard to `>=0.9.14,<0.9.21` while [upstream portable-wheel
|
||||
|
||||
### TorchCodec unavailable
|
||||
|
||||
When torchaudio requires an unavailable TorchCodec installation, VoiceStudio writes through soundfile and reads reference audio through its FFmpeg fallback. Reference amplitude is normalized using the decoded sample representation, including 8-, 24-, and 32-bit PCM.
|
||||
When torchaudio requires an unavailable TorchCodec installation, VoiceStudio writes through soundfile and reads audio (reference clips, dub segments, cached-segment headers) through soundfile or its FFmpeg fallback, so dub assembly works on torchaudio 2.9 without TorchCodec. Reference amplitude is normalized using the decoded sample representation, including 8-, 24-, and 32-bit PCM.
|
||||
|
||||
### Isolated engine timeouts
|
||||
|
||||
|
||||
@@ -0,0 +1,145 @@
|
||||
"""A missing ffmpeg / closed stdio pipe must reach the user as its own fix (#2404, #2405).
|
||||
|
||||
Per-chunk ASR failures are deliberately reduced to fixed public messages (no
|
||||
raw exception text may leave the process). That floor used to be the generic
|
||||
"Check the selected ASR engine" reply for *every* class, so a Mac without a
|
||||
resolvable ffmpeg (``[Errno 2] ... 'ffmpeg'``) or an orphaned backend whose
|
||||
stdio pipe closed (``[Errno 32] Broken pipe``) got no next step. The exception
|
||||
is now classified privately and mapped to a VoiceStudio-owned, actionable
|
||||
message per class.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import errno
|
||||
import struct
|
||||
import wave
|
||||
|
||||
import pytest
|
||||
|
||||
pytestmark = pytest.mark.usefixtures("asr_model_installed")
|
||||
|
||||
|
||||
def _cases():
|
||||
from services.ffmpeg_utils import MediaToolUnavailableError
|
||||
|
||||
return [
|
||||
(FileNotFoundError(errno.ENOENT, "No such file or directory", "ffmpeg"),
|
||||
"transcription_media_tool"),
|
||||
(RuntimeError("[Errno 2] No such file or directory: 'ffmpeg'"),
|
||||
"transcription_media_tool"),
|
||||
(MediaToolUnavailableError("Cannot transcribe: ffmpeg is missing"),
|
||||
"transcription_media_tool"),
|
||||
(BrokenPipeError(errno.EPIPE, "Broken pipe"), "transcription_pipe_lost"),
|
||||
(RuntimeError("TOKEN=chunk-secret /home/alice/a.wav"), "transcription_failed"),
|
||||
# A missing INPUT file is not a missing media engine.
|
||||
(FileNotFoundError(errno.ENOENT, "No such file or directory", "/tmp/in.wav"),
|
||||
"transcription_failed"),
|
||||
]
|
||||
|
||||
|
||||
def test_failure_code_follows_the_exception_chain():
|
||||
from core.public_errors import transcription_failure_code
|
||||
|
||||
try:
|
||||
try:
|
||||
raise BrokenPipeError(errno.EPIPE, "Broken pipe")
|
||||
except OSError as inner:
|
||||
raise RuntimeError("decode failed") from inner
|
||||
except RuntimeError as outer:
|
||||
assert transcription_failure_code(outer) == "transcription_pipe_lost"
|
||||
for exc, code in _cases():
|
||||
assert transcription_failure_code(exc) == code, repr(exc)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("index", range(6))
|
||||
def test_chunk_failure_class_reaches_the_user(tmp_path, monkeypatch, index):
|
||||
import asyncio
|
||||
|
||||
from api.routers import dub_core as dc
|
||||
from core.public_errors import stream_failure
|
||||
|
||||
exc, code = _cases()[index]
|
||||
job_id = f"t_chunk_class_{index}"
|
||||
audio = tmp_path / "a.wav"
|
||||
with wave.open(str(audio), "wb") as wf:
|
||||
wf.setnchannels(1)
|
||||
wf.setsampwidth(2)
|
||||
wf.setframerate(16000)
|
||||
wf.writeframes(struct.pack("<16000h", *([0] * 16000)))
|
||||
dc._dub_jobs[job_id] = {
|
||||
"audio_path": str(audio), "vocals_path": None, "scene_cuts": [],
|
||||
}
|
||||
|
||||
class _ASR:
|
||||
id = "fake"
|
||||
|
||||
def ensure_loaded(self):
|
||||
pass
|
||||
|
||||
def transcribe(self, path, *, word_timestamps=True):
|
||||
raise exc
|
||||
|
||||
def unload(self):
|
||||
pass
|
||||
|
||||
monkeypatch.setattr(dc, "_CHUNK_TRANSCRIBE_ATTEMPTS", 1)
|
||||
monkeypatch.setattr(dc, "offload_tts_for_asr", lambda *a, **k: None)
|
||||
monkeypatch.setattr(
|
||||
"services.asr_backend.get_active_asr_backend", lambda *a, **k: _ASR()
|
||||
)
|
||||
|
||||
async def _collect():
|
||||
resp = await dc.dub_transcribe_stream(job_id)
|
||||
parts = []
|
||||
async for chunk in resp.body_iterator:
|
||||
parts.append(chunk.decode() if isinstance(chunk, (bytes, bytearray)) else str(chunk))
|
||||
return "".join(parts)
|
||||
|
||||
try:
|
||||
body = asyncio.run(_collect())
|
||||
finally:
|
||||
dc._dub_jobs.pop(job_id, None)
|
||||
|
||||
assert f'"code": "{code}"' in body, body
|
||||
assert stream_failure(code)["detail"] in body
|
||||
assert "chunk-secret" not in body and "/home/alice" not in body
|
||||
|
||||
|
||||
def test_imageio_style_binary_is_reachable_by_bare_name(tmp_path, monkeypatch):
|
||||
"""Dependencies that exec a literal ``ffmpeg`` must find a renamed binary.
|
||||
|
||||
imageio-ffmpeg names its build ``ffmpeg-<platform>-vN``; publishing only
|
||||
that directory on PATH never satisfied ``subprocess.run(["ffmpeg", ...])``
|
||||
(parakeet-mlx, openai-whisper, pydub), the ``[Errno 2] ... 'ffmpeg'`` of
|
||||
#2404.
|
||||
"""
|
||||
import os
|
||||
import shutil
|
||||
import stat
|
||||
import sys
|
||||
|
||||
import core.config as cfg
|
||||
from services import ffmpeg_utils
|
||||
|
||||
if sys.platform == "win32":
|
||||
pytest.skip("POSIX shell stub; the shim logic is identical on Windows")
|
||||
real_dir = tmp_path / "imageio"
|
||||
real_dir.mkdir()
|
||||
real = real_dir / "ffmpeg-linux-x86_64-v7.1"
|
||||
real.write_text("#!/bin/sh\nexit 0\n")
|
||||
real.chmod(real.stat().st_mode | stat.S_IEXEC)
|
||||
|
||||
monkeypatch.setattr(cfg, "DATA_DIR", str(tmp_path / "data"))
|
||||
monkeypatch.setattr(ffmpeg_utils, "find_ffmpeg", lambda: str(real))
|
||||
monkeypatch.setattr(ffmpeg_utils, "find_ffprobe", lambda: None)
|
||||
monkeypatch.setenv("PATH", "/nonexistent")
|
||||
|
||||
added = ffmpeg_utils.ensure_media_tools_on_path()
|
||||
|
||||
assert added
|
||||
found = shutil.which("ffmpeg")
|
||||
assert found and os.path.samefile(found, real)
|
||||
# Idempotent: a second call must not break or duplicate the shim.
|
||||
monkeypatch.setenv("PATH", "/nonexistent")
|
||||
ffmpeg_utils.ensure_media_tools_on_path()
|
||||
assert os.path.samefile(shutil.which("ffmpeg"), real)
|
||||
Reference in New Issue
Block a user