mirror of
https://github.com/debpalash/VoiceStudio.git
synced 2026-10-02 09:34:38 +08:00
fix(indextts): don't claim ROCm for a sidecar venv holding a CUDA torch; docs for Windows AMD (#2371, #2468)
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 5.5
parent
c3b564a23c
commit
2bfbd6097d
@@ -31,6 +31,7 @@ import logging
|
||||
import math
|
||||
import os
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from services.subprocess_backend import SubprocessBackend
|
||||
@@ -162,6 +163,54 @@ class IndexTTS2Backend(SubprocessBackend):
|
||||
)
|
||||
return True, "ok"
|
||||
|
||||
@classmethod
|
||||
def _sidecar_torch_is_cuda_only(cls) -> bool:
|
||||
"""True only when the sidecar venv is KNOWN to hold a non-ROCm torch.
|
||||
|
||||
``gpu_compat`` claims ROCm because the one-click installer provisions a
|
||||
ROCm torch on ROCm hosts (#2371). A venv that predates that, or a
|
||||
user-managed clone, can still carry the CUDA wheel upstream resolves -
|
||||
which sees no AMD GPU, so the engine would run on the CPU while routing
|
||||
reported acceleration (the PR #2423 review finding). Unknown (venv not
|
||||
located yet, no torch wheel found) keeps the claim: this runs on every
|
||||
engine-list refresh, so it must never spawn the interpreter probe.
|
||||
"""
|
||||
try:
|
||||
from engines.indextts import bootstrap
|
||||
from services.sidecar_install import venv_torch_hip
|
||||
|
||||
python = bootstrap._resolved_python
|
||||
if python is None and os.environ.get("OMNIVOICE_INDEXTTS_DIR"):
|
||||
python = bootstrap._venv_python_path(
|
||||
Path(os.environ["OMNIVOICE_INDEXTTS_DIR"]) / ".venv"
|
||||
)
|
||||
if python is None:
|
||||
return False
|
||||
return venv_torch_hip(Path(python).parent.parent) is False
|
||||
except Exception: # noqa: BLE001 - metadata only, never break the picker
|
||||
return False
|
||||
|
||||
@classmethod
|
||||
def runtime_compute_profile(cls, caps) -> dict:
|
||||
from services.engine_routing import resolve_routing
|
||||
|
||||
compat = tuple(cls.gpu_compat)
|
||||
if caps.family == "rocm" and cls._sidecar_torch_is_cuda_only():
|
||||
# Honest on this host: the sidecar's torch cannot see the GPU.
|
||||
compat = tuple(c for c in compat if c != "rocm")
|
||||
floor = float(getattr(cls, "min_vram_gb", 0.0) or 0.0)
|
||||
return {
|
||||
"gpu_compat": compat,
|
||||
"min_vram_gb": floor,
|
||||
**resolve_routing(compat, caps, floor),
|
||||
"runtime_backend": None,
|
||||
"runtime_device_index": None,
|
||||
"runtime_device_name": None,
|
||||
"runtime_hardware_family": None,
|
||||
"runtime_vram_gb": None,
|
||||
"runtime_device_verified": None,
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def venv_python(cls):
|
||||
from engines.indextts.bootstrap import resolve_indextts_venv
|
||||
|
||||
@@ -309,6 +309,23 @@ def _rocm_pin_args() -> list[str]:
|
||||
return ["--no-sources", *(f"{pin}{tag}" for pin in ROCM_TORCH_PINS), *idx_args]
|
||||
|
||||
|
||||
def venv_torch_hip(venv_dir: Path) -> Optional[bool]:
|
||||
"""``True``/``False`` when the venv's torch is/isn't a ROCm build, ``None``
|
||||
when that cannot be told (no torch wheel found, unreadable file).
|
||||
|
||||
Reads the wheel's generated ``version.py`` instead of importing torch, so it
|
||||
is safe on every engine-list refresh."""
|
||||
matches = sorted((venv_dir / "lib").glob("python*/site-packages/torch/version.py"))
|
||||
if not matches:
|
||||
return None
|
||||
try:
|
||||
text = matches[0].read_text(encoding="utf-8", errors="replace")
|
||||
except OSError:
|
||||
return None
|
||||
hip_set = re.search(r"^hip(?:\s*:[^=]*)?\s*=\s*['\"][^'\"]+['\"]", text, re.M)
|
||||
return bool(hip_set) or "+rocm" in text
|
||||
|
||||
|
||||
def _venv_torch_is_rocm(venv_dir: Path) -> bool:
|
||||
"""True when the venv's installed torch is a ROCm build.
|
||||
|
||||
@@ -318,15 +335,7 @@ def _venv_torch_is_rocm(venv_dir: Path) -> bool:
|
||||
seconds the completion marker exists to avoid (see
|
||||
``_install_marker_valid``). Missing file = no torch = not ROCm, so a
|
||||
broken deps step is offered the repair too."""
|
||||
matches = sorted((venv_dir / "lib").glob("python*/site-packages/torch/version.py"))
|
||||
if not matches:
|
||||
return False
|
||||
try:
|
||||
text = matches[0].read_text(encoding="utf-8", errors="replace")
|
||||
except OSError:
|
||||
return False
|
||||
hip_set = re.search(r"^hip(?:\s*:[^=]*)?\s*=\s*['\"][^'\"]+['\"]", text, re.M)
|
||||
return bool(hip_set) or "+rocm" in text
|
||||
return bool(venv_torch_hip(venv_dir))
|
||||
|
||||
|
||||
def _moss_host() -> tuple[bool, str]:
|
||||
|
||||
@@ -386,9 +386,9 @@ on CPU until you opt into the ROCm variant.
|
||||
> **Running in Docker or Podman instead?** There's a prebuilt ROCm image —
|
||||
> `ghcr.io/debpalash/voicestudio:rocm` — with GPU acceleration out of the
|
||||
> box; see [docker.md](docker.md#pull-and-run-amd-gpu--rocm). The rest of this
|
||||
> section is about source/desktop installs. (On Windows there is no ROCm path
|
||||
at all — PyTorch publishes no Windows ROCm wheels; see
|
||||
[windows.md](windows.md#gpu-support).)
|
||||
> section is about source/desktop installs. (VoiceStudio has no ROCm path on
|
||||
Windows; see [windows.md](windows.md#gpu-support) for what a Radeon card can
|
||||
do there.)
|
||||
|
||||
Three ways to opt in, in order of preference:
|
||||
|
||||
|
||||
@@ -638,6 +638,32 @@ Apple-Silicon install skips this index entirely.
|
||||
|
||||
**Linked issue:** [#569](https://github.com/debpalash/VoiceStudio/issues/569)
|
||||
|
||||
## 12b. My AMD Radeon GPU is not used (CPU is busy, GPU is idle)
|
||||
|
||||
**Symptom:** generation or transcription is slow, Task Manager shows the CPU
|
||||
busy and the Radeon idle, and **Settings → About → Run self-check** says the
|
||||
compute device is `cpu`.
|
||||
|
||||
**Cause:** the default install ships the NVIDIA CUDA build of PyTorch, which
|
||||
cannot drive AMD GPUs. This is not a driver problem on your side. Open
|
||||
**Settings → Performance → GPU acceleration**: it names your card, the installed
|
||||
PyTorch build and, for every engine, whether it uses the GPU.
|
||||
|
||||
**Fix, by platform:**
|
||||
|
||||
- **Windows:** PyTorch engines stay on the CPU (no ROCm wheels exist for the
|
||||
PyTorch version VoiceStudio ships). Engines with their own GPU runtime — today
|
||||
audio.cpp (Vulkan) — do use a Radeon: install its runtime from **Settings →
|
||||
Models**. Details and the advanced, unsupported AMD-wheels route:
|
||||
[windows.md — GPU support](windows.md#gpu-support).
|
||||
- **Linux:** set `OMNIVOICE_TORCH_VARIANT=rocm` and run setup again, or use the
|
||||
ROCm Docker image — [linux.md — AMD GPU (ROCm)](linux.md#amd-gpu-rocm). If the
|
||||
panel says PyTorch has ROCm but cannot open the device, check that the `amdgpu`
|
||||
driver is loaded and your user can open `/dev/kfd` (`render` and `video`
|
||||
groups).
|
||||
|
||||
**Linked issue:** [#2468](https://github.com/debpalash/VoiceStudio/issues/2468)
|
||||
|
||||
## 13. Stuck on the download page / incomplete model cache ("only `refs/`")
|
||||
|
||||
**Symptom:** the setup screen never finishes the model download and you can't
|
||||
|
||||
+31
-8
@@ -28,7 +28,8 @@ working VoiceStudio install on Windows 10 / 11 (x64).
|
||||
- **Windows 10 (21H2 or newer) or Windows 11**, x64.
|
||||
- **~10 GB free disk** for the app, its Python environment, and model weights.
|
||||
- Optional: an **NVIDIA GPU + driver** for CUDA acceleration — see
|
||||
[GPU support on Windows](#gpu-support). AMD GPUs run CPU-only on Windows.
|
||||
[GPU support on Windows](#gpu-support). AMD GPUs do not accelerate PyTorch
|
||||
engines on Windows (audio.cpp can use them via Vulkan).
|
||||
|
||||
That's it — Python, FFmpeg, and the model weights are bundled or bootstrapped
|
||||
by the app itself on first launch. No toolchain needed.
|
||||
@@ -56,16 +57,38 @@ Everything above, plus the toolchain:
|
||||
|
||||
<a id="gpu-support"></a>
|
||||
|
||||
**GPU acceleration on Windows is NVIDIA/CUDA-only.** The Windows install
|
||||
**PyTorch GPU acceleration on Windows is NVIDIA/CUDA-only.** The Windows install
|
||||
ships the CUDA build of PyTorch; with an NVIDIA GPU and a regular NVIDIA
|
||||
driver it's picked up automatically (no CUDA Toolkit install needed).
|
||||
|
||||
**AMD GPUs — including Ryzen / Ryzen AI integrated Radeon graphics — run
|
||||
CPU-only on Windows.** ROCm is not supported on Windows: PyTorch publishes no
|
||||
Windows ROCm wheels, and VoiceStudio's ROCm option is Linux-only. (The Ryzen AI
|
||||
NPU is likewise not used.) Everything still works on CPU, just slower. If you
|
||||
have an AMD GPU and want GPU acceleration, run VoiceStudio on Linux instead —
|
||||
see [linux.md — AMD GPU (ROCm)](linux.md#amd-gpu-rocm).
|
||||
**AMD GPUs — including Ryzen / Ryzen AI integrated Radeon graphics — do not
|
||||
accelerate PyTorch engines on Windows.** The installer ships the NVIDIA CUDA
|
||||
build of PyTorch 2.8, which cannot drive a Radeon card, so those engines run on
|
||||
the CPU. pytorch.org publishes no Windows ROCm wheels, and VoiceStudio's ROCm
|
||||
option (`OMNIVOICE_TORCH_VARIANT=rocm`) is Linux-only — it is ignored on Windows
|
||||
rather than failing setup. (The Ryzen AI NPU is likewise not used.) Everything
|
||||
still works on CPU, just slower. What you can do today:
|
||||
|
||||
- **Use an engine with its own GPU runtime.** [audio.cpp](../engines/audio-cpp.md)
|
||||
(Breeze-TTS-2) ships a Vulkan build that runs on Radeon GPUs: install the
|
||||
runtime from **Settings → Models**. VoiceStudio picks the discrete GPU
|
||||
automatically.
|
||||
- **Run on Linux** (native, or the ROCm Docker image) for ROCm acceleration of
|
||||
the PyTorch engines — see [linux.md — AMD GPU (ROCm)](linux.md#amd-gpu-rocm).
|
||||
- **Advanced / unsupported: AMD's own Windows ROCm wheels.** AMD publishes
|
||||
PyTorch ROCm wheels for Windows (`https://repo.amd.com/rocm/whl-multi-arch/`,
|
||||
Python 3.11–3.14, RDNA 3 / RDNA 4 cards such as the RX 7000 and RX 9000 series).
|
||||
They are PyTorch 2.9 or newer, not the 2.8 the engines here are validated
|
||||
against, and faster-whisper (CTranslate2) needs its own separate HIP build for
|
||||
the GPU, so expect parts of the app (WhisperX is a reported example) to break. VoiceStudio does not install them, and no engine
|
||||
parity is claimed. DirectML is not an option either: `torch-directml` needs
|
||||
PyTorch 2.4.
|
||||
|
||||
**Settings → Performance → GPU acceleration** shows exactly what applies to your
|
||||
machine: the GPUs Windows reports, which PyTorch build is installed, and a
|
||||
verdict for every engine (uses the GPU, runs on the CPU and why, or CPU by
|
||||
design). **Settings → About → Run self-check** names the card instead of just
|
||||
saying "no GPU acceleration detected".
|
||||
|
||||
## Install (from source)
|
||||
|
||||
|
||||
+7
-2
@@ -41,8 +41,13 @@ Before touching any knob, check these — they account for most slowness reports
|
||||
- **Model Catalogue** shows a routing badge per engine — "GPU active",
|
||||
"CPU fallback", or "CPU" — with the *reason* shown as small text under
|
||||
the badge (full text on hover).
|
||||
Note: **GPU acceleration on Windows is NVIDIA/CUDA-only** — AMD and Intel
|
||||
GPUs run CPU-only there (see [Windows install notes](install/windows.md)).
|
||||
Note: **PyTorch GPU acceleration on Windows is NVIDIA/CUDA-only** — AMD and
|
||||
Intel GPUs run PyTorch engines on the CPU there (audio.cpp can still use a
|
||||
Radeon through Vulkan; see [Windows install notes](install/windows.md)).
|
||||
**Settings → Performance → GPU acceleration** (`GET /api/settings/gpu-report`)
|
||||
lists, per engine, whether it uses the GPU on this machine and why not
|
||||
otherwise — including "your Radeon was found but this PyTorch build is
|
||||
NVIDIA-only".
|
||||
5. **You aborted a dub earlier (fixed in v0.3.23).** Dubbing moves the TTS
|
||||
model to CPU to free VRAM for the ASR model, then moves it back when the
|
||||
transcription finishes. Before v0.3.23 that move-back only ran on the fully
|
||||
|
||||
@@ -328,6 +328,48 @@ def test_report_reasons_are_scrubbed(monkeypatch):
|
||||
assert "alice" not in (rep["engines"][0]["reason"] or "")
|
||||
|
||||
|
||||
def _fake_venv(tmp_path, version_py: str):
|
||||
site = tmp_path / ".venv" / "lib" / "python3.11" / "site-packages" / "torch"
|
||||
site.mkdir(parents=True, exist_ok=True)
|
||||
(site / "version.py").write_text(version_py)
|
||||
(tmp_path / ".venv" / "bin").mkdir(parents=True, exist_ok=True)
|
||||
(tmp_path / ".venv" / "bin" / "python").write_text("")
|
||||
|
||||
|
||||
def test_indextts_does_not_claim_rocm_when_its_venv_holds_a_cuda_torch(tmp_path, monkeypatch):
|
||||
"""PR #2423 review: gpu_compat claims ROCm because the installer provisions a
|
||||
ROCm torch - but a venv that predates that (or a user-managed clone) still
|
||||
has the CUDA wheel, which sees no AMD GPU. Routing must not report
|
||||
acceleration for it."""
|
||||
from engines.indextts import IndexTTS2Backend, bootstrap
|
||||
|
||||
rocm_caps = HostCaps(family="rocm", available_families=("rocm", "cpu"), device_name="RX 6800 XT")
|
||||
monkeypatch.setenv("OMNIVOICE_INDEXTTS_DIR", str(tmp_path))
|
||||
monkeypatch.setattr(bootstrap, "_resolved_python", None)
|
||||
|
||||
_fake_venv(tmp_path, "__version__ = '2.8.0+cu128'\ncuda = '12.8'\nhip = None\n")
|
||||
cuda_only = IndexTTS2Backend.runtime_compute_profile(rocm_caps)
|
||||
assert cuda_only["routing_status"] == "cpu_fallback"
|
||||
assert "rocm" not in cuda_only["gpu_compat"]
|
||||
|
||||
_fake_venv(tmp_path, "__version__ = '2.8.0+rocm6.4'\ncuda = None\nhip = '6.4.43482'\n")
|
||||
rocm = IndexTTS2Backend.runtime_compute_profile(rocm_caps)
|
||||
assert rocm["routing_status"] == "accelerated"
|
||||
assert rocm["effective_device"] == "rocm"
|
||||
|
||||
|
||||
def test_indextts_keeps_its_claim_when_the_venv_cannot_be_inspected(tmp_path, monkeypatch):
|
||||
from engines.indextts import IndexTTS2Backend, bootstrap
|
||||
|
||||
monkeypatch.setenv("OMNIVOICE_INDEXTTS_DIR", str(tmp_path)) # no .venv at all
|
||||
monkeypatch.setattr(bootstrap, "_resolved_python", None)
|
||||
rocm_caps = HostCaps(family="rocm", available_families=("rocm", "cpu"))
|
||||
assert IndexTTS2Backend.runtime_compute_profile(rocm_caps)["routing_status"] == "accelerated"
|
||||
# Never narrowed off a ROCm host either.
|
||||
cuda_caps = HostCaps(family="cuda", available_families=("cuda", "cpu"))
|
||||
assert IndexTTS2Backend.runtime_compute_profile(cuda_caps)["routing_status"] == "accelerated"
|
||||
|
||||
|
||||
def test_self_check_names_the_card_instead_of_saying_no_gpu(monkeypatch):
|
||||
from core import diagnose
|
||||
|
||||
|
||||
Reference in New Issue
Block a user