fix(indextts): don't claim ROCm for a sidecar venv holding a CUDA torch; docs for Windows AMD (#2371, #2468)

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
Palash Debnath
2026-10-01 19:11:01 +05:30
co-authored by Claude Sonnet 5.5
parent c3b564a23c
commit 2bfbd6097d
7 changed files with 176 additions and 22 deletions
+49
View File
@@ -31,6 +31,7 @@ import logging
import math
import os
import re
from pathlib import Path
from typing import TYPE_CHECKING
from services.subprocess_backend import SubprocessBackend
@@ -162,6 +163,54 @@ class IndexTTS2Backend(SubprocessBackend):
)
return True, "ok"
@classmethod
def _sidecar_torch_is_cuda_only(cls) -> bool:
"""True only when the sidecar venv is KNOWN to hold a non-ROCm torch.
``gpu_compat`` claims ROCm because the one-click installer provisions a
ROCm torch on ROCm hosts (#2371). A venv that predates that, or a
user-managed clone, can still carry the CUDA wheel upstream resolves -
which sees no AMD GPU, so the engine would run on the CPU while routing
reported acceleration (the PR #2423 review finding). Unknown (venv not
located yet, no torch wheel found) keeps the claim: this runs on every
engine-list refresh, so it must never spawn the interpreter probe.
"""
try:
from engines.indextts import bootstrap
from services.sidecar_install import venv_torch_hip
python = bootstrap._resolved_python
if python is None and os.environ.get("OMNIVOICE_INDEXTTS_DIR"):
python = bootstrap._venv_python_path(
Path(os.environ["OMNIVOICE_INDEXTTS_DIR"]) / ".venv"
)
if python is None:
return False
return venv_torch_hip(Path(python).parent.parent) is False
except Exception: # noqa: BLE001 - metadata only, never break the picker
return False
@classmethod
def runtime_compute_profile(cls, caps) -> dict:
from services.engine_routing import resolve_routing
compat = tuple(cls.gpu_compat)
if caps.family == "rocm" and cls._sidecar_torch_is_cuda_only():
# Honest on this host: the sidecar's torch cannot see the GPU.
compat = tuple(c for c in compat if c != "rocm")
floor = float(getattr(cls, "min_vram_gb", 0.0) or 0.0)
return {
"gpu_compat": compat,
"min_vram_gb": floor,
**resolve_routing(compat, caps, floor),
"runtime_backend": None,
"runtime_device_index": None,
"runtime_device_name": None,
"runtime_hardware_family": None,
"runtime_vram_gb": None,
"runtime_device_verified": None,
}
@classmethod
def venv_python(cls):
from engines.indextts.bootstrap import resolve_indextts_venv
+18 -9
View File
@@ -309,6 +309,23 @@ def _rocm_pin_args() -> list[str]:
return ["--no-sources", *(f"{pin}{tag}" for pin in ROCM_TORCH_PINS), *idx_args]
def venv_torch_hip(venv_dir: Path) -> Optional[bool]:
"""``True``/``False`` when the venv's torch is/isn't a ROCm build, ``None``
when that cannot be told (no torch wheel found, unreadable file).
Reads the wheel's generated ``version.py`` instead of importing torch, so it
is safe on every engine-list refresh."""
matches = sorted((venv_dir / "lib").glob("python*/site-packages/torch/version.py"))
if not matches:
return None
try:
text = matches[0].read_text(encoding="utf-8", errors="replace")
except OSError:
return None
hip_set = re.search(r"^hip(?:\s*:[^=]*)?\s*=\s*['\"][^'\"]+['\"]", text, re.M)
return bool(hip_set) or "+rocm" in text
def _venv_torch_is_rocm(venv_dir: Path) -> bool:
"""True when the venv's installed torch is a ROCm build.
@@ -318,15 +335,7 @@ def _venv_torch_is_rocm(venv_dir: Path) -> bool:
seconds the completion marker exists to avoid (see
``_install_marker_valid``). Missing file = no torch = not ROCm, so a
broken deps step is offered the repair too."""
matches = sorted((venv_dir / "lib").glob("python*/site-packages/torch/version.py"))
if not matches:
return False
try:
text = matches[0].read_text(encoding="utf-8", errors="replace")
except OSError:
return False
hip_set = re.search(r"^hip(?:\s*:[^=]*)?\s*=\s*['\"][^'\"]+['\"]", text, re.M)
return bool(hip_set) or "+rocm" in text
return bool(venv_torch_hip(venv_dir))
def _moss_host() -> tuple[bool, str]:
+3 -3
View File
@@ -386,9 +386,9 @@ on CPU until you opt into the ROCm variant.
> **Running in Docker or Podman instead?** There's a prebuilt ROCm image —
> `ghcr.io/debpalash/voicestudio:rocm` — with GPU acceleration out of the
> box; see [docker.md](docker.md#pull-and-run-amd-gpu--rocm). The rest of this
> section is about source/desktop installs. (On Windows there is no ROCm path
at all — PyTorch publishes no Windows ROCm wheels; see
[windows.md](windows.md#gpu-support).)
> section is about source/desktop installs. (VoiceStudio has no ROCm path on
Windows; see [windows.md](windows.md#gpu-support) for what a Radeon card can
do there.)
Three ways to opt in, in order of preference:
+26
View File
@@ -638,6 +638,32 @@ Apple-Silicon install skips this index entirely.
**Linked issue:** [#569](https://github.com/debpalash/VoiceStudio/issues/569)
## 12b. My AMD Radeon GPU is not used (CPU is busy, GPU is idle)
**Symptom:** generation or transcription is slow, Task Manager shows the CPU
busy and the Radeon idle, and **Settings → About → Run self-check** says the
compute device is `cpu`.
**Cause:** the default install ships the NVIDIA CUDA build of PyTorch, which
cannot drive AMD GPUs. This is not a driver problem on your side. Open
**Settings → Performance → GPU acceleration**: it names your card, the installed
PyTorch build and, for every engine, whether it uses the GPU.
**Fix, by platform:**
- **Windows:** PyTorch engines stay on the CPU (no ROCm wheels exist for the
PyTorch version VoiceStudio ships). Engines with their own GPU runtime — today
audio.cpp (Vulkan) — do use a Radeon: install its runtime from **Settings →
Models**. Details and the advanced, unsupported AMD-wheels route:
[windows.md — GPU support](windows.md#gpu-support).
- **Linux:** set `OMNIVOICE_TORCH_VARIANT=rocm` and run setup again, or use the
ROCm Docker image — [linux.md — AMD GPU (ROCm)](linux.md#amd-gpu-rocm). If the
panel says PyTorch has ROCm but cannot open the device, check that the `amdgpu`
driver is loaded and your user can open `/dev/kfd` (`render` and `video`
groups).
**Linked issue:** [#2468](https://github.com/debpalash/VoiceStudio/issues/2468)
## 13. Stuck on the download page / incomplete model cache ("only `refs/`")
**Symptom:** the setup screen never finishes the model download and you can't
+31 -8
View File
@@ -28,7 +28,8 @@ working VoiceStudio install on Windows 10 / 11 (x64).
- **Windows 10 (21H2 or newer) or Windows 11**, x64.
- **~10 GB free disk** for the app, its Python environment, and model weights.
- Optional: an **NVIDIA GPU + driver** for CUDA acceleration — see
[GPU support on Windows](#gpu-support). AMD GPUs run CPU-only on Windows.
[GPU support on Windows](#gpu-support). AMD GPUs do not accelerate PyTorch
engines on Windows (audio.cpp can use them via Vulkan).
That's it — Python, FFmpeg, and the model weights are bundled or bootstrapped
by the app itself on first launch. No toolchain needed.
@@ -56,16 +57,38 @@ Everything above, plus the toolchain:
<a id="gpu-support"></a>
**GPU acceleration on Windows is NVIDIA/CUDA-only.** The Windows install
**PyTorch GPU acceleration on Windows is NVIDIA/CUDA-only.** The Windows install
ships the CUDA build of PyTorch; with an NVIDIA GPU and a regular NVIDIA
driver it's picked up automatically (no CUDA Toolkit install needed).
**AMD GPUs — including Ryzen / Ryzen AI integrated Radeon graphics — run
CPU-only on Windows.** ROCm is not supported on Windows: PyTorch publishes no
Windows ROCm wheels, and VoiceStudio's ROCm option is Linux-only. (The Ryzen AI
NPU is likewise not used.) Everything still works on CPU, just slower. If you
have an AMD GPU and want GPU acceleration, run VoiceStudio on Linux instead —
see [linux.md — AMD GPU (ROCm)](linux.md#amd-gpu-rocm).
**AMD GPUs — including Ryzen / Ryzen AI integrated Radeon graphics — do not
accelerate PyTorch engines on Windows.** The installer ships the NVIDIA CUDA
build of PyTorch 2.8, which cannot drive a Radeon card, so those engines run on
the CPU. pytorch.org publishes no Windows ROCm wheels, and VoiceStudio's ROCm
option (`OMNIVOICE_TORCH_VARIANT=rocm`) is Linux-only — it is ignored on Windows
rather than failing setup. (The Ryzen AI NPU is likewise not used.) Everything
still works on CPU, just slower. What you can do today:
- **Use an engine with its own GPU runtime.** [audio.cpp](../engines/audio-cpp.md)
(Breeze-TTS-2) ships a Vulkan build that runs on Radeon GPUs: install the
runtime from **Settings → Models**. VoiceStudio picks the discrete GPU
automatically.
- **Run on Linux** (native, or the ROCm Docker image) for ROCm acceleration of
the PyTorch engines — see [linux.md — AMD GPU (ROCm)](linux.md#amd-gpu-rocm).
- **Advanced / unsupported: AMD's own Windows ROCm wheels.** AMD publishes
PyTorch ROCm wheels for Windows (`https://repo.amd.com/rocm/whl-multi-arch/`,
Python 3.11–3.14, RDNA 3 / RDNA 4 cards such as the RX 7000 and RX 9000 series).
They are PyTorch 2.9 or newer, not the 2.8 the engines here are validated
against, and faster-whisper (CTranslate2) needs its own separate HIP build for
the GPU, so expect parts of the app (WhisperX is a reported example) to break. VoiceStudio does not install them, and no engine
parity is claimed. DirectML is not an option either: `torch-directml` needs
PyTorch 2.4.
**Settings → Performance → GPU acceleration** shows exactly what applies to your
machine: the GPUs Windows reports, which PyTorch build is installed, and a
verdict for every engine (uses the GPU, runs on the CPU and why, or CPU by
design). **Settings → About → Run self-check** names the card instead of just
saying "no GPU acceleration detected".
## Install (from source)
+7 -2
View File
@@ -41,8 +41,13 @@ Before touching any knob, check these — they account for most slowness reports
- **Model Catalogue** shows a routing badge per engine — "GPU active",
"CPU fallback", or "CPU" — with the *reason* shown as small text under
the badge (full text on hover).
Note: **GPU acceleration on Windows is NVIDIA/CUDA-only** — AMD and Intel
GPUs run CPU-only there (see [Windows install notes](install/windows.md)).
Note: **PyTorch GPU acceleration on Windows is NVIDIA/CUDA-only** — AMD and
Intel GPUs run PyTorch engines on the CPU there (audio.cpp can still use a
Radeon through Vulkan; see [Windows install notes](install/windows.md)).
**Settings → Performance → GPU acceleration** (`GET /api/settings/gpu-report`)
lists, per engine, whether it uses the GPU on this machine and why not
otherwise — including "your Radeon was found but this PyTorch build is
NVIDIA-only".
5. **You aborted a dub earlier (fixed in v0.3.23).** Dubbing moves the TTS
model to CPU to free VRAM for the ASR model, then moves it back when the
transcription finishes. Before v0.3.23 that move-back only ran on the fully
+42
View File
@@ -328,6 +328,48 @@ def test_report_reasons_are_scrubbed(monkeypatch):
assert "alice" not in (rep["engines"][0]["reason"] or "")
def _fake_venv(tmp_path, version_py: str):
site = tmp_path / ".venv" / "lib" / "python3.11" / "site-packages" / "torch"
site.mkdir(parents=True, exist_ok=True)
(site / "version.py").write_text(version_py)
(tmp_path / ".venv" / "bin").mkdir(parents=True, exist_ok=True)
(tmp_path / ".venv" / "bin" / "python").write_text("")
def test_indextts_does_not_claim_rocm_when_its_venv_holds_a_cuda_torch(tmp_path, monkeypatch):
"""PR #2423 review: gpu_compat claims ROCm because the installer provisions a
ROCm torch - but a venv that predates that (or a user-managed clone) still
has the CUDA wheel, which sees no AMD GPU. Routing must not report
acceleration for it."""
from engines.indextts import IndexTTS2Backend, bootstrap
rocm_caps = HostCaps(family="rocm", available_families=("rocm", "cpu"), device_name="RX 6800 XT")
monkeypatch.setenv("OMNIVOICE_INDEXTTS_DIR", str(tmp_path))
monkeypatch.setattr(bootstrap, "_resolved_python", None)
_fake_venv(tmp_path, "__version__ = '2.8.0+cu128'\ncuda = '12.8'\nhip = None\n")
cuda_only = IndexTTS2Backend.runtime_compute_profile(rocm_caps)
assert cuda_only["routing_status"] == "cpu_fallback"
assert "rocm" not in cuda_only["gpu_compat"]
_fake_venv(tmp_path, "__version__ = '2.8.0+rocm6.4'\ncuda = None\nhip = '6.4.43482'\n")
rocm = IndexTTS2Backend.runtime_compute_profile(rocm_caps)
assert rocm["routing_status"] == "accelerated"
assert rocm["effective_device"] == "rocm"
def test_indextts_keeps_its_claim_when_the_venv_cannot_be_inspected(tmp_path, monkeypatch):
from engines.indextts import IndexTTS2Backend, bootstrap
monkeypatch.setenv("OMNIVOICE_INDEXTTS_DIR", str(tmp_path)) # no .venv at all
monkeypatch.setattr(bootstrap, "_resolved_python", None)
rocm_caps = HostCaps(family="rocm", available_families=("rocm", "cpu"))
assert IndexTTS2Backend.runtime_compute_profile(rocm_caps)["routing_status"] == "accelerated"
# Never narrowed off a ROCm host either.
cuda_caps = HostCaps(family="cuda", available_families=("cuda", "cpu"))
assert IndexTTS2Backend.runtime_compute_profile(cuda_caps)["routing_status"] == "accelerated"
def test_self_check_names_the_card_instead_of_saying_no_gpu(monkeypatch):
from core import diagnose