perf(voice): run STT+TTS on the GPU by default with CPU fallback

Both voice backends defaulted to CPU. Fix the "CUDA unavailable" gaps so
everything that benefits from the RTX 5050 uses it:

- MeloTTS venv had CPU-only torch (2.12.0+cpu) -> installed Blackwell-capable
  torch/torchaudio 2.11.0+cu128 (sm_120 verified with a real GPU matmul).
- faster-whisper CUDA loaded but transcribe() died with "libcublas.so.12 not
  found": installed nvidia-cublas-cu12 + nvidia-cudnn-cu12 into the whisper
  venv and inject those nvidia/*/lib dirs into the worker's LD_LIBRARY_PATH at
  spawn (the loader only honours it at exec).
- WSAI_WHISPER_DEVICE / WSAI_MELO_DEVICE now default to "auto": pick CUDA when
  present, else CPU, and each worker falls back to CPU if a CUDA load fails so
  the voice loop never dies on a GPU-less host.

Verified end-to-end through the real backend classes: both workers report
"ready on cuda"; steady-state STT ~170ms (was ~1350ms CPU), TTS ~4s first call
vs ~23s CPU. All 12 tests pass.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
EJClaw
2026-08-18 21:02:52 +09:00
parent 63fcfb7ba2
commit 51811ad251
4 changed files with 86 additions and 14 deletions

View File

@@ -19,7 +19,8 @@ Env:
WSAI_WHISPER_PYTHON interpreter with faster-whisper installed
(default: /home/claude/jarvis-stt/whisper312/bin/python)
WSAI_WHISPER_MODEL model size/name (default: small)
WSAI_WHISPER_DEVICE cpu | cuda (default cpu; cuda needs GPU approval)
WSAI_WHISPER_DEVICE cpu | cuda | auto (default auto: GPU if present,
else CPU; the worker falls back to CPU if CUDA fails)
WSAI_WHISPER_LANGUAGE forced language, e.g. ko (default ko; "" = autodetect)
"""
@@ -41,6 +42,20 @@ log = logging.getLogger("wsai.stt.whisper")
_DEFAULT_PYTHON = "/home/claude/jarvis-stt/whisper312/bin/python"
def _cuda_lib_dirs(python_exe: str) -> list[str]:
"""nvidia/*/lib dirs of the worker venv (cublas, cudnn, ...), for
LD_LIBRARY_PATH so ctranslate2 can dlopen the CUDA runtime. Empty if the
interpreter has no such packages (CPU-only install)."""
import glob
# Use the literal path, NOT .resolve(): the venv's bin/python is a symlink
# into the uv-managed interpreter, and resolving it would jump out of the
# venv and miss its site-packages/nvidia libs.
venv = Path(python_exe).parent.parent # .../bin/python -> venv root
dirs = glob.glob(str(venv / "lib" / "python*" / "site-packages" / "nvidia" / "*" / "lib"))
return sorted(set(dirs))
class WhisperSTT:
def __init__(
self,
@@ -53,7 +68,7 @@ class WhisperSTT:
) -> None:
self.python = python or os.environ.get("WSAI_WHISPER_PYTHON", _DEFAULT_PYTHON)
self.model = model or os.environ.get("WSAI_WHISPER_MODEL", "small")
self.device = device or os.environ.get("WSAI_WHISPER_DEVICE", "cpu")
self.device = device or os.environ.get("WSAI_WHISPER_DEVICE", "auto")
# "" means autodetect; a real code like "ko" forces the language.
env_lang = os.environ.get("WSAI_WHISPER_LANGUAGE", "ko")
self.language = language if language is not None else (env_lang or None)
@@ -97,6 +112,15 @@ class WhisperSTT:
"WSAI_WHISPER_MODEL": self.model,
"WSAI_WHISPER_DEVICE": self.device,
}
# ctranslate2 dlopens libcublas/libcudnn from the whisper venv's nvidia
# pip packages; the dynamic loader only honours LD_LIBRARY_PATH captured
# at exec, so inject those lib dirs into the child env here (harmless on
# CPU). Without this the CUDA model loads but transcribe() dies with
# "Library libcublas.so.12 is not found".
lib_dirs = _cuda_lib_dirs(self.python)
if lib_dirs:
prev = env.get("LD_LIBRARY_PATH", "")
env["LD_LIBRARY_PATH"] = ":".join(lib_dirs + ([prev] if prev else []))
if self.language:
env["WSAI_WHISPER_LANGUAGE"] = self.language
repo_root = str(Path(__file__).resolve().parents[2])