perf(voice): run STT+TTS on the GPU by default with CPU fallback
Both voice backends defaulted to CPU. Fix the "CUDA unavailable" gaps so everything that benefits from the RTX 5050 uses it: - MeloTTS venv had CPU-only torch (2.12.0+cpu) -> installed Blackwell-capable torch/torchaudio 2.11.0+cu128 (sm_120 verified with a real GPU matmul). - faster-whisper CUDA loaded but transcribe() died with "libcublas.so.12 not found": installed nvidia-cublas-cu12 + nvidia-cudnn-cu12 into the whisper venv and inject those nvidia/*/lib dirs into the worker's LD_LIBRARY_PATH at spawn (the loader only honours it at exec). - WSAI_WHISPER_DEVICE / WSAI_MELO_DEVICE now default to "auto": pick CUDA when present, else CPU, and each worker falls back to CPU if a CUDA load fails so the voice loop never dies on a GPU-less host. Verified end-to-end through the real backend classes: both workers report "ready on cuda"; steady-state STT ~170ms (was ~1350ms CPU), TTS ~4s first call vs ~23s CPU. All 12 tests pass. Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
@@ -38,11 +38,33 @@ def _log(*a):
|
||||
|
||||
def main() -> None:
|
||||
lang = "KR"
|
||||
device = os.environ.get("WSAI_MELO_DEVICE", "cpu") # "cpu" | "cuda" | "auto"
|
||||
t0 = time.monotonic()
|
||||
requested = os.environ.get("WSAI_MELO_DEVICE", "auto") # cpu | cuda | auto
|
||||
from melo.api import TTS # heavy import; only in the melo venv
|
||||
|
||||
tts = TTS(language=lang, device=device)
|
||||
def _has_cuda() -> bool:
|
||||
try:
|
||||
import torch
|
||||
|
||||
return torch.cuda.is_available()
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
device = requested
|
||||
if requested == "auto":
|
||||
device = "cuda" if _has_cuda() else "cpu"
|
||||
|
||||
t0 = time.monotonic()
|
||||
try:
|
||||
tts = TTS(language=lang, device=device)
|
||||
except Exception as exc:
|
||||
# CUDA picked but unusable (CPU-only torch, missing libs, OOM): fall back
|
||||
# to CPU rather than leaving the whole voice loop dead.
|
||||
if device == "cuda":
|
||||
_log(f"[melo_worker] CUDA load failed ({exc}); falling back to CPU")
|
||||
device = "cpu"
|
||||
tts = TTS(language=lang, device=device)
|
||||
else:
|
||||
raise
|
||||
speaker_id = tts.hps.data.spk2id[lang]
|
||||
load_ms = int((time.monotonic() - t0) * 1000)
|
||||
_emit({"ready": True, "ms": load_ms, "device": device})
|
||||
|
||||
Reference in New Issue
Block a user