From e81ce1ac976e387d7826e02af94bddbaa8777348 Mon Sep 17 00:00:00 2001 From: EJClaw Date: Sun, 23 Aug 2026 00:18:47 +0900 Subject: [PATCH] =?UTF-8?q?fix(tts):=20lower=20base=20speed=201.5=E2=86=92?= =?UTF-8?q?1.2=20=E2=80=94=201.5x=20slurred=20Korean=20pronunciation?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit User reported the 1.5x base made Korean pronunciation mushy/slurred. Dial the default WSAI_TTS_SPEED back to 1.2: still noticeably faster than the original 1.0, but clear. Emotion multipliers scale off base as before. Co-Authored-By: Claude Opus 4.7 --- wsai/backends/melo.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/wsai/backends/melo.py b/wsai/backends/melo.py index 81cce5d..3e15bac 100644 --- a/wsai/backends/melo.py +++ b/wsai/backends/melo.py @@ -12,7 +12,7 @@ Env: WSAI_MELO_DEVICE cpu | cuda | auto (default auto: GPU if torch sees one, else CPU; the worker falls back to CPU if CUDA fails) WSAI_TTS_OUT_DIR where wavs are written (default ~/.cache/wsai/tts) - WSAI_TTS_SPEED synthesis speed multiplier (default 1.5) + WSAI_TTS_SPEED synthesis speed multiplier (default 1.2) """ from __future__ import annotations @@ -93,7 +93,7 @@ class MeloTTS: self.out_dir = Path(out_dir or os.environ.get("WSAI_TTS_OUT_DIR") or (Path.home() / ".cache/wsai/tts")) self.speed = float(speed if speed is not None - else os.environ.get("WSAI_TTS_SPEED", "1.5")) + else os.environ.get("WSAI_TTS_SPEED", "1.2")) self.sink = sink or _log_sink self._proc: asyncio.subprocess.Process | None = None self._lock = asyncio.Lock()