feat(tts): bump base speed 1.0→1.15 for slightly faster delivery
User asked for a bit faster speech. Pitch shift is already disabled, so the earlier "monster voice" risk from stacking pitch on a fast base is gone — a modest 1.15x base is natural. Emotion multipliers scale off base, so every emotion gets the bump too. Also adds tests/gen_emotion_samples.py, a dev utility that synthesizes one clip per canonical emotion (announce name at neutral speed, then a sample sentence steered by that emotion's tag) and stitches them into a single wav for auditioning the emotion palette. Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
76
tests/gen_emotion_samples.py
Normal file
76
tests/gen_emotion_samples.py
Normal file
@@ -0,0 +1,76 @@
|
||||
"""One-off: synthesize a short Korean sample for every canonical emotion and
|
||||
stitch them into a single wav so the delivery of each emotion can be heard.
|
||||
|
||||
Each clip announces the emotion name at neutral (base) speed, then speaks a
|
||||
sample sentence steered by that emotion's ``[태그]`` — exactly the path the live
|
||||
voice server uses. Run with the orchestrator venv:
|
||||
|
||||
.venv/bin/python -m tests.gen_emotion_samples /abs/out.wav
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import sys
|
||||
import wave
|
||||
|
||||
from wsai.backends.melo import MeloTTS
|
||||
|
||||
# (announced Korean name, emotion tag word, sample sentence)
|
||||
SAMPLES: list[tuple[str, str, str]] = [
|
||||
("기쁨", "기쁨", "오늘은 정말 기분 좋은 하루예요!"),
|
||||
("신남", "신남", "우와, 이거 진짜 신난다! 빨리 하자!"),
|
||||
("희망", "희망", "우리 분명히 잘 해낼 수 있어요!"),
|
||||
("슬픔", "슬픔", "조금 속상한 일이 있었어요."),
|
||||
("화남", "화남", "정말 너무하잖아요, 화가 나요."),
|
||||
("두려움", "두려움", "어떡하지, 너무 무서워요."),
|
||||
("놀람", "놀람", "어머, 이게 정말이에요?"),
|
||||
("혐오", "혐오", "으, 이건 좀 별로예요."),
|
||||
("차분", "차분", "천천히 하나씩 정리해 볼게요."),
|
||||
("다정", "다정", "언제든지 편하게 말해 주세요."),
|
||||
("진지", "진지", "이건 정말 중요한 이야기예요."),
|
||||
("실망", "실망", "조금 아쉬운 결과네요."),
|
||||
("피곤", "피곤", "아, 오늘 너무 피곤하네요."),
|
||||
("사랑스럽게", "사랑스럽게", "당신은 정말 소중한 사람이에요."),
|
||||
("장난스럽게", "장난스럽게", "히히, 한번 맞혀 보세요!"),
|
||||
("궁금", "궁금", "그건 대체 왜 그런 걸까요?"),
|
||||
("속삭임", "속삭임", "조용히, 우리끼리만 아는 비밀이에요."),
|
||||
("외침", "외침", "다 같이 힘내자, 파이팅!"),
|
||||
("단호", "단호", "이번엔 반드시 해내겠어요."),
|
||||
("안도", "안도", "휴, 이제야 마음이 놓이네요."),
|
||||
]
|
||||
|
||||
|
||||
def stitch(paths: list[str], out: str, gap_s: float = 0.45) -> None:
|
||||
with wave.open(paths[0], "rb") as w0:
|
||||
nch, sw, fr = w0.getnchannels(), w0.getsampwidth(), w0.getframerate()
|
||||
silence = b"\x00" * (int(fr * gap_s) * sw * nch)
|
||||
with wave.open(out, "wb") as wo:
|
||||
wo.setnchannels(nch)
|
||||
wo.setsampwidth(sw)
|
||||
wo.setframerate(fr)
|
||||
for i, p in enumerate(paths):
|
||||
with wave.open(p, "rb") as w:
|
||||
wo.writeframes(w.readframes(w.getnframes()))
|
||||
if i < len(paths) - 1:
|
||||
wo.writeframes(silence)
|
||||
|
||||
|
||||
async def main(out: str) -> None:
|
||||
tts = MeloTTS() # base speed comes from the new WSAI_TTS_SPEED default (1.15)
|
||||
await tts.warmup()
|
||||
print(f"melo ready ({tts.load_ms} ms), base speed {tts.speed}")
|
||||
paths: list[str] = []
|
||||
for name, tag, sample in SAMPLES:
|
||||
text = f"{name}. [{tag}] {sample}"
|
||||
p = await tts.synth(text)
|
||||
paths.append(p)
|
||||
print(f" {name:8s} -> {p}")
|
||||
await tts.aclose()
|
||||
stitch(paths, out)
|
||||
print(f"stitched {len(paths)} clips -> {out}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
out = sys.argv[1] if len(sys.argv) > 1 else "/tmp/emotion_samples.wav"
|
||||
asyncio.run(main(out))
|
||||
@@ -12,7 +12,7 @@ Env:
|
||||
WSAI_MELO_DEVICE cpu | cuda | auto (default auto: GPU if torch sees one,
|
||||
else CPU; the worker falls back to CPU if CUDA fails)
|
||||
WSAI_TTS_OUT_DIR where wavs are written (default ~/.cache/wsai/tts)
|
||||
WSAI_TTS_SPEED synthesis speed multiplier (default 1.0)
|
||||
WSAI_TTS_SPEED synthesis speed multiplier (default 1.15)
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -93,7 +93,7 @@ class MeloTTS:
|
||||
self.out_dir = Path(out_dir or os.environ.get("WSAI_TTS_OUT_DIR")
|
||||
or (Path.home() / ".cache/wsai/tts"))
|
||||
self.speed = float(speed if speed is not None
|
||||
else os.environ.get("WSAI_TTS_SPEED", "1.0"))
|
||||
else os.environ.get("WSAI_TTS_SPEED", "1.15"))
|
||||
self.sink = sink or _log_sink
|
||||
self._proc: asyncio.subprocess.Process | None = None
|
||||
self._lock = asyncio.Lock()
|
||||
|
||||
Reference in New Issue
Block a user