Compare commits
6 Commits
2ea7d04289
...
361dce70bb
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
361dce70bb | ||
|
|
e81ce1ac97 | ||
|
|
e67cd2e93f | ||
|
|
d410d4a6b5 | ||
|
|
80a83d9944 | ||
|
|
6e20f2cd79 |
32
samples/README.md
Normal file
32
samples/README.md
Normal file
@@ -0,0 +1,32 @@
|
|||||||
|
# 음성 샘플 (voice / emotion samples)
|
||||||
|
|
||||||
|
브라우저나 로컬에서 들어보며 목소리·감정 톤을 고르기 위한 오디오 샘플 모음이다.
|
||||||
|
Discord로 올리는 대신 여기에 보관한다.
|
||||||
|
|
||||||
|
## voice/ — 다른 여자 목소리 후보 (XTTS v2)
|
||||||
|
|
||||||
|
현재 라이브 TTS(MeloTTS 한국어)는 화자가 하나뿐이라 "다른 여자 목소리"를 낼 수 없다.
|
||||||
|
대안으로 XTTS v2의 내장 여성 스튜디오 보이스로 같은 한국어 문장을 합성한 후보들이다.
|
||||||
|
문장은 모두 동일하다: "안녕하세요, 저는 새로운 목소리예요. 한국어 발음이 또렷하게
|
||||||
|
들리는지 한번 들어봐 주세요."
|
||||||
|
|
||||||
|
| 파일 | 화자 |
|
||||||
|
|------|------|
|
||||||
|
| xtts_01_ana_florence.mp3 | Ana Florence |
|
||||||
|
| xtts_02_daisy_studious.mp3 | Daisy Studious |
|
||||||
|
| xtts_03_sofia_hellen.mp3 | Sofia Hellen |
|
||||||
|
| xtts_04_alexandra_hisakawa.mp3 | Alexandra Hisakawa |
|
||||||
|
| xtts_05_nova_hogarth.mp3 | Nova Hogarth |
|
||||||
|
| xtts_06_rosemary_okafor.mp3 | Rosemary Okafor |
|
||||||
|
|
||||||
|
트레이드오프: XTTS는 MeloTTS보다 발음이 또렷하고 자연스러운 여성 음색을 고를 수 있지만,
|
||||||
|
합성이 더 무겁다(실시간 1초 예산과 충돌 가능). 라이브 채택 시 GPU 스트리밍으로 첫 소리
|
||||||
|
지연을 실측해 맞춰야 한다. 생성 스크립트: `/home/claude/jarvis-tts/gen_xtts_voices.py`.
|
||||||
|
|
||||||
|
## emotion/ — 감정별 톤 샘플 (현재 라이브 MeloTTS)
|
||||||
|
|
||||||
|
현재 엔진(MeloTTS 한국어)의 `[감정]` 태그별 델리버리를 하나씩 들어보는 샘플이다.
|
||||||
|
각 클립은 감정 이름을 기본 속도로 말한 뒤 그 감정 톤으로 예시 문장을 말한다.
|
||||||
|
감정 구분은 피치가 아니라 말 빠르기로만 표현된다(피치 변조는 잡음 때문에 비활성).
|
||||||
|
파일명이 감정을 그대로 담는다(예: `emo_02_happy.mp3` = 기쁨). 생성 스크립트:
|
||||||
|
`tests/gen_emotion_samples.py`.
|
||||||
BIN
samples/emotion/emo_01_base.mp3
Normal file
BIN
samples/emotion/emo_01_base.mp3
Normal file
Binary file not shown.
BIN
samples/emotion/emo_02_happy.mp3
Normal file
BIN
samples/emotion/emo_02_happy.mp3
Normal file
Binary file not shown.
BIN
samples/emotion/emo_03_excited.mp3
Normal file
BIN
samples/emotion/emo_03_excited.mp3
Normal file
Binary file not shown.
BIN
samples/emotion/emo_04_hopeful.mp3
Normal file
BIN
samples/emotion/emo_04_hopeful.mp3
Normal file
Binary file not shown.
BIN
samples/emotion/emo_05_sad.mp3
Normal file
BIN
samples/emotion/emo_05_sad.mp3
Normal file
Binary file not shown.
BIN
samples/emotion/emo_06_angry.mp3
Normal file
BIN
samples/emotion/emo_06_angry.mp3
Normal file
Binary file not shown.
BIN
samples/emotion/emo_07_fearful.mp3
Normal file
BIN
samples/emotion/emo_07_fearful.mp3
Normal file
Binary file not shown.
BIN
samples/emotion/emo_08_surprised.mp3
Normal file
BIN
samples/emotion/emo_08_surprised.mp3
Normal file
Binary file not shown.
BIN
samples/emotion/emo_09_disgust.mp3
Normal file
BIN
samples/emotion/emo_09_disgust.mp3
Normal file
Binary file not shown.
BIN
samples/emotion/emo_10_calm.mp3
Normal file
BIN
samples/emotion/emo_10_calm.mp3
Normal file
Binary file not shown.
BIN
samples/emotion/emo_11_friendly.mp3
Normal file
BIN
samples/emotion/emo_11_friendly.mp3
Normal file
Binary file not shown.
BIN
samples/emotion/emo_12_serious.mp3
Normal file
BIN
samples/emotion/emo_12_serious.mp3
Normal file
Binary file not shown.
BIN
samples/emotion/emo_13_disappointed.mp3
Normal file
BIN
samples/emotion/emo_13_disappointed.mp3
Normal file
Binary file not shown.
BIN
samples/emotion/emo_14_tired.mp3
Normal file
BIN
samples/emotion/emo_14_tired.mp3
Normal file
Binary file not shown.
BIN
samples/emotion/emo_15_affectionate.mp3
Normal file
BIN
samples/emotion/emo_15_affectionate.mp3
Normal file
Binary file not shown.
BIN
samples/emotion/emo_16_playful.mp3
Normal file
BIN
samples/emotion/emo_16_playful.mp3
Normal file
Binary file not shown.
BIN
samples/emotion/emo_17_curious.mp3
Normal file
BIN
samples/emotion/emo_17_curious.mp3
Normal file
Binary file not shown.
BIN
samples/emotion/emo_18_whisper.mp3
Normal file
BIN
samples/emotion/emo_18_whisper.mp3
Normal file
Binary file not shown.
BIN
samples/emotion/emo_19_shout.mp3
Normal file
BIN
samples/emotion/emo_19_shout.mp3
Normal file
Binary file not shown.
BIN
samples/emotion/emo_20_determined.mp3
Normal file
BIN
samples/emotion/emo_20_determined.mp3
Normal file
Binary file not shown.
BIN
samples/emotion/emo_21_relieved.mp3
Normal file
BIN
samples/emotion/emo_21_relieved.mp3
Normal file
Binary file not shown.
BIN
samples/voice/xtts_01_ana_florence.mp3
Normal file
BIN
samples/voice/xtts_01_ana_florence.mp3
Normal file
Binary file not shown.
BIN
samples/voice/xtts_02_daisy_studious.mp3
Normal file
BIN
samples/voice/xtts_02_daisy_studious.mp3
Normal file
Binary file not shown.
BIN
samples/voice/xtts_03_sofia_hellen.mp3
Normal file
BIN
samples/voice/xtts_03_sofia_hellen.mp3
Normal file
Binary file not shown.
BIN
samples/voice/xtts_04_alexandra_hisakawa.mp3
Normal file
BIN
samples/voice/xtts_04_alexandra_hisakawa.mp3
Normal file
Binary file not shown.
BIN
samples/voice/xtts_05_nova_hogarth.mp3
Normal file
BIN
samples/voice/xtts_05_nova_hogarth.mp3
Normal file
Binary file not shown.
BIN
samples/voice/xtts_06_rosemary_okafor.mp3
Normal file
BIN
samples/voice/xtts_06_rosemary_okafor.mp3
Normal file
Binary file not shown.
83
tests/gen_emotion_samples.py
Normal file
83
tests/gen_emotion_samples.py
Normal file
@@ -0,0 +1,83 @@
|
|||||||
|
"""One-off: synthesize a short Korean sample for every canonical emotion.
|
||||||
|
|
||||||
|
Each clip announces the emotion name at neutral (base) speed, then speaks a
|
||||||
|
sample sentence steered by that emotion's ``[태그]`` — exactly the path the live
|
||||||
|
voice server uses. Writes one wav per emotion into an output directory (plus a
|
||||||
|
stitched all-in-one) so each emotion can be auditioned separately. Run with the
|
||||||
|
orchestrator venv:
|
||||||
|
|
||||||
|
.venv/bin/python -m tests.gen_emotion_samples /abs/out_dir
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import sys
|
||||||
|
import wave
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from wsai.backends.melo import MeloTTS
|
||||||
|
|
||||||
|
# (canonical english for filename, announced Korean name, emotion tag, sample)
|
||||||
|
SAMPLES: list[tuple[str, str, str, str]] = [
|
||||||
|
("base", "기본", "기본", "이건 기본 목소리예요, 감정 없이 이렇게 말해요."),
|
||||||
|
("happy", "기쁨", "기쁨", "오늘은 정말 기분 좋은 하루예요!"),
|
||||||
|
("excited", "신남", "신남", "우와, 이거 진짜 신난다! 빨리 하자!"),
|
||||||
|
("hopeful", "희망", "희망", "우리 분명히 잘 해낼 수 있어요!"),
|
||||||
|
("sad", "슬픔", "슬픔", "조금 속상한 일이 있었어요."),
|
||||||
|
("angry", "화남", "화남", "정말 너무하잖아요, 화가 나요."),
|
||||||
|
("fearful", "두려움", "두려움", "어떡하지, 너무 무서워요."),
|
||||||
|
("surprised", "놀람", "놀람", "어머, 이게 정말이에요?"),
|
||||||
|
("disgust", "혐오", "혐오", "으, 이건 좀 별로예요."),
|
||||||
|
("calm", "차분", "차분", "천천히 하나씩 정리해 볼게요."),
|
||||||
|
("friendly", "다정", "다정", "언제든지 편하게 말해 주세요."),
|
||||||
|
("serious", "진지", "진지", "이건 정말 중요한 이야기예요."),
|
||||||
|
("disappointed", "실망", "실망", "조금 아쉬운 결과네요."),
|
||||||
|
("tired", "피곤", "피곤", "아, 오늘 너무 피곤하네요."),
|
||||||
|
("affectionate", "사랑스럽게", "사랑스럽게", "당신은 정말 소중한 사람이에요."),
|
||||||
|
("playful", "장난스럽게", "장난스럽게", "히히, 한번 맞혀 보세요!"),
|
||||||
|
("curious", "궁금", "궁금", "그건 대체 왜 그런 걸까요?"),
|
||||||
|
("whisper", "속삭임", "속삭임", "조용히, 우리끼리만 아는 비밀이에요."),
|
||||||
|
("shout", "외침", "외침", "다 같이 힘내자, 파이팅!"),
|
||||||
|
("determined", "단호", "단호", "이번엔 반드시 해내겠어요."),
|
||||||
|
("relieved", "안도", "안도", "휴, 이제야 마음이 놓이네요."),
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def stitch(paths: list[str], out: str, gap_s: float = 0.45) -> None:
|
||||||
|
with wave.open(paths[0], "rb") as w0:
|
||||||
|
nch, sw, fr = w0.getnchannels(), w0.getsampwidth(), w0.getframerate()
|
||||||
|
silence = b"\x00" * (int(fr * gap_s) * sw * nch)
|
||||||
|
with wave.open(out, "wb") as wo:
|
||||||
|
wo.setnchannels(nch)
|
||||||
|
wo.setsampwidth(sw)
|
||||||
|
wo.setframerate(fr)
|
||||||
|
for i, p in enumerate(paths):
|
||||||
|
with wave.open(p, "rb") as w:
|
||||||
|
wo.writeframes(w.readframes(w.getnframes()))
|
||||||
|
if i < len(paths) - 1:
|
||||||
|
wo.writeframes(silence)
|
||||||
|
|
||||||
|
|
||||||
|
async def main(out_dir: str) -> None:
|
||||||
|
d = Path(out_dir)
|
||||||
|
d.mkdir(parents=True, exist_ok=True)
|
||||||
|
tts = MeloTTS() # base speed comes from WSAI_TTS_SPEED default
|
||||||
|
await tts.warmup()
|
||||||
|
print(f"melo ready ({tts.load_ms} ms), base speed {tts.speed}")
|
||||||
|
paths: list[str] = []
|
||||||
|
for i, (canon, name, tag, sample) in enumerate(SAMPLES, 1):
|
||||||
|
text = f"{name}. [{tag}] {sample}"
|
||||||
|
src = await tts.synth(text)
|
||||||
|
dst = d / f"emo_{i:02d}_{canon}.wav"
|
||||||
|
Path(src).replace(dst)
|
||||||
|
paths.append(str(dst))
|
||||||
|
print(f" {name:8s} -> {dst}")
|
||||||
|
await tts.aclose()
|
||||||
|
stitch(paths, str(d / "emotion_samples_all.wav"))
|
||||||
|
print(f"wrote {len(paths)} per-emotion wavs + stitched all -> {d}")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
out_dir = sys.argv[1] if len(sys.argv) > 1 else "/tmp/emotion_samples"
|
||||||
|
asyncio.run(main(out_dir))
|
||||||
@@ -7,6 +7,7 @@ behaves with no source wired yet.
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
import asyncio
|
import asyncio
|
||||||
|
import json
|
||||||
from typing import AsyncIterator
|
from typing import AsyncIterator
|
||||||
|
|
||||||
from wsai.backends.whisper import WhisperSTT
|
from wsai.backends.whisper import WhisperSTT
|
||||||
@@ -58,3 +59,75 @@ def test_empty_transcript_is_skipped(monkeypatch):
|
|||||||
|
|
||||||
utts = _collect(stt)
|
utts = _collect(stt)
|
||||||
assert [u.text for u in utts] == ["안녕"]
|
assert [u.text for u in utts] == ["안녕"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_request_during_warmup_does_not_overlap_stdout(monkeypatch):
|
||||||
|
"""Regression: a transcribe() arriving while warmup() is still awaiting the
|
||||||
|
worker's ready line must NOT read the same stdout StreamReader concurrently.
|
||||||
|
|
||||||
|
Before the fix, _ensure()'s fast path returned as soon as the subprocess was
|
||||||
|
spawned (proc set, returncode None) even though the ready handshake was still
|
||||||
|
in flight, so the request's stdout.readline() overlapped warmup's and asyncio
|
||||||
|
raised "readuntil() called while another coroutine is already waiting for
|
||||||
|
incoming data" — the exact crash seen in the Discord voice server."""
|
||||||
|
|
||||||
|
async def run():
|
||||||
|
stt = WhisperSTT()
|
||||||
|
stdout = asyncio.StreamReader()
|
||||||
|
stderr = asyncio.StreamReader()
|
||||||
|
stderr.feed_eof() # nothing on stderr; let the drain task finish cleanly
|
||||||
|
|
||||||
|
class FakeStdin:
|
||||||
|
def write(self, _b):
|
||||||
|
pass
|
||||||
|
|
||||||
|
async def drain(self):
|
||||||
|
pass
|
||||||
|
|
||||||
|
class FakeProc:
|
||||||
|
returncode = None
|
||||||
|
|
||||||
|
def __init__(self):
|
||||||
|
self.stdin = FakeStdin()
|
||||||
|
self.stdout = stdout
|
||||||
|
self.stderr = stderr
|
||||||
|
|
||||||
|
def terminate(self):
|
||||||
|
self.returncode = 0
|
||||||
|
|
||||||
|
async def wait(self):
|
||||||
|
return 0
|
||||||
|
|
||||||
|
spawns = []
|
||||||
|
|
||||||
|
async def fake_create(*_a, **_k):
|
||||||
|
spawns.append(1)
|
||||||
|
return FakeProc()
|
||||||
|
|
||||||
|
monkeypatch.setattr(asyncio, "create_subprocess_exec", fake_create)
|
||||||
|
|
||||||
|
# warmup enters _ensure and blocks awaiting the ready line on stdout.
|
||||||
|
warm = asyncio.create_task(stt.warmup())
|
||||||
|
await asyncio.sleep(0.05)
|
||||||
|
|
||||||
|
# A concurrent request lands mid-warmup. It must wait for readiness, not
|
||||||
|
# crash and not read stdout yet.
|
||||||
|
tr = asyncio.create_task(stt.transcribe("x.wav"))
|
||||||
|
await asyncio.sleep(0.05)
|
||||||
|
assert not tr.done() # blocked on the start lock, no overlapping read
|
||||||
|
|
||||||
|
# Complete the handshake -> warmup finishes and releases the request.
|
||||||
|
stdout.feed_data(
|
||||||
|
(json.dumps({"ready": True, "ms": 1, "device": "cpu"}) + "\n").encode()
|
||||||
|
)
|
||||||
|
await asyncio.wait_for(warm, timeout=1)
|
||||||
|
await asyncio.sleep(0.02)
|
||||||
|
stdout.feed_data(
|
||||||
|
(json.dumps({"ok": True, "text": "안녕", "ms": 2}) + "\n").encode()
|
||||||
|
)
|
||||||
|
assert await asyncio.wait_for(tr, timeout=1) == "안녕"
|
||||||
|
assert sum(spawns) == 1 # one worker, not one-per-concurrent-caller
|
||||||
|
|
||||||
|
await stt.aclose()
|
||||||
|
|
||||||
|
asyncio.run(asyncio.wait_for(run(), timeout=5))
|
||||||
|
|||||||
@@ -23,27 +23,34 @@ from dataclasses import dataclass
|
|||||||
|
|
||||||
# Canonical emotion -> (speed multiplier relative to base, pitch shift in semitones).
|
# Canonical emotion -> (speed multiplier relative to base, pitch shift in semitones).
|
||||||
# Kept deliberately modest so delivery stays natural, not cartoonish.
|
# Kept deliberately modest so delivery stays natural, not cartoonish.
|
||||||
|
#
|
||||||
|
# Pitch is held at 0.0 for every emotion: librosa's post-hoc pitch_shift on Melo
|
||||||
|
# output produced a robotic, "monster"-sounding artefact (worse when stacked on a
|
||||||
|
# fast base speed). Emotion is therefore conveyed by speed only — natural and
|
||||||
|
# artefact-free. The pitch column is retained (rather than removed) so the effect
|
||||||
|
# can be re-enabled per-emotion later with a real, artefact-free pitch method.
|
||||||
EMOTION_PARAMS: dict[str, tuple[float, float]] = {
|
EMOTION_PARAMS: dict[str, tuple[float, float]] = {
|
||||||
"happy": (1.08, 2.0), # 기쁨 / cheerful
|
"base": (1.00, 0.0), # 기본 / neutral — the plain base voice, no colour
|
||||||
"excited": (1.15, 3.0), # 신남 / excited
|
"happy": (1.08, 0.0), # 기쁨 / cheerful
|
||||||
"hopeful": (1.10, 1.5), # 희망 / 힘차게
|
"excited": (1.15, 0.0), # 신남 / excited
|
||||||
"sad": (0.90, -2.5), # 슬픔 / sad
|
"hopeful": (1.10, 0.0), # 희망 / 힘차게
|
||||||
"angry": (1.12, 1.0), # 화남 / angry
|
"sad": (0.90, 0.0), # 슬픔 / sad
|
||||||
"fearful": (1.12, 2.0), # 두려움 / terrified
|
"angry": (1.12, 0.0), # 화남 / angry
|
||||||
"surprised": (1.05, 3.0), # 놀람 / surprise
|
"fearful": (1.12, 0.0), # 두려움 / terrified
|
||||||
"disgust": (0.96, -1.0), # 혐오 / disgust
|
"surprised": (1.05, 0.0), # 놀람 / surprise
|
||||||
"calm": (0.95, -1.0), # 차분 / calm
|
"disgust": (0.96, 0.0), # 혐오 / disgust
|
||||||
"friendly": (1.00, 1.0), # 다정 / friendly
|
"calm": (0.95, 0.0), # 차분 / calm
|
||||||
"serious": (0.97, -1.0), # 진지 / serious
|
"friendly": (1.00, 0.0), # 다정 / friendly
|
||||||
"disappointed":(0.92, -2.0), # 실망 / disappointed
|
"serious": (0.97, 0.0), # 진지 / serious
|
||||||
"tired": (0.90, -2.0), # 피곤 / 지침
|
"disappointed":(0.92, 0.0), # 실망 / disappointed
|
||||||
"affectionate":(0.98, 1.0), # 사랑스럽게 / affectionate
|
"tired": (0.90, 0.0), # 피곤 / 지침
|
||||||
"playful": (1.08, 2.0), # 장난스럽게 / playful
|
"affectionate":(0.98, 0.0), # 사랑스럽게 / affectionate
|
||||||
"whisper": (0.92, -1.5), # 속삭임 / whispering
|
"playful": (1.08, 0.0), # 장난스럽게 / playful
|
||||||
"shout": (1.05, 2.5), # 외침 / shouting
|
"whisper": (0.92, 0.0), # 속삭임 / whispering
|
||||||
"determined": (1.05, 0.5), # 단호 / determined
|
"shout": (1.05, 0.0), # 외침 / shouting
|
||||||
"relieved": (0.95, 0.5), # 안도 / relieved
|
"determined": (1.05, 0.0), # 단호 / determined
|
||||||
"curious": (1.03, 1.5), # 궁금 / curious
|
"relieved": (0.95, 0.0), # 안도 / relieved
|
||||||
|
"curious": (1.03, 0.0), # 궁금 / curious
|
||||||
}
|
}
|
||||||
|
|
||||||
# Every spelling Claude might realistically emit, mapped to a canonical emotion.
|
# Every spelling Claude might realistically emit, mapped to a canonical emotion.
|
||||||
@@ -63,6 +70,7 @@ def _norm(word: str) -> str:
|
|||||||
return re.sub(r"\s+", "", word).lower()
|
return re.sub(r"\s+", "", word).lower()
|
||||||
|
|
||||||
|
|
||||||
|
_register("base", "기본", "기본목소리", "기본톤", "보통", "평범", "무감정", "default", "neutral", "normal", "plain")
|
||||||
_register("happy", "기쁨", "기쁘게", "기뻐", "기뻐하며", "행복", "행복하게", "행복하게도", "즐겁게", "즐거움", "밝게", "반가움", "반갑게", "반가워", "cheerful", "happy", "joyful")
|
_register("happy", "기쁨", "기쁘게", "기뻐", "기뻐하며", "행복", "행복하게", "행복하게도", "즐겁게", "즐거움", "밝게", "반가움", "반갑게", "반가워", "cheerful", "happy", "joyful")
|
||||||
_register("excited", "신남", "신나게", "신나서", "흥분", "들뜬", "들떠서", "설렘", "설레며", "excited", "thrilled")
|
_register("excited", "신남", "신나게", "신나서", "흥분", "들뜬", "들떠서", "설렘", "설레며", "excited", "thrilled")
|
||||||
_register("hopeful", "희망", "희망차게", "힘차게", "힘내", "힘내서", "응원", "응원하며", "격려", "hopeful", "encouraging")
|
_register("hopeful", "희망", "희망차게", "힘차게", "힘내", "힘내서", "응원", "응원하며", "격려", "hopeful", "encouraging")
|
||||||
|
|||||||
@@ -12,7 +12,7 @@ Env:
|
|||||||
WSAI_MELO_DEVICE cpu | cuda | auto (default auto: GPU if torch sees one,
|
WSAI_MELO_DEVICE cpu | cuda | auto (default auto: GPU if torch sees one,
|
||||||
else CPU; the worker falls back to CPU if CUDA fails)
|
else CPU; the worker falls back to CPU if CUDA fails)
|
||||||
WSAI_TTS_OUT_DIR where wavs are written (default ~/.cache/wsai/tts)
|
WSAI_TTS_OUT_DIR where wavs are written (default ~/.cache/wsai/tts)
|
||||||
WSAI_TTS_SPEED synthesis speed multiplier (default 1.3)
|
WSAI_TTS_SPEED synthesis speed multiplier (default 1.2)
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
@@ -93,10 +93,15 @@ class MeloTTS:
|
|||||||
self.out_dir = Path(out_dir or os.environ.get("WSAI_TTS_OUT_DIR")
|
self.out_dir = Path(out_dir or os.environ.get("WSAI_TTS_OUT_DIR")
|
||||||
or (Path.home() / ".cache/wsai/tts"))
|
or (Path.home() / ".cache/wsai/tts"))
|
||||||
self.speed = float(speed if speed is not None
|
self.speed = float(speed if speed is not None
|
||||||
else os.environ.get("WSAI_TTS_SPEED", "1.3"))
|
else os.environ.get("WSAI_TTS_SPEED", "1.2"))
|
||||||
self.sink = sink or _log_sink
|
self.sink = sink or _log_sink
|
||||||
self._proc: asyncio.subprocess.Process | None = None
|
self._proc: asyncio.subprocess.Process | None = None
|
||||||
self._lock = asyncio.Lock()
|
self._lock = asyncio.Lock()
|
||||||
|
# Serialises worker (re)start + the ready handshake so a caller that
|
||||||
|
# arrives mid-warmup waits for readiness instead of reading the same
|
||||||
|
# stdout StreamReader concurrently (asyncio forbids overlapping reads).
|
||||||
|
self._start_lock = asyncio.Lock()
|
||||||
|
self._ready = False # True only after the ready handshake completes
|
||||||
self._n = 0
|
self._n = 0
|
||||||
self.load_ms: int | None = None
|
self.load_ms: int | None = None
|
||||||
# Keep the worker's most recent stderr lines so a crash reports its real
|
# Keep the worker's most recent stderr lines so a crash reports its real
|
||||||
@@ -132,42 +137,53 @@ class MeloTTS:
|
|||||||
await self._ensure()
|
await self._ensure()
|
||||||
|
|
||||||
async def _ensure(self) -> None:
|
async def _ensure(self) -> None:
|
||||||
if self._proc is not None and self._proc.returncode is None:
|
# Fast path: only skip when the worker is not just spawned but fully
|
||||||
|
# handshaked. Checking `_proc` alone would let a caller sail past while
|
||||||
|
# another coroutine (e.g. warmup) is still awaiting the ready line on
|
||||||
|
# this same stdout, causing overlapping StreamReader reads.
|
||||||
|
if self._proc is not None and self._proc.returncode is None and self._ready:
|
||||||
return
|
return
|
||||||
self.out_dir.mkdir(parents=True, exist_ok=True)
|
async with self._start_lock:
|
||||||
env = {**os.environ, "WSAI_MELO_DEVICE": self.device}
|
# Re-check under the lock: another coroutine may have finished the
|
||||||
# Run the worker module from the wsai source tree with the melo venv.
|
# (re)start + handshake while we waited.
|
||||||
repo_root = str(Path(__file__).resolve().parents[2])
|
if self._proc is not None and self._proc.returncode is None and self._ready:
|
||||||
self._proc = await asyncio.create_subprocess_exec(
|
return
|
||||||
self.python, "-m", "wsai.backends.melo_worker",
|
self._ready = False
|
||||||
cwd=repo_root, env=env,
|
self.out_dir.mkdir(parents=True, exist_ok=True)
|
||||||
stdin=asyncio.subprocess.PIPE,
|
env = {**os.environ, "WSAI_MELO_DEVICE": self.device}
|
||||||
stdout=asyncio.subprocess.PIPE,
|
# Run the worker module from the wsai source tree with the melo venv.
|
||||||
stderr=asyncio.subprocess.PIPE,
|
repo_root = str(Path(__file__).resolve().parents[2])
|
||||||
)
|
self._proc = await asyncio.create_subprocess_exec(
|
||||||
self._stderr_tail.clear()
|
self.python, "-m", "wsai.backends.melo_worker",
|
||||||
assert self._proc.stderr is not None
|
cwd=repo_root, env=env,
|
||||||
self._stderr_task = asyncio.create_task(self._drain_stderr(self._proc.stderr))
|
stdin=asyncio.subprocess.PIPE,
|
||||||
ready = await self._proc.stdout.readline()
|
stdout=asyncio.subprocess.PIPE,
|
||||||
if not ready: # worker died before signalling ready
|
stderr=asyncio.subprocess.PIPE,
|
||||||
await self._proc.wait()
|
|
||||||
raise RuntimeError(
|
|
||||||
f"melo worker exited before ready (code {self._proc.returncode})."
|
|
||||||
f"{self._stderr_hint()}"
|
|
||||||
)
|
)
|
||||||
try:
|
self._stderr_tail.clear()
|
||||||
info = json.loads(ready.decode())
|
assert self._proc.stderr is not None
|
||||||
except json.JSONDecodeError as exc:
|
self._stderr_task = asyncio.create_task(self._drain_stderr(self._proc.stderr))
|
||||||
raise RuntimeError(
|
ready = await self._proc.stdout.readline()
|
||||||
f"melo worker sent invalid ready line {ready!r}: {exc}."
|
if not ready: # worker died before signalling ready
|
||||||
f"{self._stderr_hint()}"
|
await self._proc.wait()
|
||||||
) from exc
|
raise RuntimeError(
|
||||||
if not info.get("ready"):
|
f"melo worker exited before ready (code {self._proc.returncode})."
|
||||||
raise RuntimeError(
|
f"{self._stderr_hint()}"
|
||||||
f"melo worker failed to start: {info}.{self._stderr_hint()}"
|
)
|
||||||
)
|
try:
|
||||||
self.load_ms = info.get("ms")
|
info = json.loads(ready.decode())
|
||||||
log.info("melo worker ready in %s ms on %s", self.load_ms, info.get("device"))
|
except json.JSONDecodeError as exc:
|
||||||
|
raise RuntimeError(
|
||||||
|
f"melo worker sent invalid ready line {ready!r}: {exc}."
|
||||||
|
f"{self._stderr_hint()}"
|
||||||
|
) from exc
|
||||||
|
if not info.get("ready"):
|
||||||
|
raise RuntimeError(
|
||||||
|
f"melo worker failed to start: {info}.{self._stderr_hint()}"
|
||||||
|
)
|
||||||
|
self.load_ms = info.get("ms")
|
||||||
|
self._ready = True
|
||||||
|
log.info("melo worker ready in %s ms on %s", self.load_ms, info.get("device"))
|
||||||
|
|
||||||
async def synth(self, text: str) -> str:
|
async def synth(self, text: str) -> str:
|
||||||
"""Synthesize `text` to a wav and return its path (no sink). Reusable by
|
"""Synthesize `text` to a wav and return its path (no sink). Reusable by
|
||||||
@@ -223,3 +239,4 @@ class MeloTTS:
|
|||||||
pass
|
pass
|
||||||
self._stderr_task = None
|
self._stderr_task = None
|
||||||
self._proc = None
|
self._proc = None
|
||||||
|
self._ready = False
|
||||||
|
|||||||
@@ -75,6 +75,11 @@ class WhisperSTT:
|
|||||||
self.audio_source = audio_source
|
self.audio_source = audio_source
|
||||||
self._proc: asyncio.subprocess.Process | None = None
|
self._proc: asyncio.subprocess.Process | None = None
|
||||||
self._lock = asyncio.Lock()
|
self._lock = asyncio.Lock()
|
||||||
|
# Serialises worker (re)start + the ready handshake so a caller that
|
||||||
|
# arrives mid-warmup waits for readiness instead of reading the same
|
||||||
|
# stdout StreamReader concurrently (asyncio forbids overlapping reads).
|
||||||
|
self._start_lock = asyncio.Lock()
|
||||||
|
self._ready = False # True only after the ready handshake completes
|
||||||
self.load_ms: int | None = None
|
self.load_ms: int | None = None
|
||||||
self.resolved_device: str | None = None # "cuda" | "cpu", known after start
|
self.resolved_device: str | None = None # "cuda" | "cpu", known after start
|
||||||
# Keep the worker's most recent stderr so a crash reports its real cause
|
# Keep the worker's most recent stderr so a crash reports its real cause
|
||||||
@@ -106,59 +111,70 @@ class WhisperSTT:
|
|||||||
await self._ensure()
|
await self._ensure()
|
||||||
|
|
||||||
async def _ensure(self) -> None:
|
async def _ensure(self) -> None:
|
||||||
if self._proc is not None and self._proc.returncode is None:
|
# Fast path: only skip when the worker is not just spawned but fully
|
||||||
|
# handshaked. Checking `_proc` alone would let a caller sail past while
|
||||||
|
# another coroutine (e.g. warmup) is still awaiting the ready line on
|
||||||
|
# this same stdout, causing overlapping StreamReader reads.
|
||||||
|
if self._proc is not None and self._proc.returncode is None and self._ready:
|
||||||
return
|
return
|
||||||
env = {
|
async with self._start_lock:
|
||||||
**os.environ,
|
# Re-check under the lock: another coroutine may have finished the
|
||||||
"WSAI_WHISPER_MODEL": self.model,
|
# (re)start + handshake while we waited.
|
||||||
"WSAI_WHISPER_DEVICE": self.device,
|
if self._proc is not None and self._proc.returncode is None and self._ready:
|
||||||
}
|
return
|
||||||
# ctranslate2 dlopens libcublas/libcudnn from the whisper venv's nvidia
|
self._ready = False
|
||||||
# pip packages; the dynamic loader only honours LD_LIBRARY_PATH captured
|
env = {
|
||||||
# at exec, so inject those lib dirs into the child env here (harmless on
|
**os.environ,
|
||||||
# CPU). Without this the CUDA model loads but transcribe() dies with
|
"WSAI_WHISPER_MODEL": self.model,
|
||||||
# "Library libcublas.so.12 is not found".
|
"WSAI_WHISPER_DEVICE": self.device,
|
||||||
lib_dirs = _cuda_lib_dirs(self.python)
|
}
|
||||||
if lib_dirs:
|
# ctranslate2 dlopens libcublas/libcudnn from the whisper venv's nvidia
|
||||||
prev = env.get("LD_LIBRARY_PATH", "")
|
# pip packages; the dynamic loader only honours LD_LIBRARY_PATH captured
|
||||||
env["LD_LIBRARY_PATH"] = ":".join(lib_dirs + ([prev] if prev else []))
|
# at exec, so inject those lib dirs into the child env here (harmless on
|
||||||
if self.language:
|
# CPU). Without this the CUDA model loads but transcribe() dies with
|
||||||
env["WSAI_WHISPER_LANGUAGE"] = self.language
|
# "Library libcublas.so.12 is not found".
|
||||||
repo_root = str(Path(__file__).resolve().parents[2])
|
lib_dirs = _cuda_lib_dirs(self.python)
|
||||||
self._proc = await asyncio.create_subprocess_exec(
|
if lib_dirs:
|
||||||
self.python, "-m", "wsai.backends.whisper_worker",
|
prev = env.get("LD_LIBRARY_PATH", "")
|
||||||
cwd=repo_root, env=env,
|
env["LD_LIBRARY_PATH"] = ":".join(lib_dirs + ([prev] if prev else []))
|
||||||
stdin=asyncio.subprocess.PIPE,
|
if self.language:
|
||||||
stdout=asyncio.subprocess.PIPE,
|
env["WSAI_WHISPER_LANGUAGE"] = self.language
|
||||||
stderr=asyncio.subprocess.PIPE,
|
repo_root = str(Path(__file__).resolve().parents[2])
|
||||||
)
|
self._proc = await asyncio.create_subprocess_exec(
|
||||||
self._stderr_tail.clear()
|
self.python, "-m", "wsai.backends.whisper_worker",
|
||||||
assert self._proc.stderr is not None
|
cwd=repo_root, env=env,
|
||||||
self._stderr_task = asyncio.create_task(self._drain_stderr(self._proc.stderr))
|
stdin=asyncio.subprocess.PIPE,
|
||||||
ready = await self._proc.stdout.readline()
|
stdout=asyncio.subprocess.PIPE,
|
||||||
if not ready: # worker died before signalling ready
|
stderr=asyncio.subprocess.PIPE,
|
||||||
await self._proc.wait()
|
|
||||||
raise RuntimeError(
|
|
||||||
f"whisper worker exited before ready (code {self._proc.returncode})."
|
|
||||||
f"{self._stderr_hint()}"
|
|
||||||
)
|
)
|
||||||
try:
|
self._stderr_tail.clear()
|
||||||
info = json.loads(ready.decode())
|
assert self._proc.stderr is not None
|
||||||
except json.JSONDecodeError as exc:
|
self._stderr_task = asyncio.create_task(self._drain_stderr(self._proc.stderr))
|
||||||
raise RuntimeError(
|
ready = await self._proc.stdout.readline()
|
||||||
f"whisper worker sent invalid ready line {ready!r}: {exc}."
|
if not ready: # worker died before signalling ready
|
||||||
f"{self._stderr_hint()}"
|
await self._proc.wait()
|
||||||
) from exc
|
raise RuntimeError(
|
||||||
if not info.get("ready"):
|
f"whisper worker exited before ready (code {self._proc.returncode})."
|
||||||
raise RuntimeError(
|
f"{self._stderr_hint()}"
|
||||||
f"whisper worker failed to start: {info}.{self._stderr_hint()}"
|
)
|
||||||
|
try:
|
||||||
|
info = json.loads(ready.decode())
|
||||||
|
except json.JSONDecodeError as exc:
|
||||||
|
raise RuntimeError(
|
||||||
|
f"whisper worker sent invalid ready line {ready!r}: {exc}."
|
||||||
|
f"{self._stderr_hint()}"
|
||||||
|
) from exc
|
||||||
|
if not info.get("ready"):
|
||||||
|
raise RuntimeError(
|
||||||
|
f"whisper worker failed to start: {info}.{self._stderr_hint()}"
|
||||||
|
)
|
||||||
|
self.load_ms = info.get("ms")
|
||||||
|
self.resolved_device = info.get("device")
|
||||||
|
self._ready = True
|
||||||
|
log.info(
|
||||||
|
"whisper worker ready in %s ms on %s (model %s)",
|
||||||
|
self.load_ms, info.get("device"), info.get("model"),
|
||||||
)
|
)
|
||||||
self.load_ms = info.get("ms")
|
|
||||||
self.resolved_device = info.get("device")
|
|
||||||
log.info(
|
|
||||||
"whisper worker ready in %s ms on %s (model %s)",
|
|
||||||
self.load_ms, info.get("device"), info.get("model"),
|
|
||||||
)
|
|
||||||
|
|
||||||
async def transcribe(self, wav_path: str, *, language: str | None = None) -> str:
|
async def transcribe(self, wav_path: str, *, language: str | None = None) -> str:
|
||||||
"""Transcribe one wav file to text using the warm worker."""
|
"""Transcribe one wav file to text using the warm worker."""
|
||||||
@@ -211,3 +227,4 @@ class WhisperSTT:
|
|||||||
pass
|
pass
|
||||||
self._stderr_task = None
|
self._stderr_task = None
|
||||||
self._proc = None
|
self._proc = None
|
||||||
|
self._ready = False
|
||||||
|
|||||||
Reference in New Issue
Block a user