Files
tts_site/app/engines/melo_engine.py
claude 91f8ec3533 refactor: 다중 엔진 레지스트리 구조로 전환 + 프론트 엔진 선택 UI
- app/engines: BaseEngine 인터페이스, MeloEngine(인프로세스), 레지스트리
- /api/engines 추가, /api/tts 에 engine 파라미터
- 프론트: 엔진 → 언어 → 목소리 선택, 라이선스/GPU/조절 지원 표기
- MeloTTS 동작 불변
2026-08-23 22:21:26 +09:00

141 lines
4.4 KiB
Python

"""MeloTTS 엔진 (in-process). 한국어/영어, GPU 자동 사용, 속도/피치 지원."""
from __future__ import annotations
import io
import threading
import numpy as np
import soundfile as sf
import torch
try:
import librosa # 피치 조절용
except Exception: # pragma: no cover
librosa = None
from melo.api import TTS
from .base import BaseEngine
# 화면에 노출할 언어/화자 정의 (중국어/일본어는 제외)
VOICES: dict[str, dict] = {
"KR": {
"label": "한국어",
"melo_lang": "KR",
"speakers": [
{"id": "KR", "label": "한국어 기본 목소리"},
],
},
"EN": {
"label": "English (영어)",
"melo_lang": "EN",
"speakers": [
{"id": "EN-US", "label": "영어 · 미국식"},
{"id": "EN-BR", "label": "영어 · 영국식"},
{"id": "EN_INDIA", "label": "영어 · 인도식"},
{"id": "EN-AU", "label": "영어 · 호주식"},
{"id": "EN-Default", "label": "영어 · 기본"},
],
},
}
def _pick_device() -> str:
return "cuda:0" if torch.cuda.is_available() else "cpu"
class MeloEngine(BaseEngine):
id = "melo"
label = "MeloTTS"
license = "MIT"
uses_gpu = True
notes = "자연스러운 한국어/영어 음성. 속도·피치 조절 지원."
def __init__(self) -> None:
self.device = _pick_device()
self._models: dict[str, TTS] = {}
self._locks: dict[str, threading.Lock] = {k: threading.Lock() for k in VOICES}
self._global_lock = threading.Lock()
# ---- 모델 로딩 -------------------------------------------------
def _get_model(self, language: str) -> TTS:
if language not in VOICES:
raise ValueError(f"지원하지 않는 언어입니다: {language}")
model = self._models.get(language)
if model is not None:
return model
with self._locks[language]:
model = self._models.get(language)
if model is None:
melo_lang = VOICES[language]["melo_lang"]
model = TTS(language=melo_lang, device=self.device)
self._models[language] = model
return model
def warmup(self) -> None:
for lang in VOICES:
try:
self._get_model(lang)
except Exception as exc: # pragma: no cover
print(f"[melo warmup] {lang} 모델 로드 실패: {exc}")
def _list_languages(self) -> list[dict]:
return [
{"code": code, "label": info["label"], "speakers": info["speakers"]}
for code, info in VOICES.items()
]
def describe(self) -> dict:
return {
"id": self.id,
"label": self.label,
"license": self.license,
"uses_gpu": self.uses_gpu,
"notes": self.notes,
"languages": self._list_languages(),
"supports": {"speed": True, "pitch": True},
}
def _resolve_speaker(self, model: TTS, language: str, speaker: str | None) -> int:
spk2id = model.hps.data.spk2id
if speaker and speaker in spk2id:
return spk2id[speaker]
default_id = VOICES[language]["speakers"][0]["id"]
if default_id in spk2id:
return spk2id[default_id]
return list(spk2id.values())[0]
# ---- 합성 -----------------------------------------------------
def synth_wav(
self,
text: str,
language: str,
speaker: str | None = None,
speed: float = 1.0,
pitch: float = 0.0,
) -> bytes:
text = (text or "").strip()
if not text:
raise ValueError("텍스트가 비어 있습니다.")
speed = float(max(0.5, min(2.0, speed)))
pitch = float(max(-12.0, min(12.0, pitch)))
model = self._get_model(language)
speaker_id = self._resolve_speaker(model, language, speaker)
with self._global_lock:
audio = model.tts_to_file(
text, speaker_id, output_path=None, speed=speed, quiet=True
)
sr = model.hps.data.sampling_rate
audio = np.asarray(audio, dtype=np.float32)
if abs(pitch) > 1e-3 and librosa is not None:
audio = librosa.effects.pitch_shift(audio, sr, n_steps=pitch)
buf = io.BytesIO()
sf.write(buf, audio, sr, format="WAV", subtype="PCM_16")
buf.seek(0)
return buf.read()