한국어 우선 MeloTTS 웹 서비스 추가 (GPU, 속도/피치, 로그인 없음)
- FastAPI 백엔드: /api/voices, /api/tts, /api/health - 한국어 메인 + 영어(5종 악센트), 중국어/일본어 UI 제거 - 언어별 목소리 모델 선택, 속도(0.5~2.0x)/피치(±12반음) 조절 - 로그인/사용제한/과금 유도 없음 (무제한 테스트) - NVIDIA GPU 자동 사용 (torch cu128, Blackwell sm_120) - 유저 친화 한국어 UI (frontend) - Docker/compose (GPU 예약), 8788 포트 - MeloTTS(순수 파이썬) 벤더링 + 검증된 버전 고정(constraints)
This commit is contained in:
139
app/tts_engine.py
Normal file
139
app/tts_engine.py
Normal file
@@ -0,0 +1,139 @@
|
||||
"""MeloTTS 래퍼 엔진.
|
||||
|
||||
- 언어(한국어/영어)별 모델을 lazy-load 하고 캐시한다.
|
||||
- GPU(CUDA)가 있으면 자동으로 사용한다.
|
||||
- 속도(speed)와 피치(pitch, 반음 단위)를 조절할 수 있다.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import threading
|
||||
|
||||
import numpy as np
|
||||
import soundfile as sf
|
||||
import torch
|
||||
|
||||
try:
|
||||
import librosa # 피치 조절용
|
||||
except Exception: # pragma: no cover
|
||||
librosa = None
|
||||
|
||||
from melo.api import TTS
|
||||
|
||||
|
||||
# 화면에 노출할 언어/화자 정의 (중국어/일본어는 제외)
|
||||
VOICES: dict[str, dict] = {
|
||||
"KR": {
|
||||
"label": "한국어",
|
||||
"melo_lang": "KR",
|
||||
"speakers": [
|
||||
{"id": "KR", "label": "한국어 기본 목소리"},
|
||||
],
|
||||
},
|
||||
"EN": {
|
||||
"label": "English (영어)",
|
||||
"melo_lang": "EN",
|
||||
"speakers": [
|
||||
{"id": "EN-US", "label": "영어 · 미국식"},
|
||||
{"id": "EN-BR", "label": "영어 · 영국식"},
|
||||
{"id": "EN_INDIA", "label": "영어 · 인도식"},
|
||||
{"id": "EN-AU", "label": "영어 · 호주식"},
|
||||
{"id": "EN-Default", "label": "영어 · 기본"},
|
||||
],
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _pick_device() -> str:
|
||||
return "cuda:0" if torch.cuda.is_available() else "cpu"
|
||||
|
||||
|
||||
class TTSEngine:
|
||||
def __init__(self) -> None:
|
||||
self.device = _pick_device()
|
||||
self._models: dict[str, TTS] = {}
|
||||
self._locks: dict[str, threading.Lock] = {k: threading.Lock() for k in VOICES}
|
||||
self._global_lock = threading.Lock()
|
||||
|
||||
# ---- 모델 로딩 -------------------------------------------------
|
||||
def _get_model(self, language: str) -> TTS:
|
||||
if language not in VOICES:
|
||||
raise ValueError(f"지원하지 않는 언어입니다: {language}")
|
||||
model = self._models.get(language)
|
||||
if model is not None:
|
||||
return model
|
||||
with self._locks[language]:
|
||||
model = self._models.get(language)
|
||||
if model is None:
|
||||
melo_lang = VOICES[language]["melo_lang"]
|
||||
model = TTS(language=melo_lang, device=self.device)
|
||||
self._models[language] = model
|
||||
return model
|
||||
|
||||
def warmup(self) -> None:
|
||||
"""서버 시작 시 모델을 미리 로드한다."""
|
||||
for lang in VOICES:
|
||||
try:
|
||||
self._get_model(lang)
|
||||
except Exception as exc: # pragma: no cover
|
||||
print(f"[warmup] {lang} 모델 로드 실패: {exc}")
|
||||
|
||||
def list_voices(self) -> list[dict]:
|
||||
out = []
|
||||
for code, info in VOICES.items():
|
||||
out.append(
|
||||
{
|
||||
"code": code,
|
||||
"label": info["label"],
|
||||
"speakers": info["speakers"],
|
||||
}
|
||||
)
|
||||
return out
|
||||
|
||||
def _resolve_speaker(self, model: TTS, language: str, speaker: str | None) -> int:
|
||||
spk2id = model.hps.data.spk2id
|
||||
if speaker and speaker in spk2id:
|
||||
return spk2id[speaker]
|
||||
# 기본값: 정의된 첫 화자
|
||||
default_id = VOICES[language]["speakers"][0]["id"]
|
||||
if default_id in spk2id:
|
||||
return spk2id[default_id]
|
||||
return list(spk2id.values())[0]
|
||||
|
||||
# ---- 합성 -----------------------------------------------------
|
||||
def synth_wav(
|
||||
self,
|
||||
text: str,
|
||||
language: str,
|
||||
speaker: str | None = None,
|
||||
speed: float = 1.0,
|
||||
pitch: float = 0.0,
|
||||
) -> bytes:
|
||||
text = (text or "").strip()
|
||||
if not text:
|
||||
raise ValueError("텍스트가 비어 있습니다.")
|
||||
|
||||
speed = float(max(0.5, min(2.0, speed)))
|
||||
pitch = float(max(-12.0, min(12.0, pitch)))
|
||||
|
||||
model = self._get_model(language)
|
||||
speaker_id = self._resolve_speaker(model, language, speaker)
|
||||
|
||||
# 동시 추론은 GPU 메모리 보호를 위해 직렬화
|
||||
with self._global_lock:
|
||||
audio = model.tts_to_file(
|
||||
text, speaker_id, output_path=None, speed=speed, quiet=True
|
||||
)
|
||||
sr = model.hps.data.sampling_rate
|
||||
|
||||
audio = np.asarray(audio, dtype=np.float32)
|
||||
if abs(pitch) > 1e-3 and librosa is not None:
|
||||
audio = librosa.effects.pitch_shift(audio, sr, n_steps=pitch)
|
||||
|
||||
buf = io.BytesIO()
|
||||
sf.write(buf, audio, sr, format="WAV", subtype="PCM_16")
|
||||
buf.seek(0)
|
||||
return buf.read()
|
||||
|
||||
|
||||
engine = TTSEngine()
|
||||
Reference in New Issue
Block a user