한국어 우선 MeloTTS 웹 서비스 추가 (GPU, 속도/피치, 로그인 없음)

- FastAPI 백엔드: /api/voices, /api/tts, /api/health
- 한국어 메인 + 영어(5종 악센트), 중국어/일본어 UI 제거
- 언어별 목소리 모델 선택, 속도(0.5~2.0x)/피치(±12반음) 조절
- 로그인/사용제한/과금 유도 없음 (무제한 테스트)
- NVIDIA GPU 자동 사용 (torch cu128, Blackwell sm_120)
- 유저 친화 한국어 UI (frontend)
- Docker/compose (GPU 예약), 8788 포트
- MeloTTS(순수 파이썬) 벤더링 + 검증된 버전 고정(constraints)
This commit is contained in:
claude
2026-08-23 21:57:48 +09:00
parent a6f3bac584
commit f197177ae1
77 changed files with 11621 additions and 1 deletions

0
app/__init__.py Normal file
View File

90
app/server.py Normal file
View File

@@ -0,0 +1,90 @@
"""한국어 우선 TTS 웹 서비스 (MeloTTS 기반).
- 로그인/과금 없이 무제한으로 사용 가능
- 한국어/영어 목소리 모델을 언어별로 선택
- 속도 / 피치 조절
- GPU 자동 사용
"""
from __future__ import annotations
import os
from fastapi import FastAPI, HTTPException
from fastapi.middleware.cors import CORSMiddleware
from fastapi.responses import FileResponse, Response
from fastapi.staticfiles import StaticFiles
from pydantic import BaseModel, Field
from .tts_engine import engine
FRONTEND_DIR = os.path.join(os.path.dirname(os.path.dirname(__file__)), "frontend")
app = FastAPI(title="한국어 TTS 스튜디오", version="1.0.0")
app.add_middleware(
CORSMiddleware,
allow_origins=["*"],
allow_methods=["*"],
allow_headers=["*"],
)
class TTSRequest(BaseModel):
text: str = Field(..., description="읽을 텍스트")
language: str = Field("KR", description="언어 코드 (KR / EN)")
speaker: str | None = Field(None, description="화자 ID")
speed: float = Field(1.0, ge=0.5, le=2.0, description="말하기 속도")
pitch: float = Field(0.0, ge=-12.0, le=12.0, description="피치(반음)")
@app.on_event("startup")
def _startup() -> None:
# 첫 요청 지연을 줄이기 위해 백그라운드가 아닌 즉시 워밍업
engine.warmup()
@app.get("/api/health")
def health() -> dict:
import torch
return {
"status": "ok",
"device": engine.device,
"cuda": torch.cuda.is_available(),
"gpu": torch.cuda.get_device_name(0) if torch.cuda.is_available() else None,
}
@app.get("/api/voices")
def voices() -> dict:
return {"languages": engine.list_voices()}
@app.post("/api/tts")
def tts(req: TTSRequest) -> Response:
try:
wav = engine.synth_wav(
text=req.text,
language=req.language,
speaker=req.speaker,
speed=req.speed,
pitch=req.pitch,
)
except ValueError as exc:
raise HTTPException(status_code=400, detail=str(exc))
except Exception as exc: # pragma: no cover
raise HTTPException(status_code=500, detail=f"합성 실패: {exc}")
return Response(
content=wav,
media_type="audio/wav",
headers={"Content-Disposition": 'inline; filename="tts.wav"'},
)
@app.get("/")
def index() -> FileResponse:
return FileResponse(os.path.join(FRONTEND_DIR, "index.html"))
app.mount("/static", StaticFiles(directory=FRONTEND_DIR), name="static")

139
app/tts_engine.py Normal file
View File

@@ -0,0 +1,139 @@
"""MeloTTS 래퍼 엔진.
- 언어(한국어/영어)별 모델을 lazy-load 하고 캐시한다.
- GPU(CUDA)가 있으면 자동으로 사용한다.
- 속도(speed)와 피치(pitch, 반음 단위)를 조절할 수 있다.
"""
from __future__ import annotations
import io
import threading
import numpy as np
import soundfile as sf
import torch
try:
import librosa # 피치 조절용
except Exception: # pragma: no cover
librosa = None
from melo.api import TTS
# 화면에 노출할 언어/화자 정의 (중국어/일본어는 제외)
VOICES: dict[str, dict] = {
"KR": {
"label": "한국어",
"melo_lang": "KR",
"speakers": [
{"id": "KR", "label": "한국어 기본 목소리"},
],
},
"EN": {
"label": "English (영어)",
"melo_lang": "EN",
"speakers": [
{"id": "EN-US", "label": "영어 · 미국식"},
{"id": "EN-BR", "label": "영어 · 영국식"},
{"id": "EN_INDIA", "label": "영어 · 인도식"},
{"id": "EN-AU", "label": "영어 · 호주식"},
{"id": "EN-Default", "label": "영어 · 기본"},
],
},
}
def _pick_device() -> str:
return "cuda:0" if torch.cuda.is_available() else "cpu"
class TTSEngine:
def __init__(self) -> None:
self.device = _pick_device()
self._models: dict[str, TTS] = {}
self._locks: dict[str, threading.Lock] = {k: threading.Lock() for k in VOICES}
self._global_lock = threading.Lock()
# ---- 모델 로딩 -------------------------------------------------
def _get_model(self, language: str) -> TTS:
if language not in VOICES:
raise ValueError(f"지원하지 않는 언어입니다: {language}")
model = self._models.get(language)
if model is not None:
return model
with self._locks[language]:
model = self._models.get(language)
if model is None:
melo_lang = VOICES[language]["melo_lang"]
model = TTS(language=melo_lang, device=self.device)
self._models[language] = model
return model
def warmup(self) -> None:
"""서버 시작 시 모델을 미리 로드한다."""
for lang in VOICES:
try:
self._get_model(lang)
except Exception as exc: # pragma: no cover
print(f"[warmup] {lang} 모델 로드 실패: {exc}")
def list_voices(self) -> list[dict]:
out = []
for code, info in VOICES.items():
out.append(
{
"code": code,
"label": info["label"],
"speakers": info["speakers"],
}
)
return out
def _resolve_speaker(self, model: TTS, language: str, speaker: str | None) -> int:
spk2id = model.hps.data.spk2id
if speaker and speaker in spk2id:
return spk2id[speaker]
# 기본값: 정의된 첫 화자
default_id = VOICES[language]["speakers"][0]["id"]
if default_id in spk2id:
return spk2id[default_id]
return list(spk2id.values())[0]
# ---- 합성 -----------------------------------------------------
def synth_wav(
self,
text: str,
language: str,
speaker: str | None = None,
speed: float = 1.0,
pitch: float = 0.0,
) -> bytes:
text = (text or "").strip()
if not text:
raise ValueError("텍스트가 비어 있습니다.")
speed = float(max(0.5, min(2.0, speed)))
pitch = float(max(-12.0, min(12.0, pitch)))
model = self._get_model(language)
speaker_id = self._resolve_speaker(model, language, speaker)
# 동시 추론은 GPU 메모리 보호를 위해 직렬화
with self._global_lock:
audio = model.tts_to_file(
text, speaker_id, output_path=None, speed=speed, quiet=True
)
sr = model.hps.data.sampling_rate
audio = np.asarray(audio, dtype=np.float32)
if abs(pitch) > 1e-3 and librosa is not None:
audio = librosa.effects.pitch_shift(audio, sr, n_steps=pitch)
buf = io.BytesIO()
sf.write(buf, audio, sr, format="WAV", subtype="PCM_16")
buf.seek(0)
return buf.read()
engine = TTSEngine()