feat: Supertonic 엔진 추가(자연스러운 한국어, MIT, ONNX)

- 전용 venv 워커(8002)로 격리, MeloTTS(8788)와 병행
- 남녀 10종 음성, 한국어/자동감지(다국어), 속도·피치
- 모델 ~260MB 베이킹, torch/GPU 불필요
- entrypoint로 워커+메인앱 동시 기동
This commit is contained in:
claude
2026-08-24 21:00:45 +09:00
parent e6bc6ef2be
commit ed09c6a08f
6 changed files with 220 additions and 10 deletions

View File

@@ -0,0 +1,4 @@
# Supertonic (수퍼톤) — 자연스러운 다국어 TTS, ONNX 런타임, torch 불필요
supertonic==1.3.1
fastapi==0.115.6
uvicorn==0.30.0

View File

@@ -0,0 +1,91 @@
"""Supertonic 합성 워커 (전용 venv에서 실행).
- ONNX 런타임 기반, torch 불필요, 모델 ~260MB
- 다국어(31개) 지원. 한국어는 lang="ko", 자동감지는 lang="na"
- 프리셋 음성 10종(M1~M5, F1~F5)
- 속도(speed) 네이티브, 피치(pitch)는 librosa 후처리
"""
from __future__ import annotations
import io
import numpy as np
import soundfile as sf
try:
import librosa
except Exception: # pragma: no cover
librosa = None
from fastapi import FastAPI
from fastapi.responses import Response
from pydantic import BaseModel
from supertonic import TTS
VOICES = ["M1", "M2", "M3", "M4", "M5", "F1", "F2", "F3", "F4", "F5"]
SR = 44100
_tts = None
_styles: dict = {}
def get_tts() -> TTS:
global _tts
if _tts is None:
_tts = TTS(auto_download=True)
return _tts
def get_style(name: str):
if name not in _styles:
_styles[name] = get_tts().get_voice_style(voice_name=name)
return _styles[name]
app = FastAPI(title="Supertonic Worker")
class SynthReq(BaseModel):
text: str
voice: str = "F1"
lang: str = "ko"
speed: float = 1.0
pitch: float = 0.0
@app.on_event("startup")
def _startup() -> None:
get_tts()
get_style("F1")
@app.get("/health")
def health() -> dict:
get_tts()
return {"status": "ok", "voices": VOICES, "sr": SR}
@app.post("/synth")
def synth(req: SynthReq) -> Response:
text = (req.text or "").strip()
if not text:
return Response(content="empty text", status_code=400)
tts = get_tts()
voice = req.voice if req.voice in VOICES else "F1"
style = get_style(voice)
lang = req.lang if req.lang in ("ko", "na", "en", "ja", "zh") else "ko"
speed = float(max(0.5, min(2.0, req.speed)))
pitch = float(max(-12.0, min(12.0, req.pitch)))
wav, _dur = tts.synthesize(text, voice_style=style, lang=lang, speed=speed)
audio = np.asarray(wav).reshape(-1).astype(np.float32)
if abs(pitch) > 1e-3 and librosa is not None:
audio = librosa.effects.pitch_shift(audio, sr=SR, n_steps=pitch)
buf = io.BytesIO()
sf.write(buf, audio, SR, format="WAV", subtype="PCM_16")
buf.seek(0)
return Response(content=buf.read(), media_type="audio/wav")