feat: Supertonic 엔진 추가(자연스러운 한국어, MIT, ONNX)

- 전용 venv 워커(8002)로 격리, MeloTTS(8788)와 병행
- 남녀 10종 음성, 한국어/자동감지(다국어), 속도·피치
- 모델 ~260MB 베이킹, torch/GPU 불필요
- entrypoint로 워커+메인앱 동시 기동
This commit is contained in:
claude
2026-08-24 21:00:45 +09:00
parent e6bc6ef2be
commit ed09c6a08f
6 changed files with 220 additions and 10 deletions

View File

@@ -61,9 +61,22 @@ COPY --from=builder /usr/local/lib/python3.11/site-packages /usr/local/lib/pytho
COPY --from=builder /usr/local/bin /usr/local/bin COPY --from=builder /usr/local/bin /usr/local/bin
COPY --from=builder /models /models COPY --from=builder /models /models
# ===== Supertonic 엔진 (자연스러운 한국어, MIT, ONNX, torch 불필요) =====
# 전용 venv 로 의존성 격리(메인 melo 스택의 librosa 0.9.1 과 충돌 방지).
COPY supertonic_worker /opt/supertonic_worker
RUN python -m venv /opt/supertonic-venv \
&& /opt/supertonic-venv/bin/pip install --no-cache-dir -r /opt/supertonic_worker/requirements.txt
# 모델 사전 다운로드(베이킹, ~260MB) — HF_HOME=/models 캐시에 저장
RUN HF_HOME=/models /opt/supertonic-venv/bin/python -c "from supertonic import TTS; TTS(auto_download=True)"
ENV SUPERTONIC_URL=http://127.0.0.1:8002
COPY app ./app COPY app ./app
COPY frontend ./frontend COPY frontend ./frontend
COPY entrypoint.sh /entrypoint.sh
RUN chmod +x /entrypoint.sh
# 기본 MeloTTS 단일 엔진 (자동 언어 감지: 한/영 혼합) # MeloTTS(8788, 메인) + Supertonic 워커(8002) 동시 기동
EXPOSE 8788 EXPOSE 8788
CMD ["uvicorn", "app.server:app", "--host", "0.0.0.0", "--port", "8788"] CMD ["/entrypoint.sh"]

View File

@@ -31,7 +31,15 @@ def _build() -> None:
except Exception as exc: # pragma: no cover except Exception as exc: # pragma: no cover
print(f"[registry] MeloTTS 로드 실패: {exc}") print(f"[registry] MeloTTS 로드 실패: {exc}")
# (CosyVoice 엔진은 사용자 요청에 따라 비활성화 — 기본 MeloTTS만 유지) # Supertonic (자연스러운 한국어, 전용 워커 URL 이 설정된 경우에만)
try:
from .supertonic_engine import SupertonicEngine
if SupertonicEngine.available():
e = SupertonicEngine()
_ENGINES[e.id] = e
except Exception as exc: # pragma: no cover
print(f"[registry] Supertonic 엔진 스킵: {exc}")
_INITED = True _INITED = True

View File

@@ -0,0 +1,94 @@
"""Supertonic 엔진 (프록시).
실제 합성은 전용 venv 워커(worker.py)가 담당하고, 이 엔진은 HTTP로 요청을
전달한다. 워커 URL(SUPERTONIC_URL)이 설정된 경우에만 활성화된다.
자연스러운 한국어 음성(수퍼톤), ONNX 기반이라 GPU 불필요.
"""
from __future__ import annotations
import json
import os
import urllib.error
import urllib.request
from .base import BaseEngine
WORKER_URL = os.environ.get("SUPERTONIC_URL", "").rstrip("/")
_VOICES = [
{"id": "F1", "label": "여성 1"},
{"id": "F2", "label": "여성 2"},
{"id": "F3", "label": "여성 3"},
{"id": "F4", "label": "여성 4"},
{"id": "F5", "label": "여성 5"},
{"id": "M1", "label": "남성 1"},
{"id": "M2", "label": "남성 2"},
{"id": "M3", "label": "남성 3"},
{"id": "M4", "label": "남성 4"},
{"id": "M5", "label": "남성 5"},
]
class SupertonicEngine(BaseEngine):
id = "supertonic"
label = "Supertonic (자연 음성)"
license = "MIT"
uses_gpu = False
notes = "수퍼톤의 자연스러운 다국어 음성. 한국어/자동감지, 남녀 10종, 속도·피치."
@staticmethod
def available() -> bool:
return bool(WORKER_URL)
def describe(self) -> dict:
return {
"id": self.id,
"label": self.label,
"license": self.license,
"uses_gpu": self.uses_gpu,
"notes": self.notes,
"languages": [
{"code": "KO", "label": "한국어", "speakers": _VOICES},
{"code": "AUTO", "label": "자동 감지 (다국어)", "speakers": _VOICES},
],
"default_language": "KO",
"supports": {"speed": True, "pitch": True},
}
def synth_wav(
self,
text: str,
language: str,
speaker: str | None = None,
speed: float = 1.0,
pitch: float = 0.0,
) -> bytes:
text = (text or "").strip()
if not text:
raise ValueError("텍스트가 비어 있습니다.")
if not WORKER_URL:
raise ValueError("Supertonic 워커가 설정되지 않았습니다.")
lang = "na" if (language or "KO").upper() == "AUTO" else "ko"
payload = json.dumps(
{
"text": text,
"voice": speaker or "F1",
"lang": lang,
"speed": speed,
"pitch": pitch,
}
).encode("utf-8")
request = urllib.request.Request(
WORKER_URL + "/synth",
data=payload,
headers={"Content-Type": "application/json"},
)
try:
with urllib.request.urlopen(request, timeout=120) as resp:
return resp.read()
except urllib.error.HTTPError as exc:
detail = exc.read().decode("utf-8", errors="ignore")
raise ValueError(f"Supertonic 합성 실패: {detail}")
except Exception as exc: # pragma: no cover
raise ValueError(f"Supertonic 워커 연결 실패: {exc}")

View File

@@ -1,13 +1,13 @@
#!/usr/bin/env bash #!/usr/bin/env bash
set -e set -e
# CosyVoice 워커(전용 venv)를 백그라운드로 기동 — 모델을 1회 로드해 상주 # Supertonic 워커(전용 venv)를 백그라운드로 기동 — 모델을 1회 로드해 상주
if [ -n "$COSYVOICE_URL" ] && [ -x /opt/cosyvoice-venv/bin/python ]; then if [ -n "$SUPERTONIC_URL" ] && [ -x /opt/supertonic-venv/bin/python ]; then
echo "[entrypoint] starting CosyVoice worker on :8001" echo "[entrypoint] starting Supertonic worker on :8002"
/opt/cosyvoice-venv/bin/python -m uvicorn worker:app \ /opt/supertonic-venv/bin/python -m uvicorn worker:app \
--app-dir /opt/cosyvoice_worker \ --app-dir /opt/supertonic_worker \
--host 127.0.0.1 --port 8001 \ --host 127.0.0.1 --port 8002 \
> /tmp/cosyvoice_worker.log 2>&1 & > /tmp/supertonic_worker.log 2>&1 &
fi fi
# 메인 앱(글로벌 파이썬, MeloTTS 인프로세스) # 메인 앱(글로벌 파이썬, MeloTTS 인프로세스)

View File

@@ -0,0 +1,4 @@
# Supertonic (수퍼톤) — 자연스러운 다국어 TTS, ONNX 런타임, torch 불필요
supertonic==1.3.1
fastapi==0.115.6
uvicorn==0.30.0

View File

@@ -0,0 +1,91 @@
"""Supertonic 합성 워커 (전용 venv에서 실행).
- ONNX 런타임 기반, torch 불필요, 모델 ~260MB
- 다국어(31개) 지원. 한국어는 lang="ko", 자동감지는 lang="na"
- 프리셋 음성 10종(M1~M5, F1~F5)
- 속도(speed) 네이티브, 피치(pitch)는 librosa 후처리
"""
from __future__ import annotations
import io
import numpy as np
import soundfile as sf
try:
import librosa
except Exception: # pragma: no cover
librosa = None
from fastapi import FastAPI
from fastapi.responses import Response
from pydantic import BaseModel
from supertonic import TTS
VOICES = ["M1", "M2", "M3", "M4", "M5", "F1", "F2", "F3", "F4", "F5"]
SR = 44100
_tts = None
_styles: dict = {}
def get_tts() -> TTS:
global _tts
if _tts is None:
_tts = TTS(auto_download=True)
return _tts
def get_style(name: str):
if name not in _styles:
_styles[name] = get_tts().get_voice_style(voice_name=name)
return _styles[name]
app = FastAPI(title="Supertonic Worker")
class SynthReq(BaseModel):
text: str
voice: str = "F1"
lang: str = "ko"
speed: float = 1.0
pitch: float = 0.0
@app.on_event("startup")
def _startup() -> None:
get_tts()
get_style("F1")
@app.get("/health")
def health() -> dict:
get_tts()
return {"status": "ok", "voices": VOICES, "sr": SR}
@app.post("/synth")
def synth(req: SynthReq) -> Response:
text = (req.text or "").strip()
if not text:
return Response(content="empty text", status_code=400)
tts = get_tts()
voice = req.voice if req.voice in VOICES else "F1"
style = get_style(voice)
lang = req.lang if req.lang in ("ko", "na", "en", "ja", "zh") else "ko"
speed = float(max(0.5, min(2.0, req.speed)))
pitch = float(max(-12.0, min(12.0, req.pitch)))
wav, _dur = tts.synthesize(text, voice_style=style, lang=lang, speed=speed)
audio = np.asarray(wav).reshape(-1).astype(np.float32)
if abs(pitch) > 1e-3 and librosa is not None:
audio = librosa.effects.pitch_shift(audio, sr=SR, n_steps=pitch)
buf = io.BytesIO()
sf.write(buf, audio, SR, format="WAV", subtype="PCM_16")
buf.seek(0)
return Response(content=buf.read(), media_type="audio/wav")