diff --git a/Dockerfile b/Dockerfile index ea94c59..eb35213 100644 --- a/Dockerfile +++ b/Dockerfile @@ -61,9 +61,22 @@ COPY --from=builder /usr/local/lib/python3.11/site-packages /usr/local/lib/pytho COPY --from=builder /usr/local/bin /usr/local/bin COPY --from=builder /models /models +# ===== Supertonic 엔진 (자연스러운 한국어, MIT, ONNX, torch 불필요) ===== +# 전용 venv 로 의존성 격리(메인 melo 스택의 librosa 0.9.1 과 충돌 방지). +COPY supertonic_worker /opt/supertonic_worker +RUN python -m venv /opt/supertonic-venv \ + && /opt/supertonic-venv/bin/pip install --no-cache-dir -r /opt/supertonic_worker/requirements.txt + +# 모델 사전 다운로드(베이킹, ~260MB) — HF_HOME=/models 캐시에 저장 +RUN HF_HOME=/models /opt/supertonic-venv/bin/python -c "from supertonic import TTS; TTS(auto_download=True)" + +ENV SUPERTONIC_URL=http://127.0.0.1:8002 + COPY app ./app COPY frontend ./frontend +COPY entrypoint.sh /entrypoint.sh +RUN chmod +x /entrypoint.sh -# 기본 MeloTTS 단일 엔진 (자동 언어 감지: 한/영 혼합) +# MeloTTS(8788, 메인) + Supertonic 워커(8002) 동시 기동 EXPOSE 8788 -CMD ["uvicorn", "app.server:app", "--host", "0.0.0.0", "--port", "8788"] +CMD ["/entrypoint.sh"] diff --git a/app/engines/__init__.py b/app/engines/__init__.py index 3f1a158..b4fdb0a 100644 --- a/app/engines/__init__.py +++ b/app/engines/__init__.py @@ -31,7 +31,15 @@ def _build() -> None: except Exception as exc: # pragma: no cover print(f"[registry] MeloTTS 로드 실패: {exc}") - # (CosyVoice 엔진은 사용자 요청에 따라 비활성화 — 기본 MeloTTS만 유지) + # Supertonic (자연스러운 한국어, 전용 워커 URL 이 설정된 경우에만) + try: + from .supertonic_engine import SupertonicEngine + + if SupertonicEngine.available(): + e = SupertonicEngine() + _ENGINES[e.id] = e + except Exception as exc: # pragma: no cover + print(f"[registry] Supertonic 엔진 스킵: {exc}") _INITED = True diff --git a/app/engines/supertonic_engine.py b/app/engines/supertonic_engine.py new file mode 100644 index 0000000..61f2739 --- /dev/null +++ b/app/engines/supertonic_engine.py @@ -0,0 +1,94 @@ +"""Supertonic 엔진 (프록시). + +실제 합성은 전용 venv 워커(worker.py)가 담당하고, 이 엔진은 HTTP로 요청을 +전달한다. 워커 URL(SUPERTONIC_URL)이 설정된 경우에만 활성화된다. +자연스러운 한국어 음성(수퍼톤), ONNX 기반이라 GPU 불필요. +""" +from __future__ import annotations + +import json +import os +import urllib.error +import urllib.request + +from .base import BaseEngine + +WORKER_URL = os.environ.get("SUPERTONIC_URL", "").rstrip("/") + +_VOICES = [ + {"id": "F1", "label": "여성 1"}, + {"id": "F2", "label": "여성 2"}, + {"id": "F3", "label": "여성 3"}, + {"id": "F4", "label": "여성 4"}, + {"id": "F5", "label": "여성 5"}, + {"id": "M1", "label": "남성 1"}, + {"id": "M2", "label": "남성 2"}, + {"id": "M3", "label": "남성 3"}, + {"id": "M4", "label": "남성 4"}, + {"id": "M5", "label": "남성 5"}, +] + + +class SupertonicEngine(BaseEngine): + id = "supertonic" + label = "Supertonic (자연 음성)" + license = "MIT" + uses_gpu = False + notes = "수퍼톤의 자연스러운 다국어 음성. 한국어/자동감지, 남녀 10종, 속도·피치." + + @staticmethod + def available() -> bool: + return bool(WORKER_URL) + + def describe(self) -> dict: + return { + "id": self.id, + "label": self.label, + "license": self.license, + "uses_gpu": self.uses_gpu, + "notes": self.notes, + "languages": [ + {"code": "KO", "label": "한국어", "speakers": _VOICES}, + {"code": "AUTO", "label": "자동 감지 (다국어)", "speakers": _VOICES}, + ], + "default_language": "KO", + "supports": {"speed": True, "pitch": True}, + } + + def synth_wav( + self, + text: str, + language: str, + speaker: str | None = None, + speed: float = 1.0, + pitch: float = 0.0, + ) -> bytes: + text = (text or "").strip() + if not text: + raise ValueError("텍스트가 비어 있습니다.") + if not WORKER_URL: + raise ValueError("Supertonic 워커가 설정되지 않았습니다.") + + lang = "na" if (language or "KO").upper() == "AUTO" else "ko" + payload = json.dumps( + { + "text": text, + "voice": speaker or "F1", + "lang": lang, + "speed": speed, + "pitch": pitch, + } + ).encode("utf-8") + request = urllib.request.Request( + WORKER_URL + "/synth", + data=payload, + headers={"Content-Type": "application/json"}, + ) + try: + with urllib.request.urlopen(request, timeout=120) as resp: + return resp.read() + except urllib.error.HTTPError as exc: + detail = exc.read().decode("utf-8", errors="ignore") + raise ValueError(f"Supertonic 합성 실패: {detail}") + except Exception as exc: # pragma: no cover + raise ValueError(f"Supertonic 워커 연결 실패: {exc}") diff --git a/entrypoint.sh b/entrypoint.sh index 50dcd76..0ef5a0f 100644 --- a/entrypoint.sh +++ b/entrypoint.sh @@ -1,13 +1,13 @@ #!/usr/bin/env bash set -e -# CosyVoice 워커(전용 venv)를 백그라운드로 기동 — 모델을 1회 로드해 상주 -if [ -n "$COSYVOICE_URL" ] && [ -x /opt/cosyvoice-venv/bin/python ]; then - echo "[entrypoint] starting CosyVoice worker on :8001" - /opt/cosyvoice-venv/bin/python -m uvicorn worker:app \ - --app-dir /opt/cosyvoice_worker \ - --host 127.0.0.1 --port 8001 \ - > /tmp/cosyvoice_worker.log 2>&1 & +# Supertonic 워커(전용 venv)를 백그라운드로 기동 — 모델을 1회 로드해 상주 +if [ -n "$SUPERTONIC_URL" ] && [ -x /opt/supertonic-venv/bin/python ]; then + echo "[entrypoint] starting Supertonic worker on :8002" + /opt/supertonic-venv/bin/python -m uvicorn worker:app \ + --app-dir /opt/supertonic_worker \ + --host 127.0.0.1 --port 8002 \ + > /tmp/supertonic_worker.log 2>&1 & fi # 메인 앱(글로벌 파이썬, MeloTTS 인프로세스) diff --git a/supertonic_worker/requirements.txt b/supertonic_worker/requirements.txt new file mode 100644 index 0000000..c083203 --- /dev/null +++ b/supertonic_worker/requirements.txt @@ -0,0 +1,4 @@ +# Supertonic (수퍼톤) — 자연스러운 다국어 TTS, ONNX 런타임, torch 불필요 +supertonic==1.3.1 +fastapi==0.115.6 +uvicorn==0.30.0 diff --git a/supertonic_worker/worker.py b/supertonic_worker/worker.py new file mode 100644 index 0000000..1007670 --- /dev/null +++ b/supertonic_worker/worker.py @@ -0,0 +1,91 @@ +"""Supertonic 합성 워커 (전용 venv에서 실행). + +- ONNX 런타임 기반, torch 불필요, 모델 ~260MB +- 다국어(31개) 지원. 한국어는 lang="ko", 자동감지는 lang="na" +- 프리셋 음성 10종(M1~M5, F1~F5) +- 속도(speed) 네이티브, 피치(pitch)는 librosa 후처리 +""" +from __future__ import annotations + +import io + +import numpy as np +import soundfile as sf + +try: + import librosa +except Exception: # pragma: no cover + librosa = None + +from fastapi import FastAPI +from fastapi.responses import Response +from pydantic import BaseModel + +from supertonic import TTS + +VOICES = ["M1", "M2", "M3", "M4", "M5", "F1", "F2", "F3", "F4", "F5"] +SR = 44100 + +_tts = None +_styles: dict = {} + + +def get_tts() -> TTS: + global _tts + if _tts is None: + _tts = TTS(auto_download=True) + return _tts + + +def get_style(name: str): + if name not in _styles: + _styles[name] = get_tts().get_voice_style(voice_name=name) + return _styles[name] + + +app = FastAPI(title="Supertonic Worker") + + +class SynthReq(BaseModel): + text: str + voice: str = "F1" + lang: str = "ko" + speed: float = 1.0 + pitch: float = 0.0 + + +@app.on_event("startup") +def _startup() -> None: + get_tts() + get_style("F1") + + +@app.get("/health") +def health() -> dict: + get_tts() + return {"status": "ok", "voices": VOICES, "sr": SR} + + +@app.post("/synth") +def synth(req: SynthReq) -> Response: + text = (req.text or "").strip() + if not text: + return Response(content="empty text", status_code=400) + + tts = get_tts() + voice = req.voice if req.voice in VOICES else "F1" + style = get_style(voice) + lang = req.lang if req.lang in ("ko", "na", "en", "ja", "zh") else "ko" + speed = float(max(0.5, min(2.0, req.speed))) + pitch = float(max(-12.0, min(12.0, req.pitch))) + + wav, _dur = tts.synthesize(text, voice_style=style, lang=lang, speed=speed) + audio = np.asarray(wav).reshape(-1).astype(np.float32) + + if abs(pitch) > 1e-3 and librosa is not None: + audio = librosa.effects.pitch_shift(audio, sr=SR, n_steps=pitch) + + buf = io.BytesIO() + sf.write(buf, audio, SR, format="WAV", subtype="PCM_16") + buf.seek(0) + return Response(content=buf.read(), media_type="audio/wav")