diff --git a/Dockerfile b/Dockerfile index eb35213..642926a 100644 --- a/Dockerfile +++ b/Dockerfile @@ -72,11 +72,25 @@ RUN HF_HOME=/models /opt/supertonic-venv/bin/python -c "from supertonic import T ENV SUPERTONIC_URL=http://127.0.0.1:8002 +# ===== OuteTTS 엔진 (개인용/비상업, CC-BY-NC-4.0) ===== +# Qwen2 500M LM + WavTokenizer 코덱. torch cu128 GPU. transformers 4.49(토치비전 회피). +COPY outetts_worker /opt/outetts_worker +RUN python -m venv /opt/oute-venv \ + && /opt/oute-venv/bin/pip install --no-cache-dir torch==2.11.0 torchaudio==2.11.0 \ + --index-url https://download.pytorch.org/whl/cu128 \ + && /opt/oute-venv/bin/pip install --no-cache-dir -r /opt/outetts_worker/requirements.txt +RUN /opt/oute-venv/bin/pip uninstall -y torchvision 2>/dev/null || true + +# 모델+코덱 사전 다운로드(베이킹) — 빌드 시엔 GPU 없이 CPU 로 로드 +RUN HF_HOME=/models /opt/oute-venv/bin/python -c "import outetts; cfg=outetts.HFModelConfig_v1(model_path='OuteAI/OuteTTS-0.2-500M', language='ko'); outetts.InterfaceHF(model_version='0.2', cfg=cfg); print('oute warmup ok')" + +ENV OUTETTS_URL=http://127.0.0.1:8003 + COPY app ./app COPY frontend ./frontend COPY entrypoint.sh /entrypoint.sh RUN chmod +x /entrypoint.sh -# MeloTTS(8788, 메인) + Supertonic 워커(8002) 동시 기동 +# MeloTTS(8788) + Supertonic(8002) + OuteTTS(8003) 동시 기동 EXPOSE 8788 CMD ["/entrypoint.sh"] diff --git a/app/engines/__init__.py b/app/engines/__init__.py index b4fdb0a..a776f16 100644 --- a/app/engines/__init__.py +++ b/app/engines/__init__.py @@ -41,6 +41,16 @@ def _build() -> None: except Exception as exc: # pragma: no cover print(f"[registry] Supertonic 엔진 스킵: {exc}") + # OuteTTS (개인용/비상업, 전용 워커 URL 이 설정된 경우에만) + try: + from .outetts_engine import OuteTTSEngine + + if OuteTTSEngine.available(): + e = OuteTTSEngine() + _ENGINES[e.id] = e + except Exception as exc: # pragma: no cover + print(f"[registry] OuteTTS 엔진 스킵: {exc}") + _INITED = True diff --git a/app/engines/outetts_engine.py b/app/engines/outetts_engine.py new file mode 100644 index 0000000..a0d6ce9 --- /dev/null +++ b/app/engines/outetts_engine.py @@ -0,0 +1,84 @@ +"""OuteTTS 엔진 (프록시, 개인용/비상업). + +CC-BY-NC-4.0 라이선스라 개인/비상업 용도로만 사용. 워커 URL(OUTETTS_URL)이 +설정된 경우에만 활성화된다. +""" +from __future__ import annotations + +import json +import os +import urllib.error +import urllib.request + +from .base import BaseEngine + +WORKER_URL = os.environ.get("OUTETTS_URL", "").rstrip("/") + +_VOICES = [ + {"id": "female_1", "label": "여성 1"}, + {"id": "female_2", "label": "여성 2"}, + {"id": "male_1", "label": "남성 1"}, + {"id": "male_2", "label": "남성 2"}, +] + + +class OuteTTSEngine(BaseEngine): + id = "outetts" + label = "OuteTTS (개인용·비상업)" + license = "CC-BY-NC-4.0" + uses_gpu = True + notes = "Qwen2 기반 한국어 TTS. 비상업 라이선스라 개인 사용 전용. 속도·피치 후처리." + + @staticmethod + def available() -> bool: + return bool(WORKER_URL) + + def describe(self) -> dict: + return { + "id": self.id, + "label": self.label, + "license": self.license, + "uses_gpu": self.uses_gpu, + "notes": self.notes, + "languages": [ + {"code": "KO", "label": "한국어", "speakers": _VOICES}, + ], + "default_language": "KO", + "supports": {"speed": True, "pitch": True}, + } + + def synth_wav( + self, + text: str, + language: str, + speaker: str | None = None, + speed: float = 1.0, + pitch: float = 0.0, + ) -> bytes: + text = (text or "").strip() + if not text: + raise ValueError("텍스트가 비어 있습니다.") + if not WORKER_URL: + raise ValueError("OuteTTS 워커가 설정되지 않았습니다.") + + payload = json.dumps( + { + "text": text, + "voice": speaker or "female_1", + "speed": speed, + "pitch": pitch, + } + ).encode("utf-8") + request = urllib.request.Request( + WORKER_URL + "/synth", + data=payload, + headers={"Content-Type": "application/json"}, + ) + try: + with urllib.request.urlopen(request, timeout=180) as resp: + return resp.read() + except urllib.error.HTTPError as exc: + detail = exc.read().decode("utf-8", errors="ignore") + raise ValueError(f"OuteTTS 합성 실패: {detail}") + except Exception as exc: # pragma: no cover + raise ValueError(f"OuteTTS 워커 연결 실패: {exc}") diff --git a/entrypoint.sh b/entrypoint.sh index 0ef5a0f..fb20350 100644 --- a/entrypoint.sh +++ b/entrypoint.sh @@ -10,6 +10,15 @@ if [ -n "$SUPERTONIC_URL" ] && [ -x /opt/supertonic-venv/bin/python ]; then > /tmp/supertonic_worker.log 2>&1 & fi +# OuteTTS 워커(전용 venv, 개인용) — 모델을 1회 로드해 상주 +if [ -n "$OUTETTS_URL" ] && [ -x /opt/oute-venv/bin/python ]; then + echo "[entrypoint] starting OuteTTS worker on :8003" + /opt/oute-venv/bin/python -m uvicorn worker:app \ + --app-dir /opt/outetts_worker \ + --host 127.0.0.1 --port 8003 \ + > /tmp/outetts_worker.log 2>&1 & +fi + # 메인 앱(글로벌 파이썬, MeloTTS 인프로세스) echo "[entrypoint] starting main app on :8788" exec uvicorn app.server:app --host 0.0.0.0 --port 8788 diff --git a/outetts_worker/requirements.txt b/outetts_worker/requirements.txt new file mode 100644 index 0000000..4ff778f --- /dev/null +++ b/outetts_worker/requirements.txt @@ -0,0 +1,7 @@ +# OuteTTS 워커 (torch cu128 은 Dockerfile 에서 별도 설치) +outetts==0.3.3 +transformers==4.49.0 +soundfile +librosa +fastapi==0.115.6 +uvicorn==0.30.0 diff --git a/outetts_worker/worker.py b/outetts_worker/worker.py new file mode 100644 index 0000000..3298c98 --- /dev/null +++ b/outetts_worker/worker.py @@ -0,0 +1,105 @@ +"""OuteTTS 0.2-500M 한국어 합성 워커 (전용 venv, 개인용/비상업 CC-BY-NC). + +- Qwen2 기반 500M LM + WavTokenizer 코덱, torch cu128 GPU 사용 +- 한국어 프리셋 화자 4종(male_1/2, female_1/2) +- 속도/피치는 librosa 후처리(OuteTTS 자체 speed 파라미터 없음) +""" +from __future__ import annotations + +import io + +import numpy as np +import soundfile as sf + +try: + import librosa +except Exception: # pragma: no cover + librosa = None + +from fastapi import FastAPI +from fastapi.responses import Response +from pydantic import BaseModel + +import outetts + +VOICES = ["female_1", "female_2", "male_1", "male_2"] + +_interface = None +_speakers: dict = {} + + +def get_interface(): + global _interface + if _interface is None: + cfg = outetts.HFModelConfig_v1( + model_path="OuteAI/OuteTTS-0.2-500M", language="ko" + ) + _interface = outetts.InterfaceHF(model_version="0.2", cfg=cfg) + return _interface + + +def get_speaker(name: str): + if name not in _speakers: + _speakers[name] = get_interface().load_default_speaker(name=name) + return _speakers[name] + + +app = FastAPI(title="OuteTTS Worker") + + +class SynthReq(BaseModel): + text: str + voice: str = "female_1" + speed: float = 1.0 + pitch: float = 0.0 + + +@app.on_event("startup") +def _startup() -> None: + get_speaker("female_1") + + +@app.get("/health") +def health() -> dict: + get_interface() + return {"status": "ok", "voices": VOICES} + + +@app.post("/synth") +def synth(req: SynthReq) -> Response: + text = (req.text or "").strip() + if not text: + return Response(content="empty text", status_code=400) + + it = get_interface() + voice = req.voice if req.voice in VOICES else "female_1" + speaker = get_speaker(voice) + speed = float(max(0.5, min(2.0, req.speed))) + pitch = float(max(-12.0, min(12.0, req.pitch))) + + out = it.generate( + text=text, + temperature=0.1, + repetition_penalty=1.1, + max_length=4096, + speaker=speaker, + ) + audio = out.audio.cpu().numpy().reshape(-1).astype(np.float32) + sr = int(out.sr) + + if librosa is not None: + if abs(speed - 1.0) > 1e-3: + try: + audio = librosa.effects.time_stretch(audio, rate=speed).astype(np.float32) + except Exception: + try: + audio = librosa.effects.time_stretch(audio, speed).astype(np.float32) + except Exception: + pass + if abs(pitch) > 1e-3: + audio = librosa.effects.pitch_shift(audio, sr=sr, n_steps=pitch) + + buf = io.BytesIO() + sf.write(buf, audio, sr, format="WAV", subtype="PCM_16") + buf.seek(0) + return Response(content=buf.read(), media_type="audio/wav")