feat: OuteTTS 엔진 추가(개인용/비상업 CC-BY-NC, 한국어)
- Qwen2 500M + WavTokenizer, torch cu128 GPU, 전용 venv 워커(8003) - 한국어 프리셋 4종(male/female), 속도·피치 librosa 후처리 - transformers 4.49로 torchvision 강제 import 회피, soundfile 저장 - scenema-audio는 VRAM 24GB+/Gemma12B 게이트로 8GB GPU에선 불가 → 미포함
This commit is contained in:
16
Dockerfile
16
Dockerfile
@@ -72,11 +72,25 @@ RUN HF_HOME=/models /opt/supertonic-venv/bin/python -c "from supertonic import T
|
|||||||
|
|
||||||
ENV SUPERTONIC_URL=http://127.0.0.1:8002
|
ENV SUPERTONIC_URL=http://127.0.0.1:8002
|
||||||
|
|
||||||
|
# ===== OuteTTS 엔진 (개인용/비상업, CC-BY-NC-4.0) =====
|
||||||
|
# Qwen2 500M LM + WavTokenizer 코덱. torch cu128 GPU. transformers 4.49(토치비전 회피).
|
||||||
|
COPY outetts_worker /opt/outetts_worker
|
||||||
|
RUN python -m venv /opt/oute-venv \
|
||||||
|
&& /opt/oute-venv/bin/pip install --no-cache-dir torch==2.11.0 torchaudio==2.11.0 \
|
||||||
|
--index-url https://download.pytorch.org/whl/cu128 \
|
||||||
|
&& /opt/oute-venv/bin/pip install --no-cache-dir -r /opt/outetts_worker/requirements.txt
|
||||||
|
RUN /opt/oute-venv/bin/pip uninstall -y torchvision 2>/dev/null || true
|
||||||
|
|
||||||
|
# 모델+코덱 사전 다운로드(베이킹) — 빌드 시엔 GPU 없이 CPU 로 로드
|
||||||
|
RUN HF_HOME=/models /opt/oute-venv/bin/python -c "import outetts; cfg=outetts.HFModelConfig_v1(model_path='OuteAI/OuteTTS-0.2-500M', language='ko'); outetts.InterfaceHF(model_version='0.2', cfg=cfg); print('oute warmup ok')"
|
||||||
|
|
||||||
|
ENV OUTETTS_URL=http://127.0.0.1:8003
|
||||||
|
|
||||||
COPY app ./app
|
COPY app ./app
|
||||||
COPY frontend ./frontend
|
COPY frontend ./frontend
|
||||||
COPY entrypoint.sh /entrypoint.sh
|
COPY entrypoint.sh /entrypoint.sh
|
||||||
RUN chmod +x /entrypoint.sh
|
RUN chmod +x /entrypoint.sh
|
||||||
|
|
||||||
# MeloTTS(8788, 메인) + Supertonic 워커(8002) 동시 기동
|
# MeloTTS(8788) + Supertonic(8002) + OuteTTS(8003) 동시 기동
|
||||||
EXPOSE 8788
|
EXPOSE 8788
|
||||||
CMD ["/entrypoint.sh"]
|
CMD ["/entrypoint.sh"]
|
||||||
|
|||||||
@@ -41,6 +41,16 @@ def _build() -> None:
|
|||||||
except Exception as exc: # pragma: no cover
|
except Exception as exc: # pragma: no cover
|
||||||
print(f"[registry] Supertonic 엔진 스킵: {exc}")
|
print(f"[registry] Supertonic 엔진 스킵: {exc}")
|
||||||
|
|
||||||
|
# OuteTTS (개인용/비상업, 전용 워커 URL 이 설정된 경우에만)
|
||||||
|
try:
|
||||||
|
from .outetts_engine import OuteTTSEngine
|
||||||
|
|
||||||
|
if OuteTTSEngine.available():
|
||||||
|
e = OuteTTSEngine()
|
||||||
|
_ENGINES[e.id] = e
|
||||||
|
except Exception as exc: # pragma: no cover
|
||||||
|
print(f"[registry] OuteTTS 엔진 스킵: {exc}")
|
||||||
|
|
||||||
_INITED = True
|
_INITED = True
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
84
app/engines/outetts_engine.py
Normal file
84
app/engines/outetts_engine.py
Normal file
@@ -0,0 +1,84 @@
|
|||||||
|
"""OuteTTS 엔진 (프록시, 개인용/비상업).
|
||||||
|
|
||||||
|
CC-BY-NC-4.0 라이선스라 개인/비상업 용도로만 사용. 워커 URL(OUTETTS_URL)이
|
||||||
|
설정된 경우에만 활성화된다.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import urllib.error
|
||||||
|
import urllib.request
|
||||||
|
|
||||||
|
from .base import BaseEngine
|
||||||
|
|
||||||
|
WORKER_URL = os.environ.get("OUTETTS_URL", "").rstrip("/")
|
||||||
|
|
||||||
|
_VOICES = [
|
||||||
|
{"id": "female_1", "label": "여성 1"},
|
||||||
|
{"id": "female_2", "label": "여성 2"},
|
||||||
|
{"id": "male_1", "label": "남성 1"},
|
||||||
|
{"id": "male_2", "label": "남성 2"},
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
class OuteTTSEngine(BaseEngine):
|
||||||
|
id = "outetts"
|
||||||
|
label = "OuteTTS (개인용·비상업)"
|
||||||
|
license = "CC-BY-NC-4.0"
|
||||||
|
uses_gpu = True
|
||||||
|
notes = "Qwen2 기반 한국어 TTS. 비상업 라이선스라 개인 사용 전용. 속도·피치 후처리."
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def available() -> bool:
|
||||||
|
return bool(WORKER_URL)
|
||||||
|
|
||||||
|
def describe(self) -> dict:
|
||||||
|
return {
|
||||||
|
"id": self.id,
|
||||||
|
"label": self.label,
|
||||||
|
"license": self.license,
|
||||||
|
"uses_gpu": self.uses_gpu,
|
||||||
|
"notes": self.notes,
|
||||||
|
"languages": [
|
||||||
|
{"code": "KO", "label": "한국어", "speakers": _VOICES},
|
||||||
|
],
|
||||||
|
"default_language": "KO",
|
||||||
|
"supports": {"speed": True, "pitch": True},
|
||||||
|
}
|
||||||
|
|
||||||
|
def synth_wav(
|
||||||
|
self,
|
||||||
|
text: str,
|
||||||
|
language: str,
|
||||||
|
speaker: str | None = None,
|
||||||
|
speed: float = 1.0,
|
||||||
|
pitch: float = 0.0,
|
||||||
|
) -> bytes:
|
||||||
|
text = (text or "").strip()
|
||||||
|
if not text:
|
||||||
|
raise ValueError("텍스트가 비어 있습니다.")
|
||||||
|
if not WORKER_URL:
|
||||||
|
raise ValueError("OuteTTS 워커가 설정되지 않았습니다.")
|
||||||
|
|
||||||
|
payload = json.dumps(
|
||||||
|
{
|
||||||
|
"text": text,
|
||||||
|
"voice": speaker or "female_1",
|
||||||
|
"speed": speed,
|
||||||
|
"pitch": pitch,
|
||||||
|
}
|
||||||
|
).encode("utf-8")
|
||||||
|
request = urllib.request.Request(
|
||||||
|
WORKER_URL + "/synth",
|
||||||
|
data=payload,
|
||||||
|
headers={"Content-Type": "application/json"},
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
with urllib.request.urlopen(request, timeout=180) as resp:
|
||||||
|
return resp.read()
|
||||||
|
except urllib.error.HTTPError as exc:
|
||||||
|
detail = exc.read().decode("utf-8", errors="ignore")
|
||||||
|
raise ValueError(f"OuteTTS 합성 실패: {detail}")
|
||||||
|
except Exception as exc: # pragma: no cover
|
||||||
|
raise ValueError(f"OuteTTS 워커 연결 실패: {exc}")
|
||||||
@@ -10,6 +10,15 @@ if [ -n "$SUPERTONIC_URL" ] && [ -x /opt/supertonic-venv/bin/python ]; then
|
|||||||
> /tmp/supertonic_worker.log 2>&1 &
|
> /tmp/supertonic_worker.log 2>&1 &
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
# OuteTTS 워커(전용 venv, 개인용) — 모델을 1회 로드해 상주
|
||||||
|
if [ -n "$OUTETTS_URL" ] && [ -x /opt/oute-venv/bin/python ]; then
|
||||||
|
echo "[entrypoint] starting OuteTTS worker on :8003"
|
||||||
|
/opt/oute-venv/bin/python -m uvicorn worker:app \
|
||||||
|
--app-dir /opt/outetts_worker \
|
||||||
|
--host 127.0.0.1 --port 8003 \
|
||||||
|
> /tmp/outetts_worker.log 2>&1 &
|
||||||
|
fi
|
||||||
|
|
||||||
# 메인 앱(글로벌 파이썬, MeloTTS 인프로세스)
|
# 메인 앱(글로벌 파이썬, MeloTTS 인프로세스)
|
||||||
echo "[entrypoint] starting main app on :8788"
|
echo "[entrypoint] starting main app on :8788"
|
||||||
exec uvicorn app.server:app --host 0.0.0.0 --port 8788
|
exec uvicorn app.server:app --host 0.0.0.0 --port 8788
|
||||||
|
|||||||
7
outetts_worker/requirements.txt
Normal file
7
outetts_worker/requirements.txt
Normal file
@@ -0,0 +1,7 @@
|
|||||||
|
# OuteTTS 워커 (torch cu128 은 Dockerfile 에서 별도 설치)
|
||||||
|
outetts==0.3.3
|
||||||
|
transformers==4.49.0
|
||||||
|
soundfile
|
||||||
|
librosa
|
||||||
|
fastapi==0.115.6
|
||||||
|
uvicorn==0.30.0
|
||||||
105
outetts_worker/worker.py
Normal file
105
outetts_worker/worker.py
Normal file
@@ -0,0 +1,105 @@
|
|||||||
|
"""OuteTTS 0.2-500M 한국어 합성 워커 (전용 venv, 개인용/비상업 CC-BY-NC).
|
||||||
|
|
||||||
|
- Qwen2 기반 500M LM + WavTokenizer 코덱, torch cu128 GPU 사용
|
||||||
|
- 한국어 프리셋 화자 4종(male_1/2, female_1/2)
|
||||||
|
- 속도/피치는 librosa 후처리(OuteTTS 자체 speed 파라미터 없음)
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import io
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
import soundfile as sf
|
||||||
|
|
||||||
|
try:
|
||||||
|
import librosa
|
||||||
|
except Exception: # pragma: no cover
|
||||||
|
librosa = None
|
||||||
|
|
||||||
|
from fastapi import FastAPI
|
||||||
|
from fastapi.responses import Response
|
||||||
|
from pydantic import BaseModel
|
||||||
|
|
||||||
|
import outetts
|
||||||
|
|
||||||
|
VOICES = ["female_1", "female_2", "male_1", "male_2"]
|
||||||
|
|
||||||
|
_interface = None
|
||||||
|
_speakers: dict = {}
|
||||||
|
|
||||||
|
|
||||||
|
def get_interface():
|
||||||
|
global _interface
|
||||||
|
if _interface is None:
|
||||||
|
cfg = outetts.HFModelConfig_v1(
|
||||||
|
model_path="OuteAI/OuteTTS-0.2-500M", language="ko"
|
||||||
|
)
|
||||||
|
_interface = outetts.InterfaceHF(model_version="0.2", cfg=cfg)
|
||||||
|
return _interface
|
||||||
|
|
||||||
|
|
||||||
|
def get_speaker(name: str):
|
||||||
|
if name not in _speakers:
|
||||||
|
_speakers[name] = get_interface().load_default_speaker(name=name)
|
||||||
|
return _speakers[name]
|
||||||
|
|
||||||
|
|
||||||
|
app = FastAPI(title="OuteTTS Worker")
|
||||||
|
|
||||||
|
|
||||||
|
class SynthReq(BaseModel):
|
||||||
|
text: str
|
||||||
|
voice: str = "female_1"
|
||||||
|
speed: float = 1.0
|
||||||
|
pitch: float = 0.0
|
||||||
|
|
||||||
|
|
||||||
|
@app.on_event("startup")
|
||||||
|
def _startup() -> None:
|
||||||
|
get_speaker("female_1")
|
||||||
|
|
||||||
|
|
||||||
|
@app.get("/health")
|
||||||
|
def health() -> dict:
|
||||||
|
get_interface()
|
||||||
|
return {"status": "ok", "voices": VOICES}
|
||||||
|
|
||||||
|
|
||||||
|
@app.post("/synth")
|
||||||
|
def synth(req: SynthReq) -> Response:
|
||||||
|
text = (req.text or "").strip()
|
||||||
|
if not text:
|
||||||
|
return Response(content="empty text", status_code=400)
|
||||||
|
|
||||||
|
it = get_interface()
|
||||||
|
voice = req.voice if req.voice in VOICES else "female_1"
|
||||||
|
speaker = get_speaker(voice)
|
||||||
|
speed = float(max(0.5, min(2.0, req.speed)))
|
||||||
|
pitch = float(max(-12.0, min(12.0, req.pitch)))
|
||||||
|
|
||||||
|
out = it.generate(
|
||||||
|
text=text,
|
||||||
|
temperature=0.1,
|
||||||
|
repetition_penalty=1.1,
|
||||||
|
max_length=4096,
|
||||||
|
speaker=speaker,
|
||||||
|
)
|
||||||
|
audio = out.audio.cpu().numpy().reshape(-1).astype(np.float32)
|
||||||
|
sr = int(out.sr)
|
||||||
|
|
||||||
|
if librosa is not None:
|
||||||
|
if abs(speed - 1.0) > 1e-3:
|
||||||
|
try:
|
||||||
|
audio = librosa.effects.time_stretch(audio, rate=speed).astype(np.float32)
|
||||||
|
except Exception:
|
||||||
|
try:
|
||||||
|
audio = librosa.effects.time_stretch(audio, speed).astype(np.float32)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
if abs(pitch) > 1e-3:
|
||||||
|
audio = librosa.effects.pitch_shift(audio, sr=sr, n_steps=pitch)
|
||||||
|
|
||||||
|
buf = io.BytesIO()
|
||||||
|
sf.write(buf, audio, sr, format="WAV", subtype="PCM_16")
|
||||||
|
buf.seek(0)
|
||||||
|
return Response(content=buf.read(), media_type="audio/wav")
|
||||||
Reference in New Issue
Block a user