Files
tts_site/outetts_worker/worker.py
claude febb3f3ee4 fix: 전 엔진 속도 조절을 피치보존 타임스트레치(Rubberband R3)로 통일
- Supertonic/OuteTTS도 자연 속도 합성 후 Rubberband로 배속 → 고배속에서 목소리 안 뭉개짐
- Typecast 등 AI보이스가 쓰는 방식(pitch-preserving TSM)과 동일
- pyrubberband 워커 의존성 추가(rubberband-cli는 이미 이미지에 포함)
2026-08-25 19:46:20 +09:00

131 lines
3.4 KiB
Python

"""OuteTTS 0.2-500M 한국어 합성 워커 (전용 venv, 개인용/비상업 CC-BY-NC).
- Qwen2 기반 500M LM + WavTokenizer 코덱, torch cu128 GPU 사용
- 한국어 프리셋 화자 4종(male_1/2, female_1/2)
- 속도/피치는 librosa 후처리(OuteTTS 자체 speed 파라미터 없음)
"""
from __future__ import annotations
import io
import shutil
import numpy as np
import soundfile as sf
try:
import librosa
except Exception: # pragma: no cover
librosa = None
try:
import pyrubberband as prb
except Exception: # pragma: no cover
prb = None
def time_stretch(audio: np.ndarray, sr: int, rate: float) -> np.ndarray:
"""피치 보존 배속(Rubberband R3 → librosa 폴백). rate>1 = 빠르게."""
if abs(rate - 1.0) < 1e-3:
return audio
audio = np.asarray(audio, dtype=np.float32)
if prb is not None and shutil.which("rubberband"):
try:
return np.asarray(
prb.time_stretch(audio, sr, rate, rbargs={"--engine": "finer"}),
dtype=np.float32,
)
except Exception:
try:
return np.asarray(prb.time_stretch(audio, sr, rate), dtype=np.float32)
except Exception:
pass
if librosa is not None:
try:
return librosa.effects.time_stretch(audio, rate=rate).astype(np.float32)
except Exception:
pass
return audio
from fastapi import FastAPI
from fastapi.responses import Response
from pydantic import BaseModel
import outetts
VOICES = ["female_1", "female_2", "male_1", "male_2"]
_interface = None
_speakers: dict = {}
def get_interface():
global _interface
if _interface is None:
cfg = outetts.HFModelConfig_v1(
model_path="OuteAI/OuteTTS-0.2-500M", language="ko"
)
_interface = outetts.InterfaceHF(model_version="0.2", cfg=cfg)
return _interface
def get_speaker(name: str):
if name not in _speakers:
_speakers[name] = get_interface().load_default_speaker(name=name)
return _speakers[name]
app = FastAPI(title="OuteTTS Worker")
class SynthReq(BaseModel):
text: str
voice: str = "female_1"
speed: float = 1.0
pitch: float = 0.0
@app.on_event("startup")
def _startup() -> None:
get_speaker("female_1")
@app.get("/health")
def health() -> dict:
get_interface()
return {"status": "ok", "voices": VOICES}
@app.post("/synth")
def synth(req: SynthReq) -> Response:
text = (req.text or "").strip()
if not text:
return Response(content="empty text", status_code=400)
it = get_interface()
voice = req.voice if req.voice in VOICES else "female_1"
speaker = get_speaker(voice)
speed = float(max(0.5, min(2.0, req.speed)))
pitch = float(max(-12.0, min(12.0, req.pitch)))
out = it.generate(
text=text,
temperature=0.1,
repetition_penalty=1.1,
max_length=4096,
speaker=speaker,
)
audio = out.audio.cpu().numpy().reshape(-1).astype(np.float32)
sr = int(out.sr)
# 피치 보존 배속(고배속에서도 목소리 안 뭉개짐)
if abs(speed - 1.0) > 1e-3:
audio = time_stretch(audio, sr, speed)
if abs(pitch) > 1e-3 and librosa is not None:
audio = librosa.effects.pitch_shift(audio, sr=sr, n_steps=pitch)
buf = io.BytesIO()
sf.write(buf, audio, sr, format="WAV", subtype="PCM_16")
buf.seek(0)
return Response(content=buf.read(), media_type="audio/wav")