"""OuteTTS 0.2-500M 한국어 합성 워커 (전용 venv, 개인용/비상업 CC-BY-NC). - Qwen2 기반 500M LM + WavTokenizer 코덱, torch cu128 GPU 사용 - 한국어 프리셋 화자 4종(male_1/2, female_1/2) - 속도/피치는 librosa 후처리(OuteTTS 자체 speed 파라미터 없음) """ from __future__ import annotations import io import numpy as np import soundfile as sf try: import librosa except Exception: # pragma: no cover librosa = None from fastapi import FastAPI from fastapi.responses import Response from pydantic import BaseModel import outetts VOICES = ["female_1", "female_2", "male_1", "male_2"] _interface = None _speakers: dict = {} def get_interface(): global _interface if _interface is None: cfg = outetts.HFModelConfig_v1( model_path="OuteAI/OuteTTS-0.2-500M", language="ko" ) _interface = outetts.InterfaceHF(model_version="0.2", cfg=cfg) return _interface def get_speaker(name: str): if name not in _speakers: _speakers[name] = get_interface().load_default_speaker(name=name) return _speakers[name] app = FastAPI(title="OuteTTS Worker") class SynthReq(BaseModel): text: str voice: str = "female_1" speed: float = 1.0 pitch: float = 0.0 @app.on_event("startup") def _startup() -> None: get_speaker("female_1") @app.get("/health") def health() -> dict: get_interface() return {"status": "ok", "voices": VOICES} @app.post("/synth") def synth(req: SynthReq) -> Response: text = (req.text or "").strip() if not text: return Response(content="empty text", status_code=400) it = get_interface() voice = req.voice if req.voice in VOICES else "female_1" speaker = get_speaker(voice) speed = float(max(0.5, min(2.0, req.speed))) pitch = float(max(-12.0, min(12.0, req.pitch))) out = it.generate( text=text, temperature=0.1, repetition_penalty=1.1, max_length=4096, speaker=speaker, ) audio = out.audio.cpu().numpy().reshape(-1).astype(np.float32) sr = int(out.sr) if librosa is not None: if abs(speed - 1.0) > 1e-3: try: audio = librosa.effects.time_stretch(audio, rate=speed).astype(np.float32) except Exception: try: audio = librosa.effects.time_stretch(audio, speed).astype(np.float32) except Exception: pass if abs(pitch) > 1e-3: audio = librosa.effects.pitch_shift(audio, sr=sr, n_steps=pitch) buf = io.BytesIO() sf.write(buf, audio, sr, format="WAV", subtype="PCM_16") buf.seek(0) return Response(content=buf.read(), media_type="audio/wav")