- Qwen2 500M + WavTokenizer, torch cu128 GPU, 전용 venv 워커(8003) - 한국어 프리셋 4종(male/female), 속도·피치 librosa 후처리 - transformers 4.49로 torchvision 강제 import 회피, soundfile 저장 - scenema-audio는 VRAM 24GB+/Gemma12B 게이트로 8GB GPU에선 불가 → 미포함
106 lines
2.7 KiB
Python
106 lines
2.7 KiB
Python
"""OuteTTS 0.2-500M 한국어 합성 워커 (전용 venv, 개인용/비상업 CC-BY-NC).
|
|
|
|
- Qwen2 기반 500M LM + WavTokenizer 코덱, torch cu128 GPU 사용
|
|
- 한국어 프리셋 화자 4종(male_1/2, female_1/2)
|
|
- 속도/피치는 librosa 후처리(OuteTTS 자체 speed 파라미터 없음)
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import io
|
|
|
|
import numpy as np
|
|
import soundfile as sf
|
|
|
|
try:
|
|
import librosa
|
|
except Exception: # pragma: no cover
|
|
librosa = None
|
|
|
|
from fastapi import FastAPI
|
|
from fastapi.responses import Response
|
|
from pydantic import BaseModel
|
|
|
|
import outetts
|
|
|
|
VOICES = ["female_1", "female_2", "male_1", "male_2"]
|
|
|
|
_interface = None
|
|
_speakers: dict = {}
|
|
|
|
|
|
def get_interface():
|
|
global _interface
|
|
if _interface is None:
|
|
cfg = outetts.HFModelConfig_v1(
|
|
model_path="OuteAI/OuteTTS-0.2-500M", language="ko"
|
|
)
|
|
_interface = outetts.InterfaceHF(model_version="0.2", cfg=cfg)
|
|
return _interface
|
|
|
|
|
|
def get_speaker(name: str):
|
|
if name not in _speakers:
|
|
_speakers[name] = get_interface().load_default_speaker(name=name)
|
|
return _speakers[name]
|
|
|
|
|
|
app = FastAPI(title="OuteTTS Worker")
|
|
|
|
|
|
class SynthReq(BaseModel):
|
|
text: str
|
|
voice: str = "female_1"
|
|
speed: float = 1.0
|
|
pitch: float = 0.0
|
|
|
|
|
|
@app.on_event("startup")
|
|
def _startup() -> None:
|
|
get_speaker("female_1")
|
|
|
|
|
|
@app.get("/health")
|
|
def health() -> dict:
|
|
get_interface()
|
|
return {"status": "ok", "voices": VOICES}
|
|
|
|
|
|
@app.post("/synth")
|
|
def synth(req: SynthReq) -> Response:
|
|
text = (req.text or "").strip()
|
|
if not text:
|
|
return Response(content="empty text", status_code=400)
|
|
|
|
it = get_interface()
|
|
voice = req.voice if req.voice in VOICES else "female_1"
|
|
speaker = get_speaker(voice)
|
|
speed = float(max(0.5, min(2.0, req.speed)))
|
|
pitch = float(max(-12.0, min(12.0, req.pitch)))
|
|
|
|
out = it.generate(
|
|
text=text,
|
|
temperature=0.1,
|
|
repetition_penalty=1.1,
|
|
max_length=4096,
|
|
speaker=speaker,
|
|
)
|
|
audio = out.audio.cpu().numpy().reshape(-1).astype(np.float32)
|
|
sr = int(out.sr)
|
|
|
|
if librosa is not None:
|
|
if abs(speed - 1.0) > 1e-3:
|
|
try:
|
|
audio = librosa.effects.time_stretch(audio, rate=speed).astype(np.float32)
|
|
except Exception:
|
|
try:
|
|
audio = librosa.effects.time_stretch(audio, speed).astype(np.float32)
|
|
except Exception:
|
|
pass
|
|
if abs(pitch) > 1e-3:
|
|
audio = librosa.effects.pitch_shift(audio, sr=sr, n_steps=pitch)
|
|
|
|
buf = io.BytesIO()
|
|
sf.write(buf, audio, sr, format="WAV", subtype="PCM_16")
|
|
buf.seek(0)
|
|
return Response(content=buf.read(), media_type="audio/wav")
|