- Supertonic/OuteTTS도 자연 속도 합성 후 Rubberband로 배속 → 고배속에서 목소리 안 뭉개짐 - Typecast 등 AI보이스가 쓰는 방식(pitch-preserving TSM)과 동일 - pyrubberband 워커 의존성 추가(rubberband-cli는 이미 이미지에 포함)
126 lines
3.3 KiB
Python
126 lines
3.3 KiB
Python
"""Supertonic 합성 워커 (전용 venv에서 실행).
|
|
|
|
- ONNX 런타임 기반, torch 불필요, 모델 ~260MB
|
|
- 다국어(31개) 지원. 한국어는 lang="ko", 자동감지는 lang="na"
|
|
- 프리셋 음성 10종(M1~M5, F1~F5)
|
|
- 속도(speed) 네이티브, 피치(pitch)는 librosa 후처리
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import io
|
|
|
|
import shutil
|
|
|
|
import numpy as np
|
|
import soundfile as sf
|
|
|
|
try:
|
|
import librosa
|
|
except Exception: # pragma: no cover
|
|
librosa = None
|
|
|
|
try:
|
|
import pyrubberband as prb
|
|
except Exception: # pragma: no cover
|
|
prb = None
|
|
|
|
|
|
def time_stretch(audio: np.ndarray, sr: int, rate: float) -> np.ndarray:
|
|
"""피치 보존 배속(Rubberband R3 → librosa 폴백). rate>1 = 빠르게."""
|
|
if abs(rate - 1.0) < 1e-3:
|
|
return audio
|
|
audio = np.asarray(audio, dtype=np.float32)
|
|
if prb is not None and shutil.which("rubberband"):
|
|
try:
|
|
return np.asarray(
|
|
prb.time_stretch(audio, sr, rate, rbargs={"--engine": "finer"}),
|
|
dtype=np.float32,
|
|
)
|
|
except Exception:
|
|
try:
|
|
return np.asarray(prb.time_stretch(audio, sr, rate), dtype=np.float32)
|
|
except Exception:
|
|
pass
|
|
if librosa is not None:
|
|
try:
|
|
return librosa.effects.time_stretch(audio, rate=rate).astype(np.float32)
|
|
except Exception:
|
|
pass
|
|
return audio
|
|
|
|
from fastapi import FastAPI
|
|
from fastapi.responses import Response
|
|
from pydantic import BaseModel
|
|
|
|
from supertonic import TTS
|
|
|
|
VOICES = ["M1", "M2", "M3", "M4", "M5", "F1", "F2", "F3", "F4", "F5"]
|
|
SR = 44100
|
|
|
|
_tts = None
|
|
_styles: dict = {}
|
|
|
|
|
|
def get_tts() -> TTS:
|
|
global _tts
|
|
if _tts is None:
|
|
_tts = TTS(auto_download=True)
|
|
return _tts
|
|
|
|
|
|
def get_style(name: str):
|
|
if name not in _styles:
|
|
_styles[name] = get_tts().get_voice_style(voice_name=name)
|
|
return _styles[name]
|
|
|
|
|
|
app = FastAPI(title="Supertonic Worker")
|
|
|
|
|
|
class SynthReq(BaseModel):
|
|
text: str
|
|
voice: str = "F1"
|
|
lang: str = "ko"
|
|
speed: float = 1.0
|
|
pitch: float = 0.0
|
|
|
|
|
|
@app.on_event("startup")
|
|
def _startup() -> None:
|
|
get_tts()
|
|
get_style("F1")
|
|
|
|
|
|
@app.get("/health")
|
|
def health() -> dict:
|
|
get_tts()
|
|
return {"status": "ok", "voices": VOICES, "sr": SR}
|
|
|
|
|
|
@app.post("/synth")
|
|
def synth(req: SynthReq) -> Response:
|
|
text = (req.text or "").strip()
|
|
if not text:
|
|
return Response(content="empty text", status_code=400)
|
|
|
|
tts = get_tts()
|
|
voice = req.voice if req.voice in VOICES else "F1"
|
|
style = get_style(voice)
|
|
lang = req.lang if req.lang in ("ko", "na", "en", "ja", "zh") else "ko"
|
|
speed = float(max(0.5, min(2.0, req.speed)))
|
|
pitch = float(max(-12.0, min(12.0, req.pitch)))
|
|
|
|
# 자연 속도로 합성 후 피치 보존 배속(고배속에서도 목소리 안 뭉개짐)
|
|
wav, _dur = tts.synthesize(text, voice_style=style, lang=lang, speed=1.0)
|
|
audio = np.asarray(wav).reshape(-1).astype(np.float32)
|
|
|
|
if abs(speed - 1.0) > 1e-3:
|
|
audio = time_stretch(audio, SR, speed)
|
|
if abs(pitch) > 1e-3 and librosa is not None:
|
|
audio = librosa.effects.pitch_shift(audio, sr=SR, n_steps=pitch)
|
|
|
|
buf = io.BytesIO()
|
|
sf.write(buf, audio, SR, format="WAV", subtype="PCM_16")
|
|
buf.seek(0)
|
|
return Response(content=buf.read(), media_type="audio/wav")
|