"""Supertonic 합성 워커 (전용 venv에서 실행). - ONNX 런타임 기반, torch 불필요, 모델 ~260MB - 다국어(31개) 지원. 한국어는 lang="ko", 자동감지는 lang="na" - 프리셋 음성 10종(M1~M5, F1~F5) - 속도(speed) 네이티브, 피치(pitch)는 librosa 후처리 """ from __future__ import annotations import io import shutil import numpy as np import soundfile as sf try: import librosa except Exception: # pragma: no cover librosa = None try: import pyrubberband as prb except Exception: # pragma: no cover prb = None def time_stretch(audio: np.ndarray, sr: int, rate: float) -> np.ndarray: """피치 보존 배속(Rubberband R3 → librosa 폴백). rate>1 = 빠르게.""" if abs(rate - 1.0) < 1e-3: return audio audio = np.asarray(audio, dtype=np.float32) if prb is not None and shutil.which("rubberband"): try: return np.asarray( prb.time_stretch(audio, sr, rate, rbargs={"--engine": "finer"}), dtype=np.float32, ) except Exception: try: return np.asarray(prb.time_stretch(audio, sr, rate), dtype=np.float32) except Exception: pass if librosa is not None: try: return librosa.effects.time_stretch(audio, rate=rate).astype(np.float32) except Exception: pass return audio from fastapi import FastAPI from fastapi.responses import Response from pydantic import BaseModel from supertonic import TTS VOICES = ["M1", "M2", "M3", "M4", "M5", "F1", "F2", "F3", "F4", "F5"] SR = 44100 _tts = None _styles: dict = {} def get_tts() -> TTS: global _tts if _tts is None: _tts = TTS(auto_download=True) return _tts def get_style(name: str): if name not in _styles: _styles[name] = get_tts().get_voice_style(voice_name=name) return _styles[name] app = FastAPI(title="Supertonic Worker") class SynthReq(BaseModel): text: str voice: str = "F1" lang: str = "ko" speed: float = 1.0 pitch: float = 0.0 @app.on_event("startup") def _startup() -> None: get_tts() get_style("F1") @app.get("/health") def health() -> dict: get_tts() return {"status": "ok", "voices": VOICES, "sr": SR} @app.post("/synth") def synth(req: SynthReq) -> Response: text = (req.text or "").strip() if not text: return Response(content="empty text", status_code=400) tts = get_tts() voice = req.voice if req.voice in VOICES else "F1" style = get_style(voice) lang = req.lang if req.lang in ("ko", "na", "en", "ja", "zh") else "ko" speed = float(max(0.5, min(2.0, req.speed))) pitch = float(max(-12.0, min(12.0, req.pitch))) # 자연 속도로 합성 후 피치 보존 배속(고배속에서도 목소리 안 뭉개짐) wav, _dur = tts.synthesize(text, voice_style=style, lang=lang, speed=1.0) audio = np.asarray(wav).reshape(-1).astype(np.float32) if abs(speed - 1.0) > 1e-3: audio = time_stretch(audio, SR, speed) if abs(pitch) > 1e-3 and librosa is not None: audio = librosa.effects.pitch_shift(audio, sr=SR, n_steps=pitch) buf = io.BytesIO() sf.write(buf, audio, SR, format="WAV", subtype="PCM_16") buf.seek(0) return Response(content=buf.read(), media_type="audio/wav")