fix: 전 엔진 속도 조절을 피치보존 타임스트레치(Rubberband R3)로 통일
- Supertonic/OuteTTS도 자연 속도 합성 후 Rubberband로 배속 → 고배속에서 목소리 안 뭉개짐 - Typecast 등 AI보이스가 쓰는 방식(pitch-preserving TSM)과 동일 - pyrubberband 워커 의존성 추가(rubberband-cli는 이미 이미지에 포함)
This commit is contained in:
@@ -2,3 +2,4 @@
|
||||
supertonic==1.3.1
|
||||
fastapi==0.115.6
|
||||
uvicorn==0.30.0
|
||||
pyrubberband==0.4.0
|
||||
|
||||
@@ -9,6 +9,8 @@ from __future__ import annotations
|
||||
|
||||
import io
|
||||
|
||||
import shutil
|
||||
|
||||
import numpy as np
|
||||
import soundfile as sf
|
||||
|
||||
@@ -17,6 +19,35 @@ try:
|
||||
except Exception: # pragma: no cover
|
||||
librosa = None
|
||||
|
||||
try:
|
||||
import pyrubberband as prb
|
||||
except Exception: # pragma: no cover
|
||||
prb = None
|
||||
|
||||
|
||||
def time_stretch(audio: np.ndarray, sr: int, rate: float) -> np.ndarray:
|
||||
"""피치 보존 배속(Rubberband R3 → librosa 폴백). rate>1 = 빠르게."""
|
||||
if abs(rate - 1.0) < 1e-3:
|
||||
return audio
|
||||
audio = np.asarray(audio, dtype=np.float32)
|
||||
if prb is not None and shutil.which("rubberband"):
|
||||
try:
|
||||
return np.asarray(
|
||||
prb.time_stretch(audio, sr, rate, rbargs={"--engine": "finer"}),
|
||||
dtype=np.float32,
|
||||
)
|
||||
except Exception:
|
||||
try:
|
||||
return np.asarray(prb.time_stretch(audio, sr, rate), dtype=np.float32)
|
||||
except Exception:
|
||||
pass
|
||||
if librosa is not None:
|
||||
try:
|
||||
return librosa.effects.time_stretch(audio, rate=rate).astype(np.float32)
|
||||
except Exception:
|
||||
pass
|
||||
return audio
|
||||
|
||||
from fastapi import FastAPI
|
||||
from fastapi.responses import Response
|
||||
from pydantic import BaseModel
|
||||
@@ -79,9 +110,12 @@ def synth(req: SynthReq) -> Response:
|
||||
speed = float(max(0.5, min(2.0, req.speed)))
|
||||
pitch = float(max(-12.0, min(12.0, req.pitch)))
|
||||
|
||||
wav, _dur = tts.synthesize(text, voice_style=style, lang=lang, speed=speed)
|
||||
# 자연 속도로 합성 후 피치 보존 배속(고배속에서도 목소리 안 뭉개짐)
|
||||
wav, _dur = tts.synthesize(text, voice_style=style, lang=lang, speed=1.0)
|
||||
audio = np.asarray(wav).reshape(-1).astype(np.float32)
|
||||
|
||||
if abs(speed - 1.0) > 1e-3:
|
||||
audio = time_stretch(audio, SR, speed)
|
||||
if abs(pitch) > 1e-3 and librosa is not None:
|
||||
audio = librosa.effects.pitch_shift(audio, sr=SR, n_steps=pitch)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user