fix: 전 엔진 속도 조절을 피치보존 타임스트레치(Rubberband R3)로 통일

- Supertonic/OuteTTS도 자연 속도 합성 후 Rubberband로 배속 → 고배속에서 목소리 안 뭉개짐
- Typecast 등 AI보이스가 쓰는 방식(pitch-preserving TSM)과 동일
- pyrubberband 워커 의존성 추가(rubberband-cli는 이미 이미지에 포함)
This commit is contained in:
claude
2026-08-25 19:46:20 +09:00
parent df18f64d1f
commit febb3f3ee4
4 changed files with 73 additions and 12 deletions

View File

@@ -5,3 +5,4 @@ soundfile
librosa librosa
fastapi==0.115.6 fastapi==0.115.6
uvicorn==0.30.0 uvicorn==0.30.0
pyrubberband==0.4.0

View File

@@ -8,6 +8,8 @@ from __future__ import annotations
import io import io
import shutil
import numpy as np import numpy as np
import soundfile as sf import soundfile as sf
@@ -16,6 +18,35 @@ try:
except Exception: # pragma: no cover except Exception: # pragma: no cover
librosa = None librosa = None
try:
import pyrubberband as prb
except Exception: # pragma: no cover
prb = None
def time_stretch(audio: np.ndarray, sr: int, rate: float) -> np.ndarray:
"""피치 보존 배속(Rubberband R3 → librosa 폴백). rate>1 = 빠르게."""
if abs(rate - 1.0) < 1e-3:
return audio
audio = np.asarray(audio, dtype=np.float32)
if prb is not None and shutil.which("rubberband"):
try:
return np.asarray(
prb.time_stretch(audio, sr, rate, rbargs={"--engine": "finer"}),
dtype=np.float32,
)
except Exception:
try:
return np.asarray(prb.time_stretch(audio, sr, rate), dtype=np.float32)
except Exception:
pass
if librosa is not None:
try:
return librosa.effects.time_stretch(audio, rate=rate).astype(np.float32)
except Exception:
pass
return audio
from fastapi import FastAPI from fastapi import FastAPI
from fastapi.responses import Response from fastapi.responses import Response
from pydantic import BaseModel from pydantic import BaseModel
@@ -87,17 +118,11 @@ def synth(req: SynthReq) -> Response:
audio = out.audio.cpu().numpy().reshape(-1).astype(np.float32) audio = out.audio.cpu().numpy().reshape(-1).astype(np.float32)
sr = int(out.sr) sr = int(out.sr)
if librosa is not None: # 피치 보존 배속(고배속에서도 목소리 안 뭉개짐)
if abs(speed - 1.0) > 1e-3: if abs(speed - 1.0) > 1e-3:
try: audio = time_stretch(audio, sr, speed)
audio = librosa.effects.time_stretch(audio, rate=speed).astype(np.float32) if abs(pitch) > 1e-3 and librosa is not None:
except Exception: audio = librosa.effects.pitch_shift(audio, sr=sr, n_steps=pitch)
try:
audio = librosa.effects.time_stretch(audio, speed).astype(np.float32)
except Exception:
pass
if abs(pitch) > 1e-3:
audio = librosa.effects.pitch_shift(audio, sr=sr, n_steps=pitch)
buf = io.BytesIO() buf = io.BytesIO()
sf.write(buf, audio, sr, format="WAV", subtype="PCM_16") sf.write(buf, audio, sr, format="WAV", subtype="PCM_16")

View File

@@ -2,3 +2,4 @@
supertonic==1.3.1 supertonic==1.3.1
fastapi==0.115.6 fastapi==0.115.6
uvicorn==0.30.0 uvicorn==0.30.0
pyrubberband==0.4.0

View File

@@ -9,6 +9,8 @@ from __future__ import annotations
import io import io
import shutil
import numpy as np import numpy as np
import soundfile as sf import soundfile as sf
@@ -17,6 +19,35 @@ try:
except Exception: # pragma: no cover except Exception: # pragma: no cover
librosa = None librosa = None
try:
import pyrubberband as prb
except Exception: # pragma: no cover
prb = None
def time_stretch(audio: np.ndarray, sr: int, rate: float) -> np.ndarray:
"""피치 보존 배속(Rubberband R3 → librosa 폴백). rate>1 = 빠르게."""
if abs(rate - 1.0) < 1e-3:
return audio
audio = np.asarray(audio, dtype=np.float32)
if prb is not None and shutil.which("rubberband"):
try:
return np.asarray(
prb.time_stretch(audio, sr, rate, rbargs={"--engine": "finer"}),
dtype=np.float32,
)
except Exception:
try:
return np.asarray(prb.time_stretch(audio, sr, rate), dtype=np.float32)
except Exception:
pass
if librosa is not None:
try:
return librosa.effects.time_stretch(audio, rate=rate).astype(np.float32)
except Exception:
pass
return audio
from fastapi import FastAPI from fastapi import FastAPI
from fastapi.responses import Response from fastapi.responses import Response
from pydantic import BaseModel from pydantic import BaseModel
@@ -79,9 +110,12 @@ def synth(req: SynthReq) -> Response:
speed = float(max(0.5, min(2.0, req.speed))) speed = float(max(0.5, min(2.0, req.speed)))
pitch = float(max(-12.0, min(12.0, req.pitch))) pitch = float(max(-12.0, min(12.0, req.pitch)))
wav, _dur = tts.synthesize(text, voice_style=style, lang=lang, speed=speed) # 자연 속도로 합성 후 피치 보존 배속(고배속에서도 목소리 안 뭉개짐)
wav, _dur = tts.synthesize(text, voice_style=style, lang=lang, speed=1.0)
audio = np.asarray(wav).reshape(-1).astype(np.float32) audio = np.asarray(wav).reshape(-1).astype(np.float32)
if abs(speed - 1.0) > 1e-3:
audio = time_stretch(audio, SR, speed)
if abs(pitch) > 1e-3 and librosa is not None: if abs(pitch) > 1e-3 and librosa is not None:
audio = librosa.effects.pitch_shift(audio, sr=SR, n_steps=pitch) audio = librosa.effects.pitch_shift(audio, sr=SR, n_steps=pitch)