fix: 전 엔진 속도 조절을 피치보존 타임스트레치(Rubberband R3)로 통일
- Supertonic/OuteTTS도 자연 속도 합성 후 Rubberband로 배속 → 고배속에서 목소리 안 뭉개짐 - Typecast 등 AI보이스가 쓰는 방식(pitch-preserving TSM)과 동일 - pyrubberband 워커 의존성 추가(rubberband-cli는 이미 이미지에 포함)
This commit is contained in:
@@ -5,3 +5,4 @@ soundfile
|
|||||||
librosa
|
librosa
|
||||||
fastapi==0.115.6
|
fastapi==0.115.6
|
||||||
uvicorn==0.30.0
|
uvicorn==0.30.0
|
||||||
|
pyrubberband==0.4.0
|
||||||
|
|||||||
@@ -8,6 +8,8 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import io
|
import io
|
||||||
|
|
||||||
|
import shutil
|
||||||
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
import soundfile as sf
|
import soundfile as sf
|
||||||
|
|
||||||
@@ -16,6 +18,35 @@ try:
|
|||||||
except Exception: # pragma: no cover
|
except Exception: # pragma: no cover
|
||||||
librosa = None
|
librosa = None
|
||||||
|
|
||||||
|
try:
|
||||||
|
import pyrubberband as prb
|
||||||
|
except Exception: # pragma: no cover
|
||||||
|
prb = None
|
||||||
|
|
||||||
|
|
||||||
|
def time_stretch(audio: np.ndarray, sr: int, rate: float) -> np.ndarray:
|
||||||
|
"""피치 보존 배속(Rubberband R3 → librosa 폴백). rate>1 = 빠르게."""
|
||||||
|
if abs(rate - 1.0) < 1e-3:
|
||||||
|
return audio
|
||||||
|
audio = np.asarray(audio, dtype=np.float32)
|
||||||
|
if prb is not None and shutil.which("rubberband"):
|
||||||
|
try:
|
||||||
|
return np.asarray(
|
||||||
|
prb.time_stretch(audio, sr, rate, rbargs={"--engine": "finer"}),
|
||||||
|
dtype=np.float32,
|
||||||
|
)
|
||||||
|
except Exception:
|
||||||
|
try:
|
||||||
|
return np.asarray(prb.time_stretch(audio, sr, rate), dtype=np.float32)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
if librosa is not None:
|
||||||
|
try:
|
||||||
|
return librosa.effects.time_stretch(audio, rate=rate).astype(np.float32)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
return audio
|
||||||
|
|
||||||
from fastapi import FastAPI
|
from fastapi import FastAPI
|
||||||
from fastapi.responses import Response
|
from fastapi.responses import Response
|
||||||
from pydantic import BaseModel
|
from pydantic import BaseModel
|
||||||
@@ -87,16 +118,10 @@ def synth(req: SynthReq) -> Response:
|
|||||||
audio = out.audio.cpu().numpy().reshape(-1).astype(np.float32)
|
audio = out.audio.cpu().numpy().reshape(-1).astype(np.float32)
|
||||||
sr = int(out.sr)
|
sr = int(out.sr)
|
||||||
|
|
||||||
if librosa is not None:
|
# 피치 보존 배속(고배속에서도 목소리 안 뭉개짐)
|
||||||
if abs(speed - 1.0) > 1e-3:
|
if abs(speed - 1.0) > 1e-3:
|
||||||
try:
|
audio = time_stretch(audio, sr, speed)
|
||||||
audio = librosa.effects.time_stretch(audio, rate=speed).astype(np.float32)
|
if abs(pitch) > 1e-3 and librosa is not None:
|
||||||
except Exception:
|
|
||||||
try:
|
|
||||||
audio = librosa.effects.time_stretch(audio, speed).astype(np.float32)
|
|
||||||
except Exception:
|
|
||||||
pass
|
|
||||||
if abs(pitch) > 1e-3:
|
|
||||||
audio = librosa.effects.pitch_shift(audio, sr=sr, n_steps=pitch)
|
audio = librosa.effects.pitch_shift(audio, sr=sr, n_steps=pitch)
|
||||||
|
|
||||||
buf = io.BytesIO()
|
buf = io.BytesIO()
|
||||||
|
|||||||
@@ -2,3 +2,4 @@
|
|||||||
supertonic==1.3.1
|
supertonic==1.3.1
|
||||||
fastapi==0.115.6
|
fastapi==0.115.6
|
||||||
uvicorn==0.30.0
|
uvicorn==0.30.0
|
||||||
|
pyrubberband==0.4.0
|
||||||
|
|||||||
@@ -9,6 +9,8 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import io
|
import io
|
||||||
|
|
||||||
|
import shutil
|
||||||
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
import soundfile as sf
|
import soundfile as sf
|
||||||
|
|
||||||
@@ -17,6 +19,35 @@ try:
|
|||||||
except Exception: # pragma: no cover
|
except Exception: # pragma: no cover
|
||||||
librosa = None
|
librosa = None
|
||||||
|
|
||||||
|
try:
|
||||||
|
import pyrubberband as prb
|
||||||
|
except Exception: # pragma: no cover
|
||||||
|
prb = None
|
||||||
|
|
||||||
|
|
||||||
|
def time_stretch(audio: np.ndarray, sr: int, rate: float) -> np.ndarray:
|
||||||
|
"""피치 보존 배속(Rubberband R3 → librosa 폴백). rate>1 = 빠르게."""
|
||||||
|
if abs(rate - 1.0) < 1e-3:
|
||||||
|
return audio
|
||||||
|
audio = np.asarray(audio, dtype=np.float32)
|
||||||
|
if prb is not None and shutil.which("rubberband"):
|
||||||
|
try:
|
||||||
|
return np.asarray(
|
||||||
|
prb.time_stretch(audio, sr, rate, rbargs={"--engine": "finer"}),
|
||||||
|
dtype=np.float32,
|
||||||
|
)
|
||||||
|
except Exception:
|
||||||
|
try:
|
||||||
|
return np.asarray(prb.time_stretch(audio, sr, rate), dtype=np.float32)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
if librosa is not None:
|
||||||
|
try:
|
||||||
|
return librosa.effects.time_stretch(audio, rate=rate).astype(np.float32)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
return audio
|
||||||
|
|
||||||
from fastapi import FastAPI
|
from fastapi import FastAPI
|
||||||
from fastapi.responses import Response
|
from fastapi.responses import Response
|
||||||
from pydantic import BaseModel
|
from pydantic import BaseModel
|
||||||
@@ -79,9 +110,12 @@ def synth(req: SynthReq) -> Response:
|
|||||||
speed = float(max(0.5, min(2.0, req.speed)))
|
speed = float(max(0.5, min(2.0, req.speed)))
|
||||||
pitch = float(max(-12.0, min(12.0, req.pitch)))
|
pitch = float(max(-12.0, min(12.0, req.pitch)))
|
||||||
|
|
||||||
wav, _dur = tts.synthesize(text, voice_style=style, lang=lang, speed=speed)
|
# 자연 속도로 합성 후 피치 보존 배속(고배속에서도 목소리 안 뭉개짐)
|
||||||
|
wav, _dur = tts.synthesize(text, voice_style=style, lang=lang, speed=1.0)
|
||||||
audio = np.asarray(wav).reshape(-1).astype(np.float32)
|
audio = np.asarray(wav).reshape(-1).astype(np.float32)
|
||||||
|
|
||||||
|
if abs(speed - 1.0) > 1e-3:
|
||||||
|
audio = time_stretch(audio, SR, speed)
|
||||||
if abs(pitch) > 1e-3 and librosa is not None:
|
if abs(pitch) > 1e-3 and librosa is not None:
|
||||||
audio = librosa.effects.pitch_shift(audio, sr=SR, n_steps=pitch)
|
audio = librosa.effects.pitch_shift(audio, sr=SR, n_steps=pitch)
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user