fix: 속도 조절을 length_scale에서 사후 타임스트레치로 분리(발음 뭉개짐 개선)

- 자연 속도(1.0)로 합성 후 Rubberband(R3)로 배속 → 발음/피치 보존
- 폴백: ffmpeg atempo → librosa. rubberband-cli/pyrubberband 추가
- 원인: MeloTTS speed는 length_scale 축소라 고배속에서 자음이 뭉개짐
This commit is contained in:
claude
2026-08-24 14:42:21 +09:00
parent 9fe1a5479d
commit 1942a7aecd
3 changed files with 82 additions and 6 deletions

View File

@@ -53,7 +53,7 @@ ENV PYTHONUNBUFFERED=1 \
# 런타임 필수 라이브러리만 (컴파일러 툴체인 제외) # 런타임 필수 라이브러리만 (컴파일러 툴체인 제외)
RUN apt-get update && apt-get install -y --no-install-recommends \ RUN apt-get update && apt-get install -y --no-install-recommends \
ffmpeg libsndfile1 libgomp1 ca-certificates \ ffmpeg libsndfile1 libgomp1 ca-certificates rubberband-cli \
&& rm -rf /var/lib/apt/lists/* && rm -rf /var/lib/apt/lists/*
WORKDIR /app WORKDIR /app

View File

@@ -7,7 +7,11 @@
from __future__ import annotations from __future__ import annotations
import io import io
import os
import re import re
import shutil
import subprocess
import tempfile
import threading import threading
import numpy as np import numpy as np
@@ -15,10 +19,15 @@ import soundfile as sf
import torch import torch
try: try:
import librosa # 피치 조절용 import librosa # 피치 조절 / 폴백 타임스트레치
except Exception: # pragma: no cover except Exception: # pragma: no cover
librosa = None librosa = None
try:
import pyrubberband as prb # 고품질 타임스트레치(발음 보존)
except Exception: # pragma: no cover
prb = None
from melo.api import TTS from melo.api import TTS
from .base import BaseEngine from .base import BaseEngine
@@ -175,17 +184,79 @@ class MeloEngine(BaseEngine):
return list(spk2id.values())[0] return list(spk2id.values())[0]
def _synth_segment( def _synth_segment(
self, text: str, lang: str, speaker: str | None, speed: float self, text: str, lang: str, speaker: str | None
) -> tuple[np.ndarray, int]: ) -> tuple[np.ndarray, int]:
# 항상 자연 속도(1.0)로 합성 — 속도는 사후 타임스트레치로 처리해 발음 유지
model = self._get_model(lang) model = self._get_model(lang)
speaker_id = self._resolve_speaker(model, lang, speaker) speaker_id = self._resolve_speaker(model, lang, speaker)
with self._global_lock: with self._global_lock:
audio = model.tts_to_file( audio = model.tts_to_file(
text, speaker_id, output_path=None, speed=speed, quiet=True text, speaker_id, output_path=None, speed=1.0, quiet=True
) )
sr = model.hps.data.sampling_rate sr = model.hps.data.sampling_rate
return np.asarray(audio, dtype=np.float32), sr return np.asarray(audio, dtype=np.float32), sr
def _time_stretch(self, audio: np.ndarray, sr: int, rate: float) -> np.ndarray:
"""오디오를 rate 배속으로 변경(피치 보존). rate>1 = 빠르게.
MeloTTS 의 length_scale 방식(speed 파라미터)은 배속을 올리면 발음이
뭉개지므로, 자연 속도로 합성한 뒤 여기서 고품질 타임스트레치를 적용한다.
우선순위: Rubberband(R3) → ffmpeg atempo → librosa(phase vocoder).
"""
if abs(rate - 1.0) < 1e-3:
return audio
audio = np.asarray(audio, dtype=np.float32)
# 1) Rubberband R3 (발음/포먼트 보존이 가장 좋음)
if prb is not None and shutil.which("rubberband"):
try:
out = prb.time_stretch(
audio, sr, rate, rbargs={"--engine": "finer"}
)
return np.asarray(out, dtype=np.float32)
except Exception:
try:
out = prb.time_stretch(audio, sr, rate)
return np.asarray(out, dtype=np.float32)
except Exception:
pass
# 2) ffmpeg atempo (추가 의존성 없음, 무난)
if shutil.which("ffmpeg") and 0.5 <= rate <= 2.0:
try:
return self._atempo(audio, sr, rate)
except Exception:
pass
# 3) librosa phase vocoder (최후 폴백)
if librosa is not None:
try:
return librosa.effects.time_stretch(audio, rate).astype(np.float32)
except Exception:
try:
return librosa.effects.time_stretch(
audio, rate=rate
).astype(np.float32)
except Exception:
pass
return audio
@staticmethod
def _atempo(audio: np.ndarray, sr: int, rate: float) -> np.ndarray:
with tempfile.TemporaryDirectory() as d:
src = os.path.join(d, "in.wav")
dst = os.path.join(d, "out.wav")
sf.write(src, audio, sr, format="WAV", subtype="PCM_16")
subprocess.run(
[
"ffmpeg", "-hide_banner", "-loglevel", "error", "-y",
"-i", src, "-filter:a", f"atempo={rate:.4f}", dst,
],
check=True,
)
out, _ = sf.read(dst, dtype="float32")
return np.asarray(out, dtype=np.float32)
# ---- 합성 ----------------------------------------------------- # ---- 합성 -----------------------------------------------------
def synth_wav( def synth_wav(
self, self,
@@ -215,7 +286,7 @@ class MeloEngine(BaseEngine):
gap = None gap = None
for seg_lang, seg_text in parts: for seg_lang, seg_text in parts:
seg_spk = spk_by_lang.get(seg_lang, en_speaker) seg_spk = spk_by_lang.get(seg_lang, en_speaker)
audio, seg_sr = self._synth_segment(seg_text, seg_lang, seg_spk, speed) audio, seg_sr = self._synth_segment(seg_text, seg_lang, seg_spk)
if sr is None: if sr is None:
sr = seg_sr sr = seg_sr
gap = np.zeros(int(sr * 0.06), dtype=np.float32) gap = np.zeros(int(sr * 0.06), dtype=np.float32)
@@ -229,7 +300,11 @@ class MeloEngine(BaseEngine):
audios.append(audio) audios.append(audio)
audio = np.concatenate(audios) if audios else np.zeros(1, dtype=np.float32) audio = np.concatenate(audios) if audios else np.zeros(1, dtype=np.float32)
else: else:
audio, sr = self._synth_segment(text, language, speaker, speed) audio, sr = self._synth_segment(text, language, speaker)
# 속도: 자연 속도 합성 결과에 고품질 타임스트레치 적용(발음 유지)
if abs(speed - 1.0) > 1e-3:
audio = self._time_stretch(audio, sr, speed)
if abs(pitch) > 1e-3 and librosa is not None: if abs(pitch) > 1e-3 and librosa is not None:
audio = librosa.effects.pitch_shift(audio, sr, n_steps=pitch) audio = librosa.effects.pitch_shift(audio, sr, n_steps=pitch)

View File

@@ -7,6 +7,7 @@ python-multipart==0.0.32
soundfile==0.14.0 soundfile==0.14.0
librosa==0.9.1 librosa==0.9.1
numpy==1.26.4 numpy==1.26.4
pyrubberband==0.4.0
# MeloTTS 런타임 의존성 (정확한 버전은 constraints.txt 로 고정) # MeloTTS 런타임 의존성 (정확한 버전은 constraints.txt 로 고정)
# gradio/tensorboard/unidic(full) 등 미사용 무거운 의존성은 제외한다. # gradio/tensorboard/unidic(full) 등 미사용 무거운 의존성은 제외한다.