From 1942a7aecd88839a47f31520af5c42acb8bc8e4d Mon Sep 17 00:00:00 2001 From: claude Date: Mon, 24 Aug 2026 14:42:21 +0900 Subject: [PATCH] =?UTF-8?q?fix:=20=EC=86=8D=EB=8F=84=20=EC=A1=B0=EC=A0=88?= =?UTF-8?q?=EC=9D=84=20length=5Fscale=EC=97=90=EC=84=9C=20=EC=82=AC?= =?UTF-8?q?=ED=9B=84=20=ED=83=80=EC=9E=84=EC=8A=A4=ED=8A=B8=EB=A0=88?= =?UTF-8?q?=EC=B9=98=EB=A1=9C=20=EB=B6=84=EB=A6=AC(=EB=B0=9C=EC=9D=8C=20?= =?UTF-8?q?=EB=AD=89=EA=B0=9C=EC=A7=90=20=EA=B0=9C=EC=84=A0)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 자연 속도(1.0)로 합성 후 Rubberband(R3)로 배속 → 발음/피치 보존 - 폴백: ffmpeg atempo → librosa. rubberband-cli/pyrubberband 추가 - 원인: MeloTTS speed는 length_scale 축소라 고배속에서 자음이 뭉개짐 --- Dockerfile | 2 +- app/engines/melo_engine.py | 85 +++++++++++++++++++++++++++++++++++--- requirements.txt | 1 + 3 files changed, 82 insertions(+), 6 deletions(-) diff --git a/Dockerfile b/Dockerfile index 0806571..ea94c59 100644 --- a/Dockerfile +++ b/Dockerfile @@ -53,7 +53,7 @@ ENV PYTHONUNBUFFERED=1 \ # 런타임 필수 라이브러리만 (컴파일러 툴체인 제외) RUN apt-get update && apt-get install -y --no-install-recommends \ - ffmpeg libsndfile1 libgomp1 ca-certificates \ + ffmpeg libsndfile1 libgomp1 ca-certificates rubberband-cli \ && rm -rf /var/lib/apt/lists/* WORKDIR /app diff --git a/app/engines/melo_engine.py b/app/engines/melo_engine.py index 93fcc13..02a100c 100644 --- a/app/engines/melo_engine.py +++ b/app/engines/melo_engine.py @@ -7,7 +7,11 @@ from __future__ import annotations import io +import os import re +import shutil +import subprocess +import tempfile import threading import numpy as np @@ -15,10 +19,15 @@ import soundfile as sf import torch try: - import librosa # 피치 조절용 + import librosa # 피치 조절 / 폴백 타임스트레치 except Exception: # pragma: no cover librosa = None +try: + import pyrubberband as prb # 고품질 타임스트레치(발음 보존) +except Exception: # pragma: no cover + prb = None + from melo.api import TTS from .base import BaseEngine @@ -175,17 +184,79 @@ class MeloEngine(BaseEngine): return list(spk2id.values())[0] def _synth_segment( - self, text: str, lang: str, speaker: str | None, speed: float + self, text: str, lang: str, speaker: str | None ) -> tuple[np.ndarray, int]: + # 항상 자연 속도(1.0)로 합성 — 속도는 사후 타임스트레치로 처리해 발음 유지 model = self._get_model(lang) speaker_id = self._resolve_speaker(model, lang, speaker) with self._global_lock: audio = model.tts_to_file( - text, speaker_id, output_path=None, speed=speed, quiet=True + text, speaker_id, output_path=None, speed=1.0, quiet=True ) sr = model.hps.data.sampling_rate return np.asarray(audio, dtype=np.float32), sr + def _time_stretch(self, audio: np.ndarray, sr: int, rate: float) -> np.ndarray: + """오디오를 rate 배속으로 변경(피치 보존). rate>1 = 빠르게. + + MeloTTS 의 length_scale 방식(speed 파라미터)은 배속을 올리면 발음이 + 뭉개지므로, 자연 속도로 합성한 뒤 여기서 고품질 타임스트레치를 적용한다. + 우선순위: Rubberband(R3) → ffmpeg atempo → librosa(phase vocoder). + """ + if abs(rate - 1.0) < 1e-3: + return audio + audio = np.asarray(audio, dtype=np.float32) + + # 1) Rubberband R3 (발음/포먼트 보존이 가장 좋음) + if prb is not None and shutil.which("rubberband"): + try: + out = prb.time_stretch( + audio, sr, rate, rbargs={"--engine": "finer"} + ) + return np.asarray(out, dtype=np.float32) + except Exception: + try: + out = prb.time_stretch(audio, sr, rate) + return np.asarray(out, dtype=np.float32) + except Exception: + pass + + # 2) ffmpeg atempo (추가 의존성 없음, 무난) + if shutil.which("ffmpeg") and 0.5 <= rate <= 2.0: + try: + return self._atempo(audio, sr, rate) + except Exception: + pass + + # 3) librosa phase vocoder (최후 폴백) + if librosa is not None: + try: + return librosa.effects.time_stretch(audio, rate).astype(np.float32) + except Exception: + try: + return librosa.effects.time_stretch( + audio, rate=rate + ).astype(np.float32) + except Exception: + pass + return audio + + @staticmethod + def _atempo(audio: np.ndarray, sr: int, rate: float) -> np.ndarray: + with tempfile.TemporaryDirectory() as d: + src = os.path.join(d, "in.wav") + dst = os.path.join(d, "out.wav") + sf.write(src, audio, sr, format="WAV", subtype="PCM_16") + subprocess.run( + [ + "ffmpeg", "-hide_banner", "-loglevel", "error", "-y", + "-i", src, "-filter:a", f"atempo={rate:.4f}", dst, + ], + check=True, + ) + out, _ = sf.read(dst, dtype="float32") + return np.asarray(out, dtype=np.float32) + # ---- 합성 ----------------------------------------------------- def synth_wav( self, @@ -215,7 +286,7 @@ class MeloEngine(BaseEngine): gap = None for seg_lang, seg_text in parts: seg_spk = spk_by_lang.get(seg_lang, en_speaker) - audio, seg_sr = self._synth_segment(seg_text, seg_lang, seg_spk, speed) + audio, seg_sr = self._synth_segment(seg_text, seg_lang, seg_spk) if sr is None: sr = seg_sr gap = np.zeros(int(sr * 0.06), dtype=np.float32) @@ -229,7 +300,11 @@ class MeloEngine(BaseEngine): audios.append(audio) audio = np.concatenate(audios) if audios else np.zeros(1, dtype=np.float32) else: - audio, sr = self._synth_segment(text, language, speaker, speed) + audio, sr = self._synth_segment(text, language, speaker) + + # 속도: 자연 속도 합성 결과에 고품질 타임스트레치 적용(발음 유지) + if abs(speed - 1.0) > 1e-3: + audio = self._time_stretch(audio, sr, speed) if abs(pitch) > 1e-3 and librosa is not None: audio = librosa.effects.pitch_shift(audio, sr, n_steps=pitch) diff --git a/requirements.txt b/requirements.txt index 9c18213..9d34484 100644 --- a/requirements.txt +++ b/requirements.txt @@ -7,6 +7,7 @@ python-multipart==0.0.32 soundfile==0.14.0 librosa==0.9.1 numpy==1.26.4 +pyrubberband==0.4.0 # MeloTTS 런타임 의존성 (정확한 버전은 constraints.txt 로 고정) # gradio/tensorboard/unidic(full) 등 미사용 무거운 의존성은 제외한다.