feat: 글자속도·단어간격·문장간격 독립 조절 + atempo 기본 타임스트레치

- 타임스트레치 우선순위를 ffmpeg atempo(기본) → rubberband → librosa 로 변경
- MeloEngine: 문장 단위로 분리 후 문장 사이 무음(sentence_gap), 단어(어절) 사이
  무음(word_gap)을 타임스트레치와 독립적으로 삽입. speed 는 글자(음소) 속도만 조절
- API(TTSRequest)에 word_gap/sentence_gap 필드 추가, 전 엔진 시그니처 확장
- 프론트엔드에 '글자 속도/단어 간격/문장 간격' 슬라이더 추가, supports 플래그로
  미지원 엔진(supertonic/oute/cosy)에서는 간격 슬라이더 자동 비활성화
This commit is contained in:
claude
2026-08-25 23:54:48 +09:00
parent e16f88e534
commit c0665893a7
8 changed files with 181 additions and 50 deletions

View File

@@ -39,8 +39,14 @@ class BaseEngine(ABC):
speaker: str | None = None,
speed: float = 1.0,
pitch: float = 0.0,
word_gap: float = 0.0,
sentence_gap: float | None = None,
) -> bytes:
"""WAV(PCM16) 바이트를 반환."""
"""WAV(PCM16) 바이트를 반환.
speed=글자(음소) 속도, word_gap=단어 사이 무음(초),
sentence_gap=문장 사이 무음(초). 간격을 지원하지 않는 엔진은 무시한다.
"""
raise NotImplementedError
def warmup(self) -> None:

View File

@@ -42,7 +42,12 @@ class CosyVoiceEngine(BaseEngine):
],
}
],
"supports": {"speed": True, "pitch": True},
"supports": {
"speed": True,
"pitch": True,
"word_gap": False,
"sentence_gap": False,
},
}
def synth_wav(
@@ -52,6 +57,8 @@ class CosyVoiceEngine(BaseEngine):
speaker: str | None = None,
speed: float = 1.0,
pitch: float = 0.0,
word_gap: float = 0.0,
sentence_gap: float | None = None,
) -> bytes:
text = (text or "").strip()
if not text:

View File

@@ -50,6 +50,15 @@ MELO_LANG = {"KR": "KR", "EN": "EN", "JP": "JP", "ZH": "ZH"}
DEFAULT_LANG = "EN" # 판별 불가 구간의 기본 언어
# 문장 경계: 종결부호 뒤 공백, 또는 줄바꿈
_SENT_SPLIT_RE = re.compile(r"(?<=[.!?。!?…])\s+|\n+")
def split_sentences(text: str) -> list[str]:
"""텍스트를 문장 단위로 분리(종결부호/줄바꿈 기준)."""
parts = [p.strip() for p in _SENT_SPLIT_RE.split(text) if p and p.strip()]
return parts or ([text.strip()] if text.strip() else [])
_KR_RE = re.compile(r"[가-힣ᄀ-ᇿ㄰-㆏]")
_EN_RE = re.compile(r"[A-Za-z]")
# 일본어 가나(히라가나/가타카나/반각 가타카나)
@@ -171,7 +180,12 @@ class MeloEngine(BaseEngine):
"notes": self.notes,
"languages": languages,
"default_language": "AUTO",
"supports": {"speed": True, "pitch": True},
"supports": {
"speed": True,
"pitch": True,
"word_gap": True,
"sentence_gap": True,
},
}
def _resolve_speaker(self, model: TTS, lang: str, speaker: str | None) -> int:
@@ -200,15 +214,22 @@ class MeloEngine(BaseEngine):
def _time_stretch(self, audio: np.ndarray, sr: int, rate: float) -> np.ndarray:
"""오디오를 rate 배속으로 변경(피치 보존). rate>1 = 빠르게.
MeloTTS 의 length_scale 방식(speed 파라미터)은 배속을 올리면 발음이
뭉개지므로, 자연 속도로 합성한 뒤 여기서 고품질 타임스트레치를 적용한다.
우선순위: Rubberband(R3) → ffmpeg atempo → librosa(phase vocoder).
자연 속도(1.0)로 합성한 뒤 여기서 타임스트레치를 적용해 글자 발화
속도만 조절한다(피치 보존).
우선순위: ffmpeg atempo(기본) → Rubberband(R3) → librosa(phase vocoder).
"""
if abs(rate - 1.0) < 1e-3:
return audio
audio = np.asarray(audio, dtype=np.float32)
# 1) Rubberband R3 (발음/포먼트 보존이 가장 좋음)
# 1) ffmpeg atempo (기본값: 추가 의존성 없음, 무난)
if shutil.which("ffmpeg") and 0.5 <= rate <= 2.0:
try:
return self._atempo(audio, sr, rate)
except Exception:
pass
# 2) Rubberband R3 (발음/포먼트 보존 폴백)
if prb is not None and shutil.which("rubberband"):
try:
out = prb.time_stretch(
@@ -222,13 +243,6 @@ class MeloEngine(BaseEngine):
except Exception:
pass
# 2) ffmpeg atempo (추가 의존성 없음, 무난)
if shutil.which("ffmpeg") and 0.5 <= rate <= 2.0:
try:
return self._atempo(audio, sr, rate)
except Exception:
pass
# 3) librosa phase vocoder (최후 폴백)
if librosa is not None:
try:
@@ -258,27 +272,13 @@ class MeloEngine(BaseEngine):
out, _ = sf.read(dst, dtype="float32")
return np.asarray(out, dtype=np.float32)
# ---- 합성 -----------------------------------------------------
def synth_wav(
self,
text: str,
language: str,
speaker: str | None = None,
speed: float = 1.0,
pitch: float = 0.0,
) -> bytes:
text = (text or "").strip()
if not text:
raise ValueError("텍스트가 비어 있습니다.")
speed = float(max(0.5, min(2.0, speed)))
pitch = float(max(-12.0, min(12.0, pitch)))
language = (language or "AUTO").upper()
# 한국어 발음 개선: 기호/약어/사용자사전 전처리 (한국어·자동 모드)
if language in ("AUTO", "KR"):
text = normalize_korean(text)
def _synth_natural(
self, text: str, language: str, speaker: str | None
) -> tuple[np.ndarray, int]:
"""자연 속도(1.0)로 텍스트 한 조각을 합성. AUTO 모드는 언어별로 나눠 합성.
속도/문장·단어 간격 조절은 이 결과 위에서 처리한다(여기서는 하지 않음).
"""
if language == "AUTO":
# 영어 구간엔 사용자가 고른 억양, 한국어 구간엔 KR 음성
en_speaker = speaker if (speaker and speaker.startswith("EN")) else "EN-US"
@@ -295,21 +295,89 @@ class MeloEngine(BaseEngine):
if sr is None:
sr = seg_sr
gap = np.zeros(int(sr * 0.06), dtype=np.float32)
elif seg_sr != sr:
if librosa is not None:
audio = librosa.resample(
audio, orig_sr=seg_sr, target_sr=sr
).astype(np.float32)
elif seg_sr != sr and librosa is not None:
audio = librosa.resample(
audio, orig_sr=seg_sr, target_sr=sr
).astype(np.float32)
if audios:
audios.append(gap)
audios.append(audio)
audio = np.concatenate(audios) if audios else np.zeros(1, dtype=np.float32)
else:
audio, sr = self._synth_segment(text, language, speaker)
return audio, (sr or self._get_model(DEFAULT_LANG).hps.data.sampling_rate)
return self._synth_segment(text, language, speaker)
# 속도: 자연 속도 합성 결과에 고품질 타임스트레치 적용(발음 유지)
if abs(speed - 1.0) > 1e-3:
audio = self._time_stretch(audio, sr, speed)
# ---- 합성 -----------------------------------------------------
def synth_wav(
self,
text: str,
language: str,
speaker: str | None = None,
speed: float = 1.0,
pitch: float = 0.0,
word_gap: float = 0.0,
sentence_gap: float | None = None,
) -> bytes:
"""세 가지를 독립적으로 조절:
- speed: 글자(음소) 발화 속도 — 자연 합성 후 타임스트레치(피치 보존)
- word_gap: 단어 사이 추가 무음(초). >0 이면 어절 단위로 나눠 합성
- sentence_gap: 문장 사이 무음(초)
간격은 타임스트레치의 영향을 받지 않아 글자속도와 독립적으로 유지된다.
"""
text = (text or "").strip()
if not text:
raise ValueError("텍스트가 비어 있습니다.")
speed = float(max(0.5, min(2.0, speed)))
pitch = float(max(-12.0, min(12.0, pitch)))
word_gap = float(max(0.0, min(1.0, word_gap)))
sentence_gap = 0.30 if sentence_gap is None else sentence_gap
sentence_gap = float(max(0.0, min(2.0, sentence_gap)))
language = (language or "AUTO").upper()
# 한국어 발음 개선: 기호/약어/사용자사전 전처리 (한국어·자동 모드)
if language in ("AUTO", "KR"):
text = normalize_korean(text)
def stretch(a: np.ndarray, sr: int) -> np.ndarray:
if abs(speed - 1.0) > 1e-3:
return self._time_stretch(a, sr, speed)
return a
sentences = split_sentences(text) or [text]
sr: int | None = None
pieces: list[np.ndarray] = []
for si, sent in enumerate(sentences):
if word_gap > 1e-4:
# 어절 단위로 나눠 합성 후 단어 간 무음 삽입
words = sent.split()
wpieces: list[np.ndarray] = []
for wi, word in enumerate(words):
a, s = self._synth_natural(word, language, speaker)
if sr is None:
sr = s
a = stretch(a, sr)
if wi > 0:
wpieces.append(np.zeros(int(sr * word_gap), dtype=np.float32))
wpieces.append(a)
sent_audio = (
np.concatenate(wpieces)
if wpieces
else np.zeros(1, dtype=np.float32)
)
else:
a, s = self._synth_natural(sent, language, speaker)
if sr is None:
sr = s
sent_audio = stretch(a, sr)
if pieces and sentence_gap > 1e-4:
pieces.append(np.zeros(int(sr * sentence_gap), dtype=np.float32))
pieces.append(sent_audio)
audio = np.concatenate(pieces) if pieces else np.zeros(1, dtype=np.float32)
if sr is None:
sr = self._get_model(DEFAULT_LANG).hps.data.sampling_rate
if abs(pitch) > 1e-3 and librosa is not None:
audio = librosa.effects.pitch_shift(audio, sr, n_steps=pitch)

View File

@@ -44,7 +44,12 @@ class OuteTTSEngine(BaseEngine):
{"code": "KO", "label": "한국어", "speakers": _VOICES},
],
"default_language": "KO",
"supports": {"speed": True, "pitch": True},
"supports": {
"speed": True,
"pitch": True,
"word_gap": False,
"sentence_gap": False,
},
}
def synth_wav(
@@ -54,6 +59,8 @@ class OuteTTSEngine(BaseEngine):
speaker: str | None = None,
speed: float = 1.0,
pitch: float = 0.0,
word_gap: float = 0.0,
sentence_gap: float | None = None,
) -> bytes:
text = (text or "").strip()
if not text:

View File

@@ -52,7 +52,12 @@ class SupertonicEngine(BaseEngine):
{"code": "AUTO", "label": "자동 감지 (다국어)", "speakers": _VOICES},
],
"default_language": "KO",
"supports": {"speed": True, "pitch": True},
"supports": {
"speed": True,
"pitch": True,
"word_gap": False,
"sentence_gap": False,
},
}
def synth_wav(
@@ -62,6 +67,8 @@ class SupertonicEngine(BaseEngine):
speaker: str | None = None,
speed: float = 1.0,
pitch: float = 0.0,
word_gap: float = 0.0,
sentence_gap: float | None = None,
) -> bytes:
text = (text or "").strip()
if not text:

View File

@@ -39,8 +39,10 @@ class TTSRequest(BaseModel):
engine: str | None = Field(None, description="엔진 ID (melo / coqui-kss 등)")
language: str = Field("KR", description="언어 코드")
speaker: str | None = Field(None, description="화자/모델 ID")
speed: float = Field(1.4, ge=0.5, le=2.0, description="말하기 속도")
speed: float = Field(1.4, ge=0.5, le=2.0, description="글자(음소) 발화 속도")
pitch: float = Field(0.0, ge=-12.0, le=12.0, description="피치(반음)")
word_gap: float = Field(0.0, ge=0.0, le=1.0, description="단어 사이 무음(초)")
sentence_gap: float = Field(0.3, ge=0.0, le=2.0, description="문장 사이 무음(초)")
@app.on_event("startup")
@@ -80,6 +82,8 @@ def tts(req: TTSRequest) -> Response:
speaker=req.speaker,
speed=req.speed,
pitch=req.pitch,
word_gap=req.word_gap,
sentence_gap=req.sentence_gap,
)
except ValueError as exc:
raise HTTPException(status_code=400, detail=str(exc))