feat: 글자속도·단어간격·문장간격 독립 조절 + atempo 기본 타임스트레치

- 타임스트레치 우선순위를 ffmpeg atempo(기본) → rubberband → librosa 로 변경
- MeloEngine: 문장 단위로 분리 후 문장 사이 무음(sentence_gap), 단어(어절) 사이
  무음(word_gap)을 타임스트레치와 독립적으로 삽입. speed 는 글자(음소) 속도만 조절
- API(TTSRequest)에 word_gap/sentence_gap 필드 추가, 전 엔진 시그니처 확장
- 프론트엔드에 '글자 속도/단어 간격/문장 간격' 슬라이더 추가, supports 플래그로
  미지원 엔진(supertonic/oute/cosy)에서는 간격 슬라이더 자동 비활성화
This commit is contained in:
claude
2026-08-25 23:54:48 +09:00
parent e16f88e534
commit c0665893a7
8 changed files with 181 additions and 50 deletions

View File

@@ -39,8 +39,14 @@ class BaseEngine(ABC):
speaker: str | None = None, speaker: str | None = None,
speed: float = 1.0, speed: float = 1.0,
pitch: float = 0.0, pitch: float = 0.0,
word_gap: float = 0.0,
sentence_gap: float | None = None,
) -> bytes: ) -> bytes:
"""WAV(PCM16) 바이트를 반환.""" """WAV(PCM16) 바이트를 반환.
speed=글자(음소) 속도, word_gap=단어 사이 무음(초),
sentence_gap=문장 사이 무음(초). 간격을 지원하지 않는 엔진은 무시한다.
"""
raise NotImplementedError raise NotImplementedError
def warmup(self) -> None: def warmup(self) -> None:

View File

@@ -42,7 +42,12 @@ class CosyVoiceEngine(BaseEngine):
], ],
} }
], ],
"supports": {"speed": True, "pitch": True}, "supports": {
"speed": True,
"pitch": True,
"word_gap": False,
"sentence_gap": False,
},
} }
def synth_wav( def synth_wav(
@@ -52,6 +57,8 @@ class CosyVoiceEngine(BaseEngine):
speaker: str | None = None, speaker: str | None = None,
speed: float = 1.0, speed: float = 1.0,
pitch: float = 0.0, pitch: float = 0.0,
word_gap: float = 0.0,
sentence_gap: float | None = None,
) -> bytes: ) -> bytes:
text = (text or "").strip() text = (text or "").strip()
if not text: if not text:

View File

@@ -50,6 +50,15 @@ MELO_LANG = {"KR": "KR", "EN": "EN", "JP": "JP", "ZH": "ZH"}
DEFAULT_LANG = "EN" # 판별 불가 구간의 기본 언어 DEFAULT_LANG = "EN" # 판별 불가 구간의 기본 언어
# 문장 경계: 종결부호 뒤 공백, 또는 줄바꿈
_SENT_SPLIT_RE = re.compile(r"(?<=[.!?。!?…])\s+|\n+")
def split_sentences(text: str) -> list[str]:
"""텍스트를 문장 단위로 분리(종결부호/줄바꿈 기준)."""
parts = [p.strip() for p in _SENT_SPLIT_RE.split(text) if p and p.strip()]
return parts or ([text.strip()] if text.strip() else [])
_KR_RE = re.compile(r"[가-힣ᄀ-ᇿ㄰-㆏]") _KR_RE = re.compile(r"[가-힣ᄀ-ᇿ㄰-㆏]")
_EN_RE = re.compile(r"[A-Za-z]") _EN_RE = re.compile(r"[A-Za-z]")
# 일본어 가나(히라가나/가타카나/반각 가타카나) # 일본어 가나(히라가나/가타카나/반각 가타카나)
@@ -171,7 +180,12 @@ class MeloEngine(BaseEngine):
"notes": self.notes, "notes": self.notes,
"languages": languages, "languages": languages,
"default_language": "AUTO", "default_language": "AUTO",
"supports": {"speed": True, "pitch": True}, "supports": {
"speed": True,
"pitch": True,
"word_gap": True,
"sentence_gap": True,
},
} }
def _resolve_speaker(self, model: TTS, lang: str, speaker: str | None) -> int: def _resolve_speaker(self, model: TTS, lang: str, speaker: str | None) -> int:
@@ -200,15 +214,22 @@ class MeloEngine(BaseEngine):
def _time_stretch(self, audio: np.ndarray, sr: int, rate: float) -> np.ndarray: def _time_stretch(self, audio: np.ndarray, sr: int, rate: float) -> np.ndarray:
"""오디오를 rate 배속으로 변경(피치 보존). rate>1 = 빠르게. """오디오를 rate 배속으로 변경(피치 보존). rate>1 = 빠르게.
MeloTTS 의 length_scale 방식(speed 파라미터)은 배속을 올리면 발음이 자연 속도(1.0)로 합성한 뒤 여기서 타임스트레치를 적용해 글자 발화
뭉개지므로, 자연 속도로 합성한 뒤 여기서 고품질 타임스트레치를 적용한다. 속도만 조절한다(피치 보존).
우선순위: Rubberband(R3) → ffmpeg atempo → librosa(phase vocoder). 우선순위: ffmpeg atempo(기본) → Rubberband(R3) → librosa(phase vocoder).
""" """
if abs(rate - 1.0) < 1e-3: if abs(rate - 1.0) < 1e-3:
return audio return audio
audio = np.asarray(audio, dtype=np.float32) audio = np.asarray(audio, dtype=np.float32)
# 1) Rubberband R3 (발음/포먼트 보존이 가장 좋음) # 1) ffmpeg atempo (기본값: 추가 의존성 없음, 무난)
if shutil.which("ffmpeg") and 0.5 <= rate <= 2.0:
try:
return self._atempo(audio, sr, rate)
except Exception:
pass
# 2) Rubberband R3 (발음/포먼트 보존 폴백)
if prb is not None and shutil.which("rubberband"): if prb is not None and shutil.which("rubberband"):
try: try:
out = prb.time_stretch( out = prb.time_stretch(
@@ -222,13 +243,6 @@ class MeloEngine(BaseEngine):
except Exception: except Exception:
pass pass
# 2) ffmpeg atempo (추가 의존성 없음, 무난)
if shutil.which("ffmpeg") and 0.5 <= rate <= 2.0:
try:
return self._atempo(audio, sr, rate)
except Exception:
pass
# 3) librosa phase vocoder (최후 폴백) # 3) librosa phase vocoder (최후 폴백)
if librosa is not None: if librosa is not None:
try: try:
@@ -258,27 +272,13 @@ class MeloEngine(BaseEngine):
out, _ = sf.read(dst, dtype="float32") out, _ = sf.read(dst, dtype="float32")
return np.asarray(out, dtype=np.float32) return np.asarray(out, dtype=np.float32)
# ---- 합성 ----------------------------------------------------- def _synth_natural(
def synth_wav( self, text: str, language: str, speaker: str | None
self, ) -> tuple[np.ndarray, int]:
text: str, """자연 속도(1.0)로 텍스트 한 조각을 합성. AUTO 모드는 언어별로 나눠 합성.
language: str,
speaker: str | None = None,
speed: float = 1.0,
pitch: float = 0.0,
) -> bytes:
text = (text or "").strip()
if not text:
raise ValueError("텍스트가 비어 있습니다.")
speed = float(max(0.5, min(2.0, speed)))
pitch = float(max(-12.0, min(12.0, pitch)))
language = (language or "AUTO").upper()
# 한국어 발음 개선: 기호/약어/사용자사전 전처리 (한국어·자동 모드)
if language in ("AUTO", "KR"):
text = normalize_korean(text)
속도/문장·단어 간격 조절은 이 결과 위에서 처리한다(여기서는 하지 않음).
"""
if language == "AUTO": if language == "AUTO":
# 영어 구간엔 사용자가 고른 억양, 한국어 구간엔 KR 음성 # 영어 구간엔 사용자가 고른 억양, 한국어 구간엔 KR 음성
en_speaker = speaker if (speaker and speaker.startswith("EN")) else "EN-US" en_speaker = speaker if (speaker and speaker.startswith("EN")) else "EN-US"
@@ -295,8 +295,7 @@ class MeloEngine(BaseEngine):
if sr is None: if sr is None:
sr = seg_sr sr = seg_sr
gap = np.zeros(int(sr * 0.06), dtype=np.float32) gap = np.zeros(int(sr * 0.06), dtype=np.float32)
elif seg_sr != sr: elif seg_sr != sr and librosa is not None:
if librosa is not None:
audio = librosa.resample( audio = librosa.resample(
audio, orig_sr=seg_sr, target_sr=sr audio, orig_sr=seg_sr, target_sr=sr
).astype(np.float32) ).astype(np.float32)
@@ -304,12 +303,81 @@ class MeloEngine(BaseEngine):
audios.append(gap) audios.append(gap)
audios.append(audio) audios.append(audio)
audio = np.concatenate(audios) if audios else np.zeros(1, dtype=np.float32) audio = np.concatenate(audios) if audios else np.zeros(1, dtype=np.float32)
else: return audio, (sr or self._get_model(DEFAULT_LANG).hps.data.sampling_rate)
audio, sr = self._synth_segment(text, language, speaker) return self._synth_segment(text, language, speaker)
# 속도: 자연 속도 합성 결과에 고품질 타임스트레치 적용(발음 유지) # ---- 합성 -----------------------------------------------------
def synth_wav(
self,
text: str,
language: str,
speaker: str | None = None,
speed: float = 1.0,
pitch: float = 0.0,
word_gap: float = 0.0,
sentence_gap: float | None = None,
) -> bytes:
"""세 가지를 독립적으로 조절:
- speed: 글자(음소) 발화 속도 — 자연 합성 후 타임스트레치(피치 보존)
- word_gap: 단어 사이 추가 무음(초). >0 이면 어절 단위로 나눠 합성
- sentence_gap: 문장 사이 무음(초)
간격은 타임스트레치의 영향을 받지 않아 글자속도와 독립적으로 유지된다.
"""
text = (text or "").strip()
if not text:
raise ValueError("텍스트가 비어 있습니다.")
speed = float(max(0.5, min(2.0, speed)))
pitch = float(max(-12.0, min(12.0, pitch)))
word_gap = float(max(0.0, min(1.0, word_gap)))
sentence_gap = 0.30 if sentence_gap is None else sentence_gap
sentence_gap = float(max(0.0, min(2.0, sentence_gap)))
language = (language or "AUTO").upper()
# 한국어 발음 개선: 기호/약어/사용자사전 전처리 (한국어·자동 모드)
if language in ("AUTO", "KR"):
text = normalize_korean(text)
def stretch(a: np.ndarray, sr: int) -> np.ndarray:
if abs(speed - 1.0) > 1e-3: if abs(speed - 1.0) > 1e-3:
audio = self._time_stretch(audio, sr, speed) return self._time_stretch(a, sr, speed)
return a
sentences = split_sentences(text) or [text]
sr: int | None = None
pieces: list[np.ndarray] = []
for si, sent in enumerate(sentences):
if word_gap > 1e-4:
# 어절 단위로 나눠 합성 후 단어 간 무음 삽입
words = sent.split()
wpieces: list[np.ndarray] = []
for wi, word in enumerate(words):
a, s = self._synth_natural(word, language, speaker)
if sr is None:
sr = s
a = stretch(a, sr)
if wi > 0:
wpieces.append(np.zeros(int(sr * word_gap), dtype=np.float32))
wpieces.append(a)
sent_audio = (
np.concatenate(wpieces)
if wpieces
else np.zeros(1, dtype=np.float32)
)
else:
a, s = self._synth_natural(sent, language, speaker)
if sr is None:
sr = s
sent_audio = stretch(a, sr)
if pieces and sentence_gap > 1e-4:
pieces.append(np.zeros(int(sr * sentence_gap), dtype=np.float32))
pieces.append(sent_audio)
audio = np.concatenate(pieces) if pieces else np.zeros(1, dtype=np.float32)
if sr is None:
sr = self._get_model(DEFAULT_LANG).hps.data.sampling_rate
if abs(pitch) > 1e-3 and librosa is not None: if abs(pitch) > 1e-3 and librosa is not None:
audio = librosa.effects.pitch_shift(audio, sr, n_steps=pitch) audio = librosa.effects.pitch_shift(audio, sr, n_steps=pitch)

View File

@@ -44,7 +44,12 @@ class OuteTTSEngine(BaseEngine):
{"code": "KO", "label": "한국어", "speakers": _VOICES}, {"code": "KO", "label": "한국어", "speakers": _VOICES},
], ],
"default_language": "KO", "default_language": "KO",
"supports": {"speed": True, "pitch": True}, "supports": {
"speed": True,
"pitch": True,
"word_gap": False,
"sentence_gap": False,
},
} }
def synth_wav( def synth_wav(
@@ -54,6 +59,8 @@ class OuteTTSEngine(BaseEngine):
speaker: str | None = None, speaker: str | None = None,
speed: float = 1.0, speed: float = 1.0,
pitch: float = 0.0, pitch: float = 0.0,
word_gap: float = 0.0,
sentence_gap: float | None = None,
) -> bytes: ) -> bytes:
text = (text or "").strip() text = (text or "").strip()
if not text: if not text:

View File

@@ -52,7 +52,12 @@ class SupertonicEngine(BaseEngine):
{"code": "AUTO", "label": "자동 감지 (다국어)", "speakers": _VOICES}, {"code": "AUTO", "label": "자동 감지 (다국어)", "speakers": _VOICES},
], ],
"default_language": "KO", "default_language": "KO",
"supports": {"speed": True, "pitch": True}, "supports": {
"speed": True,
"pitch": True,
"word_gap": False,
"sentence_gap": False,
},
} }
def synth_wav( def synth_wav(
@@ -62,6 +67,8 @@ class SupertonicEngine(BaseEngine):
speaker: str | None = None, speaker: str | None = None,
speed: float = 1.0, speed: float = 1.0,
pitch: float = 0.0, pitch: float = 0.0,
word_gap: float = 0.0,
sentence_gap: float | None = None,
) -> bytes: ) -> bytes:
text = (text or "").strip() text = (text or "").strip()
if not text: if not text:

View File

@@ -39,8 +39,10 @@ class TTSRequest(BaseModel):
engine: str | None = Field(None, description="엔진 ID (melo / coqui-kss 등)") engine: str | None = Field(None, description="엔진 ID (melo / coqui-kss 등)")
language: str = Field("KR", description="언어 코드") language: str = Field("KR", description="언어 코드")
speaker: str | None = Field(None, description="화자/모델 ID") speaker: str | None = Field(None, description="화자/모델 ID")
speed: float = Field(1.4, ge=0.5, le=2.0, description="말하기 속도") speed: float = Field(1.4, ge=0.5, le=2.0, description="글자(음소) 발화 속도")
pitch: float = Field(0.0, ge=-12.0, le=12.0, description="피치(반음)") pitch: float = Field(0.0, ge=-12.0, le=12.0, description="피치(반음)")
word_gap: float = Field(0.0, ge=0.0, le=1.0, description="단어 사이 무음(초)")
sentence_gap: float = Field(0.3, ge=0.0, le=2.0, description="문장 사이 무음(초)")
@app.on_event("startup") @app.on_event("startup")
@@ -80,6 +82,8 @@ def tts(req: TTSRequest) -> Response:
speaker=req.speaker, speaker=req.speaker,
speed=req.speed, speed=req.speed,
pitch=req.pitch, pitch=req.pitch,
word_gap=req.word_gap,
sentence_gap=req.sentence_gap,
) )
except ValueError as exc: except ValueError as exc:
raise HTTPException(status_code=400, detail=str(exc)) raise HTTPException(status_code=400, detail=str(exc))

View File

@@ -9,6 +9,10 @@ const els = {
speaker: $("speaker"), speaker: $("speaker"),
speed: $("speed"), speed: $("speed"),
speedVal: $("speedVal"), speedVal: $("speedVal"),
wordGap: $("wordGap"),
wordGapVal: $("wordGapVal"),
sentGap: $("sentGap"),
sentGapVal: $("sentGapVal"),
pitch: $("pitch"), pitch: $("pitch"),
pitchVal: $("pitchVal"), pitchVal: $("pitchVal"),
generate: $("generate"), generate: $("generate"),
@@ -76,7 +80,9 @@ function updateEngineInfo() {
: '<span class="cpu">CPU</span>'; : '<span class="cpu">CPU</span>';
const supports = eng.supports || {}; const supports = eng.supports || {};
const ctrl = [ const ctrl = [
supports.speed ? "속도" : null, supports.speed ? "글자속도" : null,
supports.word_gap ? "단어간격" : null,
supports.sentence_gap ? "문장간격" : null,
supports.pitch ? "피치" : null, supports.pitch ? "피치" : null,
] ]
.filter(Boolean) .filter(Boolean)
@@ -87,9 +93,11 @@ function updateEngineInfo() {
`· ${gpu}` + `· ${gpu}` +
(ctrl ? ` · 조절: ${ctrl}` : ""); (ctrl ? ` · 조절: ${ctrl}` : "");
// 지원하지 않는 컨트롤 비활성화 // 지원하지 않는 컨트롤 비활성화 (플래그가 명시적으로 false면 끔)
els.speed.disabled = supports.speed === false; els.speed.disabled = supports.speed === false;
els.pitch.disabled = supports.pitch === false; els.pitch.disabled = supports.pitch === false;
els.wordGap.disabled = supports.word_gap !== true;
els.sentGap.disabled = supports.sentence_gap !== true;
} }
async function loadEngines() { async function loadEngines() {
@@ -159,6 +167,8 @@ async function generate() {
speaker: els.speaker.value, speaker: els.speaker.value,
speed: parseFloat(els.speed.value), speed: parseFloat(els.speed.value),
pitch: parseFloat(els.pitch.value), pitch: parseFloat(els.pitch.value),
word_gap: parseFloat(els.wordGap.value),
sentence_gap: parseFloat(els.sentGap.value),
}), }),
}); });
if (!res.ok) { if (!res.ok) {
@@ -189,6 +199,12 @@ els.language.addEventListener("change", populateSpeakers);
els.speed.addEventListener("input", () => { els.speed.addEventListener("input", () => {
els.speedVal.textContent = parseFloat(els.speed.value).toFixed(2) + "x"; els.speedVal.textContent = parseFloat(els.speed.value).toFixed(2) + "x";
}); });
els.wordGap.addEventListener("input", () => {
els.wordGapVal.textContent = Math.round(parseFloat(els.wordGap.value) * 1000) + " ms";
});
els.sentGap.addEventListener("input", () => {
els.sentGapVal.textContent = parseFloat(els.sentGap.value).toFixed(2) + " s";
});
els.pitch.addEventListener("input", () => { els.pitch.addEventListener("input", () => {
const v = parseInt(els.pitch.value, 10); const v = parseInt(els.pitch.value, 10);
els.pitchVal.textContent = (v > 0 ? "+" + v : v) + " 반음"; els.pitchVal.textContent = (v > 0 ? "+" + v : v) + " 반음";

View File

@@ -42,12 +42,28 @@
<div class="sliders"> <div class="sliders">
<div class="slider-row"> <div class="slider-row">
<div class="slider-head"> <div class="slider-head">
<span>속도</span> <span>글자 속도</span>
<span class="slider-val" id="speedVal">1.40x</span> <span class="slider-val" id="speedVal">1.40x</span>
</div> </div>
<input type="range" id="speed" min="0.5" max="2.0" step="0.05" value="1.4" /> <input type="range" id="speed" min="0.5" max="2.0" step="0.05" value="1.4" />
<div class="slider-scale"><span>느리게</span><span>빠르게</span></div> <div class="slider-scale"><span>느리게</span><span>빠르게</span></div>
</div> </div>
<div class="slider-row">
<div class="slider-head">
<span>단어 간격</span>
<span class="slider-val" id="wordGapVal">0 ms</span>
</div>
<input type="range" id="wordGap" min="0" max="1.0" step="0.02" value="0" />
<div class="slider-scale"><span>붙여서</span><span>띄어서</span></div>
</div>
<div class="slider-row">
<div class="slider-head">
<span>문장 간격</span>
<span class="slider-val" id="sentGapVal">0.30 s</span>
</div>
<input type="range" id="sentGap" min="0" max="1.5" step="0.05" value="0.3" />
<div class="slider-scale"><span>짧게</span><span>길게</span></div>
</div>
<div class="slider-row"> <div class="slider-row">
<div class="slider-head"> <div class="slider-head">
<span>피치</span> <span>피치</span>