Compare commits

..

6 Commits

Author SHA1 Message Date
claude
47e8af00bb docs: TTS 설정(속도·간격·피치) 사용자용 사용법 매뉴얼 추가 2026-08-26 22:13:54 +09:00
claude
dc21acf032 docs: TTS 속도·간격·피치 제어 이식 매뉴얼(watch_screen_ai 적용용) 추가 2026-08-26 21:31:31 +09:00
claude
5ed2d5fa8a fix: 단어 간격을 무음 스케일링으로 전환(자연 0 기준 아래로도 축소)
- 어절 개별 합성(억양 저하, 0 미만 불가) 대신, 문장을 통째로 자연 합성한 뒤
  내부 어절 무음 구간을 _scale_word_gaps 로 늘이거나 줄인다
- word_gap<0 이 자연 상태(0)보다 더 촘촘해짐(요청 충족), 억양 보존
- min_pause(80ms) 미만 짧은 무음/경계 무음은 건드리지 않아 발음 안전
2026-08-26 20:42:24 +09:00
claude
c0fbc4d881 feat: 단어·문장 간격을 음수까지 허용(0 미만은 경계 무음 트리밍)
- word_gap 범위 -0.2~0.5, sentence_gap 범위 -0.5~1.5 로 확장(음수=더 붙임)
- 음수 간격은 이전 조각 끝 + 다음 조각 시작의 무음을 |gap| 초까지 잘라 붙이는
  _trim_end/start_silence + _append_unit 로 처리
- API Field 범위/설명, 프론트 슬라이더 min/스케일 라벨 갱신
2026-08-26 20:37:31 +09:00
claude
fae52f67c5 fix: 글자 속도를 생성 단계 length_scale(음절 발화 속도)로 변경
- 사후 배속(atempo) 대신 model.tts_to_file(speed=) 로 각 음절(안/녕/하)의
  발음 속도 자체를 조절. 피치 보존, TSM 아티팩트 없음
- _synth_segment/_synth_natural 에 speed 인자 전달, synth_wav 의 사후
  타임스트레치 제거. 단어·문장 간격은 여전히 발화 속도와 독립
2026-08-26 00:20:57 +09:00
claude
dde57aafda tune: 속도/간격 슬라이더 기본값을 범위 가운데로, 양 끝을 최소·최대치로 정렬
- 글자 속도 기본 1.4→1.25(범위 0.5~2.0의 중앙)
- 단어 간격 기본 0→0.25s(범위 0~0.5s의 중앙)
- 문장 간격 기본 0.3→0.75s(범위 0~1.5s의 중앙)
- 슬라이더 스케일 라벨에 최소/최대 수치 표기
2026-08-26 00:16:41 +09:00
5 changed files with 520 additions and 56 deletions

View File

@@ -199,14 +199,15 @@ class MeloEngine(BaseEngine):
return list(spk2id.values())[0]
def _synth_segment(
self, text: str, lang: str, speaker: str | None
self, text: str, lang: str, speaker: str | None, speed: float = 1.0
) -> tuple[np.ndarray, int]:
# 항상 자연 속도(1.0)로 합성 — 속도는 사후 타임스트레치로 처리해 발음 유지
# 글자(음절) 발화 속도는 모델 length_scale(=1/speed)로 생성 단계에서 조절.
# 각 음절의 발음 자체가 빨라/느려지며(피치 보존), 사후 배속이 아니다.
model = self._get_model(lang)
speaker_id = self._resolve_speaker(model, lang, speaker)
with self._global_lock:
audio = model.tts_to_file(
text, speaker_id, output_path=None, speed=1.0, quiet=True
text, speaker_id, output_path=None, speed=speed, quiet=True
)
sr = model.hps.data.sampling_rate
return np.asarray(audio, dtype=np.float32), sr
@@ -273,11 +274,12 @@ class MeloEngine(BaseEngine):
return np.asarray(out, dtype=np.float32)
def _synth_natural(
self, text: str, language: str, speaker: str | None
self, text: str, language: str, speaker: str | None, speed: float = 1.0
) -> tuple[np.ndarray, int]:
"""자연 속도(1.0)로 텍스트 한 조각을 합성. AUTO 모드는 언어별로 나눠 합성.
"""텍스트 한 조각을 합성. AUTO 모드는 언어별로 나눠 합성.
속도/문장·단어 간격 조절은 이 결과 위에서 처리한다(여기서는 하지 않음).
speed 는 모델 length_scale 로 각 음절의 발화 속도를 조절한다(생성 단계).
문장·단어 간격은 이 결과 위에서 삽입한다(여기서는 하지 않음).
"""
if language == "AUTO":
# 영어 구간엔 사용자가 고른 억양, 한국어 구간엔 KR 음성
@@ -291,7 +293,7 @@ class MeloEngine(BaseEngine):
gap = None
for seg_lang, seg_text in parts:
seg_spk = spk_by_lang.get(seg_lang, en_speaker)
audio, seg_sr = self._synth_segment(seg_text, seg_lang, seg_spk)
audio, seg_sr = self._synth_segment(seg_text, seg_lang, seg_spk, speed)
if sr is None:
sr = seg_sr
gap = np.zeros(int(sr * 0.06), dtype=np.float32)
@@ -304,7 +306,107 @@ class MeloEngine(BaseEngine):
audios.append(audio)
audio = np.concatenate(audios) if audios else np.zeros(1, dtype=np.float32)
return audio, (sr or self._get_model(DEFAULT_LANG).hps.data.sampling_rate)
return self._synth_segment(text, language, speaker)
return self._synth_segment(text, language, speaker, speed)
# ---- 간격 처리(양수=무음 삽입, 음수=경계 무음 트리밍) --------------
@staticmethod
def _trim_end_silence(
a: np.ndarray, sr: int, max_sec: float, thresh: float = 0.02
) -> tuple[np.ndarray, float]:
n = a.size
if n == 0 or max_sec <= 0:
return a, 0.0
max_n = min(n, int(sr * max_sec))
if max_n <= 0:
return a, 0.0
tail = np.abs(a[n - max_n:])
nz = np.where(tail >= thresh)[0]
cut = max_n if nz.size == 0 else (max_n - 1 - int(nz[-1]))
if cut <= 0:
return a, 0.0
return a[: n - cut], cut / sr
@staticmethod
def _trim_start_silence(
a: np.ndarray, sr: int, max_sec: float, thresh: float = 0.02
) -> tuple[np.ndarray, float]:
n = a.size
if n == 0 or max_sec <= 0:
return a, 0.0
max_n = min(n, int(sr * max_sec))
if max_n <= 0:
return a, 0.0
head = np.abs(a[:max_n])
nz = np.where(head >= thresh)[0]
cut = max_n if nz.size == 0 else int(nz[0])
if cut <= 0:
return a, 0.0
return a[cut:], cut / sr
def _append_unit(
self, pieces: list[np.ndarray], unit: np.ndarray, gap: float, sr: int
) -> None:
"""이전 조각들 뒤에 unit 을 붙인다.
gap>=0 이면 그 만큼 무음을 삽입, gap<0 이면 |gap| 초만큼 경계의 무음을
(이전 조각 끝 + 다음 조각 시작 순서로) 잘라내 더 붙인다.
"""
if not pieces:
pieces.append(unit)
return
if gap >= 0:
if gap > 1e-4:
pieces.append(np.zeros(int(sr * gap), dtype=np.float32))
pieces.append(unit)
return
budget = -gap
prev, removed = self._trim_end_silence(pieces[-1], sr, budget)
pieces[-1] = prev
budget -= removed
if budget > 1e-4:
unit, _ = self._trim_start_silence(unit, sr, budget)
pieces.append(unit)
def _scale_word_gaps(
self,
a: np.ndarray,
sr: int,
delta: float,
thresh: float = 0.02,
min_pause: float = 0.08,
) -> np.ndarray:
"""자연 합성된 문장 오디오에서 '어절 사이 무음'을 delta 초만큼 조절.
- 억양 보존: 문장을 통째로 합성한 결과의 내부 무음 구간만 늘리거나 줄인다.
- delta>0 이면 무음을 늘려 단어를 벌리고, delta<0 이면 무음을 줄여 붙인다.
- 앞/뒤 경계 무음과 min_pause 미만의 짧은 무음(자음 폐쇄 등)은 건드리지 않아
발음이 깨지지 않는다.
"""
n = a.size
if abs(delta) < 1e-4 or n == 0:
return a
silent = np.abs(a) < thresh
changes = np.flatnonzero(np.diff(silent.astype(np.int8)) != 0) + 1
bounds = [0, *changes.tolist(), n]
min_n = int(sr * min_pause)
floor_n = int(sr * 0.015)
add_n = int(delta * sr)
out: list[np.ndarray] = []
last = len(bounds) - 2
for k in range(len(bounds) - 1):
s, e = bounds[k], bounds[k + 1]
seg = a[s:e]
is_internal = 0 < k < last # 앞/뒤 경계 구간은 제외
if silent[s] and is_internal and (e - s) >= min_n:
new_n = max(floor_n, (e - s) + add_n)
if new_n >= (e - s):
seg = np.concatenate(
[seg, np.zeros(new_n - (e - s), dtype=np.float32)]
)
else:
seg = seg[:new_n]
out.append(seg)
return np.concatenate(out) if out else a
# ---- 합성 -----------------------------------------------------
def synth_wav(
@@ -318,10 +420,12 @@ class MeloEngine(BaseEngine):
sentence_gap: float | None = None,
) -> bytes:
"""세 가지를 독립적으로 조절:
- speed: 글자(음소) 발화 속도 — 자연 합성 후 타임스트레치(피치 보존)
- word_gap: 단어 사이 추가 무음(초). >0 이면 어절 단위로 나눠 합성
- sentence_gap: 문장 사이 무음(초)
간격은 타임스트레치의 영향을 받지 않아 글자속도와 독립적으로 유지된다.
- speed: 글자(음절) 발화 속도 — 모델 length_scale 로 각 음절의 발음
속도를 생성 단계에서 조절(피치 보존, 사후 배속 아님)
- word_gap: 단어 사이 간격(초). 문장을 통째로 합성한 뒤 어절 사이 무음을
>0 이면 늘리고 <0 이면 줄인다(억양 보존, 0 미만으로도 붙일 수 있음)
- sentence_gap: 문장 사이 간격(초). >0 무음 삽입, <0 경계 무음 트리밍
간격은 발화 속도와 무관하게 유지되어 서로 독립적이다.
"""
text = (text or "").strip()
if not text:
@@ -329,51 +433,32 @@ class MeloEngine(BaseEngine):
speed = float(max(0.5, min(2.0, speed)))
pitch = float(max(-12.0, min(12.0, pitch)))
word_gap = float(max(0.0, min(1.0, word_gap)))
word_gap = float(max(-0.2, min(0.5, word_gap)))
sentence_gap = 0.30 if sentence_gap is None else sentence_gap
sentence_gap = float(max(0.0, min(2.0, sentence_gap)))
sentence_gap = float(max(-0.5, min(1.5, sentence_gap)))
language = (language or "AUTO").upper()
# 한국어 발음 개선: 기호/약어/사용자사전 전처리 (한국어·자동 모드)
if language in ("AUTO", "KR"):
text = normalize_korean(text)
def stretch(a: np.ndarray, sr: int) -> np.ndarray:
if abs(speed - 1.0) > 1e-3:
return self._time_stretch(a, sr, speed)
return a
sentences = split_sentences(text) or [text]
sr: int | None = None
pieces: list[np.ndarray] = []
first = True
for si, sent in enumerate(sentences):
if word_gap > 1e-4:
# 어절 단위로 나눠 합성 후 단어 간 무음 삽입
words = sent.split()
wpieces: list[np.ndarray] = []
for wi, word in enumerate(words):
a, s = self._synth_natural(word, language, speaker)
if sr is None:
sr = s
a = stretch(a, sr)
if wi > 0:
wpieces.append(np.zeros(int(sr * word_gap), dtype=np.float32))
wpieces.append(a)
sent_audio = (
np.concatenate(wpieces)
if wpieces
else np.zeros(1, dtype=np.float32)
)
# 문장을 통째로 자연 합성(억양 유지) 후 어절 사이 무음만 조절
a, s = self._synth_natural(sent, language, speaker, speed)
if sr is None:
sr = s
if abs(word_gap) > 1e-4:
a = self._scale_word_gaps(a, sr, word_gap)
if first:
pieces.append(a)
first = False
else:
a, s = self._synth_natural(sent, language, speaker)
if sr is None:
sr = s
sent_audio = stretch(a, sr)
if pieces and sentence_gap > 1e-4:
pieces.append(np.zeros(int(sr * sentence_gap), dtype=np.float32))
pieces.append(sent_audio)
self._append_unit(pieces, a, sentence_gap, sr)
audio = np.concatenate(pieces) if pieces else np.zeros(1, dtype=np.float32)
if sr is None:

View File

@@ -39,10 +39,11 @@ class TTSRequest(BaseModel):
engine: str | None = Field(None, description="엔진 ID (melo / coqui-kss 등)")
language: str = Field("KR", description="언어 코드")
speaker: str | None = Field(None, description="화자/모델 ID")
speed: float = Field(1.4, ge=0.5, le=2.0, description="글자(음소) 발화 속도")
# 기본값은 각 범위의 가운데. 슬라이더 양 끝이 조절 가능한 최소/최대치.
speed: float = Field(1.25, ge=0.5, le=2.0, description="글자(음소) 발화 속도(0.5~2.0, 기본 1.25)")
pitch: float = Field(0.0, ge=-12.0, le=12.0, description="피치(반음)")
word_gap: float = Field(0.0, ge=0.0, le=1.0, description="단어 사이 무음(초)")
sentence_gap: float = Field(0.3, ge=0.0, le=2.0, description="문장 사이 무음(초)")
word_gap: float = Field(0.25, ge=-0.2, le=0.5, description="단어 사이 간격 초(-0.2~0.5, 음수=더 붙임, 기본 0.25)")
sentence_gap: float = Field(0.75, ge=-0.5, le=1.5, description="문장 사이 간격 초(-0.5~1.5, 음수=더 붙임, 기본 0.75)")
@app.on_event("startup")

View File

@@ -0,0 +1,88 @@
# TTS 설정 사용법 (속도·간격·피치)
음성 생성 화면의 설정 4가지 — **글자 속도 · 단어 간격 · 문장 간격 · 피치** — 를
무엇을 바꾸는지, 어떤 값으로 두면 좋은지 설명하는 사용 매뉴얼이다.
(개발자용 이식/코드 매뉴얼은 `docs/TTS_속도_간격_적용_매뉴얼.md` 참고.)
---
## 한눈에 보기
| 설정 | 바꾸는 것 | 범위 | 기본값 | 왼쪽 끝(작게) | 오른쪽 끝(크게) |
|---|---|---|---|---|---|
| 글자 속도 | 한 글자(음절)를 발음하는 빠르기 | 0.5x ~ 2.0x | 1.25x | 0.5x 또박또박 느리게 | 2.0x 빠르게 |
| 단어 간격 | 단어와 단어 사이 띄는 정도 | -200ms ~ 500ms | 0.25초 | -200ms 더 붙여 | 500ms 띄어서 |
| 문장 간격 | 문장과 문장 사이 쉬는 시간 | -0.5초 ~ 1.5초 | 0.75초 | -0.5초 더 붙여 | 1.5초 길게 |
| 피치 | 목소리 높낮이 | -12 ~ +12 반음 | 0 | 낮게 | 높게 |
세 가지 속도/간격은 **서로 독립**이다. 글자 속도를 올려도 단어/문장 간격은 설정한
시간 그대로 유지된다.
---
## 1. 글자 속도
각 글자(음절 "안·녕·하")를 얼마나 빠르게 발음할지 정한다. 전체를 빨리 감는 "배속"이
아니라 발음 자체의 속도라서, 목소리 톤이 그대로 유지되고 갈라짐·기계음이 없다.
- 왼쪽(0.5x): 아주 또박또박, 느리게. 안내·학습용.
- 가운데(1.0~1.25x): 자연스러운 속도.
- 오른쪽(2.0x): 빠르게. 긴 글 요약 낭독용. 너무 올리면 알아듣기 어려워짐.
추천: 일반 1.1~1.25x, 또박또박 0.9x, 빠른 낭독 1.5x.
## 2. 단어 간격
단어(어절) 사이를 얼마나 띄울지 정한다. 문장은 자연스럽게 통째로 읽되, 단어 사이
쉼만 늘리거나 줄인다(억양은 유지됨).
- 0 (기본에 가까움): 가장 자연스럽고 촘촘하다.
- 양수(오른쪽): 단어 사이를 벌려 또박또박 끊어 읽는 느낌.
- 음수(왼쪽): 단어를 자연 상태보다 더 붙여 빠르고 유창하게.
주의: 원래 쉼이 있는 자리(쉼표·구 경계)에서 효과가 크고, 완전히 이어 읽는 구간은
조절 여지가 적다. 아주 짧은 발음 사이 공백(자음)은 건드리지 않아 발음이 깨지지 않는다.
추천: 또박또박 +0.15초, 자연 0초, 유창·빠르게 -0.1초.
## 3. 문장 간격
문장과 문장 사이 쉬는 시간이다.
- 양수(오른쪽): 문장 사이를 길게 쉰다. 발표·안내 방송처럼 또렷한 호흡.
- 0: 짧게 붙는다.
- 음수(왼쪽): 문장 경계의 쉼까지 줄여 더 빠르게 이어 읽는다.
추천: 안내 0.6~0.8초, 일반 0.3~0.5초, 빠른 낭독 0초~-0.2초.
## 4. 피치
목소리 높낮이를 반음 단위로 올리고 내린다. 0이 원래 목소리.
- 양수: 밝고 높은 목소리.
- 음수: 낮고 묵직한 목소리.
- 팁: ±2~3 반음 정도가 자연스럽다. 크게 바꾸면 어색해질 수 있다.
---
## 상황별 추천 프리셋
| 용도 | 글자 속도 | 단어 간격 | 문장 간격 | 피치 |
|---|---|---|---|---|
| 자연스러운 기본 | 1.15x | 0초 | 0.4초 | 0 |
| 또박또박 안내/학습 | 0.9x | +0.15초 | 0.7초 | 0 |
| 빠른 요약 낭독 | 1.5x | -0.1초 | -0.1초 | 0 |
| 차분한 내레이션 | 1.0x | +0.05초 | 0.6초 | -2 |
| 발랄한 톤 | 1.2x | 0초 | 0.4초 | +2 |
---
## 팁 & 주의
- 세 속도/간격은 독립이라 자유롭게 조합하면 된다. 예: 글자는 빠르게(1.5x) + 문장은
길게 쉬기(0.8초) → 빠르지만 문장 구분은 또렷.
- 같은 설정이라도 문장마다 길이가 미세하게 다를 수 있다(모델 특성). 정상이다.
- 간격 조절(단어/문장)은 현재 **MeloTTS 엔진에서만** 지원된다. 다른 엔진을 고르면 해당
슬라이더가 비활성화된다.
- 글자 속도를 아주 낮추면(0.5x) 늘어져 들리고, 아주 높이면(2.0x) 발음이 뭉개질 수
있으니 극단값은 미리 들어보고 쓰는 것을 권장한다.

View File

@@ -0,0 +1,290 @@
# TTS 속도·간격·피치 제어 적용 매뉴얼 (watch_screen_ai 이식용)
이 문서는 tts_site에 구현된 4개 음성 제어(글자 속도 · 단어 간격 · 문장 간격 · 피치)를
다른 프로젝트(예: `watch_screen_ai`)에 그대로 이식하기 위한 매뉴얼이다.
MeloTTS(한국어) 기준의 레퍼런스 구현과, API/UI 계약, 적용 절차, 주의점을 담는다.
- 원본 구현: `tts_site` (repo `git.tkrmagid.kr/tkrmagid/tts_site`), 커밋 `5ed2d5f` 기준
- 핵심 파일: `app/engines/melo_engine.py`, `app/server.py`, `frontend/index.html`, `frontend/app.js`
---
## 1. 4개 컨트롤 요약
| 컨트롤 | 파라미터 | 범위 | 기본값 | 동작 원리 |
|---|---|---|---|---|
| 글자 속도 | `speed` | 0.5 ~ 2.0 | 1.25 | 생성 단계 `length_scale = 1/speed` 로 각 음절 발화 길이 조절 (피치 보존) |
| 단어 간격 | `word_gap` (초) | -0.2 ~ 0.5 | 0.25 | 자연 합성 결과의 **어절 사이 무음**을 늘리거나(양수) 줄임(음수) |
| 문장 간격 | `sentence_gap` (초) | -0.5 ~ 1.5 | 0.75 | 문장 경계에 무음 삽입(양수) 또는 경계 무음 트리밍(음수) |
| 피치 | `pitch` (반음) | -12 ~ +12 | 0 | `librosa.effects.pitch_shift` (선택) |
핵심 설계 원칙:
- **글자 속도**는 후처리 배속(atempo)이 아니라 **생성 단계 length_scale**로 처리한다.
각 음절("안·녕·하")의 발음 자체가 빨라지고 느려지며, 배속 특유의 뭉개짐/금속성이 없다.
- **단어/문장 간격은 글자 속도와 독립**이다. 간격은 삽입/삭제한 "초" 단위 그대로 유지되고
발화 속도 변경에 영향받지 않는다.
- **음수 간격**은 경계/어절 사이 무음을 잘라 자연 상태(0)보다 **더 촘촘하게** 만든다.
---
## 2. 왜 이렇게 하는가 (원리)
### 2.1 글자 속도 = length_scale (후처리 배속 금지)
빠르게 만드는 방법은 크게 3가지이고 왜곡 성격이 다르다.
- 단순 리샘플링: 피치까지 올라가 "다람쥐 소리" → **사용 금지**
- 후처리 타임스트레치(atempo/rubberband): 피치는 보존되나 자음 뭉개짐·금속성 아티팩트
- **생성 단계 length_scale**: 모델이 처음부터 빠른/느린 발화를 합성 → 음색 보존, 아티팩트 없음 → **채택**
MeloTTS는 `tts_to_file(..., speed=s)` 내부에서 `length_scale = 1/s`로 모든 음소 길이를 스케일한다.
### 2.2 간격은 오디오 레벨에서 (무음 삽입/트리밍)
- 문장 간격: 문장을 개별 합성 → 사이에 무음(zeros)을 넣거나(양수), 경계 무음을 잘라낸다(음수).
- 단어 간격: 문장을 **통째로 자연 합성**(억양 보존)한 뒤, 내부 어절 무음 구간만 늘리거나 줄인다.
- 어절을 개별 합성하면 억양이 밋밋해지고 자연 상태보다 더 붙일 수 없으므로 쓰지 않는다.
- 자음 폐쇄음 같은 짧은 무음(< 80ms)과 문장 앞뒤 경계 무음은 건드리지 않아 발음이 안전하다.
---
## 3. 레퍼런스 구현 (그대로 복붙 가능)
MeloTTS `TTS` 객체(`from melo.api import TTS`)와 `numpy`만 있으면 된다.
`model.tts_to_file(text, speaker_id, output_path=None, speed=s, quiet=True)` 는
`float32` 파형(numpy)을 반환하고, `model.hps.data.sampling_rate` 가 sr 이다.
```python
import re
import numpy as np
# ── 문장 분리 ────────────────────────────────────────────────
_SENT_SPLIT_RE = re.compile(r"(?<=[.!?。!?…])\s+|\n+")
def split_sentences(text: str) -> list[str]:
parts = [p.strip() for p in _SENT_SPLIT_RE.split(text) if p and p.strip()]
return parts or ([text.strip()] if text.strip() else [])
# ── 경계 무음 트리밍(문장 간격 음수용) ──────────────────────────
def trim_end_silence(a, sr, max_sec, thresh=0.02):
n = a.size
if n == 0 or max_sec <= 0:
return a, 0.0
max_n = min(n, int(sr * max_sec))
if max_n <= 0:
return a, 0.0
tail = np.abs(a[n - max_n:])
nz = np.where(tail >= thresh)[0]
cut = max_n if nz.size == 0 else (max_n - 1 - int(nz[-1]))
return (a, 0.0) if cut <= 0 else (a[: n - cut], cut / sr)
def trim_start_silence(a, sr, max_sec, thresh=0.02):
n = a.size
if n == 0 or max_sec <= 0:
return a, 0.0
max_n = min(n, int(sr * max_sec))
if max_n <= 0:
return a, 0.0
head = np.abs(a[:max_n])
nz = np.where(head >= thresh)[0]
cut = max_n if nz.size == 0 else int(nz[0])
return (a, 0.0) if cut <= 0 else (a[cut:], cut / sr)
# ── 조각 이어붙이기(문장 간격: 양수=무음삽입, 음수=경계 트리밍) ──────
def append_unit(pieces, unit, gap, sr):
if not pieces:
pieces.append(unit); return
if gap >= 0:
if gap > 1e-4:
pieces.append(np.zeros(int(sr * gap), dtype=np.float32))
pieces.append(unit); return
budget = -gap
prev, removed = trim_end_silence(pieces[-1], sr, budget)
pieces[-1] = prev
budget -= removed
if budget > 1e-4:
unit, _ = trim_start_silence(unit, sr, budget)
pieces.append(unit)
# ── 단어 간격: 문장 내부 무음 스케일링(양수=늘림, 음수=줄임) ─────────
def scale_word_gaps(a, sr, delta, thresh=0.02, min_pause=0.08):
n = a.size
if abs(delta) < 1e-4 or n == 0:
return a
silent = np.abs(a) < thresh
changes = np.flatnonzero(np.diff(silent.astype(np.int8)) != 0) + 1
bounds = [0, *changes.tolist(), n]
min_n = int(sr * min_pause)
floor_n = int(sr * 0.015)
add_n = int(delta * sr)
out, last = [], len(bounds) - 2
for k in range(len(bounds) - 1):
s, e = bounds[k], bounds[k + 1]
seg = a[s:e]
is_internal = 0 < k < last # 앞/뒤 경계 무음은 제외
if silent[s] and is_internal and (e - s) >= min_n:
new_n = max(floor_n, (e - s) + add_n)
if new_n >= (e - s):
seg = np.concatenate([seg, np.zeros(new_n - (e - s), dtype=np.float32)])
else:
seg = seg[:new_n]
out.append(seg)
return np.concatenate(out) if out else a
# ── 메인 렌더 함수 ──────────────────────────────────────────────
def render(model, text, speaker_id, *,
speed=1.25, word_gap=0.25, sentence_gap=0.75, pitch=0.0):
"""model: melo.api.TTS 인스턴스. float32 파형(numpy)과 sr 을 만든다."""
speed = float(max(0.5, min(2.0, speed)))
word_gap = float(max(-0.2, min(0.5, word_gap)))
sentence_gap = float(max(-0.5, min(1.5, sentence_gap)))
pitch = float(max(-12., min(12., pitch)))
sr = model.hps.data.sampling_rate
pieces, first = [], True
for sent in (split_sentences(text) or [text]):
# 글자 속도 = 생성 단계 length_scale
a = np.asarray(
model.tts_to_file(sent, speaker_id, output_path=None, speed=speed, quiet=True),
dtype=np.float32,
)
if abs(word_gap) > 1e-4:
a = scale_word_gaps(a, sr, word_gap)
if first:
pieces.append(a); first = False
else:
append_unit(pieces, a, sentence_gap, sr)
audio = np.concatenate(pieces) if pieces else np.zeros(1, dtype=np.float32)
if abs(pitch) > 1e-3:
import librosa
audio = librosa.effects.pitch_shift(audio, sr, n_steps=pitch)
return audio, sr
```
WAV(PCM16) 바이트로 만들려면:
```python
import io, soundfile as sf
buf = io.BytesIO(); sf.write(buf, audio, sr, format="WAV", subtype="PCM_16"); buf.seek(0)
wav_bytes = buf.read()
```
> AUTO(한/영/일/중 혼합) 언어 분기가 필요하면 원본 `melo_engine.py`의 `split_by_language`
> + `_synth_natural` 를 참고. 단일 언어면 위 `render` 로 충분하다.
---
## 4. API 계약
요청(JSON, `POST /api/tts`):
```json
{
"text": "읽을 내용",
"engine": "melo",
"language": "KR",
"speaker": "KR",
"speed": 1.25,
"pitch": 0.0,
"word_gap": 0.25,
"sentence_gap": 0.75
}
```
FastAPI/Pydantic 필드 정의(범위·기본값 포함):
```python
speed: float = Field(1.25, ge=0.5, le=2.0)
pitch: float = Field(0.0, ge=-12.0, le=12.0)
word_gap: float = Field(0.25, ge=-0.2, le=0.5) # 음수=더 붙임
sentence_gap: float = Field(0.75, ge=-0.5, le=1.5) # 음수=더 붙임
```
응답: `audio/wav` (PCM16) 바이트.
---
## 5. UI 슬라이더 스펙
```html
<!-- 글자 속도 -->
<input type="range" id="speed" min="0.5" max="2.0" step="0.05" value="1.25" />
<!-- 단어 간격(초) -->
<input type="range" id="wordGap" min="-0.2" max="0.5" step="0.01" value="0.25" />
<!-- 문장 간격(초) -->
<input type="range" id="sentGap" min="-0.5" max="1.5" step="0.05" value="0.75" />
<!-- 피치(반음) -->
<input type="range" id="pitch" min="-12" max="12" step="1" value="0" />
```
표시 포맷(예):
```js
speedVal.textContent = parseFloat(speed.value).toFixed(2) + "x"; // 1.25x
wordGapVal.textContent = Math.round(parseFloat(wordGap.value)*1000) + " ms"; // -70 ms
sentGapVal.textContent = parseFloat(sentGap.value).toFixed(2) + " s"; // -0.30 s
pitchVal.textContent = (v>0? "+"+v : v) + " 반음";
```
전송 시 `word_gap`, `sentence_gap` 은 **초 단위 float** 로 보낸다(ms 아님).
엔진 지원 플래그로 슬라이더 활성/비활성 처리(선택):
```js
wordGap.disabled = supports.word_gap !== true;
sentGap.disabled = supports.sentence_gap !== true;
```
엔진 `describe().supports` 예: `{"speed": true, "pitch": true, "word_gap": true, "sentence_gap": true}`
---
## 6. watch_screen_ai 적용 절차
1. **의존성**: `melo`(MeloTTS), `numpy`, `soundfile`, (피치 쓰면) `librosa`. GPU면 torch cu128.
2. **모델 로드**: `model = TTS(language="KR", device="cuda:0" if torch.cuda.is_available() else "cpu")`
- `speaker_id = model.hps.data.spk2id["KR"]`
3. **렌더 함수 이식**: 위 §3 코드를 그대로 넣고, TTS 호출부를 `render(model, text, speaker_id, ...)` 로 교체.
4. **파라미터 노출**:
- 기존에 "속도" 하나만 있었다면 `speed`(글자 속도)로 매핑하고, `word_gap`/`sentence_gap` 을 추가.
- API/설정에 §4 필드를, UI가 있으면 §5 슬라이더를 추가.
5. **검증**: §8 스니펫으로 각 파라미터가 독립적으로 duration을 바꾸는지 확인.
6. **배포**: 이미지/서비스 재빌드·재기동 후 실제 합성으로 확인.
기존에 후처리 배속(atempo/rubberband/리샘플)으로 속도를 주고 있었다면 그 코드는 제거하고
`speed`(length_scale) 경로로 교체할 것. 배속과 length_scale을 동시에 걸면 이중 왜곡이 된다.
---
## 7. 주의점 / 한계
- **글자 속도 vs 전체 배속**: 이 방식은 "음절 발화 속도"다. 완성 음성을 통째로 빠르게(전체 배속)
하고 싶으면 그건 별도의 atempo 슬라이더로 분리해야 한다(두 개념은 한 슬라이더로 공존 불가).
- **단어 간격의 효과 범위**: 문장 내부에 실제로 존재하는 무음(어절/구 경계 pause)만 조절한다.
쉼표 등으로 pause가 있으면 효과가 크고, 완전 연속 발화 구간은 조절 여지가 적다.
음수는 그 pause를 자연 상태보다 더 줄인다.
- **min_pause(기본 80ms)**: 이보다 짧은 무음은 자음 폐쇄음일 수 있어 건드리지 않는다.
더 촘촘히 줄이고 싶으면 낮추되, 너무 낮추면 파열음이 뭉개질 수 있다.
- **thresh(기본 0.02)**: 무음 판정 임계값(파형 진폭, float32 [-1,1] 기준 ~1~2%). 배경 잡음이
있는 음성이면 올리고, 아주 조용하면 내린다.
- **MeloTTS 확률성**: 내부 duration predictor가 확률적이라 같은 문장도 길이가 미세하게 다르다.
검증 시 여러 번 평균으로 비교할 것.
- **문장 간격 트리밍 한도**: 경계에 존재하는 무음 이상으로는 못 줄인다(겹침/크로스페이드 미구현).
---
## 8. 검증 스니펫
```python
import io, wave, statistics
def dur(wav_bytes):
w = wave.open(io.BytesIO(wav_bytes)); return w.getnframes()/w.getframerate()
# 글자 속도(간격 0): 1.5배가 1.0배보다 짧아야
# 문장 간격: 0.75 > 0.0 > -0.5 순으로 짧아져야
# 단어 간격: +0.3 > 0.0 > -0.2 순으로 짧아져야 (확률성 있어 3~4회 평균)
```
측정 예(라이브, 참고값):
- 글자 속도 0.7 / 1.0 / 1.5 → 3.82 / 2.74 / 1.96 s
- 문장 간격 0.75 / 0.0 / -0.5 → 4.64 / 3.86 / 3.36 s
- 단어 간격 +0.3 / 0.0 / -0.2 → 6.82 / 6.00 / 5.64 s (같은 문장)
---
## 9. 파라미터 치트시트
- 또박또박 천천히: `speed 0.9`, `word_gap 0.15`, `sentence_gap 0.6`
- 빠르고 촘촘히(요약 낭독): `speed 1.5`, `word_gap -0.1`, `sentence_gap -0.2`
- 자연스러운 기본: `speed 1.1~1.25`, `word_gap 0.0`, `sentence_gap 0.3~0.5`

View File

@@ -43,26 +43,26 @@
<div class="slider-row">
<div class="slider-head">
<span>글자 속도</span>
<span class="slider-val" id="speedVal">1.40x</span>
<span class="slider-val" id="speedVal">1.25x</span>
</div>
<input type="range" id="speed" min="0.5" max="2.0" step="0.05" value="1.4" />
<div class="slider-scale"><span>느리게</span><span>빠르게</span></div>
<input type="range" id="speed" min="0.5" max="2.0" step="0.05" value="1.25" />
<div class="slider-scale"><span>0.5x 느리게</span><span>2.0x 빠르게</span></div>
</div>
<div class="slider-row">
<div class="slider-head">
<span>단어 간격</span>
<span class="slider-val" id="wordGapVal">0 ms</span>
<span class="slider-val" id="wordGapVal">250 ms</span>
</div>
<input type="range" id="wordGap" min="0" max="1.0" step="0.02" value="0" />
<div class="slider-scale"><span>붙여서</span><span>띄어서</span></div>
<input type="range" id="wordGap" min="-0.2" max="0.5" step="0.01" value="0.25" />
<div class="slider-scale"><span>-200 ms 더 붙여</span><span>500 ms 띄어서</span></div>
</div>
<div class="slider-row">
<div class="slider-head">
<span>문장 간격</span>
<span class="slider-val" id="sentGapVal">0.30 s</span>
<span class="slider-val" id="sentGapVal">0.75 s</span>
</div>
<input type="range" id="sentGap" min="0" max="1.5" step="0.05" value="0.3" />
<div class="slider-scale"><span>짧게</span><span>길게</span></div>
<input type="range" id="sentGap" min="-0.5" max="1.5" step="0.05" value="0.75" />
<div class="slider-scale"><span>-0.5 s 더 붙여</span><span>1.5 s 길게</span></div>
</div>
<div class="slider-row">
<div class="slider-head">