diff --git a/app/engines/melo_engine.py b/app/engines/melo_engine.py index ccbd663..3e661d0 100644 --- a/app/engines/melo_engine.py +++ b/app/engines/melo_engine.py @@ -308,6 +308,65 @@ class MeloEngine(BaseEngine): return audio, (sr or self._get_model(DEFAULT_LANG).hps.data.sampling_rate) return self._synth_segment(text, language, speaker, speed) + # ---- 간격 처리(양수=무음 삽입, 음수=경계 무음 트리밍) -------------- + @staticmethod + def _trim_end_silence( + a: np.ndarray, sr: int, max_sec: float, thresh: float = 0.02 + ) -> tuple[np.ndarray, float]: + n = a.size + if n == 0 or max_sec <= 0: + return a, 0.0 + max_n = min(n, int(sr * max_sec)) + if max_n <= 0: + return a, 0.0 + tail = np.abs(a[n - max_n:]) + nz = np.where(tail >= thresh)[0] + cut = max_n if nz.size == 0 else (max_n - 1 - int(nz[-1])) + if cut <= 0: + return a, 0.0 + return a[: n - cut], cut / sr + + @staticmethod + def _trim_start_silence( + a: np.ndarray, sr: int, max_sec: float, thresh: float = 0.02 + ) -> tuple[np.ndarray, float]: + n = a.size + if n == 0 or max_sec <= 0: + return a, 0.0 + max_n = min(n, int(sr * max_sec)) + if max_n <= 0: + return a, 0.0 + head = np.abs(a[:max_n]) + nz = np.where(head >= thresh)[0] + cut = max_n if nz.size == 0 else int(nz[0]) + if cut <= 0: + return a, 0.0 + return a[cut:], cut / sr + + def _append_unit( + self, pieces: list[np.ndarray], unit: np.ndarray, gap: float, sr: int + ) -> None: + """이전 조각들 뒤에 unit 을 붙인다. + + gap>=0 이면 그 만큼 무음을 삽입, gap<0 이면 |gap| 초만큼 경계의 무음을 + (이전 조각 끝 + 다음 조각 시작 순서로) 잘라내 더 붙인다. + """ + if not pieces: + pieces.append(unit) + return + if gap >= 0: + if gap > 1e-4: + pieces.append(np.zeros(int(sr * gap), dtype=np.float32)) + pieces.append(unit) + return + budget = -gap + prev, removed = self._trim_end_silence(pieces[-1], sr, budget) + pieces[-1] = prev + budget -= removed + if budget > 1e-4: + unit, _ = self._trim_start_silence(unit, sr, budget) + pieces.append(unit) + # ---- 합성 ----------------------------------------------------- def synth_wav( self, @@ -322,9 +381,10 @@ class MeloEngine(BaseEngine): """세 가지를 독립적으로 조절: - speed: 글자(음절) 발화 속도 — 모델 length_scale 로 각 음절의 발음 속도를 생성 단계에서 조절(피치 보존, 사후 배속 아님) - - word_gap: 단어 사이 추가 무음(초). >0 이면 어절 단위로 나눠 합성 - - sentence_gap: 문장 사이 무음(초) - 간격은 발화 속도와 무관하게 지정한 초만큼 유지되어 서로 독립적이다. + - word_gap: 단어 사이 간격(초). >0 무음 삽입(어절 단위 합성), + <0 경계 무음을 잘라 더 붙임, 0 자연스러운 문장 합성 + - sentence_gap: 문장 사이 간격(초). >0 무음 삽입, <0 경계 무음 트리밍 + 간격은 발화 속도와 무관하게 유지되어 서로 독립적이다. """ text = (text or "").strip() if not text: @@ -332,45 +392,45 @@ class MeloEngine(BaseEngine): speed = float(max(0.5, min(2.0, speed))) pitch = float(max(-12.0, min(12.0, pitch))) - word_gap = float(max(0.0, min(1.0, word_gap))) + word_gap = float(max(-0.2, min(0.5, word_gap))) sentence_gap = 0.30 if sentence_gap is None else sentence_gap - sentence_gap = float(max(0.0, min(2.0, sentence_gap))) + sentence_gap = float(max(-0.5, min(1.5, sentence_gap))) language = (language or "AUTO").upper() # 한국어 발음 개선: 기호/약어/사용자사전 전처리 (한국어·자동 모드) if language in ("AUTO", "KR"): text = normalize_korean(text) + # word_gap 이 0 이 아니면(양/음 모두) 어절 단위로 나눠 간격을 조절 + use_word_mode = abs(word_gap) > 1e-4 sentences = split_sentences(text) or [text] sr: int | None = None pieces: list[np.ndarray] = [] + first = True for si, sent in enumerate(sentences): - if word_gap > 1e-4: - # 어절 단위로 나눠 합성 후 단어 간 무음 삽입 - words = sent.split() - wpieces: list[np.ndarray] = [] + sent_gap = 0.0 if first else sentence_gap + if use_word_mode: + words = sent.split() or [sent] for wi, word in enumerate(words): a, s = self._synth_natural(word, language, speaker, speed) if sr is None: sr = s - if wi > 0: - wpieces.append(np.zeros(int(sr * word_gap), dtype=np.float32)) - wpieces.append(a) - sent_audio = ( - np.concatenate(wpieces) - if wpieces - else np.zeros(1, dtype=np.float32) - ) + gap = sent_gap if wi == 0 else word_gap + if first: + pieces.append(a) + first = False + else: + self._append_unit(pieces, a, gap, sr) else: a, s = self._synth_natural(sent, language, speaker, speed) if sr is None: sr = s - sent_audio = a - - if pieces and sentence_gap > 1e-4: - pieces.append(np.zeros(int(sr * sentence_gap), dtype=np.float32)) - pieces.append(sent_audio) + if first: + pieces.append(a) + first = False + else: + self._append_unit(pieces, a, sent_gap, sr) audio = np.concatenate(pieces) if pieces else np.zeros(1, dtype=np.float32) if sr is None: diff --git a/app/server.py b/app/server.py index 2e2f5ca..cb90991 100644 --- a/app/server.py +++ b/app/server.py @@ -42,8 +42,8 @@ class TTSRequest(BaseModel): # 기본값은 각 범위의 가운데. 슬라이더 양 끝이 조절 가능한 최소/최대치. speed: float = Field(1.25, ge=0.5, le=2.0, description="글자(음소) 발화 속도(0.5~2.0, 기본 1.25)") pitch: float = Field(0.0, ge=-12.0, le=12.0, description="피치(반음)") - word_gap: float = Field(0.25, ge=0.0, le=0.5, description="단어 사이 무음 초(0~0.5, 기본 0.25)") - sentence_gap: float = Field(0.75, ge=0.0, le=1.5, description="문장 사이 무음 초(0~1.5, 기본 0.75)") + word_gap: float = Field(0.25, ge=-0.2, le=0.5, description="단어 사이 간격 초(-0.2~0.5, 음수=더 붙임, 기본 0.25)") + sentence_gap: float = Field(0.75, ge=-0.5, le=1.5, description="문장 사이 간격 초(-0.5~1.5, 음수=더 붙임, 기본 0.75)") @app.on_event("startup") diff --git a/frontend/index.html b/frontend/index.html index f7ba760..373e9e7 100644 --- a/frontend/index.html +++ b/frontend/index.html @@ -53,16 +53,16 @@ 단어 간격 250 ms - -
0 ms 붙여서500 ms 띄어서
+ +
-200 ms 더 붙여500 ms 띄어서
문장 간격 0.75 s
- -
0 s 짧게1.5 s 길게
+ +
-0.5 s 더 붙여1.5 s 길게