feat: 단어·문장 간격을 음수까지 허용(0 미만은 경계 무음 트리밍)
- word_gap 범위 -0.2~0.5, sentence_gap 범위 -0.5~1.5 로 확장(음수=더 붙임) - 음수 간격은 이전 조각 끝 + 다음 조각 시작의 무음을 |gap| 초까지 잘라 붙이는 _trim_end/start_silence + _append_unit 로 처리 - API Field 범위/설명, 프론트 슬라이더 min/스케일 라벨 갱신
This commit is contained in:
@@ -308,6 +308,65 @@ class MeloEngine(BaseEngine):
|
|||||||
return audio, (sr or self._get_model(DEFAULT_LANG).hps.data.sampling_rate)
|
return audio, (sr or self._get_model(DEFAULT_LANG).hps.data.sampling_rate)
|
||||||
return self._synth_segment(text, language, speaker, speed)
|
return self._synth_segment(text, language, speaker, speed)
|
||||||
|
|
||||||
|
# ---- 간격 처리(양수=무음 삽입, 음수=경계 무음 트리밍) --------------
|
||||||
|
@staticmethod
|
||||||
|
def _trim_end_silence(
|
||||||
|
a: np.ndarray, sr: int, max_sec: float, thresh: float = 0.02
|
||||||
|
) -> tuple[np.ndarray, float]:
|
||||||
|
n = a.size
|
||||||
|
if n == 0 or max_sec <= 0:
|
||||||
|
return a, 0.0
|
||||||
|
max_n = min(n, int(sr * max_sec))
|
||||||
|
if max_n <= 0:
|
||||||
|
return a, 0.0
|
||||||
|
tail = np.abs(a[n - max_n:])
|
||||||
|
nz = np.where(tail >= thresh)[0]
|
||||||
|
cut = max_n if nz.size == 0 else (max_n - 1 - int(nz[-1]))
|
||||||
|
if cut <= 0:
|
||||||
|
return a, 0.0
|
||||||
|
return a[: n - cut], cut / sr
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _trim_start_silence(
|
||||||
|
a: np.ndarray, sr: int, max_sec: float, thresh: float = 0.02
|
||||||
|
) -> tuple[np.ndarray, float]:
|
||||||
|
n = a.size
|
||||||
|
if n == 0 or max_sec <= 0:
|
||||||
|
return a, 0.0
|
||||||
|
max_n = min(n, int(sr * max_sec))
|
||||||
|
if max_n <= 0:
|
||||||
|
return a, 0.0
|
||||||
|
head = np.abs(a[:max_n])
|
||||||
|
nz = np.where(head >= thresh)[0]
|
||||||
|
cut = max_n if nz.size == 0 else int(nz[0])
|
||||||
|
if cut <= 0:
|
||||||
|
return a, 0.0
|
||||||
|
return a[cut:], cut / sr
|
||||||
|
|
||||||
|
def _append_unit(
|
||||||
|
self, pieces: list[np.ndarray], unit: np.ndarray, gap: float, sr: int
|
||||||
|
) -> None:
|
||||||
|
"""이전 조각들 뒤에 unit 을 붙인다.
|
||||||
|
|
||||||
|
gap>=0 이면 그 만큼 무음을 삽입, gap<0 이면 |gap| 초만큼 경계의 무음을
|
||||||
|
(이전 조각 끝 + 다음 조각 시작 순서로) 잘라내 더 붙인다.
|
||||||
|
"""
|
||||||
|
if not pieces:
|
||||||
|
pieces.append(unit)
|
||||||
|
return
|
||||||
|
if gap >= 0:
|
||||||
|
if gap > 1e-4:
|
||||||
|
pieces.append(np.zeros(int(sr * gap), dtype=np.float32))
|
||||||
|
pieces.append(unit)
|
||||||
|
return
|
||||||
|
budget = -gap
|
||||||
|
prev, removed = self._trim_end_silence(pieces[-1], sr, budget)
|
||||||
|
pieces[-1] = prev
|
||||||
|
budget -= removed
|
||||||
|
if budget > 1e-4:
|
||||||
|
unit, _ = self._trim_start_silence(unit, sr, budget)
|
||||||
|
pieces.append(unit)
|
||||||
|
|
||||||
# ---- 합성 -----------------------------------------------------
|
# ---- 합성 -----------------------------------------------------
|
||||||
def synth_wav(
|
def synth_wav(
|
||||||
self,
|
self,
|
||||||
@@ -322,9 +381,10 @@ class MeloEngine(BaseEngine):
|
|||||||
"""세 가지를 독립적으로 조절:
|
"""세 가지를 독립적으로 조절:
|
||||||
- speed: 글자(음절) 발화 속도 — 모델 length_scale 로 각 음절의 발음
|
- speed: 글자(음절) 발화 속도 — 모델 length_scale 로 각 음절의 발음
|
||||||
속도를 생성 단계에서 조절(피치 보존, 사후 배속 아님)
|
속도를 생성 단계에서 조절(피치 보존, 사후 배속 아님)
|
||||||
- word_gap: 단어 사이 추가 무음(초). >0 이면 어절 단위로 나눠 합성
|
- word_gap: 단어 사이 간격(초). >0 무음 삽입(어절 단위 합성),
|
||||||
- sentence_gap: 문장 사이 무음(초)
|
<0 경계 무음을 잘라 더 붙임, 0 자연스러운 문장 합성
|
||||||
간격은 발화 속도와 무관하게 지정한 초만큼 유지되어 서로 독립적이다.
|
- sentence_gap: 문장 사이 간격(초). >0 무음 삽입, <0 경계 무음 트리밍
|
||||||
|
간격은 발화 속도와 무관하게 유지되어 서로 독립적이다.
|
||||||
"""
|
"""
|
||||||
text = (text or "").strip()
|
text = (text or "").strip()
|
||||||
if not text:
|
if not text:
|
||||||
@@ -332,45 +392,45 @@ class MeloEngine(BaseEngine):
|
|||||||
|
|
||||||
speed = float(max(0.5, min(2.0, speed)))
|
speed = float(max(0.5, min(2.0, speed)))
|
||||||
pitch = float(max(-12.0, min(12.0, pitch)))
|
pitch = float(max(-12.0, min(12.0, pitch)))
|
||||||
word_gap = float(max(0.0, min(1.0, word_gap)))
|
word_gap = float(max(-0.2, min(0.5, word_gap)))
|
||||||
sentence_gap = 0.30 if sentence_gap is None else sentence_gap
|
sentence_gap = 0.30 if sentence_gap is None else sentence_gap
|
||||||
sentence_gap = float(max(0.0, min(2.0, sentence_gap)))
|
sentence_gap = float(max(-0.5, min(1.5, sentence_gap)))
|
||||||
language = (language or "AUTO").upper()
|
language = (language or "AUTO").upper()
|
||||||
|
|
||||||
# 한국어 발음 개선: 기호/약어/사용자사전 전처리 (한국어·자동 모드)
|
# 한국어 발음 개선: 기호/약어/사용자사전 전처리 (한국어·자동 모드)
|
||||||
if language in ("AUTO", "KR"):
|
if language in ("AUTO", "KR"):
|
||||||
text = normalize_korean(text)
|
text = normalize_korean(text)
|
||||||
|
|
||||||
|
# word_gap 이 0 이 아니면(양/음 모두) 어절 단위로 나눠 간격을 조절
|
||||||
|
use_word_mode = abs(word_gap) > 1e-4
|
||||||
sentences = split_sentences(text) or [text]
|
sentences = split_sentences(text) or [text]
|
||||||
sr: int | None = None
|
sr: int | None = None
|
||||||
pieces: list[np.ndarray] = []
|
pieces: list[np.ndarray] = []
|
||||||
|
first = True
|
||||||
|
|
||||||
for si, sent in enumerate(sentences):
|
for si, sent in enumerate(sentences):
|
||||||
if word_gap > 1e-4:
|
sent_gap = 0.0 if first else sentence_gap
|
||||||
# 어절 단위로 나눠 합성 후 단어 간 무음 삽입
|
if use_word_mode:
|
||||||
words = sent.split()
|
words = sent.split() or [sent]
|
||||||
wpieces: list[np.ndarray] = []
|
|
||||||
for wi, word in enumerate(words):
|
for wi, word in enumerate(words):
|
||||||
a, s = self._synth_natural(word, language, speaker, speed)
|
a, s = self._synth_natural(word, language, speaker, speed)
|
||||||
if sr is None:
|
if sr is None:
|
||||||
sr = s
|
sr = s
|
||||||
if wi > 0:
|
gap = sent_gap if wi == 0 else word_gap
|
||||||
wpieces.append(np.zeros(int(sr * word_gap), dtype=np.float32))
|
if first:
|
||||||
wpieces.append(a)
|
pieces.append(a)
|
||||||
sent_audio = (
|
first = False
|
||||||
np.concatenate(wpieces)
|
else:
|
||||||
if wpieces
|
self._append_unit(pieces, a, gap, sr)
|
||||||
else np.zeros(1, dtype=np.float32)
|
|
||||||
)
|
|
||||||
else:
|
else:
|
||||||
a, s = self._synth_natural(sent, language, speaker, speed)
|
a, s = self._synth_natural(sent, language, speaker, speed)
|
||||||
if sr is None:
|
if sr is None:
|
||||||
sr = s
|
sr = s
|
||||||
sent_audio = a
|
if first:
|
||||||
|
pieces.append(a)
|
||||||
if pieces and sentence_gap > 1e-4:
|
first = False
|
||||||
pieces.append(np.zeros(int(sr * sentence_gap), dtype=np.float32))
|
else:
|
||||||
pieces.append(sent_audio)
|
self._append_unit(pieces, a, sent_gap, sr)
|
||||||
|
|
||||||
audio = np.concatenate(pieces) if pieces else np.zeros(1, dtype=np.float32)
|
audio = np.concatenate(pieces) if pieces else np.zeros(1, dtype=np.float32)
|
||||||
if sr is None:
|
if sr is None:
|
||||||
|
|||||||
@@ -42,8 +42,8 @@ class TTSRequest(BaseModel):
|
|||||||
# 기본값은 각 범위의 가운데. 슬라이더 양 끝이 조절 가능한 최소/최대치.
|
# 기본값은 각 범위의 가운데. 슬라이더 양 끝이 조절 가능한 최소/최대치.
|
||||||
speed: float = Field(1.25, ge=0.5, le=2.0, description="글자(음소) 발화 속도(0.5~2.0, 기본 1.25)")
|
speed: float = Field(1.25, ge=0.5, le=2.0, description="글자(음소) 발화 속도(0.5~2.0, 기본 1.25)")
|
||||||
pitch: float = Field(0.0, ge=-12.0, le=12.0, description="피치(반음)")
|
pitch: float = Field(0.0, ge=-12.0, le=12.0, description="피치(반음)")
|
||||||
word_gap: float = Field(0.25, ge=0.0, le=0.5, description="단어 사이 무음 초(0~0.5, 기본 0.25)")
|
word_gap: float = Field(0.25, ge=-0.2, le=0.5, description="단어 사이 간격 초(-0.2~0.5, 음수=더 붙임, 기본 0.25)")
|
||||||
sentence_gap: float = Field(0.75, ge=0.0, le=1.5, description="문장 사이 무음 초(0~1.5, 기본 0.75)")
|
sentence_gap: float = Field(0.75, ge=-0.5, le=1.5, description="문장 사이 간격 초(-0.5~1.5, 음수=더 붙임, 기본 0.75)")
|
||||||
|
|
||||||
|
|
||||||
@app.on_event("startup")
|
@app.on_event("startup")
|
||||||
|
|||||||
@@ -53,16 +53,16 @@
|
|||||||
<span>단어 간격</span>
|
<span>단어 간격</span>
|
||||||
<span class="slider-val" id="wordGapVal">250 ms</span>
|
<span class="slider-val" id="wordGapVal">250 ms</span>
|
||||||
</div>
|
</div>
|
||||||
<input type="range" id="wordGap" min="0" max="0.5" step="0.01" value="0.25" />
|
<input type="range" id="wordGap" min="-0.2" max="0.5" step="0.01" value="0.25" />
|
||||||
<div class="slider-scale"><span>0 ms 붙여서</span><span>500 ms 띄어서</span></div>
|
<div class="slider-scale"><span>-200 ms 더 붙여</span><span>500 ms 띄어서</span></div>
|
||||||
</div>
|
</div>
|
||||||
<div class="slider-row">
|
<div class="slider-row">
|
||||||
<div class="slider-head">
|
<div class="slider-head">
|
||||||
<span>문장 간격</span>
|
<span>문장 간격</span>
|
||||||
<span class="slider-val" id="sentGapVal">0.75 s</span>
|
<span class="slider-val" id="sentGapVal">0.75 s</span>
|
||||||
</div>
|
</div>
|
||||||
<input type="range" id="sentGap" min="0" max="1.5" step="0.05" value="0.75" />
|
<input type="range" id="sentGap" min="-0.5" max="1.5" step="0.05" value="0.75" />
|
||||||
<div class="slider-scale"><span>0 s 짧게</span><span>1.5 s 길게</span></div>
|
<div class="slider-scale"><span>-0.5 s 더 붙여</span><span>1.5 s 길게</span></div>
|
||||||
</div>
|
</div>
|
||||||
<div class="slider-row">
|
<div class="slider-row">
|
||||||
<div class="slider-head">
|
<div class="slider-head">
|
||||||
|
|||||||
Reference in New Issue
Block a user