feat(tts): per-emotion voice controls (speed/word-gap/sentence-gap/pitch)
Each emotion can now be tuned independently. parse_segments resolves the four
controls per segment from a base (공통) dict plus an optional per-emotion
override; a missing override key inherits base. By default there are no
overrides, so every emotion delivers with the base values (모든 감정 = 기본값).
- emotion.py: Segment now carries all 4 controls + the canonical emotion name;
parse_segments(text, base, overrides). Adds EMOTION_LABELS/EMOTIONS for the UI.
- melo.py: MeloTTS.emotion_overrides store; synth resolves per-segment controls
and sends them per segment.
- melo_worker.py: _render applies each segment's own word_gap/sentence_gap/pitch
(previously reply-global).
- dashboard.py: emotion dropdown in the TTS panel; GET returns base + overrides
+ emotion list; POST {emotion,...} stores an override (or {reset:true} clears
it); base is set when emotion is omitted/"base".
Verified: an override on one emotion slows only that emotion (happy@0.7=5.66s vs
base 3.02s; sad unchanged at 3.06s); dashboard store/reset/base all work; 34
tests pass.
Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
@@ -1,56 +1,82 @@
|
|||||||
"""Emotion-tag parsing for expressive TTS. A ``[감정]`` tag must steer the
|
"""Emotion-tag parsing for expressive TTS. A ``[감정]`` tag must steer the
|
||||||
pitch/speed of the text that follows without being spoken; a bracket that is NOT
|
delivery of the text that follows without being spoken; a bracket that is NOT a
|
||||||
a known emotion word must be kept as ordinary spoken content."""
|
known emotion word must be kept as ordinary spoken content.
|
||||||
|
|
||||||
|
Delivery is now driven by a base (공통) control dict plus optional per-emotion
|
||||||
|
overrides. By default there are no overrides, so every emotion delivers with the
|
||||||
|
base controls (모든 감정 = 기본값)."""
|
||||||
|
|
||||||
from wsai.backends.emotion import (
|
from wsai.backends.emotion import (
|
||||||
EMOTION_PARAMS,
|
EMOTION_LABELS,
|
||||||
|
_SYNONYMS,
|
||||||
match_emotion,
|
match_emotion,
|
||||||
parse_segments,
|
parse_segments,
|
||||||
)
|
)
|
||||||
from wsai.dashboard import _speech_text
|
from wsai.dashboard import _speech_text
|
||||||
|
|
||||||
BASE = 1.3
|
BASE = {"speed": 1.35, "word_gap": -0.07, "sentence_gap": -0.30, "pitch": 0.0}
|
||||||
|
|
||||||
|
|
||||||
def test_emotion_tag_is_not_spoken_and_sets_delivery():
|
def test_emotion_tag_is_not_spoken_and_tags_segment():
|
||||||
segs = parse_segments("[기쁨] 오늘 날씨 좋다", BASE)
|
segs = parse_segments("[기쁨] 오늘 날씨 좋다", BASE)
|
||||||
assert len(segs) == 1
|
assert len(segs) == 1
|
||||||
assert "기쁨" not in segs[0].text # the tag word is dropped
|
assert "기쁨" not in segs[0].text # the tag word is dropped
|
||||||
assert segs[0].text == "오늘 날씨 좋다"
|
assert segs[0].text == "오늘 날씨 좋다"
|
||||||
mult, semis = EMOTION_PARAMS["happy"]
|
assert segs[0].emotion == "happy"
|
||||||
assert segs[0].speed == BASE * mult
|
|
||||||
assert segs[0].pitch == semis
|
|
||||||
|
def test_default_all_emotions_use_base():
|
||||||
|
# No overrides -> a tagged emotion delivers with exactly the base controls.
|
||||||
|
segs = parse_segments("[기쁨] 좋아", BASE)
|
||||||
|
s = segs[0]
|
||||||
|
assert (s.speed, s.word_gap, s.sentence_gap, s.pitch) == (1.35, -0.07, -0.30, 0.0)
|
||||||
|
|
||||||
|
|
||||||
|
def test_per_emotion_override_applies_and_inherits_missing_keys():
|
||||||
|
segs = parse_segments("[기쁨] 좋아", BASE, {"happy": {"speed": 1.6, "pitch": 3.0}})
|
||||||
|
s = segs[0]
|
||||||
|
assert s.speed == 1.6 and s.pitch == 3.0 # overridden
|
||||||
|
assert s.word_gap == -0.07 and s.sentence_gap == -0.30 # inherited from base
|
||||||
|
|
||||||
|
|
||||||
|
def test_override_only_affects_its_own_emotion():
|
||||||
|
segs = parse_segments("[기쁨] 가. [슬픔] 나.", BASE, {"sad": {"speed": 0.8}})
|
||||||
|
assert segs[0].emotion == "happy" and segs[0].speed == 1.35 # untouched
|
||||||
|
assert segs[1].emotion == "sad" and segs[1].speed == 0.8 # overridden
|
||||||
|
|
||||||
|
|
||||||
def test_midreply_emotion_change_splits_segments():
|
def test_midreply_emotion_change_splits_segments():
|
||||||
segs = parse_segments("[속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!", BASE)
|
segs = parse_segments("[속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!", BASE)
|
||||||
assert len(segs) == 2
|
assert len(segs) == 2
|
||||||
assert segs[0].text == "정말 힘들었겠다."
|
assert segs[0].text == "정말 힘들었겠다." and segs[0].emotion == "sad"
|
||||||
assert segs[1].text == "하지만 넌 할 수 있어!"
|
assert segs[1].text == "하지만 넌 할 수 있어!" and segs[1].emotion == "hopeful"
|
||||||
assert segs[0].pitch == EMOTION_PARAMS["sad"][1]
|
|
||||||
assert segs[1].pitch == EMOTION_PARAMS["hopeful"][1]
|
|
||||||
|
|
||||||
|
|
||||||
def test_non_emotion_bracket_is_spoken_without_brackets():
|
def test_non_emotion_bracket_is_spoken_without_brackets():
|
||||||
segs = parse_segments("[기쁨] 첫째는 [1번] 항목이야", BASE)
|
segs = parse_segments("[기쁨] 첫째는 [1번] 항목이야", BASE)
|
||||||
assert len(segs) == 1
|
assert len(segs) == 1
|
||||||
# "1번" is not an emotion -> read it; brackets themselves are gone.
|
|
||||||
assert "1번" in segs[0].text
|
assert "1번" in segs[0].text
|
||||||
assert "[" not in segs[0].text and "]" not in segs[0].text
|
assert "[" not in segs[0].text and "]" not in segs[0].text
|
||||||
assert segs[0].pitch == EMOTION_PARAMS["happy"][1]
|
assert segs[0].emotion == "happy"
|
||||||
|
|
||||||
|
|
||||||
def test_text_before_first_tag_is_neutral():
|
def test_text_before_first_tag_is_neutral():
|
||||||
segs = parse_segments("잠깐만. [신남] 찾았다!", BASE)
|
segs = parse_segments("잠깐만. [신남] 찾았다!", BASE)
|
||||||
assert segs[0].text == "잠깐만."
|
assert segs[0].text == "잠깐만." and segs[0].emotion is None
|
||||||
assert segs[0].speed == BASE and segs[0].pitch == 0.0
|
assert segs[0].speed == 1.35 and segs[0].pitch == 0.0
|
||||||
assert segs[1].pitch == EMOTION_PARAMS["excited"][1]
|
assert segs[1].emotion == "excited"
|
||||||
|
|
||||||
|
|
||||||
def test_plain_text_is_one_neutral_segment():
|
def test_plain_text_is_one_neutral_segment():
|
||||||
segs = parse_segments("그냥 평범한 문장이야", BASE)
|
segs = parse_segments("그냥 평범한 문장이야", BASE)
|
||||||
assert len(segs) == 1
|
assert len(segs) == 1
|
||||||
assert segs[0].speed == BASE and segs[0].pitch == 0.0
|
assert segs[0].emotion is None and segs[0].speed == 1.35
|
||||||
|
|
||||||
|
|
||||||
|
def test_backward_compat_float_base():
|
||||||
|
# A bare float base still works (speed only; other controls neutral).
|
||||||
|
segs = parse_segments("그냥 문장", 1.2)
|
||||||
|
assert segs[0].speed == 1.2 and segs[0].word_gap == 0.0
|
||||||
|
|
||||||
|
|
||||||
def test_empty_input_yields_no_segments():
|
def test_empty_input_yields_no_segments():
|
||||||
@@ -77,6 +103,13 @@ def test_persona_examples_are_all_recognised():
|
|||||||
assert match_emotion(word) is not None, word
|
assert match_emotion(word) is not None, word
|
||||||
|
|
||||||
|
|
||||||
|
def test_every_canonical_emotion_has_a_ui_label():
|
||||||
|
# The dashboard dropdown lists EMOTION_LABELS; every emotion a tag can
|
||||||
|
# resolve to must appear there or it would be untunable.
|
||||||
|
for canon in set(_SYNONYMS.values()):
|
||||||
|
assert canon in EMOTION_LABELS, canon
|
||||||
|
|
||||||
|
|
||||||
def test_voice_turn_keeps_leading_emotion_tag_for_tts_parser():
|
def test_voice_turn_keeps_leading_emotion_tag_for_tts_parser():
|
||||||
text = "[속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!"
|
text = "[속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!"
|
||||||
|
|
||||||
@@ -84,7 +117,5 @@ def test_voice_turn_keeps_leading_emotion_tag_for_tts_parser():
|
|||||||
segs = parse_segments(spoken, BASE)
|
segs = parse_segments(spoken, BASE)
|
||||||
|
|
||||||
assert spoken == text
|
assert spoken == text
|
||||||
assert segs[0].text == "정말 힘들었겠다."
|
assert segs[0].text == "정말 힘들었겠다." and segs[0].emotion == "sad"
|
||||||
assert segs[0].pitch == EMOTION_PARAMS["sad"][1]
|
assert segs[1].text == "하지만 넌 할 수 있어!" and segs[1].emotion == "hopeful"
|
||||||
assert segs[1].text == "하지만 넌 할 수 있어!"
|
|
||||||
assert segs[1].pitch == EMOTION_PARAMS["hopeful"][1]
|
|
||||||
|
|||||||
@@ -99,33 +99,75 @@ def match_emotion(inner: str) -> str | None:
|
|||||||
return _SYNONYMS.get(_norm(inner))
|
return _SYNONYMS.get(_norm(inner))
|
||||||
|
|
||||||
|
|
||||||
|
# The four adjustable TTS controls, per segment.
|
||||||
|
PARAM_KEYS = ("speed", "word_gap", "sentence_gap", "pitch")
|
||||||
|
|
||||||
|
# Canonical emotion -> Korean UI label, in display order. "base" == 기본/공통
|
||||||
|
# (the value used for untagged text and inherited by any emotion with no
|
||||||
|
# override). Every canonical emotion the synonym table resolves to must appear
|
||||||
|
# here so the dashboard dropdown can list it.
|
||||||
|
EMOTION_LABELS: dict[str, str] = {
|
||||||
|
"base": "기본(공통)", "happy": "기쁨", "excited": "신남", "hopeful": "희망",
|
||||||
|
"sad": "슬픔", "angry": "화남", "fearful": "두려움", "surprised": "놀람",
|
||||||
|
"disgust": "혐오", "calm": "차분", "friendly": "다정", "serious": "진지",
|
||||||
|
"disappointed": "실망", "tired": "피곤", "affectionate": "사랑", "playful": "장난",
|
||||||
|
"curious": "궁금", "whisper": "속삭임", "shout": "외침", "determined": "단호",
|
||||||
|
"relieved": "안도",
|
||||||
|
}
|
||||||
|
EMOTIONS: list[str] = list(EMOTION_LABELS)
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
class Segment:
|
class Segment:
|
||||||
text: str
|
text: str
|
||||||
speed: float
|
speed: float
|
||||||
|
word_gap: float
|
||||||
|
sentence_gap: float
|
||||||
pitch: float # semitones; 0.0 == no shift
|
pitch: float # semitones; 0.0 == no shift
|
||||||
|
emotion: str | None = None # canonical emotion, or None for neutral/base
|
||||||
|
|
||||||
|
|
||||||
_TAG_RE = re.compile(r"\[([^\[\]]*)\]")
|
_TAG_RE = re.compile(r"\[([^\[\]]*)\]")
|
||||||
|
|
||||||
|
|
||||||
def parse_segments(text: str, base_speed: float) -> list[Segment]:
|
def _resolve(base: dict, overrides: dict | None, emotion: str | None) -> dict:
|
||||||
"""Split ``text`` into consecutive spoken segments, each carrying the speed
|
"""The 4 controls for one segment: the base values, with any per-emotion
|
||||||
and pitch implied by the most recent emotion tag.
|
override applied on top (a missing override key inherits base)."""
|
||||||
|
p = {k: float(base[k]) for k in PARAM_KEYS}
|
||||||
|
if emotion and overrides:
|
||||||
|
ov = overrides.get(emotion)
|
||||||
|
if ov:
|
||||||
|
for k in PARAM_KEYS:
|
||||||
|
if ov.get(k) is not None:
|
||||||
|
p[k] = float(ov[k])
|
||||||
|
return p
|
||||||
|
|
||||||
|
|
||||||
|
def parse_segments(text: str, base, overrides: dict | None = None) -> list[Segment]:
|
||||||
|
"""Split ``text`` into consecutive spoken segments, each carrying the four
|
||||||
|
TTS controls implied by the most recent emotion tag.
|
||||||
|
|
||||||
|
* ``base`` is the 공통 control dict ``{speed, word_gap, sentence_gap, pitch}``
|
||||||
|
used for untagged text and inherited by any emotion without an override.
|
||||||
|
A bare float is also accepted (speed only) for backward compatibility.
|
||||||
|
* ``overrides`` maps a canonical emotion -> a partial control dict; only the
|
||||||
|
keys present override the base. Empty/None means every emotion delivers
|
||||||
|
with the base controls (the default: 모든 감정 = 기본값).
|
||||||
* An emotion tag switches the active emotion for everything after it and is
|
* An emotion tag switches the active emotion for everything after it and is
|
||||||
not spoken.
|
not spoken. A non-emotion bracket keeps its inner words as spoken text.
|
||||||
* A non-emotion bracket keeps its inner words as spoken text (brackets gone).
|
|
||||||
* Text before any tag is spoken with neutral delivery (base speed, no shift).
|
|
||||||
"""
|
"""
|
||||||
|
if not isinstance(base, dict):
|
||||||
|
base = {"speed": float(base), "word_gap": 0.0, "sentence_gap": 0.0, "pitch": 0.0}
|
||||||
segments: list[Segment] = []
|
segments: list[Segment] = []
|
||||||
cur_speed, cur_pitch = base_speed, 0.0
|
cur_emotion: str | None = None
|
||||||
buf: list[str] = []
|
buf: list[str] = []
|
||||||
|
|
||||||
def flush() -> None:
|
def flush() -> None:
|
||||||
joined = "".join(buf).strip()
|
joined = "".join(buf).strip()
|
||||||
if joined:
|
if joined:
|
||||||
segments.append(Segment(joined, cur_speed, cur_pitch))
|
p = _resolve(base, overrides, cur_emotion)
|
||||||
|
segments.append(Segment(joined, p["speed"], p["word_gap"],
|
||||||
|
p["sentence_gap"], p["pitch"], cur_emotion))
|
||||||
buf.clear()
|
buf.clear()
|
||||||
|
|
||||||
pos = 0
|
pos = 0
|
||||||
@@ -140,8 +182,7 @@ def parse_segments(text: str, base_speed: float) -> list[Segment]:
|
|||||||
# Emotion tag — everything so far belongs to the previous emotion;
|
# Emotion tag — everything so far belongs to the previous emotion;
|
||||||
# flush it, then switch delivery for what follows.
|
# flush it, then switch delivery for what follows.
|
||||||
flush()
|
flush()
|
||||||
mult, semis = EMOTION_PARAMS[emotion]
|
cur_emotion = emotion
|
||||||
cur_speed, cur_pitch = base_speed * mult, semis
|
|
||||||
buf.append(text[pos:])
|
buf.append(text[pos:])
|
||||||
flush()
|
flush()
|
||||||
return segments # empty when there is nothing speakable (blank or all-tags)
|
return segments # empty when there is nothing speakable (blank or all-tags)
|
||||||
|
|||||||
@@ -107,6 +107,10 @@ class MeloTTS:
|
|||||||
else os.environ.get("WSAI_TTS_SENTENCE_GAP", "-0.30"))
|
else os.environ.get("WSAI_TTS_SENTENCE_GAP", "-0.30"))
|
||||||
self.pitch = float(pitch if pitch is not None
|
self.pitch = float(pitch if pitch is not None
|
||||||
else os.environ.get("WSAI_TTS_PITCH", "0.0"))
|
else os.environ.get("WSAI_TTS_PITCH", "0.0"))
|
||||||
|
# Per-emotion control overrides: canonical emotion -> partial dict of
|
||||||
|
# {speed, word_gap, sentence_gap, pitch}. Empty by default, so every
|
||||||
|
# emotion inherits the base controls above (모든 감정 = 기본값).
|
||||||
|
self.emotion_overrides: dict[str, dict] = {}
|
||||||
self.sink = sink or _log_sink
|
self.sink = sink or _log_sink
|
||||||
self._proc: asyncio.subprocess.Process | None = None
|
self._proc: asyncio.subprocess.Process | None = None
|
||||||
self._lock = asyncio.Lock()
|
self._lock = asyncio.Lock()
|
||||||
@@ -214,35 +218,36 @@ class MeloTTS:
|
|||||||
per call (used by the dashboard preview so tuning does not disturb the
|
per call (used by the dashboard preview so tuning does not disturb the
|
||||||
live bot voice until explicitly applied)."""
|
live bot voice until explicitly applied)."""
|
||||||
await self._ensure()
|
await self._ensure()
|
||||||
spd = self.speed if speed is None else float(speed)
|
base = {
|
||||||
wg = self.word_gap if word_gap is None else float(word_gap)
|
"speed": self.speed if speed is None else float(speed),
|
||||||
sg = self.sentence_gap if sentence_gap is None else float(sentence_gap)
|
"word_gap": self.word_gap if word_gap is None else float(word_gap),
|
||||||
pt = self.pitch if pitch is None else float(pitch)
|
"sentence_gap": self.sentence_gap if sentence_gap is None else float(sentence_gap),
|
||||||
|
"pitch": self.pitch if pitch is None else float(pitch),
|
||||||
|
}
|
||||||
text = normalize_for_speech(text)
|
text = normalize_for_speech(text)
|
||||||
# Split on [감정] tags: each tag steers pitch/speed for the text that
|
# Split on [감정] tags: each tag switches delivery for the text that
|
||||||
# follows (and is itself not spoken); non-emotion brackets stay as words.
|
# follows (and is itself not spoken); non-emotion brackets stay as words.
|
||||||
segments = parse_segments(text, spd)
|
# Each segment carries its own 4 controls (base + per-emotion override).
|
||||||
|
segments = parse_segments(text, base, self.emotion_overrides)
|
||||||
self._n += 1
|
self._n += 1
|
||||||
out = str(self.out_dir / f"tts-{self._n:06d}.wav")
|
out = str(self.out_dir / f"tts-{self._n:06d}.wav")
|
||||||
if segments:
|
if segments:
|
||||||
payload = {
|
payload = {
|
||||||
"segments": [
|
"segments": [
|
||||||
{"text": s.text, "speed": s.speed, "pitch": s.pitch}
|
{"text": s.text, "speed": s.speed, "word_gap": s.word_gap,
|
||||||
|
"sentence_gap": s.sentence_gap, "pitch": s.pitch}
|
||||||
for s in segments
|
for s in segments
|
||||||
],
|
],
|
||||||
"out": out,
|
"out": out,
|
||||||
"word_gap": wg,
|
|
||||||
"sentence_gap": sg,
|
|
||||||
"pitch": pt,
|
|
||||||
}
|
}
|
||||||
else: # empty/whitespace reply: keep legacy single-utterance behaviour
|
else: # empty/whitespace reply: keep legacy single-utterance behaviour
|
||||||
payload = {
|
payload = {
|
||||||
"text": text,
|
"text": text,
|
||||||
"out": out,
|
"out": out,
|
||||||
"speed": spd,
|
"speed": base["speed"],
|
||||||
"word_gap": wg,
|
"word_gap": base["word_gap"],
|
||||||
"sentence_gap": sg,
|
"sentence_gap": base["sentence_gap"],
|
||||||
"pitch": pt,
|
"pitch": base["pitch"],
|
||||||
}
|
}
|
||||||
req = json.dumps(payload)
|
req = json.dumps(payload)
|
||||||
s = time.monotonic()
|
s = time.monotonic()
|
||||||
|
|||||||
@@ -11,28 +11,27 @@ library chatter lands on stderr instead.
|
|||||||
|
|
||||||
Protocol (one JSON object per line, on the protocol channel):
|
Protocol (one JSON object per line, on the protocol channel):
|
||||||
<- {"text": "...", "out": "/abs/path.wav", "speed": 1.3,
|
<- {"text": "...", "out": "/abs/path.wav", "speed": 1.3,
|
||||||
"word_gap": 0.25, "sentence_gap": 0.75, "pitch": 0.0}
|
"word_gap": -0.07, "sentence_gap": -0.30, "pitch": 0.0}
|
||||||
<- {"segments": [{"text": "...", "speed": 1.3, "pitch": 2.0}, ...],
|
<- {"segments": [{"text": "...", "speed": 1.3, "word_gap": -0.07,
|
||||||
"out": "/abs/path.wav",
|
"sentence_gap": -0.30, "pitch": 2.0}, ...], "out": "/abs/path.wav"}
|
||||||
"word_gap": 0.25, "sentence_gap": 0.75, "pitch": 0.0}
|
|
||||||
-> {"ok": true, "out": "/abs/path.wav", "ms": 123}
|
-> {"ok": true, "out": "/abs/path.wav", "ms": 123}
|
||||||
-> {"ok": false, "error": "..."}
|
-> {"ok": false, "error": "..."}
|
||||||
On startup, once the model is ready, it emits exactly one line:
|
On startup, once the model is ready, it emits exactly one line:
|
||||||
-> {"ready": true, "ms": <load-ms>, "device": "cpu"}
|
-> {"ready": true, "ms": <load-ms>, "device": "cpu"}
|
||||||
|
|
||||||
Four independent voice controls (ported from tts_site, see docs manual):
|
Four independent voice controls (ported from tts_site, see docs manual). They
|
||||||
speed per-segment glyph speed -> generation-stage length_scale (1/speed)
|
are PER SEGMENT: the caller resolves each segment's controls from the base
|
||||||
word_gap reply-global, sec (-0.2..0.5): grow/shrink intra-sentence pauses
|
values plus that emotion's override, so different emotions can have different
|
||||||
sentence_gap reply-global, sec (-0.5..1.5): insert/trim silence at sentence
|
speed/gaps/pitch within one reply.
|
||||||
boundaries (also used between emotion segments)
|
speed glyph speed -> generation-stage length_scale (1/speed)
|
||||||
pitch reply-global semitone offset, added on top of each segment's own
|
word_gap sec (-0.2..0.5): grow/shrink intra-sentence pauses
|
||||||
``pitch`` (from its emotion tag; 0 == no shift)
|
sentence_gap sec (-0.5..1.5): insert/trim silence at sentence boundaries
|
||||||
|
(within a segment, and before the next segment)
|
||||||
|
pitch semitone offset applied to the segment's wav (0 == no shift)
|
||||||
|
|
||||||
Each segment's ``pitch`` is a semitone offset applied to that segment's wav so
|
A reply is split into sentences, each sentence synthesised at its segment's
|
||||||
emotion tags can raise/lower the voice without changing the words. A reply is
|
speed, and the pieces concatenated with that segment's ``sentence_gap`` so one
|
||||||
split into sentences, each sentence synthesised at its segment's speed, and the
|
reply can carry several emotions each with its own rhythm and pitch.
|
||||||
pieces concatenated with ``sentence_gap`` so one reply can carry several emotions
|
|
||||||
and a controllable rhythm.
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import json
|
import json
|
||||||
@@ -188,23 +187,21 @@ def main() -> None:
|
|||||||
tts.tts_to_file(text, speaker_id, None, speed=speed), dtype=np.float32
|
tts.tts_to_file(text, speaker_id, None, speed=speed), dtype=np.float32
|
||||||
)
|
)
|
||||||
|
|
||||||
def _render(segments, out, word_gap, sentence_gap, pitch_offset):
|
def _render(segments, out):
|
||||||
"""Render one wav from emotion segments applying the 4 controls.
|
"""Render one wav from emotion segments. Each segment carries its own
|
||||||
|
four controls (speed, word_gap, sentence_gap, pitch) — resolved on the
|
||||||
Each segment carries its own glyph ``speed`` and ``pitch`` (from emotion
|
caller side from the base values plus that emotion's override. Sentences
|
||||||
tags); ``word_gap``/``sentence_gap``/``pitch_offset`` are reply-global.
|
within a segment join with that segment's ``sentence_gap``; between
|
||||||
Sentences within a segment, and the segments themselves, are joined with
|
segments the incoming segment's ``sentence_gap`` sets the pause."""
|
||||||
``sentence_gap`` (positive inserts silence, negative trims it)."""
|
|
||||||
word_gap = _clamp(word_gap, -0.2, 0.5)
|
|
||||||
sentence_gap = _clamp(sentence_gap, -0.5, 1.5)
|
|
||||||
pitch_offset = _clamp(pitch_offset, -12.0, 12.0)
|
|
||||||
pieces = []
|
pieces = []
|
||||||
for seg in segments:
|
for seg in segments:
|
||||||
text = seg.get("text", "")
|
text = seg.get("text", "")
|
||||||
if not text.strip():
|
if not text.strip():
|
||||||
continue
|
continue
|
||||||
speed = _clamp(seg.get("speed", 1.0), 0.5, 2.0)
|
speed = _clamp(seg.get("speed", 1.0), 0.5, 2.0)
|
||||||
pitch = _clamp(float(seg.get("pitch", 0.0)) + pitch_offset, -12.0, 12.0)
|
word_gap = _clamp(seg.get("word_gap", 0.0), -0.2, 0.5)
|
||||||
|
sentence_gap = _clamp(seg.get("sentence_gap", 0.0), -0.5, 1.5)
|
||||||
|
pitch = _clamp(seg.get("pitch", 0.0), -12.0, 12.0)
|
||||||
sent_pieces = []
|
sent_pieces = []
|
||||||
for sent in split_sentences(text):
|
for sent in split_sentences(text):
|
||||||
a = _synth_one(sent, speed)
|
a = _synth_one(sent, speed)
|
||||||
@@ -254,14 +251,17 @@ def main() -> None:
|
|||||||
if out.startswith("/tmp") or out.startswith("/dev/shm"):
|
if out.startswith("/tmp") or out.startswith("/dev/shm"):
|
||||||
raise ValueError(f"refusing RAM-backed tmpfs path: {out}")
|
raise ValueError(f"refusing RAM-backed tmpfs path: {out}")
|
||||||
s = time.monotonic()
|
s = time.monotonic()
|
||||||
word_gap = float(req.get("word_gap", 0.0))
|
|
||||||
sentence_gap = float(req.get("sentence_gap", 0.0))
|
|
||||||
pitch_offset = float(req.get("pitch", 0.0))
|
|
||||||
if "segments" in req:
|
if "segments" in req:
|
||||||
_render(req["segments"], out, word_gap, sentence_gap, pitch_offset)
|
_render(req["segments"], out)
|
||||||
else: # legacy single-utterance form
|
else: # legacy single-utterance form: one segment carrying all controls
|
||||||
seg = {"text": req["text"], "speed": float(req.get("speed", 1.0)), "pitch": 0.0}
|
seg = {
|
||||||
_render([seg], out, word_gap, sentence_gap, pitch_offset)
|
"text": req["text"],
|
||||||
|
"speed": float(req.get("speed", 1.0)),
|
||||||
|
"word_gap": float(req.get("word_gap", 0.0)),
|
||||||
|
"sentence_gap": float(req.get("sentence_gap", 0.0)),
|
||||||
|
"pitch": float(req.get("pitch", 0.0)),
|
||||||
|
}
|
||||||
|
_render([seg], out)
|
||||||
ms = int((time.monotonic() - s) * 1000)
|
ms = int((time.monotonic() - s) * 1000)
|
||||||
_emit({"ok": True, "out": out, "ms": ms})
|
_emit({"ok": True, "out": out, "ms": ms})
|
||||||
except Exception as exc: # keep the worker alive across bad requests
|
except Exception as exc: # keep the worker alive across bad requests
|
||||||
|
|||||||
@@ -460,7 +460,7 @@ class Dashboard:
|
|||||||
"pitch": (-12.0, 12.0, 1.0),
|
"pitch": (-12.0, 12.0, 1.0),
|
||||||
}
|
}
|
||||||
|
|
||||||
def _tts_values(self) -> dict:
|
def _tts_base(self) -> dict:
|
||||||
t = self.tts
|
t = self.tts
|
||||||
return {
|
return {
|
||||||
"speed": float(getattr(t, "speed", 1.35)),
|
"speed": float(getattr(t, "speed", 1.35)),
|
||||||
@@ -470,15 +470,37 @@ class Dashboard:
|
|||||||
}
|
}
|
||||||
|
|
||||||
def tts_settings(self) -> dict:
|
def tts_settings(self) -> dict:
|
||||||
return {"ok": True, "settings": self._tts_values(),
|
"""Base (공통) controls + the per-emotion overrides + the emotion list
|
||||||
"ranges": {k: list(v) for k, v in self._TTS_RANGES.items()}}
|
(canonical key + Korean label) so the dashboard can offer per-emotion
|
||||||
|
tuning. Emotions with no override inherit base (모든 감정 = 기본값)."""
|
||||||
|
from .backends.emotion import EMOTION_LABELS
|
||||||
|
overrides = dict(getattr(self.tts, "emotion_overrides", {}) or {})
|
||||||
|
return {
|
||||||
|
"ok": True,
|
||||||
|
"base": self._tts_base(),
|
||||||
|
"overrides": overrides,
|
||||||
|
"emotions": [{"key": k, "label": v} for k, v in EMOTION_LABELS.items()],
|
||||||
|
"ranges": {k: list(v) for k, v in self._TTS_RANGES.items()},
|
||||||
|
}
|
||||||
|
|
||||||
def set_tts_settings(self, data: dict) -> dict:
|
def set_tts_settings(self, data: dict) -> dict:
|
||||||
"""Clamp and apply provided controls to the live TTS instance."""
|
"""Apply controls. Without ``emotion`` (or emotion == "base"), set the
|
||||||
|
base/공통 values on the live TTS instance. With a specific emotion, store
|
||||||
|
(or, if ``reset``, clear) that emotion's override."""
|
||||||
|
emotion = data.get("emotion")
|
||||||
|
if not emotion or emotion == "base":
|
||||||
for key, (lo, hi, _step) in self._TTS_RANGES.items():
|
for key, (lo, hi, _step) in self._TTS_RANGES.items():
|
||||||
v = data.get(key)
|
v = data.get(key)
|
||||||
if v is not None:
|
if v is not None:
|
||||||
setattr(self.tts, key, float(max(lo, min(hi, float(v)))))
|
setattr(self.tts, key, float(max(lo, min(hi, float(v)))))
|
||||||
|
else:
|
||||||
|
store = getattr(self.tts, "emotion_overrides", None)
|
||||||
|
if store is None:
|
||||||
|
store = self.tts.emotion_overrides = {}
|
||||||
|
if data.get("reset"):
|
||||||
|
store.pop(emotion, None)
|
||||||
|
else:
|
||||||
|
store[emotion] = self._tts_overrides(data)
|
||||||
return self.tts_settings()
|
return self.tts_settings()
|
||||||
|
|
||||||
def _tts_overrides(self, data: dict) -> dict:
|
def _tts_overrides(self, data: dict) -> dict:
|
||||||
@@ -736,6 +758,7 @@ PAGE = r"""<!DOCTYPE html>
|
|||||||
.ttsgrid label b{color:var(--fg);font-weight:600}
|
.ttsgrid label b{color:var(--fg);font-weight:600}
|
||||||
.ttsgrid input[type=range]{width:100%;accent-color:#3aa0ff}
|
.ttsgrid input[type=range]{width:100%;accent-color:#3aa0ff}
|
||||||
.ttstext{flex:1;min-width:180px;background:#0f2333;border:1px solid #234a63;color:var(--fg);border-radius:10px;padding:8px 12px;font-size:13px}
|
.ttstext{flex:1;min-width:180px;background:#0f2333;border:1px solid #234a63;color:var(--fg);border-radius:10px;padding:8px 12px;font-size:13px}
|
||||||
|
.ttssel{background:#0f2333;border:1px solid #234a63;color:var(--fg);border-radius:10px;padding:7px 10px;font-size:13px;margin-left:8px}
|
||||||
.sttres .txt{color:var(--heard);font-weight:600;line-height:1.5}
|
.sttres .txt{color:var(--heard);font-weight:600;line-height:1.5}
|
||||||
.sttres .meta{color:var(--muted);font-size:12px;margin-top:5px}
|
.sttres .meta{color:var(--muted);font-size:12px;margin-top:5px}
|
||||||
.hbtn{padding:6px 12px;font-size:12.5px}
|
.hbtn{padding:6px 12px;font-size:12.5px}
|
||||||
@@ -869,6 +892,11 @@ PAGE = r"""<!DOCTYPE html>
|
|||||||
<section class="sttbox" id="ttsbox" style="display:none">
|
<section class="sttbox" id="ttsbox" style="display:none">
|
||||||
<h2 id="ttsToggle" class="collap"><span id="ttsCaret">▸</span> 🔊 봇 목소리(TTS) 조절 · MeloTTS</h2>
|
<h2 id="ttsToggle" class="collap"><span id="ttsCaret">▸</span> 🔊 봇 목소리(TTS) 조절 · MeloTTS</h2>
|
||||||
<div id="ttsBody" style="display:none">
|
<div id="ttsBody" style="display:none">
|
||||||
|
<div class="sttrow" style="margin-bottom:10px">
|
||||||
|
<label style="font-size:12.5px;color:var(--muted)">감정 선택
|
||||||
|
<select id="tEmotion" class="ttssel"></select></label>
|
||||||
|
<span id="tEmoNote" class="sttstat">감정을 고르면 그 감정만 따로 조절합니다. 기본(공통)은 태그 없는 말투에 적용됩니다.</span>
|
||||||
|
</div>
|
||||||
<div class="ttsgrid">
|
<div class="ttsgrid">
|
||||||
<label>글자 속도 <b id="tvSpeed">1.35x</b>
|
<label>글자 속도 <b id="tvSpeed">1.35x</b>
|
||||||
<input type="range" id="tSpeed" min="0.5" max="2.0" step="0.05" value="1.35"></label>
|
<input type="range" id="tSpeed" min="0.5" max="2.0" step="0.05" value="1.35"></label>
|
||||||
@@ -1171,8 +1199,10 @@ async function sendBlob(blob){
|
|||||||
};
|
};
|
||||||
})();
|
})();
|
||||||
|
|
||||||
// --- 봇 목소리(TTS) 조절: 슬라이더 → 미리듣기 → 봇 적용 -------------------- #
|
// --- 봇 목소리(TTS) 조절: 감정별 슬라이더 → 미리듣기 → 봇 적용 -------------- #
|
||||||
let ttsInited = false;
|
let ttsInited = false;
|
||||||
|
let ttsState = {base:{}, overrides:{}, emotions:[]}; // last-known server state
|
||||||
|
const TTS_DEFAULT = {speed:1.35, word_gap:-0.07, sentence_gap:-0.3, pitch:0};
|
||||||
function ttsLabels(){
|
function ttsLabels(){
|
||||||
const sp=parseFloat($('tSpeed').value), wg=parseFloat($('tWord').value),
|
const sp=parseFloat($('tSpeed').value), wg=parseFloat($('tWord').value),
|
||||||
sg=parseFloat($('tSent').value), pt=parseInt($('tPitch').value,10);
|
sg=parseFloat($('tSent').value), pt=parseInt($('tPitch').value,10);
|
||||||
@@ -1187,8 +1217,34 @@ function ttsVals(){
|
|||||||
}
|
}
|
||||||
function ttsSet(s){
|
function ttsSet(s){
|
||||||
if(!s) return;
|
if(!s) return;
|
||||||
$('tSpeed').value=s.speed; $('tWord').value=s.word_gap;
|
if(s.speed!=null) $('tSpeed').value=s.speed;
|
||||||
$('tSent').value=s.sentence_gap; $('tPitch').value=s.pitch; ttsLabels();
|
if(s.word_gap!=null) $('tWord').value=s.word_gap;
|
||||||
|
if(s.sentence_gap!=null) $('tSent').value=s.sentence_gap;
|
||||||
|
if(s.pitch!=null) $('tPitch').value=s.pitch;
|
||||||
|
ttsLabels();
|
||||||
|
}
|
||||||
|
// Effective values for the picked emotion: its override merged over base
|
||||||
|
// (base itself when 기본 or when the emotion has no override).
|
||||||
|
function ttsValuesFor(key){
|
||||||
|
const base = Object.assign({}, TTS_DEFAULT, ttsState.base);
|
||||||
|
if(key && key!=='base' && ttsState.overrides && ttsState.overrides[key])
|
||||||
|
return Object.assign({}, base, ttsState.overrides[key]);
|
||||||
|
return base;
|
||||||
|
}
|
||||||
|
function ttsApplyState(j){
|
||||||
|
if(!j || !j.ok) return;
|
||||||
|
ttsState = {base:j.base||{}, overrides:j.overrides||{}, emotions:j.emotions||ttsState.emotions};
|
||||||
|
// Refresh dropdown labels to mark which emotions are customised.
|
||||||
|
const sel=$('tEmotion'); const cur=sel.value||'base';
|
||||||
|
sel.innerHTML='';
|
||||||
|
ttsState.emotions.forEach(e=>{
|
||||||
|
const o=document.createElement('option'); o.value=e.key;
|
||||||
|
const custom = e.key!=='base' && ttsState.overrides[e.key];
|
||||||
|
o.textContent = e.label + (custom ? ' ●' : '');
|
||||||
|
sel.appendChild(o);
|
||||||
|
});
|
||||||
|
sel.value = cur;
|
||||||
|
ttsSet(ttsValuesFor(sel.value));
|
||||||
}
|
}
|
||||||
async function initTts(){
|
async function initTts(){
|
||||||
if(ttsInited) return; ttsInited = true;
|
if(ttsInited) return; ttsInited = true;
|
||||||
@@ -1198,6 +1254,9 @@ async function initTts(){
|
|||||||
$('ttsCaret').textContent = open ? '▾' : '▸';
|
$('ttsCaret').textContent = open ? '▾' : '▸';
|
||||||
};
|
};
|
||||||
['tSpeed','tWord','tSent','tPitch'].forEach(id => $(id).addEventListener('input', ttsLabels));
|
['tSpeed','tWord','tSent','tPitch'].forEach(id => $(id).addEventListener('input', ttsLabels));
|
||||||
|
$('tEmotion').onchange = ()=>{ ttsSet(ttsValuesFor($('tEmotion').value));
|
||||||
|
$('tStat').textContent = $('tEmotion').value==='base'
|
||||||
|
? '기본(공통) 값을 조절 중입니다.' : '"'+$('tEmotion').selectedOptions[0].textContent.replace(' ●','')+'" 감정만 조절 중입니다.'; };
|
||||||
$('tPreview').onclick = async ()=>{
|
$('tPreview').onclick = async ()=>{
|
||||||
const text=($('tText').value||'').trim();
|
const text=($('tText').value||'').trim();
|
||||||
if(!text){ $('tStat').textContent='미리들을 문장을 입력하세요.'; return; }
|
if(!text){ $('tStat').textContent='미리들을 문장을 입력하세요.'; return; }
|
||||||
@@ -1212,20 +1271,27 @@ async function initTts(){
|
|||||||
}catch(e){ $('tStat').textContent='미리듣기 오류: '+e; }
|
}catch(e){ $('tStat').textContent='미리듣기 오류: '+e; }
|
||||||
};
|
};
|
||||||
$('tApply').onclick = async ()=>{
|
$('tApply').onclick = async ()=>{
|
||||||
|
const key=$('tEmotion').value||'base';
|
||||||
try{
|
try{
|
||||||
const r=await fetch('/api/tts/settings',{method:'POST',headers:{'Content-Type':'application/json'},
|
const r=await fetch('/api/tts/settings',{method:'POST',headers:{'Content-Type':'application/json'},
|
||||||
body:JSON.stringify(ttsVals())});
|
body:JSON.stringify(Object.assign({emotion:key}, ttsVals()))});
|
||||||
const j=await r.json();
|
const j=await r.json();
|
||||||
if(j.ok){ ttsSet(j.settings); toast('봇 목소리에 적용됨 · 다음 답변부터 반영'); }
|
if(j.ok){ ttsApplyState(j); toast((key==='base'?'기본(공통)':'해당 감정')+' 적용됨 · 다음 답변부터 반영'); }
|
||||||
else { $('tStat').textContent='적용 실패: '+(j.error||''); }
|
else { $('tStat').textContent='적용 실패: '+(j.error||''); }
|
||||||
}catch(e){ $('tStat').textContent='적용 오류: '+e; }
|
}catch(e){ $('tStat').textContent='적용 오류: '+e; }
|
||||||
};
|
};
|
||||||
$('tReset').onclick = ()=>{ ttsSet({speed:1.35,word_gap:-0.07,sentence_gap:-0.3,pitch:0});
|
$('tReset').onclick = async ()=>{
|
||||||
$('tStat').textContent='기본값으로 되돌림 (미리듣기/적용을 눌러 반영).'; };
|
const key=$('tEmotion').value||'base';
|
||||||
try{
|
if(key==='base'){ ttsSet(TTS_DEFAULT);
|
||||||
const j=await (await fetch('/api/tts/settings')).json();
|
$('tStat').textContent='기본값으로 세팅됨 (적용을 눌러 반영).'; return; }
|
||||||
if(j.ok) ttsSet(j.settings);
|
try{ // clear this emotion's override -> it inherits 기본(공통) again
|
||||||
}catch(e){}
|
const r=await fetch('/api/tts/settings',{method:'POST',headers:{'Content-Type':'application/json'},
|
||||||
|
body:JSON.stringify({emotion:key, reset:true})});
|
||||||
|
const j=await r.json();
|
||||||
|
if(j.ok){ ttsApplyState(j); toast('이 감정을 기본(공통)값으로 되돌림'); }
|
||||||
|
}catch(e){ $('tStat').textContent='되돌리기 오류: '+e; }
|
||||||
|
};
|
||||||
|
try{ ttsApplyState(await (await fetch('/api/tts/settings')).json()); }catch(e){}
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- Reusable popup/modal (프롬프트, 화이트/블랙리스트 등이 공유) ---------- #
|
// --- Reusable popup/modal (프롬프트, 화이트/블랙리스트 등이 공유) ---------- #
|
||||||
|
|||||||
Reference in New Issue
Block a user