From 067efc7abedc89d230cb60f873b3976fb8d480e0 Mon Sep 17 00:00:00 2001 From: EJClaw Date: Wed, 26 Aug 2026 23:10:33 +0900 Subject: [PATCH] feat(tts): per-emotion voice controls (speed/word-gap/sentence-gap/pitch) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Each emotion can now be tuned independently. parse_segments resolves the four controls per segment from a base (공통) dict plus an optional per-emotion override; a missing override key inherits base. By default there are no overrides, so every emotion delivers with the base values (모든 감정 = 기본값). - emotion.py: Segment now carries all 4 controls + the canonical emotion name; parse_segments(text, base, overrides). Adds EMOTION_LABELS/EMOTIONS for the UI. - melo.py: MeloTTS.emotion_overrides store; synth resolves per-segment controls and sends them per segment. - melo_worker.py: _render applies each segment's own word_gap/sentence_gap/pitch (previously reply-global). - dashboard.py: emotion dropdown in the TTS panel; GET returns base + overrides + emotion list; POST {emotion,...} stores an override (or {reset:true} clears it); base is set when emotion is omitted/"base". Verified: an override on one emotion slows only that emotion (happy@0.7=5.66s vs base 3.02s; sad unchanged at 3.06s); dashboard store/reset/base all work; 34 tests pass. Co-Authored-By: Claude Opus 4.7 --- tests/test_emotion.py | 75 +++++++++++++++++-------- wsai/backends/emotion.py | 61 ++++++++++++++++---- wsai/backends/melo.py | 33 ++++++----- wsai/backends/melo_worker.py | 68 +++++++++++------------ wsai/dashboard.py | 104 ++++++++++++++++++++++++++++------- 5 files changed, 242 insertions(+), 99 deletions(-) diff --git a/tests/test_emotion.py b/tests/test_emotion.py index 083c5c4..534b1e0 100644 --- a/tests/test_emotion.py +++ b/tests/test_emotion.py @@ -1,56 +1,82 @@ """Emotion-tag parsing for expressive TTS. A ``[감정]`` tag must steer the -pitch/speed of the text that follows without being spoken; a bracket that is NOT -a known emotion word must be kept as ordinary spoken content.""" +delivery of the text that follows without being spoken; a bracket that is NOT a +known emotion word must be kept as ordinary spoken content. + +Delivery is now driven by a base (공통) control dict plus optional per-emotion +overrides. By default there are no overrides, so every emotion delivers with the +base controls (모든 감정 = 기본값).""" from wsai.backends.emotion import ( - EMOTION_PARAMS, + EMOTION_LABELS, + _SYNONYMS, match_emotion, parse_segments, ) from wsai.dashboard import _speech_text -BASE = 1.3 +BASE = {"speed": 1.35, "word_gap": -0.07, "sentence_gap": -0.30, "pitch": 0.0} -def test_emotion_tag_is_not_spoken_and_sets_delivery(): +def test_emotion_tag_is_not_spoken_and_tags_segment(): segs = parse_segments("[기쁨] 오늘 날씨 좋다", BASE) assert len(segs) == 1 assert "기쁨" not in segs[0].text # the tag word is dropped assert segs[0].text == "오늘 날씨 좋다" - mult, semis = EMOTION_PARAMS["happy"] - assert segs[0].speed == BASE * mult - assert segs[0].pitch == semis + assert segs[0].emotion == "happy" + + +def test_default_all_emotions_use_base(): + # No overrides -> a tagged emotion delivers with exactly the base controls. + segs = parse_segments("[기쁨] 좋아", BASE) + s = segs[0] + assert (s.speed, s.word_gap, s.sentence_gap, s.pitch) == (1.35, -0.07, -0.30, 0.0) + + +def test_per_emotion_override_applies_and_inherits_missing_keys(): + segs = parse_segments("[기쁨] 좋아", BASE, {"happy": {"speed": 1.6, "pitch": 3.0}}) + s = segs[0] + assert s.speed == 1.6 and s.pitch == 3.0 # overridden + assert s.word_gap == -0.07 and s.sentence_gap == -0.30 # inherited from base + + +def test_override_only_affects_its_own_emotion(): + segs = parse_segments("[기쁨] 가. [슬픔] 나.", BASE, {"sad": {"speed": 0.8}}) + assert segs[0].emotion == "happy" and segs[0].speed == 1.35 # untouched + assert segs[1].emotion == "sad" and segs[1].speed == 0.8 # overridden def test_midreply_emotion_change_splits_segments(): segs = parse_segments("[속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!", BASE) assert len(segs) == 2 - assert segs[0].text == "정말 힘들었겠다." - assert segs[1].text == "하지만 넌 할 수 있어!" - assert segs[0].pitch == EMOTION_PARAMS["sad"][1] - assert segs[1].pitch == EMOTION_PARAMS["hopeful"][1] + assert segs[0].text == "정말 힘들었겠다." and segs[0].emotion == "sad" + assert segs[1].text == "하지만 넌 할 수 있어!" and segs[1].emotion == "hopeful" def test_non_emotion_bracket_is_spoken_without_brackets(): segs = parse_segments("[기쁨] 첫째는 [1번] 항목이야", BASE) assert len(segs) == 1 - # "1번" is not an emotion -> read it; brackets themselves are gone. assert "1번" in segs[0].text assert "[" not in segs[0].text and "]" not in segs[0].text - assert segs[0].pitch == EMOTION_PARAMS["happy"][1] + assert segs[0].emotion == "happy" def test_text_before_first_tag_is_neutral(): segs = parse_segments("잠깐만. [신남] 찾았다!", BASE) - assert segs[0].text == "잠깐만." - assert segs[0].speed == BASE and segs[0].pitch == 0.0 - assert segs[1].pitch == EMOTION_PARAMS["excited"][1] + assert segs[0].text == "잠깐만." and segs[0].emotion is None + assert segs[0].speed == 1.35 and segs[0].pitch == 0.0 + assert segs[1].emotion == "excited" def test_plain_text_is_one_neutral_segment(): segs = parse_segments("그냥 평범한 문장이야", BASE) assert len(segs) == 1 - assert segs[0].speed == BASE and segs[0].pitch == 0.0 + assert segs[0].emotion is None and segs[0].speed == 1.35 + + +def test_backward_compat_float_base(): + # A bare float base still works (speed only; other controls neutral). + segs = parse_segments("그냥 문장", 1.2) + assert segs[0].speed == 1.2 and segs[0].word_gap == 0.0 def test_empty_input_yields_no_segments(): @@ -77,6 +103,13 @@ def test_persona_examples_are_all_recognised(): assert match_emotion(word) is not None, word +def test_every_canonical_emotion_has_a_ui_label(): + # The dashboard dropdown lists EMOTION_LABELS; every emotion a tag can + # resolve to must appear there or it would be untunable. + for canon in set(_SYNONYMS.values()): + assert canon in EMOTION_LABELS, canon + + def test_voice_turn_keeps_leading_emotion_tag_for_tts_parser(): text = "[속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!" @@ -84,7 +117,5 @@ def test_voice_turn_keeps_leading_emotion_tag_for_tts_parser(): segs = parse_segments(spoken, BASE) assert spoken == text - assert segs[0].text == "정말 힘들었겠다." - assert segs[0].pitch == EMOTION_PARAMS["sad"][1] - assert segs[1].text == "하지만 넌 할 수 있어!" - assert segs[1].pitch == EMOTION_PARAMS["hopeful"][1] + assert segs[0].text == "정말 힘들었겠다." and segs[0].emotion == "sad" + assert segs[1].text == "하지만 넌 할 수 있어!" and segs[1].emotion == "hopeful" diff --git a/wsai/backends/emotion.py b/wsai/backends/emotion.py index 8ed175b..c0beb67 100644 --- a/wsai/backends/emotion.py +++ b/wsai/backends/emotion.py @@ -99,33 +99,75 @@ def match_emotion(inner: str) -> str | None: return _SYNONYMS.get(_norm(inner)) +# The four adjustable TTS controls, per segment. +PARAM_KEYS = ("speed", "word_gap", "sentence_gap", "pitch") + +# Canonical emotion -> Korean UI label, in display order. "base" == 기본/공통 +# (the value used for untagged text and inherited by any emotion with no +# override). Every canonical emotion the synonym table resolves to must appear +# here so the dashboard dropdown can list it. +EMOTION_LABELS: dict[str, str] = { + "base": "기본(공통)", "happy": "기쁨", "excited": "신남", "hopeful": "희망", + "sad": "슬픔", "angry": "화남", "fearful": "두려움", "surprised": "놀람", + "disgust": "혐오", "calm": "차분", "friendly": "다정", "serious": "진지", + "disappointed": "실망", "tired": "피곤", "affectionate": "사랑", "playful": "장난", + "curious": "궁금", "whisper": "속삭임", "shout": "외침", "determined": "단호", + "relieved": "안도", +} +EMOTIONS: list[str] = list(EMOTION_LABELS) + + @dataclass class Segment: text: str speed: float + word_gap: float + sentence_gap: float pitch: float # semitones; 0.0 == no shift + emotion: str | None = None # canonical emotion, or None for neutral/base _TAG_RE = re.compile(r"\[([^\[\]]*)\]") -def parse_segments(text: str, base_speed: float) -> list[Segment]: - """Split ``text`` into consecutive spoken segments, each carrying the speed - and pitch implied by the most recent emotion tag. +def _resolve(base: dict, overrides: dict | None, emotion: str | None) -> dict: + """The 4 controls for one segment: the base values, with any per-emotion + override applied on top (a missing override key inherits base).""" + p = {k: float(base[k]) for k in PARAM_KEYS} + if emotion and overrides: + ov = overrides.get(emotion) + if ov: + for k in PARAM_KEYS: + if ov.get(k) is not None: + p[k] = float(ov[k]) + return p + +def parse_segments(text: str, base, overrides: dict | None = None) -> list[Segment]: + """Split ``text`` into consecutive spoken segments, each carrying the four + TTS controls implied by the most recent emotion tag. + + * ``base`` is the 공통 control dict ``{speed, word_gap, sentence_gap, pitch}`` + used for untagged text and inherited by any emotion without an override. + A bare float is also accepted (speed only) for backward compatibility. + * ``overrides`` maps a canonical emotion -> a partial control dict; only the + keys present override the base. Empty/None means every emotion delivers + with the base controls (the default: 모든 감정 = 기본값). * An emotion tag switches the active emotion for everything after it and is - not spoken. - * A non-emotion bracket keeps its inner words as spoken text (brackets gone). - * Text before any tag is spoken with neutral delivery (base speed, no shift). + not spoken. A non-emotion bracket keeps its inner words as spoken text. """ + if not isinstance(base, dict): + base = {"speed": float(base), "word_gap": 0.0, "sentence_gap": 0.0, "pitch": 0.0} segments: list[Segment] = [] - cur_speed, cur_pitch = base_speed, 0.0 + cur_emotion: str | None = None buf: list[str] = [] def flush() -> None: joined = "".join(buf).strip() if joined: - segments.append(Segment(joined, cur_speed, cur_pitch)) + p = _resolve(base, overrides, cur_emotion) + segments.append(Segment(joined, p["speed"], p["word_gap"], + p["sentence_gap"], p["pitch"], cur_emotion)) buf.clear() pos = 0 @@ -140,8 +182,7 @@ def parse_segments(text: str, base_speed: float) -> list[Segment]: # Emotion tag — everything so far belongs to the previous emotion; # flush it, then switch delivery for what follows. flush() - mult, semis = EMOTION_PARAMS[emotion] - cur_speed, cur_pitch = base_speed * mult, semis + cur_emotion = emotion buf.append(text[pos:]) flush() return segments # empty when there is nothing speakable (blank or all-tags) diff --git a/wsai/backends/melo.py b/wsai/backends/melo.py index 903873e..dadfdaa 100644 --- a/wsai/backends/melo.py +++ b/wsai/backends/melo.py @@ -107,6 +107,10 @@ class MeloTTS: else os.environ.get("WSAI_TTS_SENTENCE_GAP", "-0.30")) self.pitch = float(pitch if pitch is not None else os.environ.get("WSAI_TTS_PITCH", "0.0")) + # Per-emotion control overrides: canonical emotion -> partial dict of + # {speed, word_gap, sentence_gap, pitch}. Empty by default, so every + # emotion inherits the base controls above (모든 감정 = 기본값). + self.emotion_overrides: dict[str, dict] = {} self.sink = sink or _log_sink self._proc: asyncio.subprocess.Process | None = None self._lock = asyncio.Lock() @@ -214,35 +218,36 @@ class MeloTTS: per call (used by the dashboard preview so tuning does not disturb the live bot voice until explicitly applied).""" await self._ensure() - spd = self.speed if speed is None else float(speed) - wg = self.word_gap if word_gap is None else float(word_gap) - sg = self.sentence_gap if sentence_gap is None else float(sentence_gap) - pt = self.pitch if pitch is None else float(pitch) + base = { + "speed": self.speed if speed is None else float(speed), + "word_gap": self.word_gap if word_gap is None else float(word_gap), + "sentence_gap": self.sentence_gap if sentence_gap is None else float(sentence_gap), + "pitch": self.pitch if pitch is None else float(pitch), + } text = normalize_for_speech(text) - # Split on [감정] tags: each tag steers pitch/speed for the text that + # Split on [감정] tags: each tag switches delivery for the text that # follows (and is itself not spoken); non-emotion brackets stay as words. - segments = parse_segments(text, spd) + # Each segment carries its own 4 controls (base + per-emotion override). + segments = parse_segments(text, base, self.emotion_overrides) self._n += 1 out = str(self.out_dir / f"tts-{self._n:06d}.wav") if segments: payload = { "segments": [ - {"text": s.text, "speed": s.speed, "pitch": s.pitch} + {"text": s.text, "speed": s.speed, "word_gap": s.word_gap, + "sentence_gap": s.sentence_gap, "pitch": s.pitch} for s in segments ], "out": out, - "word_gap": wg, - "sentence_gap": sg, - "pitch": pt, } else: # empty/whitespace reply: keep legacy single-utterance behaviour payload = { "text": text, "out": out, - "speed": spd, - "word_gap": wg, - "sentence_gap": sg, - "pitch": pt, + "speed": base["speed"], + "word_gap": base["word_gap"], + "sentence_gap": base["sentence_gap"], + "pitch": base["pitch"], } req = json.dumps(payload) s = time.monotonic() diff --git a/wsai/backends/melo_worker.py b/wsai/backends/melo_worker.py index 158b69c..94fba7f 100644 --- a/wsai/backends/melo_worker.py +++ b/wsai/backends/melo_worker.py @@ -11,28 +11,27 @@ library chatter lands on stderr instead. Protocol (one JSON object per line, on the protocol channel): <- {"text": "...", "out": "/abs/path.wav", "speed": 1.3, - "word_gap": 0.25, "sentence_gap": 0.75, "pitch": 0.0} - <- {"segments": [{"text": "...", "speed": 1.3, "pitch": 2.0}, ...], - "out": "/abs/path.wav", - "word_gap": 0.25, "sentence_gap": 0.75, "pitch": 0.0} + "word_gap": -0.07, "sentence_gap": -0.30, "pitch": 0.0} + <- {"segments": [{"text": "...", "speed": 1.3, "word_gap": -0.07, + "sentence_gap": -0.30, "pitch": 2.0}, ...], "out": "/abs/path.wav"} -> {"ok": true, "out": "/abs/path.wav", "ms": 123} -> {"ok": false, "error": "..."} On startup, once the model is ready, it emits exactly one line: -> {"ready": true, "ms": , "device": "cpu"} -Four independent voice controls (ported from tts_site, see docs manual): - speed per-segment glyph speed -> generation-stage length_scale (1/speed) - word_gap reply-global, sec (-0.2..0.5): grow/shrink intra-sentence pauses - sentence_gap reply-global, sec (-0.5..1.5): insert/trim silence at sentence - boundaries (also used between emotion segments) - pitch reply-global semitone offset, added on top of each segment's own - ``pitch`` (from its emotion tag; 0 == no shift) +Four independent voice controls (ported from tts_site, see docs manual). They +are PER SEGMENT: the caller resolves each segment's controls from the base +values plus that emotion's override, so different emotions can have different +speed/gaps/pitch within one reply. + speed glyph speed -> generation-stage length_scale (1/speed) + word_gap sec (-0.2..0.5): grow/shrink intra-sentence pauses + sentence_gap sec (-0.5..1.5): insert/trim silence at sentence boundaries + (within a segment, and before the next segment) + pitch semitone offset applied to the segment's wav (0 == no shift) -Each segment's ``pitch`` is a semitone offset applied to that segment's wav so -emotion tags can raise/lower the voice without changing the words. A reply is -split into sentences, each sentence synthesised at its segment's speed, and the -pieces concatenated with ``sentence_gap`` so one reply can carry several emotions -and a controllable rhythm. +A reply is split into sentences, each sentence synthesised at its segment's +speed, and the pieces concatenated with that segment's ``sentence_gap`` so one +reply can carry several emotions each with its own rhythm and pitch. """ import json @@ -188,23 +187,21 @@ def main() -> None: tts.tts_to_file(text, speaker_id, None, speed=speed), dtype=np.float32 ) - def _render(segments, out, word_gap, sentence_gap, pitch_offset): - """Render one wav from emotion segments applying the 4 controls. - - Each segment carries its own glyph ``speed`` and ``pitch`` (from emotion - tags); ``word_gap``/``sentence_gap``/``pitch_offset`` are reply-global. - Sentences within a segment, and the segments themselves, are joined with - ``sentence_gap`` (positive inserts silence, negative trims it).""" - word_gap = _clamp(word_gap, -0.2, 0.5) - sentence_gap = _clamp(sentence_gap, -0.5, 1.5) - pitch_offset = _clamp(pitch_offset, -12.0, 12.0) + def _render(segments, out): + """Render one wav from emotion segments. Each segment carries its own + four controls (speed, word_gap, sentence_gap, pitch) — resolved on the + caller side from the base values plus that emotion's override. Sentences + within a segment join with that segment's ``sentence_gap``; between + segments the incoming segment's ``sentence_gap`` sets the pause.""" pieces = [] for seg in segments: text = seg.get("text", "") if not text.strip(): continue speed = _clamp(seg.get("speed", 1.0), 0.5, 2.0) - pitch = _clamp(float(seg.get("pitch", 0.0)) + pitch_offset, -12.0, 12.0) + word_gap = _clamp(seg.get("word_gap", 0.0), -0.2, 0.5) + sentence_gap = _clamp(seg.get("sentence_gap", 0.0), -0.5, 1.5) + pitch = _clamp(seg.get("pitch", 0.0), -12.0, 12.0) sent_pieces = [] for sent in split_sentences(text): a = _synth_one(sent, speed) @@ -254,14 +251,17 @@ def main() -> None: if out.startswith("/tmp") or out.startswith("/dev/shm"): raise ValueError(f"refusing RAM-backed tmpfs path: {out}") s = time.monotonic() - word_gap = float(req.get("word_gap", 0.0)) - sentence_gap = float(req.get("sentence_gap", 0.0)) - pitch_offset = float(req.get("pitch", 0.0)) if "segments" in req: - _render(req["segments"], out, word_gap, sentence_gap, pitch_offset) - else: # legacy single-utterance form - seg = {"text": req["text"], "speed": float(req.get("speed", 1.0)), "pitch": 0.0} - _render([seg], out, word_gap, sentence_gap, pitch_offset) + _render(req["segments"], out) + else: # legacy single-utterance form: one segment carrying all controls + seg = { + "text": req["text"], + "speed": float(req.get("speed", 1.0)), + "word_gap": float(req.get("word_gap", 0.0)), + "sentence_gap": float(req.get("sentence_gap", 0.0)), + "pitch": float(req.get("pitch", 0.0)), + } + _render([seg], out) ms = int((time.monotonic() - s) * 1000) _emit({"ok": True, "out": out, "ms": ms}) except Exception as exc: # keep the worker alive across bad requests diff --git a/wsai/dashboard.py b/wsai/dashboard.py index b300f68..208a07e 100644 --- a/wsai/dashboard.py +++ b/wsai/dashboard.py @@ -460,7 +460,7 @@ class Dashboard: "pitch": (-12.0, 12.0, 1.0), } - def _tts_values(self) -> dict: + def _tts_base(self) -> dict: t = self.tts return { "speed": float(getattr(t, "speed", 1.35)), @@ -470,15 +470,37 @@ class Dashboard: } def tts_settings(self) -> dict: - return {"ok": True, "settings": self._tts_values(), - "ranges": {k: list(v) for k, v in self._TTS_RANGES.items()}} + """Base (공통) controls + the per-emotion overrides + the emotion list + (canonical key + Korean label) so the dashboard can offer per-emotion + tuning. Emotions with no override inherit base (모든 감정 = 기본값).""" + from .backends.emotion import EMOTION_LABELS + overrides = dict(getattr(self.tts, "emotion_overrides", {}) or {}) + return { + "ok": True, + "base": self._tts_base(), + "overrides": overrides, + "emotions": [{"key": k, "label": v} for k, v in EMOTION_LABELS.items()], + "ranges": {k: list(v) for k, v in self._TTS_RANGES.items()}, + } def set_tts_settings(self, data: dict) -> dict: - """Clamp and apply provided controls to the live TTS instance.""" - for key, (lo, hi, _step) in self._TTS_RANGES.items(): - v = data.get(key) - if v is not None: - setattr(self.tts, key, float(max(lo, min(hi, float(v))))) + """Apply controls. Without ``emotion`` (or emotion == "base"), set the + base/공통 values on the live TTS instance. With a specific emotion, store + (or, if ``reset``, clear) that emotion's override.""" + emotion = data.get("emotion") + if not emotion or emotion == "base": + for key, (lo, hi, _step) in self._TTS_RANGES.items(): + v = data.get(key) + if v is not None: + setattr(self.tts, key, float(max(lo, min(hi, float(v))))) + else: + store = getattr(self.tts, "emotion_overrides", None) + if store is None: + store = self.tts.emotion_overrides = {} + if data.get("reset"): + store.pop(emotion, None) + else: + store[emotion] = self._tts_overrides(data) return self.tts_settings() def _tts_overrides(self, data: dict) -> dict: @@ -736,6 +758,7 @@ PAGE = r""" .ttsgrid label b{color:var(--fg);font-weight:600} .ttsgrid input[type=range]{width:100%;accent-color:#3aa0ff} .ttstext{flex:1;min-width:180px;background:#0f2333;border:1px solid #234a63;color:var(--fg);border-radius:10px;padding:8px 12px;font-size:13px} + .ttssel{background:#0f2333;border:1px solid #234a63;color:var(--fg);border-radius:10px;padding:7px 10px;font-size:13px;margin-left:8px} .sttres .txt{color:var(--heard);font-weight:600;line-height:1.5} .sttres .meta{color:var(--muted);font-size:12px;margin-top:5px} .hbtn{padding:6px 12px;font-size:12.5px} @@ -869,6 +892,11 @@ PAGE = r"""