feat(tts): per-emotion voice controls (speed/word-gap/sentence-gap/pitch)

Each emotion can now be tuned independently. parse_segments resolves the four
controls per segment from a base (공통) dict plus an optional per-emotion
override; a missing override key inherits base. By default there are no
overrides, so every emotion delivers with the base values (모든 감정 = 기본값).

- emotion.py: Segment now carries all 4 controls + the canonical emotion name;
  parse_segments(text, base, overrides). Adds EMOTION_LABELS/EMOTIONS for the UI.
- melo.py: MeloTTS.emotion_overrides store; synth resolves per-segment controls
  and sends them per segment.
- melo_worker.py: _render applies each segment's own word_gap/sentence_gap/pitch
  (previously reply-global).
- dashboard.py: emotion dropdown in the TTS panel; GET returns base + overrides
  + emotion list; POST {emotion,...} stores an override (or {reset:true} clears
  it); base is set when emotion is omitted/"base".

Verified: an override on one emotion slows only that emotion (happy@0.7=5.66s vs
base 3.02s; sad unchanged at 3.06s); dashboard store/reset/base all work; 34
tests pass.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
EJClaw
2026-08-26 23:10:33 +09:00
parent 39b743d976
commit 067efc7abe
5 changed files with 242 additions and 99 deletions

View File

@@ -1,56 +1,82 @@
"""Emotion-tag parsing for expressive TTS. A ``[감정]`` tag must steer the
pitch/speed of the text that follows without being spoken; a bracket that is NOT
a known emotion word must be kept as ordinary spoken content."""
delivery of the text that follows without being spoken; a bracket that is NOT a
known emotion word must be kept as ordinary spoken content.
Delivery is now driven by a base (공통) control dict plus optional per-emotion
overrides. By default there are no overrides, so every emotion delivers with the
base controls (모든 감정 = 기본값)."""
from wsai.backends.emotion import (
EMOTION_PARAMS,
EMOTION_LABELS,
_SYNONYMS,
match_emotion,
parse_segments,
)
from wsai.dashboard import _speech_text
BASE = 1.3
BASE = {"speed": 1.35, "word_gap": -0.07, "sentence_gap": -0.30, "pitch": 0.0}
def test_emotion_tag_is_not_spoken_and_sets_delivery():
def test_emotion_tag_is_not_spoken_and_tags_segment():
segs = parse_segments("[기쁨] 오늘 날씨 좋다", BASE)
assert len(segs) == 1
assert "기쁨" not in segs[0].text # the tag word is dropped
assert segs[0].text == "오늘 날씨 좋다"
mult, semis = EMOTION_PARAMS["happy"]
assert segs[0].speed == BASE * mult
assert segs[0].pitch == semis
assert segs[0].emotion == "happy"
def test_default_all_emotions_use_base():
# No overrides -> a tagged emotion delivers with exactly the base controls.
segs = parse_segments("[기쁨] 좋아", BASE)
s = segs[0]
assert (s.speed, s.word_gap, s.sentence_gap, s.pitch) == (1.35, -0.07, -0.30, 0.0)
def test_per_emotion_override_applies_and_inherits_missing_keys():
segs = parse_segments("[기쁨] 좋아", BASE, {"happy": {"speed": 1.6, "pitch": 3.0}})
s = segs[0]
assert s.speed == 1.6 and s.pitch == 3.0 # overridden
assert s.word_gap == -0.07 and s.sentence_gap == -0.30 # inherited from base
def test_override_only_affects_its_own_emotion():
segs = parse_segments("[기쁨] 가. [슬픔] 나.", BASE, {"sad": {"speed": 0.8}})
assert segs[0].emotion == "happy" and segs[0].speed == 1.35 # untouched
assert segs[1].emotion == "sad" and segs[1].speed == 0.8 # overridden
def test_midreply_emotion_change_splits_segments():
segs = parse_segments("[속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!", BASE)
assert len(segs) == 2
assert segs[0].text == "정말 힘들었겠다."
assert segs[1].text == "하지만 넌 할 수 있어!"
assert segs[0].pitch == EMOTION_PARAMS["sad"][1]
assert segs[1].pitch == EMOTION_PARAMS["hopeful"][1]
assert segs[0].text == "정말 힘들었겠다." and segs[0].emotion == "sad"
assert segs[1].text == "하지만 넌 할 수 있어!" and segs[1].emotion == "hopeful"
def test_non_emotion_bracket_is_spoken_without_brackets():
segs = parse_segments("[기쁨] 첫째는 [1번] 항목이야", BASE)
assert len(segs) == 1
# "1번" is not an emotion -> read it; brackets themselves are gone.
assert "1번" in segs[0].text
assert "[" not in segs[0].text and "]" not in segs[0].text
assert segs[0].pitch == EMOTION_PARAMS["happy"][1]
assert segs[0].emotion == "happy"
def test_text_before_first_tag_is_neutral():
segs = parse_segments("잠깐만. [신남] 찾았다!", BASE)
assert segs[0].text == "잠깐만."
assert segs[0].speed == BASE and segs[0].pitch == 0.0
assert segs[1].pitch == EMOTION_PARAMS["excited"][1]
assert segs[0].text == "잠깐만." and segs[0].emotion is None
assert segs[0].speed == 1.35 and segs[0].pitch == 0.0
assert segs[1].emotion == "excited"
def test_plain_text_is_one_neutral_segment():
segs = parse_segments("그냥 평범한 문장이야", BASE)
assert len(segs) == 1
assert segs[0].speed == BASE and segs[0].pitch == 0.0
assert segs[0].emotion is None and segs[0].speed == 1.35
def test_backward_compat_float_base():
# A bare float base still works (speed only; other controls neutral).
segs = parse_segments("그냥 문장", 1.2)
assert segs[0].speed == 1.2 and segs[0].word_gap == 0.0
def test_empty_input_yields_no_segments():
@@ -77,6 +103,13 @@ def test_persona_examples_are_all_recognised():
assert match_emotion(word) is not None, word
def test_every_canonical_emotion_has_a_ui_label():
# The dashboard dropdown lists EMOTION_LABELS; every emotion a tag can
# resolve to must appear there or it would be untunable.
for canon in set(_SYNONYMS.values()):
assert canon in EMOTION_LABELS, canon
def test_voice_turn_keeps_leading_emotion_tag_for_tts_parser():
text = "[속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!"
@@ -84,7 +117,5 @@ def test_voice_turn_keeps_leading_emotion_tag_for_tts_parser():
segs = parse_segments(spoken, BASE)
assert spoken == text
assert segs[0].text == "정말 힘들었겠다."
assert segs[0].pitch == EMOTION_PARAMS["sad"][1]
assert segs[1].text == "하지만 넌 할 수 있어!"
assert segs[1].pitch == EMOTION_PARAMS["hopeful"][1]
assert segs[0].text == "정말 힘들었겠다." and segs[0].emotion == "sad"
assert segs[1].text == "하지만 넌 할 수 있어!" and segs[1].emotion == "hopeful"