feat(tts): per-emotion voice controls (speed/word-gap/sentence-gap/pitch)
Each emotion can now be tuned independently. parse_segments resolves the four
controls per segment from a base (공통) dict plus an optional per-emotion
override; a missing override key inherits base. By default there are no
overrides, so every emotion delivers with the base values (모든 감정 = 기본값).
- emotion.py: Segment now carries all 4 controls + the canonical emotion name;
parse_segments(text, base, overrides). Adds EMOTION_LABELS/EMOTIONS for the UI.
- melo.py: MeloTTS.emotion_overrides store; synth resolves per-segment controls
and sends them per segment.
- melo_worker.py: _render applies each segment's own word_gap/sentence_gap/pitch
(previously reply-global).
- dashboard.py: emotion dropdown in the TTS panel; GET returns base + overrides
+ emotion list; POST {emotion,...} stores an override (or {reset:true} clears
it); base is set when emotion is omitted/"base".
Verified: an override on one emotion slows only that emotion (happy@0.7=5.66s vs
base 3.02s; sad unchanged at 3.06s); dashboard store/reset/base all work; 34
tests pass.
Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
@@ -1,56 +1,82 @@
|
||||
"""Emotion-tag parsing for expressive TTS. A ``[감정]`` tag must steer the
|
||||
pitch/speed of the text that follows without being spoken; a bracket that is NOT
|
||||
a known emotion word must be kept as ordinary spoken content."""
|
||||
delivery of the text that follows without being spoken; a bracket that is NOT a
|
||||
known emotion word must be kept as ordinary spoken content.
|
||||
|
||||
Delivery is now driven by a base (공통) control dict plus optional per-emotion
|
||||
overrides. By default there are no overrides, so every emotion delivers with the
|
||||
base controls (모든 감정 = 기본값)."""
|
||||
|
||||
from wsai.backends.emotion import (
|
||||
EMOTION_PARAMS,
|
||||
EMOTION_LABELS,
|
||||
_SYNONYMS,
|
||||
match_emotion,
|
||||
parse_segments,
|
||||
)
|
||||
from wsai.dashboard import _speech_text
|
||||
|
||||
BASE = 1.3
|
||||
BASE = {"speed": 1.35, "word_gap": -0.07, "sentence_gap": -0.30, "pitch": 0.0}
|
||||
|
||||
|
||||
def test_emotion_tag_is_not_spoken_and_sets_delivery():
|
||||
def test_emotion_tag_is_not_spoken_and_tags_segment():
|
||||
segs = parse_segments("[기쁨] 오늘 날씨 좋다", BASE)
|
||||
assert len(segs) == 1
|
||||
assert "기쁨" not in segs[0].text # the tag word is dropped
|
||||
assert segs[0].text == "오늘 날씨 좋다"
|
||||
mult, semis = EMOTION_PARAMS["happy"]
|
||||
assert segs[0].speed == BASE * mult
|
||||
assert segs[0].pitch == semis
|
||||
assert segs[0].emotion == "happy"
|
||||
|
||||
|
||||
def test_default_all_emotions_use_base():
|
||||
# No overrides -> a tagged emotion delivers with exactly the base controls.
|
||||
segs = parse_segments("[기쁨] 좋아", BASE)
|
||||
s = segs[0]
|
||||
assert (s.speed, s.word_gap, s.sentence_gap, s.pitch) == (1.35, -0.07, -0.30, 0.0)
|
||||
|
||||
|
||||
def test_per_emotion_override_applies_and_inherits_missing_keys():
|
||||
segs = parse_segments("[기쁨] 좋아", BASE, {"happy": {"speed": 1.6, "pitch": 3.0}})
|
||||
s = segs[0]
|
||||
assert s.speed == 1.6 and s.pitch == 3.0 # overridden
|
||||
assert s.word_gap == -0.07 and s.sentence_gap == -0.30 # inherited from base
|
||||
|
||||
|
||||
def test_override_only_affects_its_own_emotion():
|
||||
segs = parse_segments("[기쁨] 가. [슬픔] 나.", BASE, {"sad": {"speed": 0.8}})
|
||||
assert segs[0].emotion == "happy" and segs[0].speed == 1.35 # untouched
|
||||
assert segs[1].emotion == "sad" and segs[1].speed == 0.8 # overridden
|
||||
|
||||
|
||||
def test_midreply_emotion_change_splits_segments():
|
||||
segs = parse_segments("[속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!", BASE)
|
||||
assert len(segs) == 2
|
||||
assert segs[0].text == "정말 힘들었겠다."
|
||||
assert segs[1].text == "하지만 넌 할 수 있어!"
|
||||
assert segs[0].pitch == EMOTION_PARAMS["sad"][1]
|
||||
assert segs[1].pitch == EMOTION_PARAMS["hopeful"][1]
|
||||
assert segs[0].text == "정말 힘들었겠다." and segs[0].emotion == "sad"
|
||||
assert segs[1].text == "하지만 넌 할 수 있어!" and segs[1].emotion == "hopeful"
|
||||
|
||||
|
||||
def test_non_emotion_bracket_is_spoken_without_brackets():
|
||||
segs = parse_segments("[기쁨] 첫째는 [1번] 항목이야", BASE)
|
||||
assert len(segs) == 1
|
||||
# "1번" is not an emotion -> read it; brackets themselves are gone.
|
||||
assert "1번" in segs[0].text
|
||||
assert "[" not in segs[0].text and "]" not in segs[0].text
|
||||
assert segs[0].pitch == EMOTION_PARAMS["happy"][1]
|
||||
assert segs[0].emotion == "happy"
|
||||
|
||||
|
||||
def test_text_before_first_tag_is_neutral():
|
||||
segs = parse_segments("잠깐만. [신남] 찾았다!", BASE)
|
||||
assert segs[0].text == "잠깐만."
|
||||
assert segs[0].speed == BASE and segs[0].pitch == 0.0
|
||||
assert segs[1].pitch == EMOTION_PARAMS["excited"][1]
|
||||
assert segs[0].text == "잠깐만." and segs[0].emotion is None
|
||||
assert segs[0].speed == 1.35 and segs[0].pitch == 0.0
|
||||
assert segs[1].emotion == "excited"
|
||||
|
||||
|
||||
def test_plain_text_is_one_neutral_segment():
|
||||
segs = parse_segments("그냥 평범한 문장이야", BASE)
|
||||
assert len(segs) == 1
|
||||
assert segs[0].speed == BASE and segs[0].pitch == 0.0
|
||||
assert segs[0].emotion is None and segs[0].speed == 1.35
|
||||
|
||||
|
||||
def test_backward_compat_float_base():
|
||||
# A bare float base still works (speed only; other controls neutral).
|
||||
segs = parse_segments("그냥 문장", 1.2)
|
||||
assert segs[0].speed == 1.2 and segs[0].word_gap == 0.0
|
||||
|
||||
|
||||
def test_empty_input_yields_no_segments():
|
||||
@@ -77,6 +103,13 @@ def test_persona_examples_are_all_recognised():
|
||||
assert match_emotion(word) is not None, word
|
||||
|
||||
|
||||
def test_every_canonical_emotion_has_a_ui_label():
|
||||
# The dashboard dropdown lists EMOTION_LABELS; every emotion a tag can
|
||||
# resolve to must appear there or it would be untunable.
|
||||
for canon in set(_SYNONYMS.values()):
|
||||
assert canon in EMOTION_LABELS, canon
|
||||
|
||||
|
||||
def test_voice_turn_keeps_leading_emotion_tag_for_tts_parser():
|
||||
text = "[속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!"
|
||||
|
||||
@@ -84,7 +117,5 @@ def test_voice_turn_keeps_leading_emotion_tag_for_tts_parser():
|
||||
segs = parse_segments(spoken, BASE)
|
||||
|
||||
assert spoken == text
|
||||
assert segs[0].text == "정말 힘들었겠다."
|
||||
assert segs[0].pitch == EMOTION_PARAMS["sad"][1]
|
||||
assert segs[1].text == "하지만 넌 할 수 있어!"
|
||||
assert segs[1].pitch == EMOTION_PARAMS["hopeful"][1]
|
||||
assert segs[0].text == "정말 힘들었겠다." and segs[0].emotion == "sad"
|
||||
assert segs[1].text == "하지만 넌 할 수 있어!" and segs[1].emotion == "hopeful"
|
||||
|
||||
Reference in New Issue
Block a user