Files
watch_sceen_ai/tests/test_emotion.py
EJClaw 067efc7abe feat(tts): per-emotion voice controls (speed/word-gap/sentence-gap/pitch)
Each emotion can now be tuned independently. parse_segments resolves the four
controls per segment from a base (공통) dict plus an optional per-emotion
override; a missing override key inherits base. By default there are no
overrides, so every emotion delivers with the base values (모든 감정 = 기본값).

- emotion.py: Segment now carries all 4 controls + the canonical emotion name;
  parse_segments(text, base, overrides). Adds EMOTION_LABELS/EMOTIONS for the UI.
- melo.py: MeloTTS.emotion_overrides store; synth resolves per-segment controls
  and sends them per segment.
- melo_worker.py: _render applies each segment's own word_gap/sentence_gap/pitch
  (previously reply-global).
- dashboard.py: emotion dropdown in the TTS panel; GET returns base + overrides
  + emotion list; POST {emotion,...} stores an override (or {reset:true} clears
  it); base is set when emotion is omitted/"base".

Verified: an override on one emotion slows only that emotion (happy@0.7=5.66s vs
base 3.02s; sad unchanged at 3.06s); dashboard store/reset/base all work; 34
tests pass.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-08-26 23:10:33 +09:00

122 lines
4.8 KiB
Python

"""Emotion-tag parsing for expressive TTS. A ``[감정]`` tag must steer the
delivery of the text that follows without being spoken; a bracket that is NOT a
known emotion word must be kept as ordinary spoken content.
Delivery is now driven by a base (공통) control dict plus optional per-emotion
overrides. By default there are no overrides, so every emotion delivers with the
base controls (모든 감정 = 기본값)."""
from wsai.backends.emotion import (
EMOTION_LABELS,
_SYNONYMS,
match_emotion,
parse_segments,
)
from wsai.dashboard import _speech_text
BASE = {"speed": 1.35, "word_gap": -0.07, "sentence_gap": -0.30, "pitch": 0.0}
def test_emotion_tag_is_not_spoken_and_tags_segment():
segs = parse_segments("[기쁨] 오늘 날씨 좋다", BASE)
assert len(segs) == 1
assert "기쁨" not in segs[0].text # the tag word is dropped
assert segs[0].text == "오늘 날씨 좋다"
assert segs[0].emotion == "happy"
def test_default_all_emotions_use_base():
# No overrides -> a tagged emotion delivers with exactly the base controls.
segs = parse_segments("[기쁨] 좋아", BASE)
s = segs[0]
assert (s.speed, s.word_gap, s.sentence_gap, s.pitch) == (1.35, -0.07, -0.30, 0.0)
def test_per_emotion_override_applies_and_inherits_missing_keys():
segs = parse_segments("[기쁨] 좋아", BASE, {"happy": {"speed": 1.6, "pitch": 3.0}})
s = segs[0]
assert s.speed == 1.6 and s.pitch == 3.0 # overridden
assert s.word_gap == -0.07 and s.sentence_gap == -0.30 # inherited from base
def test_override_only_affects_its_own_emotion():
segs = parse_segments("[기쁨] 가. [슬픔] 나.", BASE, {"sad": {"speed": 0.8}})
assert segs[0].emotion == "happy" and segs[0].speed == 1.35 # untouched
assert segs[1].emotion == "sad" and segs[1].speed == 0.8 # overridden
def test_midreply_emotion_change_splits_segments():
segs = parse_segments("[속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!", BASE)
assert len(segs) == 2
assert segs[0].text == "정말 힘들었겠다." and segs[0].emotion == "sad"
assert segs[1].text == "하지만 넌 할 수 있어!" and segs[1].emotion == "hopeful"
def test_non_emotion_bracket_is_spoken_without_brackets():
segs = parse_segments("[기쁨] 첫째는 [1번] 항목이야", BASE)
assert len(segs) == 1
assert "1번" in segs[0].text
assert "[" not in segs[0].text and "]" not in segs[0].text
assert segs[0].emotion == "happy"
def test_text_before_first_tag_is_neutral():
segs = parse_segments("잠깐만. [신남] 찾았다!", BASE)
assert segs[0].text == "잠깐만." and segs[0].emotion is None
assert segs[0].speed == 1.35 and segs[0].pitch == 0.0
assert segs[1].emotion == "excited"
def test_plain_text_is_one_neutral_segment():
segs = parse_segments("그냥 평범한 문장이야", BASE)
assert len(segs) == 1
assert segs[0].emotion is None and segs[0].speed == 1.35
def test_backward_compat_float_base():
# A bare float base still works (speed only; other controls neutral).
segs = parse_segments("그냥 문장", 1.2)
assert segs[0].speed == 1.2 and segs[0].word_gap == 0.0
def test_empty_input_yields_no_segments():
assert parse_segments("", BASE) == []
assert parse_segments(" ", BASE) == []
def test_only_emotion_tags_yield_no_segments():
# A reply that is nothing but tags has nothing to say.
assert parse_segments("[기쁨][신남]", BASE) == []
def test_match_emotion_is_synonym_and_space_tolerant():
assert match_emotion("속상함") == "sad"
assert match_emotion(" 힘 차게 ") == "hopeful" # squeezed + trimmed
assert match_emotion("행복하게") == "happy"
assert match_emotion("메모") is None # not an emotion
def test_persona_examples_are_all_recognised():
# Every emotion the brain persona advertises must resolve, or it would be
# read aloud instead of shaping the voice.
for word in ["힘차게", "궁금", "반가움", "차분하게", "웃으며", "속상함"]:
assert match_emotion(word) is not None, word
def test_every_canonical_emotion_has_a_ui_label():
# The dashboard dropdown lists EMOTION_LABELS; every emotion a tag can
# resolve to must appear there or it would be untunable.
for canon in set(_SYNONYMS.values()):
assert canon in EMOTION_LABELS, canon
def test_voice_turn_keeps_leading_emotion_tag_for_tts_parser():
text = "[속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!"
spoken = _speech_text(text)
segs = parse_segments(spoken, BASE)
assert spoken == text
assert segs[0].text == "정말 힘들었겠다." and segs[0].emotion == "sad"
assert segs[1].text == "하지만 넌 할 수 있어!" and segs[1].emotion == "hopeful"