Files
watch_sceen_ai/wsai/backends/emotion.py
EJClaw 067efc7abe feat(tts): per-emotion voice controls (speed/word-gap/sentence-gap/pitch)
Each emotion can now be tuned independently. parse_segments resolves the four
controls per segment from a base (공통) dict plus an optional per-emotion
override; a missing override key inherits base. By default there are no
overrides, so every emotion delivers with the base values (모든 감정 = 기본값).

- emotion.py: Segment now carries all 4 controls + the canonical emotion name;
  parse_segments(text, base, overrides). Adds EMOTION_LABELS/EMOTIONS for the UI.
- melo.py: MeloTTS.emotion_overrides store; synth resolves per-segment controls
  and sends them per segment.
- melo_worker.py: _render applies each segment's own word_gap/sentence_gap/pitch
  (previously reply-global).
- dashboard.py: emotion dropdown in the TTS panel; GET returns base + overrides
  + emotion list; POST {emotion,...} stores an override (or {reset:true} clears
  it); base is set when emotion is omitted/"base".

Verified: an override on one emotion slows only that emotion (happy@0.7=5.66s vs
base 3.02s; sad unchanged at 3.06s); dashboard store/reset/base all work; 34
tests pass.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-08-26 23:10:33 +09:00

189 lines
9.8 KiB
Python

"""Emotion tags for expressive TTS.
Claude can sprinkle ``[감정]`` tags through a reply to colour the delivery, e.g.
"[속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!"
An emotion tag is NOT spoken — instead it shifts the *following* text's pitch and
speed until the next tag. A bracketed word that is NOT a known emotion is left as
ordinary spoken content (the brackets are dropped, the words are read).
The emotion vocabulary is grounded in the de-facto industry set used by Azure
Neural TTS speaking styles (cheerful, sad, angry, excited, friendly, hopeful,
terrified, shouting, whispering) together with Ekman's six basic emotions
(happiness, sadness, anger, fear, surprise, disgust). Each canonical emotion maps
to a ``(speed_multiplier, pitch_semitones)`` pair; the multiplier scales the base
synthesis speed and the semitone offset is applied as a pitch shift on the wav.
"""
from __future__ import annotations
import re
from dataclasses import dataclass
# Canonical emotion -> (speed multiplier relative to base, pitch shift in semitones).
# Kept deliberately modest so delivery stays natural, not cartoonish.
#
# Pitch is held at 0.0 for every emotion: librosa's post-hoc pitch_shift on Melo
# output produced a robotic, "monster"-sounding artefact (worse when stacked on a
# fast base speed). Emotion is therefore conveyed by speed only — natural and
# artefact-free. The pitch column is retained (rather than removed) so the effect
# can be re-enabled per-emotion later with a real, artefact-free pitch method.
EMOTION_PARAMS: dict[str, tuple[float, float]] = {
"base": (1.00, 0.0), # 기본 / neutral — the plain base voice, no colour
"happy": (1.08, 0.0), # 기쁨 / cheerful
"excited": (1.15, 0.0), # 신남 / excited
"hopeful": (1.10, 0.0), # 희망 / 힘차게
"sad": (0.90, 0.0), # 슬픔 / sad
"angry": (1.12, 0.0), # 화남 / angry
"fearful": (1.12, 0.0), # 두려움 / terrified
"surprised": (1.05, 0.0), # 놀람 / surprise
"disgust": (0.96, 0.0), # 혐오 / disgust
"calm": (0.95, 0.0), # 차분 / calm
"friendly": (1.00, 0.0), # 다정 / friendly
"serious": (0.97, 0.0), # 진지 / serious
"disappointed":(0.92, 0.0), # 실망 / disappointed
"tired": (0.90, 0.0), # 피곤 / 지침
"affectionate":(0.98, 0.0), # 사랑스럽게 / affectionate
"playful": (1.08, 0.0), # 장난스럽게 / playful
"whisper": (0.92, 0.0), # 속삭임 / whispering
"shout": (1.05, 0.0), # 외침 / shouting
"determined": (1.05, 0.0), # 단호 / determined
"relieved": (0.95, 0.0), # 안도 / relieved
"curious": (1.03, 0.0), # 궁금 / curious
}
# Every spelling Claude might realistically emit, mapped to a canonical emotion.
# False negatives (reading an emotion word aloud) are harmless; false positives
# (silently dropping real content) are not — so match spellings exactly rather
# than fuzzily.
_SYNONYMS: dict[str, str] = {}
def _register(canonical: str, *words: str) -> None:
for w in words:
_SYNONYMS[_norm(w)] = canonical
def _norm(word: str) -> str:
# Compare on a squeezed, lower-cased form so "힘 차게" == "힘차게".
return re.sub(r"\s+", "", word).lower()
_register("base", "기본", "기본목소리", "기본톤", "보통", "평범", "무감정", "default", "neutral", "normal", "plain")
_register("happy", "기쁨", "기쁘게", "기뻐", "기뻐하며", "행복", "행복하게", "행복하게도", "즐겁게", "즐거움", "밝게", "반가움", "반갑게", "반가워", "cheerful", "happy", "joyful")
_register("excited", "신남", "신나게", "신나서", "흥분", "들뜬", "들떠서", "설렘", "설레며", "excited", "thrilled")
_register("hopeful", "희망", "희망차게", "힘차게", "힘내", "힘내서", "응원", "응원하며", "격려", "hopeful", "encouraging")
_register("sad", "슬픔", "슬프게", "슬퍼", "슬퍼하며", "속상함", "속상하게", "속상해", "우울", "우울하게", "안타깝게", "울먹이며", "sad", "sorrowful")
_register("angry", "화남", "화나게", "화나서", "화가남", "분노", "분노하며", "짜증", "짜증내며", "angry", "furious")
_register("fearful", "두려움", "두렵게", "무섭게", "무서워하며", "불안", "불안하게", "겁먹은", "겁먹고", "떨리는", "fearful", "terrified", "anxious")
_register("surprised", "놀람", "놀라며", "놀랍게", "놀라서", "깜짝", "경악", "surprised", "shocked")
_register("disgust", "혐오", "역겹게", "질색", "disgust", "disgusted")
_register("calm", "차분", "차분하게", "침착", "침착하게", "담담하게", "잔잔하게", "calm", "gentle")
_register("friendly", "다정", "다정하게", "친근", "친근하게", "부드럽게", "따뜻하게", "friendly", "warm")
_register("serious", "진지", "진지하게", "무겁게", "엄숙하게", "serious", "solemn")
_register("disappointed", "실망", "실망스럽게", "실망하며", "낙담", "disappointed")
_register("tired", "피곤", "피곤하게", "지침", "지쳐서", "지친", "힘없이", "tired", "weary", "exhausted")
_register("affectionate", "사랑스럽게", "애정", "애정어린", "다정스럽게", "affectionate", "loving")
_register("playful", "장난스럽게", "장난치며", "유쾌하게", "익살스럽게", "웃으며", "웃으면서", "playful", "teasing")
_register("curious", "궁금", "궁금하게", "궁금해하며", "궁금해서", "호기심", "curious", "inquisitive")
_register("whisper", "속삭임", "속삭이며", "조용히", "나지막이", "whisper", "whispering")
_register("shout", "외침", "외치며", "큰소리로", "소리치며", "우렁차게", "shout", "shouting")
_register("determined", "단호", "단호하게", "결연하게", "당당하게", "determined", "confident")
_register("relieved", "안도", "안도하며", "안심", "안심하며", "relieved")
def match_emotion(inner: str) -> str | None:
"""Return the canonical emotion for a bracket's inner text, or None if the
text is not a recognised emotion word (and should therefore be spoken)."""
return _SYNONYMS.get(_norm(inner))
# The four adjustable TTS controls, per segment.
PARAM_KEYS = ("speed", "word_gap", "sentence_gap", "pitch")
# Canonical emotion -> Korean UI label, in display order. "base" == 기본/공통
# (the value used for untagged text and inherited by any emotion with no
# override). Every canonical emotion the synonym table resolves to must appear
# here so the dashboard dropdown can list it.
EMOTION_LABELS: dict[str, str] = {
"base": "기본(공통)", "happy": "기쁨", "excited": "신남", "hopeful": "희망",
"sad": "슬픔", "angry": "화남", "fearful": "두려움", "surprised": "놀람",
"disgust": "혐오", "calm": "차분", "friendly": "다정", "serious": "진지",
"disappointed": "실망", "tired": "피곤", "affectionate": "사랑", "playful": "장난",
"curious": "궁금", "whisper": "속삭임", "shout": "외침", "determined": "단호",
"relieved": "안도",
}
EMOTIONS: list[str] = list(EMOTION_LABELS)
@dataclass
class Segment:
text: str
speed: float
word_gap: float
sentence_gap: float
pitch: float # semitones; 0.0 == no shift
emotion: str | None = None # canonical emotion, or None for neutral/base
_TAG_RE = re.compile(r"\[([^\[\]]*)\]")
def _resolve(base: dict, overrides: dict | None, emotion: str | None) -> dict:
"""The 4 controls for one segment: the base values, with any per-emotion
override applied on top (a missing override key inherits base)."""
p = {k: float(base[k]) for k in PARAM_KEYS}
if emotion and overrides:
ov = overrides.get(emotion)
if ov:
for k in PARAM_KEYS:
if ov.get(k) is not None:
p[k] = float(ov[k])
return p
def parse_segments(text: str, base, overrides: dict | None = None) -> list[Segment]:
"""Split ``text`` into consecutive spoken segments, each carrying the four
TTS controls implied by the most recent emotion tag.
* ``base`` is the 공통 control dict ``{speed, word_gap, sentence_gap, pitch}``
used for untagged text and inherited by any emotion without an override.
A bare float is also accepted (speed only) for backward compatibility.
* ``overrides`` maps a canonical emotion -> a partial control dict; only the
keys present override the base. Empty/None means every emotion delivers
with the base controls (the default: 모든 감정 = 기본값).
* An emotion tag switches the active emotion for everything after it and is
not spoken. A non-emotion bracket keeps its inner words as spoken text.
"""
if not isinstance(base, dict):
base = {"speed": float(base), "word_gap": 0.0, "sentence_gap": 0.0, "pitch": 0.0}
segments: list[Segment] = []
cur_emotion: str | None = None
buf: list[str] = []
def flush() -> None:
joined = "".join(buf).strip()
if joined:
p = _resolve(base, overrides, cur_emotion)
segments.append(Segment(joined, p["speed"], p["word_gap"],
p["sentence_gap"], p["pitch"], cur_emotion))
buf.clear()
pos = 0
for m in _TAG_RE.finditer(text):
emotion = match_emotion(m.group(1))
buf.append(text[pos:m.start()])
pos = m.end()
if emotion is None:
# Not an emotion — read the bracket's contents, drop the brackets.
buf.append(m.group(1))
else:
# Emotion tag — everything so far belongs to the previous emotion;
# flush it, then switch delivery for what follows.
flush()
cur_emotion = emotion
buf.append(text[pos:])
flush()
return segments # empty when there is nothing speakable (blank or all-tags)