feat(tts): per-emotion voice controls (speed/word-gap/sentence-gap/pitch)

Each emotion can now be tuned independently. parse_segments resolves the four
controls per segment from a base (공통) dict plus an optional per-emotion
override; a missing override key inherits base. By default there are no
overrides, so every emotion delivers with the base values (모든 감정 = 기본값).

- emotion.py: Segment now carries all 4 controls + the canonical emotion name;
  parse_segments(text, base, overrides). Adds EMOTION_LABELS/EMOTIONS for the UI.
- melo.py: MeloTTS.emotion_overrides store; synth resolves per-segment controls
  and sends them per segment.
- melo_worker.py: _render applies each segment's own word_gap/sentence_gap/pitch
  (previously reply-global).
- dashboard.py: emotion dropdown in the TTS panel; GET returns base + overrides
  + emotion list; POST {emotion,...} stores an override (or {reset:true} clears
  it); base is set when emotion is omitted/"base".

Verified: an override on one emotion slows only that emotion (happy@0.7=5.66s vs
base 3.02s; sad unchanged at 3.06s); dashboard store/reset/base all work; 34
tests pass.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
EJClaw
2026-08-26 23:10:33 +09:00
parent 39b743d976
commit 067efc7abe
5 changed files with 242 additions and 99 deletions

View File

@@ -99,33 +99,75 @@ def match_emotion(inner: str) -> str | None:
return _SYNONYMS.get(_norm(inner))
# The four adjustable TTS controls, per segment.
PARAM_KEYS = ("speed", "word_gap", "sentence_gap", "pitch")
# Canonical emotion -> Korean UI label, in display order. "base" == 기본/공통
# (the value used for untagged text and inherited by any emotion with no
# override). Every canonical emotion the synonym table resolves to must appear
# here so the dashboard dropdown can list it.
EMOTION_LABELS: dict[str, str] = {
"base": "기본(공통)", "happy": "기쁨", "excited": "신남", "hopeful": "희망",
"sad": "슬픔", "angry": "화남", "fearful": "두려움", "surprised": "놀람",
"disgust": "혐오", "calm": "차분", "friendly": "다정", "serious": "진지",
"disappointed": "실망", "tired": "피곤", "affectionate": "사랑", "playful": "장난",
"curious": "궁금", "whisper": "속삭임", "shout": "외침", "determined": "단호",
"relieved": "안도",
}
EMOTIONS: list[str] = list(EMOTION_LABELS)
@dataclass
class Segment:
text: str
speed: float
word_gap: float
sentence_gap: float
pitch: float # semitones; 0.0 == no shift
emotion: str | None = None # canonical emotion, or None for neutral/base
_TAG_RE = re.compile(r"\[([^\[\]]*)\]")
def parse_segments(text: str, base_speed: float) -> list[Segment]:
"""Split ``text`` into consecutive spoken segments, each carrying the speed
and pitch implied by the most recent emotion tag.
def _resolve(base: dict, overrides: dict | None, emotion: str | None) -> dict:
"""The 4 controls for one segment: the base values, with any per-emotion
override applied on top (a missing override key inherits base)."""
p = {k: float(base[k]) for k in PARAM_KEYS}
if emotion and overrides:
ov = overrides.get(emotion)
if ov:
for k in PARAM_KEYS:
if ov.get(k) is not None:
p[k] = float(ov[k])
return p
def parse_segments(text: str, base, overrides: dict | None = None) -> list[Segment]:
"""Split ``text`` into consecutive spoken segments, each carrying the four
TTS controls implied by the most recent emotion tag.
* ``base`` is the 공통 control dict ``{speed, word_gap, sentence_gap, pitch}``
used for untagged text and inherited by any emotion without an override.
A bare float is also accepted (speed only) for backward compatibility.
* ``overrides`` maps a canonical emotion -> a partial control dict; only the
keys present override the base. Empty/None means every emotion delivers
with the base controls (the default: 모든 감정 = 기본값).
* An emotion tag switches the active emotion for everything after it and is
not spoken.
* A non-emotion bracket keeps its inner words as spoken text (brackets gone).
* Text before any tag is spoken with neutral delivery (base speed, no shift).
not spoken. A non-emotion bracket keeps its inner words as spoken text.
"""
if not isinstance(base, dict):
base = {"speed": float(base), "word_gap": 0.0, "sentence_gap": 0.0, "pitch": 0.0}
segments: list[Segment] = []
cur_speed, cur_pitch = base_speed, 0.0
cur_emotion: str | None = None
buf: list[str] = []
def flush() -> None:
joined = "".join(buf).strip()
if joined:
segments.append(Segment(joined, cur_speed, cur_pitch))
p = _resolve(base, overrides, cur_emotion)
segments.append(Segment(joined, p["speed"], p["word_gap"],
p["sentence_gap"], p["pitch"], cur_emotion))
buf.clear()
pos = 0
@@ -140,8 +182,7 @@ def parse_segments(text: str, base_speed: float) -> list[Segment]:
# Emotion tag — everything so far belongs to the previous emotion;
# flush it, then switch delivery for what follows.
flush()
mult, semis = EMOTION_PARAMS[emotion]
cur_speed, cur_pitch = base_speed * mult, semis
cur_emotion = emotion
buf.append(text[pos:])
flush()
return segments # empty when there is nothing speakable (blank or all-tags)