feat(tts): per-emotion voice controls (speed/word-gap/sentence-gap/pitch)
Each emotion can now be tuned independently. parse_segments resolves the four
controls per segment from a base (공통) dict plus an optional per-emotion
override; a missing override key inherits base. By default there are no
overrides, so every emotion delivers with the base values (모든 감정 = 기본값).
- emotion.py: Segment now carries all 4 controls + the canonical emotion name;
parse_segments(text, base, overrides). Adds EMOTION_LABELS/EMOTIONS for the UI.
- melo.py: MeloTTS.emotion_overrides store; synth resolves per-segment controls
and sends them per segment.
- melo_worker.py: _render applies each segment's own word_gap/sentence_gap/pitch
(previously reply-global).
- dashboard.py: emotion dropdown in the TTS panel; GET returns base + overrides
+ emotion list; POST {emotion,...} stores an override (or {reset:true} clears
it); base is set when emotion is omitted/"base".
Verified: an override on one emotion slows only that emotion (happy@0.7=5.66s vs
base 3.02s; sad unchanged at 3.06s); dashboard store/reset/base all work; 34
tests pass.
Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
@@ -99,33 +99,75 @@ def match_emotion(inner: str) -> str | None:
|
||||
return _SYNONYMS.get(_norm(inner))
|
||||
|
||||
|
||||
# The four adjustable TTS controls, per segment.
|
||||
PARAM_KEYS = ("speed", "word_gap", "sentence_gap", "pitch")
|
||||
|
||||
# Canonical emotion -> Korean UI label, in display order. "base" == 기본/공통
|
||||
# (the value used for untagged text and inherited by any emotion with no
|
||||
# override). Every canonical emotion the synonym table resolves to must appear
|
||||
# here so the dashboard dropdown can list it.
|
||||
EMOTION_LABELS: dict[str, str] = {
|
||||
"base": "기본(공통)", "happy": "기쁨", "excited": "신남", "hopeful": "희망",
|
||||
"sad": "슬픔", "angry": "화남", "fearful": "두려움", "surprised": "놀람",
|
||||
"disgust": "혐오", "calm": "차분", "friendly": "다정", "serious": "진지",
|
||||
"disappointed": "실망", "tired": "피곤", "affectionate": "사랑", "playful": "장난",
|
||||
"curious": "궁금", "whisper": "속삭임", "shout": "외침", "determined": "단호",
|
||||
"relieved": "안도",
|
||||
}
|
||||
EMOTIONS: list[str] = list(EMOTION_LABELS)
|
||||
|
||||
|
||||
@dataclass
|
||||
class Segment:
|
||||
text: str
|
||||
speed: float
|
||||
word_gap: float
|
||||
sentence_gap: float
|
||||
pitch: float # semitones; 0.0 == no shift
|
||||
emotion: str | None = None # canonical emotion, or None for neutral/base
|
||||
|
||||
|
||||
_TAG_RE = re.compile(r"\[([^\[\]]*)\]")
|
||||
|
||||
|
||||
def parse_segments(text: str, base_speed: float) -> list[Segment]:
|
||||
"""Split ``text`` into consecutive spoken segments, each carrying the speed
|
||||
and pitch implied by the most recent emotion tag.
|
||||
def _resolve(base: dict, overrides: dict | None, emotion: str | None) -> dict:
|
||||
"""The 4 controls for one segment: the base values, with any per-emotion
|
||||
override applied on top (a missing override key inherits base)."""
|
||||
p = {k: float(base[k]) for k in PARAM_KEYS}
|
||||
if emotion and overrides:
|
||||
ov = overrides.get(emotion)
|
||||
if ov:
|
||||
for k in PARAM_KEYS:
|
||||
if ov.get(k) is not None:
|
||||
p[k] = float(ov[k])
|
||||
return p
|
||||
|
||||
|
||||
def parse_segments(text: str, base, overrides: dict | None = None) -> list[Segment]:
|
||||
"""Split ``text`` into consecutive spoken segments, each carrying the four
|
||||
TTS controls implied by the most recent emotion tag.
|
||||
|
||||
* ``base`` is the 공통 control dict ``{speed, word_gap, sentence_gap, pitch}``
|
||||
used for untagged text and inherited by any emotion without an override.
|
||||
A bare float is also accepted (speed only) for backward compatibility.
|
||||
* ``overrides`` maps a canonical emotion -> a partial control dict; only the
|
||||
keys present override the base. Empty/None means every emotion delivers
|
||||
with the base controls (the default: 모든 감정 = 기본값).
|
||||
* An emotion tag switches the active emotion for everything after it and is
|
||||
not spoken.
|
||||
* A non-emotion bracket keeps its inner words as spoken text (brackets gone).
|
||||
* Text before any tag is spoken with neutral delivery (base speed, no shift).
|
||||
not spoken. A non-emotion bracket keeps its inner words as spoken text.
|
||||
"""
|
||||
if not isinstance(base, dict):
|
||||
base = {"speed": float(base), "word_gap": 0.0, "sentence_gap": 0.0, "pitch": 0.0}
|
||||
segments: list[Segment] = []
|
||||
cur_speed, cur_pitch = base_speed, 0.0
|
||||
cur_emotion: str | None = None
|
||||
buf: list[str] = []
|
||||
|
||||
def flush() -> None:
|
||||
joined = "".join(buf).strip()
|
||||
if joined:
|
||||
segments.append(Segment(joined, cur_speed, cur_pitch))
|
||||
p = _resolve(base, overrides, cur_emotion)
|
||||
segments.append(Segment(joined, p["speed"], p["word_gap"],
|
||||
p["sentence_gap"], p["pitch"], cur_emotion))
|
||||
buf.clear()
|
||||
|
||||
pos = 0
|
||||
@@ -140,8 +182,7 @@ def parse_segments(text: str, base_speed: float) -> list[Segment]:
|
||||
# Emotion tag — everything so far belongs to the previous emotion;
|
||||
# flush it, then switch delivery for what follows.
|
||||
flush()
|
||||
mult, semis = EMOTION_PARAMS[emotion]
|
||||
cur_speed, cur_pitch = base_speed * mult, semis
|
||||
cur_emotion = emotion
|
||||
buf.append(text[pos:])
|
||||
flush()
|
||||
return segments # empty when there is nothing speakable (blank or all-tags)
|
||||
|
||||
Reference in New Issue
Block a user