feat(tts): per-emotion voice controls (speed/word-gap/sentence-gap/pitch)

Each emotion can now be tuned independently. parse_segments resolves the four
controls per segment from a base (공통) dict plus an optional per-emotion
override; a missing override key inherits base. By default there are no
overrides, so every emotion delivers with the base values (모든 감정 = 기본값).

- emotion.py: Segment now carries all 4 controls + the canonical emotion name;
  parse_segments(text, base, overrides). Adds EMOTION_LABELS/EMOTIONS for the UI.
- melo.py: MeloTTS.emotion_overrides store; synth resolves per-segment controls
  and sends them per segment.
- melo_worker.py: _render applies each segment's own word_gap/sentence_gap/pitch
  (previously reply-global).
- dashboard.py: emotion dropdown in the TTS panel; GET returns base + overrides
  + emotion list; POST {emotion,...} stores an override (or {reset:true} clears
  it); base is set when emotion is omitted/"base".

Verified: an override on one emotion slows only that emotion (happy@0.7=5.66s vs
base 3.02s; sad unchanged at 3.06s); dashboard store/reset/base all work; 34
tests pass.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
EJClaw
2026-08-26 23:10:33 +09:00
parent 39b743d976
commit 067efc7abe
5 changed files with 242 additions and 99 deletions

View File

@@ -1,56 +1,82 @@
"""Emotion-tag parsing for expressive TTS. A ``[감정]`` tag must steer the """Emotion-tag parsing for expressive TTS. A ``[감정]`` tag must steer the
pitch/speed of the text that follows without being spoken; a bracket that is NOT delivery of the text that follows without being spoken; a bracket that is NOT a
a known emotion word must be kept as ordinary spoken content.""" known emotion word must be kept as ordinary spoken content.
Delivery is now driven by a base (공통) control dict plus optional per-emotion
overrides. By default there are no overrides, so every emotion delivers with the
base controls (모든 감정 = 기본값)."""
from wsai.backends.emotion import ( from wsai.backends.emotion import (
EMOTION_PARAMS, EMOTION_LABELS,
_SYNONYMS,
match_emotion, match_emotion,
parse_segments, parse_segments,
) )
from wsai.dashboard import _speech_text from wsai.dashboard import _speech_text
BASE = 1.3 BASE = {"speed": 1.35, "word_gap": -0.07, "sentence_gap": -0.30, "pitch": 0.0}
def test_emotion_tag_is_not_spoken_and_sets_delivery(): def test_emotion_tag_is_not_spoken_and_tags_segment():
segs = parse_segments("[기쁨] 오늘 날씨 좋다", BASE) segs = parse_segments("[기쁨] 오늘 날씨 좋다", BASE)
assert len(segs) == 1 assert len(segs) == 1
assert "기쁨" not in segs[0].text # the tag word is dropped assert "기쁨" not in segs[0].text # the tag word is dropped
assert segs[0].text == "오늘 날씨 좋다" assert segs[0].text == "오늘 날씨 좋다"
mult, semis = EMOTION_PARAMS["happy"] assert segs[0].emotion == "happy"
assert segs[0].speed == BASE * mult
assert segs[0].pitch == semis
def test_default_all_emotions_use_base():
# No overrides -> a tagged emotion delivers with exactly the base controls.
segs = parse_segments("[기쁨] 좋아", BASE)
s = segs[0]
assert (s.speed, s.word_gap, s.sentence_gap, s.pitch) == (1.35, -0.07, -0.30, 0.0)
def test_per_emotion_override_applies_and_inherits_missing_keys():
segs = parse_segments("[기쁨] 좋아", BASE, {"happy": {"speed": 1.6, "pitch": 3.0}})
s = segs[0]
assert s.speed == 1.6 and s.pitch == 3.0 # overridden
assert s.word_gap == -0.07 and s.sentence_gap == -0.30 # inherited from base
def test_override_only_affects_its_own_emotion():
segs = parse_segments("[기쁨] 가. [슬픔] 나.", BASE, {"sad": {"speed": 0.8}})
assert segs[0].emotion == "happy" and segs[0].speed == 1.35 # untouched
assert segs[1].emotion == "sad" and segs[1].speed == 0.8 # overridden
def test_midreply_emotion_change_splits_segments(): def test_midreply_emotion_change_splits_segments():
segs = parse_segments("[속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!", BASE) segs = parse_segments("[속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!", BASE)
assert len(segs) == 2 assert len(segs) == 2
assert segs[0].text == "정말 힘들었겠다." assert segs[0].text == "정말 힘들었겠다." and segs[0].emotion == "sad"
assert segs[1].text == "하지만 넌 할 수 있어!" assert segs[1].text == "하지만 넌 할 수 있어!" and segs[1].emotion == "hopeful"
assert segs[0].pitch == EMOTION_PARAMS["sad"][1]
assert segs[1].pitch == EMOTION_PARAMS["hopeful"][1]
def test_non_emotion_bracket_is_spoken_without_brackets(): def test_non_emotion_bracket_is_spoken_without_brackets():
segs = parse_segments("[기쁨] 첫째는 [1번] 항목이야", BASE) segs = parse_segments("[기쁨] 첫째는 [1번] 항목이야", BASE)
assert len(segs) == 1 assert len(segs) == 1
# "1번" is not an emotion -> read it; brackets themselves are gone.
assert "1번" in segs[0].text assert "1번" in segs[0].text
assert "[" not in segs[0].text and "]" not in segs[0].text assert "[" not in segs[0].text and "]" not in segs[0].text
assert segs[0].pitch == EMOTION_PARAMS["happy"][1] assert segs[0].emotion == "happy"
def test_text_before_first_tag_is_neutral(): def test_text_before_first_tag_is_neutral():
segs = parse_segments("잠깐만. [신남] 찾았다!", BASE) segs = parse_segments("잠깐만. [신남] 찾았다!", BASE)
assert segs[0].text == "잠깐만." assert segs[0].text == "잠깐만." and segs[0].emotion is None
assert segs[0].speed == BASE and segs[0].pitch == 0.0 assert segs[0].speed == 1.35 and segs[0].pitch == 0.0
assert segs[1].pitch == EMOTION_PARAMS["excited"][1] assert segs[1].emotion == "excited"
def test_plain_text_is_one_neutral_segment(): def test_plain_text_is_one_neutral_segment():
segs = parse_segments("그냥 평범한 문장이야", BASE) segs = parse_segments("그냥 평범한 문장이야", BASE)
assert len(segs) == 1 assert len(segs) == 1
assert segs[0].speed == BASE and segs[0].pitch == 0.0 assert segs[0].emotion is None and segs[0].speed == 1.35
def test_backward_compat_float_base():
# A bare float base still works (speed only; other controls neutral).
segs = parse_segments("그냥 문장", 1.2)
assert segs[0].speed == 1.2 and segs[0].word_gap == 0.0
def test_empty_input_yields_no_segments(): def test_empty_input_yields_no_segments():
@@ -77,6 +103,13 @@ def test_persona_examples_are_all_recognised():
assert match_emotion(word) is not None, word assert match_emotion(word) is not None, word
def test_every_canonical_emotion_has_a_ui_label():
# The dashboard dropdown lists EMOTION_LABELS; every emotion a tag can
# resolve to must appear there or it would be untunable.
for canon in set(_SYNONYMS.values()):
assert canon in EMOTION_LABELS, canon
def test_voice_turn_keeps_leading_emotion_tag_for_tts_parser(): def test_voice_turn_keeps_leading_emotion_tag_for_tts_parser():
text = "[속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!" text = "[속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!"
@@ -84,7 +117,5 @@ def test_voice_turn_keeps_leading_emotion_tag_for_tts_parser():
segs = parse_segments(spoken, BASE) segs = parse_segments(spoken, BASE)
assert spoken == text assert spoken == text
assert segs[0].text == "정말 힘들었겠다." assert segs[0].text == "정말 힘들었겠다." and segs[0].emotion == "sad"
assert segs[0].pitch == EMOTION_PARAMS["sad"][1] assert segs[1].text == "하지만 넌 할 수 있어!" and segs[1].emotion == "hopeful"
assert segs[1].text == "하지만 넌 할 수 있어!"
assert segs[1].pitch == EMOTION_PARAMS["hopeful"][1]

View File

@@ -99,33 +99,75 @@ def match_emotion(inner: str) -> str | None:
return _SYNONYMS.get(_norm(inner)) return _SYNONYMS.get(_norm(inner))
# The four adjustable TTS controls, per segment.
PARAM_KEYS = ("speed", "word_gap", "sentence_gap", "pitch")
# Canonical emotion -> Korean UI label, in display order. "base" == 기본/공통
# (the value used for untagged text and inherited by any emotion with no
# override). Every canonical emotion the synonym table resolves to must appear
# here so the dashboard dropdown can list it.
EMOTION_LABELS: dict[str, str] = {
"base": "기본(공통)", "happy": "기쁨", "excited": "신남", "hopeful": "희망",
"sad": "슬픔", "angry": "화남", "fearful": "두려움", "surprised": "놀람",
"disgust": "혐오", "calm": "차분", "friendly": "다정", "serious": "진지",
"disappointed": "실망", "tired": "피곤", "affectionate": "사랑", "playful": "장난",
"curious": "궁금", "whisper": "속삭임", "shout": "외침", "determined": "단호",
"relieved": "안도",
}
EMOTIONS: list[str] = list(EMOTION_LABELS)
@dataclass @dataclass
class Segment: class Segment:
text: str text: str
speed: float speed: float
word_gap: float
sentence_gap: float
pitch: float # semitones; 0.0 == no shift pitch: float # semitones; 0.0 == no shift
emotion: str | None = None # canonical emotion, or None for neutral/base
_TAG_RE = re.compile(r"\[([^\[\]]*)\]") _TAG_RE = re.compile(r"\[([^\[\]]*)\]")
def parse_segments(text: str, base_speed: float) -> list[Segment]: def _resolve(base: dict, overrides: dict | None, emotion: str | None) -> dict:
"""Split ``text`` into consecutive spoken segments, each carrying the speed """The 4 controls for one segment: the base values, with any per-emotion
and pitch implied by the most recent emotion tag. override applied on top (a missing override key inherits base)."""
p = {k: float(base[k]) for k in PARAM_KEYS}
if emotion and overrides:
ov = overrides.get(emotion)
if ov:
for k in PARAM_KEYS:
if ov.get(k) is not None:
p[k] = float(ov[k])
return p
def parse_segments(text: str, base, overrides: dict | None = None) -> list[Segment]:
"""Split ``text`` into consecutive spoken segments, each carrying the four
TTS controls implied by the most recent emotion tag.
* ``base`` is the 공통 control dict ``{speed, word_gap, sentence_gap, pitch}``
used for untagged text and inherited by any emotion without an override.
A bare float is also accepted (speed only) for backward compatibility.
* ``overrides`` maps a canonical emotion -> a partial control dict; only the
keys present override the base. Empty/None means every emotion delivers
with the base controls (the default: 모든 감정 = 기본값).
* An emotion tag switches the active emotion for everything after it and is * An emotion tag switches the active emotion for everything after it and is
not spoken. not spoken. A non-emotion bracket keeps its inner words as spoken text.
* A non-emotion bracket keeps its inner words as spoken text (brackets gone).
* Text before any tag is spoken with neutral delivery (base speed, no shift).
""" """
if not isinstance(base, dict):
base = {"speed": float(base), "word_gap": 0.0, "sentence_gap": 0.0, "pitch": 0.0}
segments: list[Segment] = [] segments: list[Segment] = []
cur_speed, cur_pitch = base_speed, 0.0 cur_emotion: str | None = None
buf: list[str] = [] buf: list[str] = []
def flush() -> None: def flush() -> None:
joined = "".join(buf).strip() joined = "".join(buf).strip()
if joined: if joined:
segments.append(Segment(joined, cur_speed, cur_pitch)) p = _resolve(base, overrides, cur_emotion)
segments.append(Segment(joined, p["speed"], p["word_gap"],
p["sentence_gap"], p["pitch"], cur_emotion))
buf.clear() buf.clear()
pos = 0 pos = 0
@@ -140,8 +182,7 @@ def parse_segments(text: str, base_speed: float) -> list[Segment]:
# Emotion tag — everything so far belongs to the previous emotion; # Emotion tag — everything so far belongs to the previous emotion;
# flush it, then switch delivery for what follows. # flush it, then switch delivery for what follows.
flush() flush()
mult, semis = EMOTION_PARAMS[emotion] cur_emotion = emotion
cur_speed, cur_pitch = base_speed * mult, semis
buf.append(text[pos:]) buf.append(text[pos:])
flush() flush()
return segments # empty when there is nothing speakable (blank or all-tags) return segments # empty when there is nothing speakable (blank or all-tags)

View File

@@ -107,6 +107,10 @@ class MeloTTS:
else os.environ.get("WSAI_TTS_SENTENCE_GAP", "-0.30")) else os.environ.get("WSAI_TTS_SENTENCE_GAP", "-0.30"))
self.pitch = float(pitch if pitch is not None self.pitch = float(pitch if pitch is not None
else os.environ.get("WSAI_TTS_PITCH", "0.0")) else os.environ.get("WSAI_TTS_PITCH", "0.0"))
# Per-emotion control overrides: canonical emotion -> partial dict of
# {speed, word_gap, sentence_gap, pitch}. Empty by default, so every
# emotion inherits the base controls above (모든 감정 = 기본값).
self.emotion_overrides: dict[str, dict] = {}
self.sink = sink or _log_sink self.sink = sink or _log_sink
self._proc: asyncio.subprocess.Process | None = None self._proc: asyncio.subprocess.Process | None = None
self._lock = asyncio.Lock() self._lock = asyncio.Lock()
@@ -214,35 +218,36 @@ class MeloTTS:
per call (used by the dashboard preview so tuning does not disturb the per call (used by the dashboard preview so tuning does not disturb the
live bot voice until explicitly applied).""" live bot voice until explicitly applied)."""
await self._ensure() await self._ensure()
spd = self.speed if speed is None else float(speed) base = {
wg = self.word_gap if word_gap is None else float(word_gap) "speed": self.speed if speed is None else float(speed),
sg = self.sentence_gap if sentence_gap is None else float(sentence_gap) "word_gap": self.word_gap if word_gap is None else float(word_gap),
pt = self.pitch if pitch is None else float(pitch) "sentence_gap": self.sentence_gap if sentence_gap is None else float(sentence_gap),
"pitch": self.pitch if pitch is None else float(pitch),
}
text = normalize_for_speech(text) text = normalize_for_speech(text)
# Split on [감정] tags: each tag steers pitch/speed for the text that # Split on [감정] tags: each tag switches delivery for the text that
# follows (and is itself not spoken); non-emotion brackets stay as words. # follows (and is itself not spoken); non-emotion brackets stay as words.
segments = parse_segments(text, spd) # Each segment carries its own 4 controls (base + per-emotion override).
segments = parse_segments(text, base, self.emotion_overrides)
self._n += 1 self._n += 1
out = str(self.out_dir / f"tts-{self._n:06d}.wav") out = str(self.out_dir / f"tts-{self._n:06d}.wav")
if segments: if segments:
payload = { payload = {
"segments": [ "segments": [
{"text": s.text, "speed": s.speed, "pitch": s.pitch} {"text": s.text, "speed": s.speed, "word_gap": s.word_gap,
"sentence_gap": s.sentence_gap, "pitch": s.pitch}
for s in segments for s in segments
], ],
"out": out, "out": out,
"word_gap": wg,
"sentence_gap": sg,
"pitch": pt,
} }
else: # empty/whitespace reply: keep legacy single-utterance behaviour else: # empty/whitespace reply: keep legacy single-utterance behaviour
payload = { payload = {
"text": text, "text": text,
"out": out, "out": out,
"speed": spd, "speed": base["speed"],
"word_gap": wg, "word_gap": base["word_gap"],
"sentence_gap": sg, "sentence_gap": base["sentence_gap"],
"pitch": pt, "pitch": base["pitch"],
} }
req = json.dumps(payload) req = json.dumps(payload)
s = time.monotonic() s = time.monotonic()

View File

@@ -11,28 +11,27 @@ library chatter lands on stderr instead.
Protocol (one JSON object per line, on the protocol channel): Protocol (one JSON object per line, on the protocol channel):
<- {"text": "...", "out": "/abs/path.wav", "speed": 1.3, <- {"text": "...", "out": "/abs/path.wav", "speed": 1.3,
"word_gap": 0.25, "sentence_gap": 0.75, "pitch": 0.0} "word_gap": -0.07, "sentence_gap": -0.30, "pitch": 0.0}
<- {"segments": [{"text": "...", "speed": 1.3, "pitch": 2.0}, ...], <- {"segments": [{"text": "...", "speed": 1.3, "word_gap": -0.07,
"out": "/abs/path.wav", "sentence_gap": -0.30, "pitch": 2.0}, ...], "out": "/abs/path.wav"}
"word_gap": 0.25, "sentence_gap": 0.75, "pitch": 0.0}
-> {"ok": true, "out": "/abs/path.wav", "ms": 123} -> {"ok": true, "out": "/abs/path.wav", "ms": 123}
-> {"ok": false, "error": "..."} -> {"ok": false, "error": "..."}
On startup, once the model is ready, it emits exactly one line: On startup, once the model is ready, it emits exactly one line:
-> {"ready": true, "ms": <load-ms>, "device": "cpu"} -> {"ready": true, "ms": <load-ms>, "device": "cpu"}
Four independent voice controls (ported from tts_site, see docs manual): Four independent voice controls (ported from tts_site, see docs manual). They
speed per-segment glyph speed -> generation-stage length_scale (1/speed) are PER SEGMENT: the caller resolves each segment's controls from the base
word_gap reply-global, sec (-0.2..0.5): grow/shrink intra-sentence pauses values plus that emotion's override, so different emotions can have different
sentence_gap reply-global, sec (-0.5..1.5): insert/trim silence at sentence speed/gaps/pitch within one reply.
boundaries (also used between emotion segments) speed glyph speed -> generation-stage length_scale (1/speed)
pitch reply-global semitone offset, added on top of each segment's own word_gap sec (-0.2..0.5): grow/shrink intra-sentence pauses
``pitch`` (from its emotion tag; 0 == no shift) sentence_gap sec (-0.5..1.5): insert/trim silence at sentence boundaries
(within a segment, and before the next segment)
pitch semitone offset applied to the segment's wav (0 == no shift)
Each segment's ``pitch`` is a semitone offset applied to that segment's wav so A reply is split into sentences, each sentence synthesised at its segment's
emotion tags can raise/lower the voice without changing the words. A reply is speed, and the pieces concatenated with that segment's ``sentence_gap`` so one
split into sentences, each sentence synthesised at its segment's speed, and the reply can carry several emotions each with its own rhythm and pitch.
pieces concatenated with ``sentence_gap`` so one reply can carry several emotions
and a controllable rhythm.
""" """
import json import json
@@ -188,23 +187,21 @@ def main() -> None:
tts.tts_to_file(text, speaker_id, None, speed=speed), dtype=np.float32 tts.tts_to_file(text, speaker_id, None, speed=speed), dtype=np.float32
) )
def _render(segments, out, word_gap, sentence_gap, pitch_offset): def _render(segments, out):
"""Render one wav from emotion segments applying the 4 controls. """Render one wav from emotion segments. Each segment carries its own
four controls (speed, word_gap, sentence_gap, pitch) — resolved on the
Each segment carries its own glyph ``speed`` and ``pitch`` (from emotion caller side from the base values plus that emotion's override. Sentences
tags); ``word_gap``/``sentence_gap``/``pitch_offset`` are reply-global. within a segment join with that segment's ``sentence_gap``; between
Sentences within a segment, and the segments themselves, are joined with segments the incoming segment's ``sentence_gap`` sets the pause."""
``sentence_gap`` (positive inserts silence, negative trims it)."""
word_gap = _clamp(word_gap, -0.2, 0.5)
sentence_gap = _clamp(sentence_gap, -0.5, 1.5)
pitch_offset = _clamp(pitch_offset, -12.0, 12.0)
pieces = [] pieces = []
for seg in segments: for seg in segments:
text = seg.get("text", "") text = seg.get("text", "")
if not text.strip(): if not text.strip():
continue continue
speed = _clamp(seg.get("speed", 1.0), 0.5, 2.0) speed = _clamp(seg.get("speed", 1.0), 0.5, 2.0)
pitch = _clamp(float(seg.get("pitch", 0.0)) + pitch_offset, -12.0, 12.0) word_gap = _clamp(seg.get("word_gap", 0.0), -0.2, 0.5)
sentence_gap = _clamp(seg.get("sentence_gap", 0.0), -0.5, 1.5)
pitch = _clamp(seg.get("pitch", 0.0), -12.0, 12.0)
sent_pieces = [] sent_pieces = []
for sent in split_sentences(text): for sent in split_sentences(text):
a = _synth_one(sent, speed) a = _synth_one(sent, speed)
@@ -254,14 +251,17 @@ def main() -> None:
if out.startswith("/tmp") or out.startswith("/dev/shm"): if out.startswith("/tmp") or out.startswith("/dev/shm"):
raise ValueError(f"refusing RAM-backed tmpfs path: {out}") raise ValueError(f"refusing RAM-backed tmpfs path: {out}")
s = time.monotonic() s = time.monotonic()
word_gap = float(req.get("word_gap", 0.0))
sentence_gap = float(req.get("sentence_gap", 0.0))
pitch_offset = float(req.get("pitch", 0.0))
if "segments" in req: if "segments" in req:
_render(req["segments"], out, word_gap, sentence_gap, pitch_offset) _render(req["segments"], out)
else: # legacy single-utterance form else: # legacy single-utterance form: one segment carrying all controls
seg = {"text": req["text"], "speed": float(req.get("speed", 1.0)), "pitch": 0.0} seg = {
_render([seg], out, word_gap, sentence_gap, pitch_offset) "text": req["text"],
"speed": float(req.get("speed", 1.0)),
"word_gap": float(req.get("word_gap", 0.0)),
"sentence_gap": float(req.get("sentence_gap", 0.0)),
"pitch": float(req.get("pitch", 0.0)),
}
_render([seg], out)
ms = int((time.monotonic() - s) * 1000) ms = int((time.monotonic() - s) * 1000)
_emit({"ok": True, "out": out, "ms": ms}) _emit({"ok": True, "out": out, "ms": ms})
except Exception as exc: # keep the worker alive across bad requests except Exception as exc: # keep the worker alive across bad requests

View File

@@ -460,7 +460,7 @@ class Dashboard:
"pitch": (-12.0, 12.0, 1.0), "pitch": (-12.0, 12.0, 1.0),
} }
def _tts_values(self) -> dict: def _tts_base(self) -> dict:
t = self.tts t = self.tts
return { return {
"speed": float(getattr(t, "speed", 1.35)), "speed": float(getattr(t, "speed", 1.35)),
@@ -470,15 +470,37 @@ class Dashboard:
} }
def tts_settings(self) -> dict: def tts_settings(self) -> dict:
return {"ok": True, "settings": self._tts_values(), """Base (공통) controls + the per-emotion overrides + the emotion list
"ranges": {k: list(v) for k, v in self._TTS_RANGES.items()}} (canonical key + Korean label) so the dashboard can offer per-emotion
tuning. Emotions with no override inherit base (모든 감정 = 기본값)."""
from .backends.emotion import EMOTION_LABELS
overrides = dict(getattr(self.tts, "emotion_overrides", {}) or {})
return {
"ok": True,
"base": self._tts_base(),
"overrides": overrides,
"emotions": [{"key": k, "label": v} for k, v in EMOTION_LABELS.items()],
"ranges": {k: list(v) for k, v in self._TTS_RANGES.items()},
}
def set_tts_settings(self, data: dict) -> dict: def set_tts_settings(self, data: dict) -> dict:
"""Clamp and apply provided controls to the live TTS instance.""" """Apply controls. Without ``emotion`` (or emotion == "base"), set the
base/공통 values on the live TTS instance. With a specific emotion, store
(or, if ``reset``, clear) that emotion's override."""
emotion = data.get("emotion")
if not emotion or emotion == "base":
for key, (lo, hi, _step) in self._TTS_RANGES.items(): for key, (lo, hi, _step) in self._TTS_RANGES.items():
v = data.get(key) v = data.get(key)
if v is not None: if v is not None:
setattr(self.tts, key, float(max(lo, min(hi, float(v))))) setattr(self.tts, key, float(max(lo, min(hi, float(v)))))
else:
store = getattr(self.tts, "emotion_overrides", None)
if store is None:
store = self.tts.emotion_overrides = {}
if data.get("reset"):
store.pop(emotion, None)
else:
store[emotion] = self._tts_overrides(data)
return self.tts_settings() return self.tts_settings()
def _tts_overrides(self, data: dict) -> dict: def _tts_overrides(self, data: dict) -> dict:
@@ -736,6 +758,7 @@ PAGE = r"""<!DOCTYPE html>
.ttsgrid label b{color:var(--fg);font-weight:600} .ttsgrid label b{color:var(--fg);font-weight:600}
.ttsgrid input[type=range]{width:100%;accent-color:#3aa0ff} .ttsgrid input[type=range]{width:100%;accent-color:#3aa0ff}
.ttstext{flex:1;min-width:180px;background:#0f2333;border:1px solid #234a63;color:var(--fg);border-radius:10px;padding:8px 12px;font-size:13px} .ttstext{flex:1;min-width:180px;background:#0f2333;border:1px solid #234a63;color:var(--fg);border-radius:10px;padding:8px 12px;font-size:13px}
.ttssel{background:#0f2333;border:1px solid #234a63;color:var(--fg);border-radius:10px;padding:7px 10px;font-size:13px;margin-left:8px}
.sttres .txt{color:var(--heard);font-weight:600;line-height:1.5} .sttres .txt{color:var(--heard);font-weight:600;line-height:1.5}
.sttres .meta{color:var(--muted);font-size:12px;margin-top:5px} .sttres .meta{color:var(--muted);font-size:12px;margin-top:5px}
.hbtn{padding:6px 12px;font-size:12.5px} .hbtn{padding:6px 12px;font-size:12.5px}
@@ -869,6 +892,11 @@ PAGE = r"""<!DOCTYPE html>
<section class="sttbox" id="ttsbox" style="display:none"> <section class="sttbox" id="ttsbox" style="display:none">
<h2 id="ttsToggle" class="collap"><span id="ttsCaret">▸</span> 🔊 봇 목소리(TTS) 조절 · MeloTTS</h2> <h2 id="ttsToggle" class="collap"><span id="ttsCaret">▸</span> 🔊 봇 목소리(TTS) 조절 · MeloTTS</h2>
<div id="ttsBody" style="display:none"> <div id="ttsBody" style="display:none">
<div class="sttrow" style="margin-bottom:10px">
<label style="font-size:12.5px;color:var(--muted)">감정 선택
<select id="tEmotion" class="ttssel"></select></label>
<span id="tEmoNote" class="sttstat">감정을 고르면 그 감정만 따로 조절합니다. 기본(공통)은 태그 없는 말투에 적용됩니다.</span>
</div>
<div class="ttsgrid"> <div class="ttsgrid">
<label>글자 속도 <b id="tvSpeed">1.35x</b> <label>글자 속도 <b id="tvSpeed">1.35x</b>
<input type="range" id="tSpeed" min="0.5" max="2.0" step="0.05" value="1.35"></label> <input type="range" id="tSpeed" min="0.5" max="2.0" step="0.05" value="1.35"></label>
@@ -1171,8 +1199,10 @@ async function sendBlob(blob){
}; };
})(); })();
// --- 봇 목소리(TTS) 조절: 슬라이더 → 미리듣기 → 봇 적용 -------------------- # // --- 봇 목소리(TTS) 조절: 감정별 슬라이더 → 미리듣기 → 봇 적용 -------------- #
let ttsInited = false; let ttsInited = false;
let ttsState = {base:{}, overrides:{}, emotions:[]}; // last-known server state
const TTS_DEFAULT = {speed:1.35, word_gap:-0.07, sentence_gap:-0.3, pitch:0};
function ttsLabels(){ function ttsLabels(){
const sp=parseFloat($('tSpeed').value), wg=parseFloat($('tWord').value), const sp=parseFloat($('tSpeed').value), wg=parseFloat($('tWord').value),
sg=parseFloat($('tSent').value), pt=parseInt($('tPitch').value,10); sg=parseFloat($('tSent').value), pt=parseInt($('tPitch').value,10);
@@ -1187,8 +1217,34 @@ function ttsVals(){
} }
function ttsSet(s){ function ttsSet(s){
if(!s) return; if(!s) return;
$('tSpeed').value=s.speed; $('tWord').value=s.word_gap; if(s.speed!=null) $('tSpeed').value=s.speed;
$('tSent').value=s.sentence_gap; $('tPitch').value=s.pitch; ttsLabels(); if(s.word_gap!=null) $('tWord').value=s.word_gap;
if(s.sentence_gap!=null) $('tSent').value=s.sentence_gap;
if(s.pitch!=null) $('tPitch').value=s.pitch;
ttsLabels();
}
// Effective values for the picked emotion: its override merged over base
// (base itself when 기본 or when the emotion has no override).
function ttsValuesFor(key){
const base = Object.assign({}, TTS_DEFAULT, ttsState.base);
if(key && key!=='base' && ttsState.overrides && ttsState.overrides[key])
return Object.assign({}, base, ttsState.overrides[key]);
return base;
}
function ttsApplyState(j){
if(!j || !j.ok) return;
ttsState = {base:j.base||{}, overrides:j.overrides||{}, emotions:j.emotions||ttsState.emotions};
// Refresh dropdown labels to mark which emotions are customised.
const sel=$('tEmotion'); const cur=sel.value||'base';
sel.innerHTML='';
ttsState.emotions.forEach(e=>{
const o=document.createElement('option'); o.value=e.key;
const custom = e.key!=='base' && ttsState.overrides[e.key];
o.textContent = e.label + (custom ? ' ●' : '');
sel.appendChild(o);
});
sel.value = cur;
ttsSet(ttsValuesFor(sel.value));
} }
async function initTts(){ async function initTts(){
if(ttsInited) return; ttsInited = true; if(ttsInited) return; ttsInited = true;
@@ -1198,6 +1254,9 @@ async function initTts(){
$('ttsCaret').textContent = open ? '▾' : '▸'; $('ttsCaret').textContent = open ? '▾' : '▸';
}; };
['tSpeed','tWord','tSent','tPitch'].forEach(id => $(id).addEventListener('input', ttsLabels)); ['tSpeed','tWord','tSent','tPitch'].forEach(id => $(id).addEventListener('input', ttsLabels));
$('tEmotion').onchange = ()=>{ ttsSet(ttsValuesFor($('tEmotion').value));
$('tStat').textContent = $('tEmotion').value==='base'
? '기본(공통) 값을 조절 중입니다.' : '"'+$('tEmotion').selectedOptions[0].textContent.replace(' ●','')+'" 감정만 조절 중입니다.'; };
$('tPreview').onclick = async ()=>{ $('tPreview').onclick = async ()=>{
const text=($('tText').value||'').trim(); const text=($('tText').value||'').trim();
if(!text){ $('tStat').textContent='미리들을 문장을 입력하세요.'; return; } if(!text){ $('tStat').textContent='미리들을 문장을 입력하세요.'; return; }
@@ -1212,20 +1271,27 @@ async function initTts(){
}catch(e){ $('tStat').textContent='미리듣기 오류: '+e; } }catch(e){ $('tStat').textContent='미리듣기 오류: '+e; }
}; };
$('tApply').onclick = async ()=>{ $('tApply').onclick = async ()=>{
const key=$('tEmotion').value||'base';
try{ try{
const r=await fetch('/api/tts/settings',{method:'POST',headers:{'Content-Type':'application/json'}, const r=await fetch('/api/tts/settings',{method:'POST',headers:{'Content-Type':'application/json'},
body:JSON.stringify(ttsVals())}); body:JSON.stringify(Object.assign({emotion:key}, ttsVals()))});
const j=await r.json(); const j=await r.json();
if(j.ok){ ttsSet(j.settings); toast('봇 목소리에 적용됨 · 다음 답변부터 반영'); } if(j.ok){ ttsApplyState(j); toast((key==='base'?'기본(공통)':'해당 감정')+' 적용됨 · 다음 답변부터 반영'); }
else { $('tStat').textContent='적용 실패: '+(j.error||''); } else { $('tStat').textContent='적용 실패: '+(j.error||''); }
}catch(e){ $('tStat').textContent='적용 오류: '+e; } }catch(e){ $('tStat').textContent='적용 오류: '+e; }
}; };
$('tReset').onclick = ()=>{ ttsSet({speed:1.35,word_gap:-0.07,sentence_gap:-0.3,pitch:0}); $('tReset').onclick = async ()=>{
$('tStat').textContent='기본값으로 되돌림 (미리듣기/적용을 눌러 반영).'; }; const key=$('tEmotion').value||'base';
try{ if(key==='base'){ ttsSet(TTS_DEFAULT);
const j=await (await fetch('/api/tts/settings')).json(); $('tStat').textContent='기본값으로 세팅됨 (적용을 눌러 반영).'; return; }
if(j.ok) ttsSet(j.settings); try{ // clear this emotion's override -> it inherits 기본(공통) again
}catch(e){} const r=await fetch('/api/tts/settings',{method:'POST',headers:{'Content-Type':'application/json'},
body:JSON.stringify({emotion:key, reset:true})});
const j=await r.json();
if(j.ok){ ttsApplyState(j); toast('이 감정을 기본(공통)값으로 되돌림'); }
}catch(e){ $('tStat').textContent='되돌리기 오류: '+e; }
};
try{ ttsApplyState(await (await fetch('/api/tts/settings')).json()); }catch(e){}
} }
// --- Reusable popup/modal (프롬프트, 화이트/블랙리스트 등이 공유) ---------- # // --- Reusable popup/modal (프롬프트, 화이트/블랙리스트 등이 공유) ---------- #