feat(tts): per-emotion voice controls (speed/word-gap/sentence-gap/pitch)

Each emotion can now be tuned independently. parse_segments resolves the four
controls per segment from a base (공통) dict plus an optional per-emotion
override; a missing override key inherits base. By default there are no
overrides, so every emotion delivers with the base values (모든 감정 = 기본값).

- emotion.py: Segment now carries all 4 controls + the canonical emotion name;
  parse_segments(text, base, overrides). Adds EMOTION_LABELS/EMOTIONS for the UI.
- melo.py: MeloTTS.emotion_overrides store; synth resolves per-segment controls
  and sends them per segment.
- melo_worker.py: _render applies each segment's own word_gap/sentence_gap/pitch
  (previously reply-global).
- dashboard.py: emotion dropdown in the TTS panel; GET returns base + overrides
  + emotion list; POST {emotion,...} stores an override (or {reset:true} clears
  it); base is set when emotion is omitted/"base".

Verified: an override on one emotion slows only that emotion (happy@0.7=5.66s vs
base 3.02s; sad unchanged at 3.06s); dashboard store/reset/base all work; 34
tests pass.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
EJClaw
2026-08-26 23:10:33 +09:00
parent 39b743d976
commit 067efc7abe
5 changed files with 242 additions and 99 deletions

View File

@@ -99,33 +99,75 @@ def match_emotion(inner: str) -> str | None:
return _SYNONYMS.get(_norm(inner))
# The four adjustable TTS controls, per segment.
PARAM_KEYS = ("speed", "word_gap", "sentence_gap", "pitch")
# Canonical emotion -> Korean UI label, in display order. "base" == 기본/공통
# (the value used for untagged text and inherited by any emotion with no
# override). Every canonical emotion the synonym table resolves to must appear
# here so the dashboard dropdown can list it.
EMOTION_LABELS: dict[str, str] = {
"base": "기본(공통)", "happy": "기쁨", "excited": "신남", "hopeful": "희망",
"sad": "슬픔", "angry": "화남", "fearful": "두려움", "surprised": "놀람",
"disgust": "혐오", "calm": "차분", "friendly": "다정", "serious": "진지",
"disappointed": "실망", "tired": "피곤", "affectionate": "사랑", "playful": "장난",
"curious": "궁금", "whisper": "속삭임", "shout": "외침", "determined": "단호",
"relieved": "안도",
}
EMOTIONS: list[str] = list(EMOTION_LABELS)
@dataclass
class Segment:
text: str
speed: float
word_gap: float
sentence_gap: float
pitch: float # semitones; 0.0 == no shift
emotion: str | None = None # canonical emotion, or None for neutral/base
_TAG_RE = re.compile(r"\[([^\[\]]*)\]")
def parse_segments(text: str, base_speed: float) -> list[Segment]:
"""Split ``text`` into consecutive spoken segments, each carrying the speed
and pitch implied by the most recent emotion tag.
def _resolve(base: dict, overrides: dict | None, emotion: str | None) -> dict:
"""The 4 controls for one segment: the base values, with any per-emotion
override applied on top (a missing override key inherits base)."""
p = {k: float(base[k]) for k in PARAM_KEYS}
if emotion and overrides:
ov = overrides.get(emotion)
if ov:
for k in PARAM_KEYS:
if ov.get(k) is not None:
p[k] = float(ov[k])
return p
def parse_segments(text: str, base, overrides: dict | None = None) -> list[Segment]:
"""Split ``text`` into consecutive spoken segments, each carrying the four
TTS controls implied by the most recent emotion tag.
* ``base`` is the 공통 control dict ``{speed, word_gap, sentence_gap, pitch}``
used for untagged text and inherited by any emotion without an override.
A bare float is also accepted (speed only) for backward compatibility.
* ``overrides`` maps a canonical emotion -> a partial control dict; only the
keys present override the base. Empty/None means every emotion delivers
with the base controls (the default: 모든 감정 = 기본값).
* An emotion tag switches the active emotion for everything after it and is
not spoken.
* A non-emotion bracket keeps its inner words as spoken text (brackets gone).
* Text before any tag is spoken with neutral delivery (base speed, no shift).
not spoken. A non-emotion bracket keeps its inner words as spoken text.
"""
if not isinstance(base, dict):
base = {"speed": float(base), "word_gap": 0.0, "sentence_gap": 0.0, "pitch": 0.0}
segments: list[Segment] = []
cur_speed, cur_pitch = base_speed, 0.0
cur_emotion: str | None = None
buf: list[str] = []
def flush() -> None:
joined = "".join(buf).strip()
if joined:
segments.append(Segment(joined, cur_speed, cur_pitch))
p = _resolve(base, overrides, cur_emotion)
segments.append(Segment(joined, p["speed"], p["word_gap"],
p["sentence_gap"], p["pitch"], cur_emotion))
buf.clear()
pos = 0
@@ -140,8 +182,7 @@ def parse_segments(text: str, base_speed: float) -> list[Segment]:
# Emotion tag — everything so far belongs to the previous emotion;
# flush it, then switch delivery for what follows.
flush()
mult, semis = EMOTION_PARAMS[emotion]
cur_speed, cur_pitch = base_speed * mult, semis
cur_emotion = emotion
buf.append(text[pos:])
flush()
return segments # empty when there is nothing speakable (blank or all-tags)

View File

@@ -107,6 +107,10 @@ class MeloTTS:
else os.environ.get("WSAI_TTS_SENTENCE_GAP", "-0.30"))
self.pitch = float(pitch if pitch is not None
else os.environ.get("WSAI_TTS_PITCH", "0.0"))
# Per-emotion control overrides: canonical emotion -> partial dict of
# {speed, word_gap, sentence_gap, pitch}. Empty by default, so every
# emotion inherits the base controls above (모든 감정 = 기본값).
self.emotion_overrides: dict[str, dict] = {}
self.sink = sink or _log_sink
self._proc: asyncio.subprocess.Process | None = None
self._lock = asyncio.Lock()
@@ -214,35 +218,36 @@ class MeloTTS:
per call (used by the dashboard preview so tuning does not disturb the
live bot voice until explicitly applied)."""
await self._ensure()
spd = self.speed if speed is None else float(speed)
wg = self.word_gap if word_gap is None else float(word_gap)
sg = self.sentence_gap if sentence_gap is None else float(sentence_gap)
pt = self.pitch if pitch is None else float(pitch)
base = {
"speed": self.speed if speed is None else float(speed),
"word_gap": self.word_gap if word_gap is None else float(word_gap),
"sentence_gap": self.sentence_gap if sentence_gap is None else float(sentence_gap),
"pitch": self.pitch if pitch is None else float(pitch),
}
text = normalize_for_speech(text)
# Split on [감정] tags: each tag steers pitch/speed for the text that
# Split on [감정] tags: each tag switches delivery for the text that
# follows (and is itself not spoken); non-emotion brackets stay as words.
segments = parse_segments(text, spd)
# Each segment carries its own 4 controls (base + per-emotion override).
segments = parse_segments(text, base, self.emotion_overrides)
self._n += 1
out = str(self.out_dir / f"tts-{self._n:06d}.wav")
if segments:
payload = {
"segments": [
{"text": s.text, "speed": s.speed, "pitch": s.pitch}
{"text": s.text, "speed": s.speed, "word_gap": s.word_gap,
"sentence_gap": s.sentence_gap, "pitch": s.pitch}
for s in segments
],
"out": out,
"word_gap": wg,
"sentence_gap": sg,
"pitch": pt,
}
else: # empty/whitespace reply: keep legacy single-utterance behaviour
payload = {
"text": text,
"out": out,
"speed": spd,
"word_gap": wg,
"sentence_gap": sg,
"pitch": pt,
"speed": base["speed"],
"word_gap": base["word_gap"],
"sentence_gap": base["sentence_gap"],
"pitch": base["pitch"],
}
req = json.dumps(payload)
s = time.monotonic()

View File

@@ -11,28 +11,27 @@ library chatter lands on stderr instead.
Protocol (one JSON object per line, on the protocol channel):
<- {"text": "...", "out": "/abs/path.wav", "speed": 1.3,
"word_gap": 0.25, "sentence_gap": 0.75, "pitch": 0.0}
<- {"segments": [{"text": "...", "speed": 1.3, "pitch": 2.0}, ...],
"out": "/abs/path.wav",
"word_gap": 0.25, "sentence_gap": 0.75, "pitch": 0.0}
"word_gap": -0.07, "sentence_gap": -0.30, "pitch": 0.0}
<- {"segments": [{"text": "...", "speed": 1.3, "word_gap": -0.07,
"sentence_gap": -0.30, "pitch": 2.0}, ...], "out": "/abs/path.wav"}
-> {"ok": true, "out": "/abs/path.wav", "ms": 123}
-> {"ok": false, "error": "..."}
On startup, once the model is ready, it emits exactly one line:
-> {"ready": true, "ms": <load-ms>, "device": "cpu"}
Four independent voice controls (ported from tts_site, see docs manual):
speed per-segment glyph speed -> generation-stage length_scale (1/speed)
word_gap reply-global, sec (-0.2..0.5): grow/shrink intra-sentence pauses
sentence_gap reply-global, sec (-0.5..1.5): insert/trim silence at sentence
boundaries (also used between emotion segments)
pitch reply-global semitone offset, added on top of each segment's own
``pitch`` (from its emotion tag; 0 == no shift)
Four independent voice controls (ported from tts_site, see docs manual). They
are PER SEGMENT: the caller resolves each segment's controls from the base
values plus that emotion's override, so different emotions can have different
speed/gaps/pitch within one reply.
speed glyph speed -> generation-stage length_scale (1/speed)
word_gap sec (-0.2..0.5): grow/shrink intra-sentence pauses
sentence_gap sec (-0.5..1.5): insert/trim silence at sentence boundaries
(within a segment, and before the next segment)
pitch semitone offset applied to the segment's wav (0 == no shift)
Each segment's ``pitch`` is a semitone offset applied to that segment's wav so
emotion tags can raise/lower the voice without changing the words. A reply is
split into sentences, each sentence synthesised at its segment's speed, and the
pieces concatenated with ``sentence_gap`` so one reply can carry several emotions
and a controllable rhythm.
A reply is split into sentences, each sentence synthesised at its segment's
speed, and the pieces concatenated with that segment's ``sentence_gap`` so one
reply can carry several emotions each with its own rhythm and pitch.
"""
import json
@@ -188,23 +187,21 @@ def main() -> None:
tts.tts_to_file(text, speaker_id, None, speed=speed), dtype=np.float32
)
def _render(segments, out, word_gap, sentence_gap, pitch_offset):
"""Render one wav from emotion segments applying the 4 controls.
Each segment carries its own glyph ``speed`` and ``pitch`` (from emotion
tags); ``word_gap``/``sentence_gap``/``pitch_offset`` are reply-global.
Sentences within a segment, and the segments themselves, are joined with
``sentence_gap`` (positive inserts silence, negative trims it)."""
word_gap = _clamp(word_gap, -0.2, 0.5)
sentence_gap = _clamp(sentence_gap, -0.5, 1.5)
pitch_offset = _clamp(pitch_offset, -12.0, 12.0)
def _render(segments, out):
"""Render one wav from emotion segments. Each segment carries its own
four controls (speed, word_gap, sentence_gap, pitch) — resolved on the
caller side from the base values plus that emotion's override. Sentences
within a segment join with that segment's ``sentence_gap``; between
segments the incoming segment's ``sentence_gap`` sets the pause."""
pieces = []
for seg in segments:
text = seg.get("text", "")
if not text.strip():
continue
speed = _clamp(seg.get("speed", 1.0), 0.5, 2.0)
pitch = _clamp(float(seg.get("pitch", 0.0)) + pitch_offset, -12.0, 12.0)
word_gap = _clamp(seg.get("word_gap", 0.0), -0.2, 0.5)
sentence_gap = _clamp(seg.get("sentence_gap", 0.0), -0.5, 1.5)
pitch = _clamp(seg.get("pitch", 0.0), -12.0, 12.0)
sent_pieces = []
for sent in split_sentences(text):
a = _synth_one(sent, speed)
@@ -254,14 +251,17 @@ def main() -> None:
if out.startswith("/tmp") or out.startswith("/dev/shm"):
raise ValueError(f"refusing RAM-backed tmpfs path: {out}")
s = time.monotonic()
word_gap = float(req.get("word_gap", 0.0))
sentence_gap = float(req.get("sentence_gap", 0.0))
pitch_offset = float(req.get("pitch", 0.0))
if "segments" in req:
_render(req["segments"], out, word_gap, sentence_gap, pitch_offset)
else: # legacy single-utterance form
seg = {"text": req["text"], "speed": float(req.get("speed", 1.0)), "pitch": 0.0}
_render([seg], out, word_gap, sentence_gap, pitch_offset)
_render(req["segments"], out)
else: # legacy single-utterance form: one segment carrying all controls
seg = {
"text": req["text"],
"speed": float(req.get("speed", 1.0)),
"word_gap": float(req.get("word_gap", 0.0)),
"sentence_gap": float(req.get("sentence_gap", 0.0)),
"pitch": float(req.get("pitch", 0.0)),
}
_render([seg], out)
ms = int((time.monotonic() - s) * 1000)
_emit({"ok": True, "out": out, "ms": ms})
except Exception as exc: # keep the worker alive across bad requests

View File

@@ -460,7 +460,7 @@ class Dashboard:
"pitch": (-12.0, 12.0, 1.0),
}
def _tts_values(self) -> dict:
def _tts_base(self) -> dict:
t = self.tts
return {
"speed": float(getattr(t, "speed", 1.35)),
@@ -470,15 +470,37 @@ class Dashboard:
}
def tts_settings(self) -> dict:
return {"ok": True, "settings": self._tts_values(),
"ranges": {k: list(v) for k, v in self._TTS_RANGES.items()}}
"""Base (공통) controls + the per-emotion overrides + the emotion list
(canonical key + Korean label) so the dashboard can offer per-emotion
tuning. Emotions with no override inherit base (모든 감정 = 기본값)."""
from .backends.emotion import EMOTION_LABELS
overrides = dict(getattr(self.tts, "emotion_overrides", {}) or {})
return {
"ok": True,
"base": self._tts_base(),
"overrides": overrides,
"emotions": [{"key": k, "label": v} for k, v in EMOTION_LABELS.items()],
"ranges": {k: list(v) for k, v in self._TTS_RANGES.items()},
}
def set_tts_settings(self, data: dict) -> dict:
"""Clamp and apply provided controls to the live TTS instance."""
for key, (lo, hi, _step) in self._TTS_RANGES.items():
v = data.get(key)
if v is not None:
setattr(self.tts, key, float(max(lo, min(hi, float(v)))))
"""Apply controls. Without ``emotion`` (or emotion == "base"), set the
base/공통 values on the live TTS instance. With a specific emotion, store
(or, if ``reset``, clear) that emotion's override."""
emotion = data.get("emotion")
if not emotion or emotion == "base":
for key, (lo, hi, _step) in self._TTS_RANGES.items():
v = data.get(key)
if v is not None:
setattr(self.tts, key, float(max(lo, min(hi, float(v)))))
else:
store = getattr(self.tts, "emotion_overrides", None)
if store is None:
store = self.tts.emotion_overrides = {}
if data.get("reset"):
store.pop(emotion, None)
else:
store[emotion] = self._tts_overrides(data)
return self.tts_settings()
def _tts_overrides(self, data: dict) -> dict:
@@ -736,6 +758,7 @@ PAGE = r"""<!DOCTYPE html>
.ttsgrid label b{color:var(--fg);font-weight:600}
.ttsgrid input[type=range]{width:100%;accent-color:#3aa0ff}
.ttstext{flex:1;min-width:180px;background:#0f2333;border:1px solid #234a63;color:var(--fg);border-radius:10px;padding:8px 12px;font-size:13px}
.ttssel{background:#0f2333;border:1px solid #234a63;color:var(--fg);border-radius:10px;padding:7px 10px;font-size:13px;margin-left:8px}
.sttres .txt{color:var(--heard);font-weight:600;line-height:1.5}
.sttres .meta{color:var(--muted);font-size:12px;margin-top:5px}
.hbtn{padding:6px 12px;font-size:12.5px}
@@ -869,6 +892,11 @@ PAGE = r"""<!DOCTYPE html>
<section class="sttbox" id="ttsbox" style="display:none">
<h2 id="ttsToggle" class="collap"><span id="ttsCaret">▸</span> 🔊 봇 목소리(TTS) 조절 · MeloTTS</h2>
<div id="ttsBody" style="display:none">
<div class="sttrow" style="margin-bottom:10px">
<label style="font-size:12.5px;color:var(--muted)">감정 선택
<select id="tEmotion" class="ttssel"></select></label>
<span id="tEmoNote" class="sttstat">감정을 고르면 그 감정만 따로 조절합니다. 기본(공통)은 태그 없는 말투에 적용됩니다.</span>
</div>
<div class="ttsgrid">
<label>글자 속도 <b id="tvSpeed">1.35x</b>
<input type="range" id="tSpeed" min="0.5" max="2.0" step="0.05" value="1.35"></label>
@@ -1171,8 +1199,10 @@ async function sendBlob(blob){
};
})();
// --- 봇 목소리(TTS) 조절: 슬라이더 → 미리듣기 → 봇 적용 -------------------- #
// --- 봇 목소리(TTS) 조절: 감정별 슬라이더 → 미리듣기 → 봇 적용 -------------- #
let ttsInited = false;
let ttsState = {base:{}, overrides:{}, emotions:[]}; // last-known server state
const TTS_DEFAULT = {speed:1.35, word_gap:-0.07, sentence_gap:-0.3, pitch:0};
function ttsLabels(){
const sp=parseFloat($('tSpeed').value), wg=parseFloat($('tWord').value),
sg=parseFloat($('tSent').value), pt=parseInt($('tPitch').value,10);
@@ -1187,8 +1217,34 @@ function ttsVals(){
}
function ttsSet(s){
if(!s) return;
$('tSpeed').value=s.speed; $('tWord').value=s.word_gap;
$('tSent').value=s.sentence_gap; $('tPitch').value=s.pitch; ttsLabels();
if(s.speed!=null) $('tSpeed').value=s.speed;
if(s.word_gap!=null) $('tWord').value=s.word_gap;
if(s.sentence_gap!=null) $('tSent').value=s.sentence_gap;
if(s.pitch!=null) $('tPitch').value=s.pitch;
ttsLabels();
}
// Effective values for the picked emotion: its override merged over base
// (base itself when 기본 or when the emotion has no override).
function ttsValuesFor(key){
const base = Object.assign({}, TTS_DEFAULT, ttsState.base);
if(key && key!=='base' && ttsState.overrides && ttsState.overrides[key])
return Object.assign({}, base, ttsState.overrides[key]);
return base;
}
function ttsApplyState(j){
if(!j || !j.ok) return;
ttsState = {base:j.base||{}, overrides:j.overrides||{}, emotions:j.emotions||ttsState.emotions};
// Refresh dropdown labels to mark which emotions are customised.
const sel=$('tEmotion'); const cur=sel.value||'base';
sel.innerHTML='';
ttsState.emotions.forEach(e=>{
const o=document.createElement('option'); o.value=e.key;
const custom = e.key!=='base' && ttsState.overrides[e.key];
o.textContent = e.label + (custom ? ' ●' : '');
sel.appendChild(o);
});
sel.value = cur;
ttsSet(ttsValuesFor(sel.value));
}
async function initTts(){
if(ttsInited) return; ttsInited = true;
@@ -1198,6 +1254,9 @@ async function initTts(){
$('ttsCaret').textContent = open ? '▾' : '▸';
};
['tSpeed','tWord','tSent','tPitch'].forEach(id => $(id).addEventListener('input', ttsLabels));
$('tEmotion').onchange = ()=>{ ttsSet(ttsValuesFor($('tEmotion').value));
$('tStat').textContent = $('tEmotion').value==='base'
? '기본(공통) 값을 조절 중입니다.' : '"'+$('tEmotion').selectedOptions[0].textContent.replace(' ●','')+'" 감정만 조절 중입니다.'; };
$('tPreview').onclick = async ()=>{
const text=($('tText').value||'').trim();
if(!text){ $('tStat').textContent='미리들을 문장을 입력하세요.'; return; }
@@ -1212,20 +1271,27 @@ async function initTts(){
}catch(e){ $('tStat').textContent='미리듣기 오류: '+e; }
};
$('tApply').onclick = async ()=>{
const key=$('tEmotion').value||'base';
try{
const r=await fetch('/api/tts/settings',{method:'POST',headers:{'Content-Type':'application/json'},
body:JSON.stringify(ttsVals())});
body:JSON.stringify(Object.assign({emotion:key}, ttsVals()))});
const j=await r.json();
if(j.ok){ ttsSet(j.settings); toast('봇 목소리에 적용됨 · 다음 답변부터 반영'); }
if(j.ok){ ttsApplyState(j); toast((key==='base'?'기본(공통)':'해당 감정')+' 적용됨 · 다음 답변부터 반영'); }
else { $('tStat').textContent='적용 실패: '+(j.error||''); }
}catch(e){ $('tStat').textContent='적용 오류: '+e; }
};
$('tReset').onclick = ()=>{ ttsSet({speed:1.35,word_gap:-0.07,sentence_gap:-0.3,pitch:0});
$('tStat').textContent='기본값으로 되돌림 (미리듣기/적용을 눌러 반영).'; };
try{
const j=await (await fetch('/api/tts/settings')).json();
if(j.ok) ttsSet(j.settings);
}catch(e){}
$('tReset').onclick = async ()=>{
const key=$('tEmotion').value||'base';
if(key==='base'){ ttsSet(TTS_DEFAULT);
$('tStat').textContent='기본값으로 세팅됨 (적용을 눌러 반영).'; return; }
try{ // clear this emotion's override -> it inherits 기본(공통) again
const r=await fetch('/api/tts/settings',{method:'POST',headers:{'Content-Type':'application/json'},
body:JSON.stringify({emotion:key, reset:true})});
const j=await r.json();
if(j.ok){ ttsApplyState(j); toast('이 감정을 기본(공통)값으로 되돌림'); }
}catch(e){ $('tStat').textContent='되돌리기 오류: '+e; }
};
try{ ttsApplyState(await (await fetch('/api/tts/settings')).json()); }catch(e){}
}
// --- Reusable popup/modal (프롬프트, 화이트/블랙리스트 등이 공유) ---------- #