feat(tts): per-emotion voice controls (speed/word-gap/sentence-gap/pitch)
Each emotion can now be tuned independently. parse_segments resolves the four
controls per segment from a base (공통) dict plus an optional per-emotion
override; a missing override key inherits base. By default there are no
overrides, so every emotion delivers with the base values (모든 감정 = 기본값).
- emotion.py: Segment now carries all 4 controls + the canonical emotion name;
parse_segments(text, base, overrides). Adds EMOTION_LABELS/EMOTIONS for the UI.
- melo.py: MeloTTS.emotion_overrides store; synth resolves per-segment controls
and sends them per segment.
- melo_worker.py: _render applies each segment's own word_gap/sentence_gap/pitch
(previously reply-global).
- dashboard.py: emotion dropdown in the TTS panel; GET returns base + overrides
+ emotion list; POST {emotion,...} stores an override (or {reset:true} clears
it); base is set when emotion is omitted/"base".
Verified: an override on one emotion slows only that emotion (happy@0.7=5.66s vs
base 3.02s; sad unchanged at 3.06s); dashboard store/reset/base all work; 34
tests pass.
Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
@@ -460,7 +460,7 @@ class Dashboard:
|
||||
"pitch": (-12.0, 12.0, 1.0),
|
||||
}
|
||||
|
||||
def _tts_values(self) -> dict:
|
||||
def _tts_base(self) -> dict:
|
||||
t = self.tts
|
||||
return {
|
||||
"speed": float(getattr(t, "speed", 1.35)),
|
||||
@@ -470,15 +470,37 @@ class Dashboard:
|
||||
}
|
||||
|
||||
def tts_settings(self) -> dict:
|
||||
return {"ok": True, "settings": self._tts_values(),
|
||||
"ranges": {k: list(v) for k, v in self._TTS_RANGES.items()}}
|
||||
"""Base (공통) controls + the per-emotion overrides + the emotion list
|
||||
(canonical key + Korean label) so the dashboard can offer per-emotion
|
||||
tuning. Emotions with no override inherit base (모든 감정 = 기본값)."""
|
||||
from .backends.emotion import EMOTION_LABELS
|
||||
overrides = dict(getattr(self.tts, "emotion_overrides", {}) or {})
|
||||
return {
|
||||
"ok": True,
|
||||
"base": self._tts_base(),
|
||||
"overrides": overrides,
|
||||
"emotions": [{"key": k, "label": v} for k, v in EMOTION_LABELS.items()],
|
||||
"ranges": {k: list(v) for k, v in self._TTS_RANGES.items()},
|
||||
}
|
||||
|
||||
def set_tts_settings(self, data: dict) -> dict:
|
||||
"""Clamp and apply provided controls to the live TTS instance."""
|
||||
for key, (lo, hi, _step) in self._TTS_RANGES.items():
|
||||
v = data.get(key)
|
||||
if v is not None:
|
||||
setattr(self.tts, key, float(max(lo, min(hi, float(v)))))
|
||||
"""Apply controls. Without ``emotion`` (or emotion == "base"), set the
|
||||
base/공통 values on the live TTS instance. With a specific emotion, store
|
||||
(or, if ``reset``, clear) that emotion's override."""
|
||||
emotion = data.get("emotion")
|
||||
if not emotion or emotion == "base":
|
||||
for key, (lo, hi, _step) in self._TTS_RANGES.items():
|
||||
v = data.get(key)
|
||||
if v is not None:
|
||||
setattr(self.tts, key, float(max(lo, min(hi, float(v)))))
|
||||
else:
|
||||
store = getattr(self.tts, "emotion_overrides", None)
|
||||
if store is None:
|
||||
store = self.tts.emotion_overrides = {}
|
||||
if data.get("reset"):
|
||||
store.pop(emotion, None)
|
||||
else:
|
||||
store[emotion] = self._tts_overrides(data)
|
||||
return self.tts_settings()
|
||||
|
||||
def _tts_overrides(self, data: dict) -> dict:
|
||||
@@ -736,6 +758,7 @@ PAGE = r"""<!DOCTYPE html>
|
||||
.ttsgrid label b{color:var(--fg);font-weight:600}
|
||||
.ttsgrid input[type=range]{width:100%;accent-color:#3aa0ff}
|
||||
.ttstext{flex:1;min-width:180px;background:#0f2333;border:1px solid #234a63;color:var(--fg);border-radius:10px;padding:8px 12px;font-size:13px}
|
||||
.ttssel{background:#0f2333;border:1px solid #234a63;color:var(--fg);border-radius:10px;padding:7px 10px;font-size:13px;margin-left:8px}
|
||||
.sttres .txt{color:var(--heard);font-weight:600;line-height:1.5}
|
||||
.sttres .meta{color:var(--muted);font-size:12px;margin-top:5px}
|
||||
.hbtn{padding:6px 12px;font-size:12.5px}
|
||||
@@ -869,6 +892,11 @@ PAGE = r"""<!DOCTYPE html>
|
||||
<section class="sttbox" id="ttsbox" style="display:none">
|
||||
<h2 id="ttsToggle" class="collap"><span id="ttsCaret">▸</span> 🔊 봇 목소리(TTS) 조절 · MeloTTS</h2>
|
||||
<div id="ttsBody" style="display:none">
|
||||
<div class="sttrow" style="margin-bottom:10px">
|
||||
<label style="font-size:12.5px;color:var(--muted)">감정 선택
|
||||
<select id="tEmotion" class="ttssel"></select></label>
|
||||
<span id="tEmoNote" class="sttstat">감정을 고르면 그 감정만 따로 조절합니다. 기본(공통)은 태그 없는 말투에 적용됩니다.</span>
|
||||
</div>
|
||||
<div class="ttsgrid">
|
||||
<label>글자 속도 <b id="tvSpeed">1.35x</b>
|
||||
<input type="range" id="tSpeed" min="0.5" max="2.0" step="0.05" value="1.35"></label>
|
||||
@@ -1171,8 +1199,10 @@ async function sendBlob(blob){
|
||||
};
|
||||
})();
|
||||
|
||||
// --- 봇 목소리(TTS) 조절: 슬라이더 → 미리듣기 → 봇 적용 -------------------- #
|
||||
// --- 봇 목소리(TTS) 조절: 감정별 슬라이더 → 미리듣기 → 봇 적용 -------------- #
|
||||
let ttsInited = false;
|
||||
let ttsState = {base:{}, overrides:{}, emotions:[]}; // last-known server state
|
||||
const TTS_DEFAULT = {speed:1.35, word_gap:-0.07, sentence_gap:-0.3, pitch:0};
|
||||
function ttsLabels(){
|
||||
const sp=parseFloat($('tSpeed').value), wg=parseFloat($('tWord').value),
|
||||
sg=parseFloat($('tSent').value), pt=parseInt($('tPitch').value,10);
|
||||
@@ -1187,8 +1217,34 @@ function ttsVals(){
|
||||
}
|
||||
function ttsSet(s){
|
||||
if(!s) return;
|
||||
$('tSpeed').value=s.speed; $('tWord').value=s.word_gap;
|
||||
$('tSent').value=s.sentence_gap; $('tPitch').value=s.pitch; ttsLabels();
|
||||
if(s.speed!=null) $('tSpeed').value=s.speed;
|
||||
if(s.word_gap!=null) $('tWord').value=s.word_gap;
|
||||
if(s.sentence_gap!=null) $('tSent').value=s.sentence_gap;
|
||||
if(s.pitch!=null) $('tPitch').value=s.pitch;
|
||||
ttsLabels();
|
||||
}
|
||||
// Effective values for the picked emotion: its override merged over base
|
||||
// (base itself when 기본 or when the emotion has no override).
|
||||
function ttsValuesFor(key){
|
||||
const base = Object.assign({}, TTS_DEFAULT, ttsState.base);
|
||||
if(key && key!=='base' && ttsState.overrides && ttsState.overrides[key])
|
||||
return Object.assign({}, base, ttsState.overrides[key]);
|
||||
return base;
|
||||
}
|
||||
function ttsApplyState(j){
|
||||
if(!j || !j.ok) return;
|
||||
ttsState = {base:j.base||{}, overrides:j.overrides||{}, emotions:j.emotions||ttsState.emotions};
|
||||
// Refresh dropdown labels to mark which emotions are customised.
|
||||
const sel=$('tEmotion'); const cur=sel.value||'base';
|
||||
sel.innerHTML='';
|
||||
ttsState.emotions.forEach(e=>{
|
||||
const o=document.createElement('option'); o.value=e.key;
|
||||
const custom = e.key!=='base' && ttsState.overrides[e.key];
|
||||
o.textContent = e.label + (custom ? ' ●' : '');
|
||||
sel.appendChild(o);
|
||||
});
|
||||
sel.value = cur;
|
||||
ttsSet(ttsValuesFor(sel.value));
|
||||
}
|
||||
async function initTts(){
|
||||
if(ttsInited) return; ttsInited = true;
|
||||
@@ -1198,6 +1254,9 @@ async function initTts(){
|
||||
$('ttsCaret').textContent = open ? '▾' : '▸';
|
||||
};
|
||||
['tSpeed','tWord','tSent','tPitch'].forEach(id => $(id).addEventListener('input', ttsLabels));
|
||||
$('tEmotion').onchange = ()=>{ ttsSet(ttsValuesFor($('tEmotion').value));
|
||||
$('tStat').textContent = $('tEmotion').value==='base'
|
||||
? '기본(공통) 값을 조절 중입니다.' : '"'+$('tEmotion').selectedOptions[0].textContent.replace(' ●','')+'" 감정만 조절 중입니다.'; };
|
||||
$('tPreview').onclick = async ()=>{
|
||||
const text=($('tText').value||'').trim();
|
||||
if(!text){ $('tStat').textContent='미리들을 문장을 입력하세요.'; return; }
|
||||
@@ -1212,20 +1271,27 @@ async function initTts(){
|
||||
}catch(e){ $('tStat').textContent='미리듣기 오류: '+e; }
|
||||
};
|
||||
$('tApply').onclick = async ()=>{
|
||||
const key=$('tEmotion').value||'base';
|
||||
try{
|
||||
const r=await fetch('/api/tts/settings',{method:'POST',headers:{'Content-Type':'application/json'},
|
||||
body:JSON.stringify(ttsVals())});
|
||||
body:JSON.stringify(Object.assign({emotion:key}, ttsVals()))});
|
||||
const j=await r.json();
|
||||
if(j.ok){ ttsSet(j.settings); toast('봇 목소리에 적용됨 · 다음 답변부터 반영'); }
|
||||
if(j.ok){ ttsApplyState(j); toast((key==='base'?'기본(공통)':'해당 감정')+' 적용됨 · 다음 답변부터 반영'); }
|
||||
else { $('tStat').textContent='적용 실패: '+(j.error||''); }
|
||||
}catch(e){ $('tStat').textContent='적용 오류: '+e; }
|
||||
};
|
||||
$('tReset').onclick = ()=>{ ttsSet({speed:1.35,word_gap:-0.07,sentence_gap:-0.3,pitch:0});
|
||||
$('tStat').textContent='기본값으로 되돌림 (미리듣기/적용을 눌러 반영).'; };
|
||||
try{
|
||||
const j=await (await fetch('/api/tts/settings')).json();
|
||||
if(j.ok) ttsSet(j.settings);
|
||||
}catch(e){}
|
||||
$('tReset').onclick = async ()=>{
|
||||
const key=$('tEmotion').value||'base';
|
||||
if(key==='base'){ ttsSet(TTS_DEFAULT);
|
||||
$('tStat').textContent='기본값으로 세팅됨 (적용을 눌러 반영).'; return; }
|
||||
try{ // clear this emotion's override -> it inherits 기본(공통) again
|
||||
const r=await fetch('/api/tts/settings',{method:'POST',headers:{'Content-Type':'application/json'},
|
||||
body:JSON.stringify({emotion:key, reset:true})});
|
||||
const j=await r.json();
|
||||
if(j.ok){ ttsApplyState(j); toast('이 감정을 기본(공통)값으로 되돌림'); }
|
||||
}catch(e){ $('tStat').textContent='되돌리기 오류: '+e; }
|
||||
};
|
||||
try{ ttsApplyState(await (await fetch('/api/tts/settings')).json()); }catch(e){}
|
||||
}
|
||||
|
||||
// --- Reusable popup/modal (프롬프트, 화이트/블랙리스트 등이 공유) ---------- #
|
||||
|
||||
Reference in New Issue
Block a user