feat(dashboard): live TTS voice controls (speed/word-gap/sentence-gap/pitch)

Adds a "봇 목소리(TTS) 조절" panel to the voice-server dashboard so the four
controls can be tuned from the browser instead of only via env vars:

- 4 sliders (glyph speed, word gap, sentence gap, pitch) with live labels
- 미리듣기: synthesises a sample with the slider values WITHOUT changing the
  live bot voice (new per-call overrides on MeloTTS.synth)
- 봇에 적용: commits the slider values to the live TTS instance; next reply uses
  them. 기본값 button resets to the manual defaults.

Backend: GET/POST /api/tts/settings (clamped to the manual ranges) and POST
/api/tts/preview (returns audio/wav). Panel shows only when tts is a real
(non-mock) backend.

Verified end-to-end against a live dashboard instance: page renders the panel,
GET returns defaults, POST applies+clamps (pitch 99->12), preview returns a
valid wav and leaves the live settings unchanged; 29 tests pass.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
EJClaw
2026-08-26 22:48:00 +09:00
parent 0b92284ff8
commit 2ce2806102
2 changed files with 215 additions and 10 deletions

View File

@@ -83,6 +83,8 @@ def _make_handler(dash: "Dashboard"):
q = _up.parse_qs(self.path.split("?", 1)[1] if "?" in self.path else "")
gid = (q.get("guildId", [""])[0])
self._send_json({"ok": True, "guildId": gid, "lists": dash.bot.get_lists(gid)})
elif path == "/api/tts/settings":
self._handle_tts_settings_get()
elif path == "/events":
self._stream_events()
else:
@@ -109,6 +111,10 @@ def _make_handler(dash: "Dashboard"):
self._handle_bot_select()
elif path == "/api/bot/lists":
self._handle_bot_lists()
elif path == "/api/tts/settings":
self._handle_tts_settings_post()
elif path == "/api/tts/preview":
self._handle_tts_preview()
else:
self._send(404, b"not found", "text/plain; charset=utf-8")
@@ -276,6 +282,54 @@ def _make_handler(dash: "Dashboard"):
monitor.log("info", f"청취 화이트/블랙리스트 업데이트 (guild={guild_id})")
self._send_json({"ok": True, "guildId": guild_id, "lists": saved})
def _handle_tts_settings_get(self) -> None:
"""Return the live TTS controls (speed/word_gap/sentence_gap/pitch)
and their slider ranges so the page can render the panel."""
if dash.tts is None:
self._send_json({"ok": False, "error": "TTS not enabled"}, 503)
return
self._send_json(dash.tts_settings())
def _handle_tts_settings_post(self) -> None:
"""Apply new TTS controls to the live bot voice (next reply uses them)."""
if dash.tts is None:
self._send_json({"ok": False, "error": "TTS not enabled"}, 503)
return
raw = self._read_body()
try:
data = json.loads(raw.decode("utf-8")) if raw else {}
except (ValueError, AttributeError):
self._send_json({"ok": False, "error": "invalid JSON"}, 400)
return
res = dash.set_tts_settings(data)
monitor.log("info", "봇 목소리 파라미터 적용: "
+ json.dumps(res["settings"], ensure_ascii=False))
self._send_json(res)
def _handle_tts_preview(self) -> None:
"""Synthesize a short sample with the supplied controls WITHOUT
changing the live bot voice, and return the wav for playback."""
if dash.tts is None:
self._send_json({"ok": False, "error": "TTS not enabled"}, 503)
return
raw = self._read_body()
try:
data = json.loads(raw.decode("utf-8")) if raw else {}
text = (data.get("text") or "").strip()
except (ValueError, AttributeError):
self._send_json({"ok": False, "error": "invalid JSON"}, 400)
return
if not text:
self._send_json({"ok": False, "error": "text required"}, 400)
return
try:
wav = dash.tts_preview(text, data)
except Exception as exc: # noqa: BLE001 — surface the reason to the page
log.exception("TTS preview failed")
self._send_json({"ok": False, "error": f"{type(exc).__name__}: {exc}"}, 500)
return
self._send(200, wav, "audio/wav")
def _handle_log_mutate(self, action: str) -> None:
"""Per-line log delete/edit by event id."""
raw = self._read_body()
@@ -397,6 +451,60 @@ class Dashboard:
if self.tts is not None:
self._submit(self.tts._ensure())
# -- TTS voice controls (dashboard sliders) -------------------------- #
# (min, max, step) for each slider — mirrors the tts_site manual ranges.
_TTS_RANGES = {
"speed": (0.5, 2.0, 0.05),
"word_gap": (-0.2, 0.5, 0.01),
"sentence_gap": (-0.5, 1.5, 0.05),
"pitch": (-12.0, 12.0, 1.0),
}
def _tts_values(self) -> dict:
t = self.tts
return {
"speed": float(getattr(t, "speed", 1.25)),
"word_gap": float(getattr(t, "word_gap", 0.25)),
"sentence_gap": float(getattr(t, "sentence_gap", 0.75)),
"pitch": float(getattr(t, "pitch", 0.0)),
}
def tts_settings(self) -> dict:
return {"ok": True, "settings": self._tts_values(),
"ranges": {k: list(v) for k, v in self._TTS_RANGES.items()}}
def set_tts_settings(self, data: dict) -> dict:
"""Clamp and apply provided controls to the live TTS instance."""
for key, (lo, hi, _step) in self._TTS_RANGES.items():
v = data.get(key)
if v is not None:
setattr(self.tts, key, float(max(lo, min(hi, float(v)))))
return self.tts_settings()
def _tts_overrides(self, data: dict) -> dict:
out = {}
for key, (lo, hi, _step) in self._TTS_RANGES.items():
v = data.get(key)
if v is not None:
out[key] = float(max(lo, min(hi, float(v))))
return out
def tts_preview(self, text: str, data: dict | None = None) -> bytes:
"""Synthesize `text` with per-call control overrides (no change to the
live bot voice) and return the wav bytes."""
import os
if self.tts is None:
raise RuntimeError("TTS not enabled")
overrides = self._tts_overrides(data or {})
out_path = self._submit(self.tts.synth(text, **overrides))
with open(out_path, "rb") as f:
wav = f.read()
try:
os.remove(out_path)
except OSError:
pass
return wav
def voice_turn(self, audio_bytes: bytes, speaker: str = "",
guild: str = "", channel: str = "") -> dict:
"""One Discord voice turn: decode the uploaded utterance, recognise it
@@ -621,6 +729,11 @@ PAGE = r"""<!DOCTYPE html>
.btn.rec{background:#4a1f27;border-color:#7a2531;color:#ffb3bb}
.sttstat{color:var(--muted);font-size:12.5px}
.sttres{margin-top:12px;font-size:15px;min-height:1px}
.ttsgrid{display:grid;grid-template-columns:repeat(auto-fit,minmax(210px,1fr));gap:12px 18px;margin:2px 0 12px}
.ttsgrid label{display:flex;flex-direction:column;gap:6px;font-size:12.5px;color:var(--muted)}
.ttsgrid label b{color:var(--fg);font-weight:600}
.ttsgrid input[type=range]{width:100%;accent-color:#3aa0ff}
.ttstext{flex:1;min-width:180px;background:#0f2333;border:1px solid #234a63;color:var(--fg);border-radius:10px;padding:8px 12px;font-size:13px}
.sttres .txt{color:var(--heard);font-weight:600;line-height:1.5}
.sttres .meta{color:var(--muted);font-size:12px;margin-top:5px}
.hbtn{padding:6px 12px;font-size:12.5px}
@@ -751,6 +864,27 @@ PAGE = r"""<!DOCTYPE html>
</div>
<div id="sttres" class="sttres"></div>
</section>
<section class="sttbox" id="ttsbox" style="display:none">
<h2>🔊 봇 목소리(TTS) 조절 · MeloTTS</h2>
<div class="ttsgrid">
<label>글자 속도 <b id="tvSpeed">1.25x</b>
<input type="range" id="tSpeed" min="0.5" max="2.0" step="0.05" value="1.25"></label>
<label>단어 간격 <b id="tvWord">250 ms</b>
<input type="range" id="tWord" min="-0.2" max="0.5" step="0.01" value="0.25"></label>
<label>문장 간격 <b id="tvSent">0.75 s</b>
<input type="range" id="tSent" min="-0.5" max="1.5" step="0.05" value="0.75"></label>
<label>피치 <b id="tvPitch">0 반음</b>
<input type="range" id="tPitch" min="-12" max="12" step="1" value="0"></label>
</div>
<div class="sttrow">
<input id="tText" class="ttstext" placeholder="미리들을 문장" value="안녕하세요, 목소리 미리듣기입니다.">
<button id="tPreview" class="btn">▶ 미리듣기</button>
<button id="tApply" class="btn">💾 봇에 적용</button>
<button id="tReset" class="btn hbtn">기본값</button>
<span id="tStat" class="sttstat">슬라이더로 조절 → 미리듣기로 확인 → 봇에 적용하면 다음 답변부터 반영됩니다.</span>
</div>
<audio id="tAudio" controls style="display:none;margin-top:10px;width:100%"></audio>
</section>
<div class="demobar" id="demobar" style="display:none"></div>
<div class="comp" id="comp"></div>
<div class="tfilter" id="tfilter">
@@ -839,6 +973,9 @@ function renderStatus(s){
// replayed sample data, not a real conversation. Say so loudly.
const sttReal = comps.stt && comps.stt!=='mock' && comps.stt!=='none';
$('sttbox').style.display = sttReal ? 'block' : 'none';
const ttsReal = comps.tts && comps.tts!=='mock' && comps.tts!=='none';
$('ttsbox').style.display = ttsReal ? 'block' : 'none';
if(ttsReal) initTts();
const mockParts = ['stt','brain','tts'].filter(k => comps[k]==='mock');
const bar = $('demobar');
if(mockParts.length){
@@ -1030,6 +1167,58 @@ async function sendBlob(blob){
};
})();
// --- 봇 목소리(TTS) 조절: 슬라이더 → 미리듣기 → 봇 적용 -------------------- #
let ttsInited = false;
function ttsLabels(){
const sp=parseFloat($('tSpeed').value), wg=parseFloat($('tWord').value),
sg=parseFloat($('tSent').value), pt=parseInt($('tPitch').value,10);
$('tvSpeed').textContent = sp.toFixed(2)+'x';
$('tvWord').textContent = Math.round(wg*1000)+' ms';
$('tvSent').textContent = sg.toFixed(2)+' s';
$('tvPitch').textContent = (pt>0?'+'+pt:pt)+' 반음';
}
function ttsVals(){
return {speed:parseFloat($('tSpeed').value), word_gap:parseFloat($('tWord').value),
sentence_gap:parseFloat($('tSent').value), pitch:parseInt($('tPitch').value,10)};
}
function ttsSet(s){
if(!s) return;
$('tSpeed').value=s.speed; $('tWord').value=s.word_gap;
$('tSent').value=s.sentence_gap; $('tPitch').value=s.pitch; ttsLabels();
}
async function initTts(){
if(ttsInited) return; ttsInited = true;
['tSpeed','tWord','tSent','tPitch'].forEach(id => $(id).addEventListener('input', ttsLabels));
$('tPreview').onclick = async ()=>{
const text=($('tText').value||'').trim();
if(!text){ $('tStat').textContent='미리들을 문장을 입력하세요.'; return; }
$('tStat').textContent='합성 중…';
try{
const r=await fetch('/api/tts/preview',{method:'POST',headers:{'Content-Type':'application/json'},
body:JSON.stringify(Object.assign({text},ttsVals()))});
if(!r.ok){ $('tStat').textContent='미리듣기 실패: '+(await r.text()); return; }
const blob=await r.blob(); const a=$('tAudio');
a.src=URL.createObjectURL(blob); a.style.display='block'; a.play();
$('tStat').textContent='미리듣기 재생 중 (봇에는 아직 미적용).';
}catch(e){ $('tStat').textContent='미리듣기 오류: '+e; }
};
$('tApply').onclick = async ()=>{
try{
const r=await fetch('/api/tts/settings',{method:'POST',headers:{'Content-Type':'application/json'},
body:JSON.stringify(ttsVals())});
const j=await r.json();
if(j.ok){ ttsSet(j.settings); toast('봇 목소리에 적용됨 · 다음 답변부터 반영'); }
else { $('tStat').textContent='적용 실패: '+(j.error||''); }
}catch(e){ $('tStat').textContent='적용 오류: '+e; }
};
$('tReset').onclick = ()=>{ ttsSet({speed:1.25,word_gap:0.25,sentence_gap:0.75,pitch:0});
$('tStat').textContent='기본값으로 되돌림 (미리듣기/적용을 눌러 반영).'; };
try{
const j=await (await fetch('/api/tts/settings')).json();
if(j.ok) ttsSet(j.settings);
}catch(e){}
}
// --- Reusable popup/modal (프롬프트, 화이트/블랙리스트 등이 공유) ---------- #
function openModal(title, actionsHtml){
$('modalTitle').textContent = title;