feat(dashboard): live TTS voice controls (speed/word-gap/sentence-gap/pitch)

Adds a "봇 목소리(TTS) 조절" panel to the voice-server dashboard so the four
controls can be tuned from the browser instead of only via env vars:

- 4 sliders (glyph speed, word gap, sentence gap, pitch) with live labels
- 미리듣기: synthesises a sample with the slider values WITHOUT changing the
  live bot voice (new per-call overrides on MeloTTS.synth)
- 봇에 적용: commits the slider values to the live TTS instance; next reply uses
  them. 기본값 button resets to the manual defaults.

Backend: GET/POST /api/tts/settings (clamped to the manual ranges) and POST
/api/tts/preview (returns audio/wav). Panel shows only when tts is a real
(non-mock) backend.

Verified end-to-end against a live dashboard instance: page renders the panel,
GET returns defaults, POST applies+clamps (pitch 99->12), preview returns a
valid wav and leaves the live settings unchanged; 29 tests pass.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
EJClaw
2026-08-26 22:48:00 +09:00
parent 0b92284ff8
commit 2ce2806102
2 changed files with 215 additions and 10 deletions

View File

@@ -198,14 +198,30 @@ class MeloTTS:
self._ready = True self._ready = True
log.info("melo worker ready in %s ms on %s", self.load_ms, info.get("device")) log.info("melo worker ready in %s ms on %s", self.load_ms, info.get("device"))
async def synth(self, text: str) -> str: async def synth(
self,
text: str,
*,
speed: float | None = None,
word_gap: float | None = None,
sentence_gap: float | None = None,
pitch: float | None = None,
) -> str:
"""Synthesize `text` to a wav and return its path (no sink). Reusable by """Synthesize `text` to a wav and return its path (no sink). Reusable by
callers that want the wav directly (e.g. the Discord voice bridge).""" callers that want the wav directly (e.g. the Discord voice bridge).
The four controls default to the instance settings but may be overridden
per call (used by the dashboard preview so tuning does not disturb the
live bot voice until explicitly applied)."""
await self._ensure() await self._ensure()
spd = self.speed if speed is None else float(speed)
wg = self.word_gap if word_gap is None else float(word_gap)
sg = self.sentence_gap if sentence_gap is None else float(sentence_gap)
pt = self.pitch if pitch is None else float(pitch)
text = normalize_for_speech(text) text = normalize_for_speech(text)
# Split on [감정] tags: each tag steers pitch/speed for the text that # Split on [감정] tags: each tag steers pitch/speed for the text that
# follows (and is itself not spoken); non-emotion brackets stay as words. # follows (and is itself not spoken); non-emotion brackets stay as words.
segments = parse_segments(text, self.speed) segments = parse_segments(text, spd)
self._n += 1 self._n += 1
out = str(self.out_dir / f"tts-{self._n:06d}.wav") out = str(self.out_dir / f"tts-{self._n:06d}.wav")
if segments: if segments:
@@ -215,18 +231,18 @@ class MeloTTS:
for s in segments for s in segments
], ],
"out": out, "out": out,
"word_gap": self.word_gap, "word_gap": wg,
"sentence_gap": self.sentence_gap, "sentence_gap": sg,
"pitch": self.pitch, "pitch": pt,
} }
else: # empty/whitespace reply: keep legacy single-utterance behaviour else: # empty/whitespace reply: keep legacy single-utterance behaviour
payload = { payload = {
"text": text, "text": text,
"out": out, "out": out,
"speed": self.speed, "speed": spd,
"word_gap": self.word_gap, "word_gap": wg,
"sentence_gap": self.sentence_gap, "sentence_gap": sg,
"pitch": self.pitch, "pitch": pt,
} }
req = json.dumps(payload) req = json.dumps(payload)
s = time.monotonic() s = time.monotonic()

View File

@@ -83,6 +83,8 @@ def _make_handler(dash: "Dashboard"):
q = _up.parse_qs(self.path.split("?", 1)[1] if "?" in self.path else "") q = _up.parse_qs(self.path.split("?", 1)[1] if "?" in self.path else "")
gid = (q.get("guildId", [""])[0]) gid = (q.get("guildId", [""])[0])
self._send_json({"ok": True, "guildId": gid, "lists": dash.bot.get_lists(gid)}) self._send_json({"ok": True, "guildId": gid, "lists": dash.bot.get_lists(gid)})
elif path == "/api/tts/settings":
self._handle_tts_settings_get()
elif path == "/events": elif path == "/events":
self._stream_events() self._stream_events()
else: else:
@@ -109,6 +111,10 @@ def _make_handler(dash: "Dashboard"):
self._handle_bot_select() self._handle_bot_select()
elif path == "/api/bot/lists": elif path == "/api/bot/lists":
self._handle_bot_lists() self._handle_bot_lists()
elif path == "/api/tts/settings":
self._handle_tts_settings_post()
elif path == "/api/tts/preview":
self._handle_tts_preview()
else: else:
self._send(404, b"not found", "text/plain; charset=utf-8") self._send(404, b"not found", "text/plain; charset=utf-8")
@@ -276,6 +282,54 @@ def _make_handler(dash: "Dashboard"):
monitor.log("info", f"청취 화이트/블랙리스트 업데이트 (guild={guild_id})") monitor.log("info", f"청취 화이트/블랙리스트 업데이트 (guild={guild_id})")
self._send_json({"ok": True, "guildId": guild_id, "lists": saved}) self._send_json({"ok": True, "guildId": guild_id, "lists": saved})
def _handle_tts_settings_get(self) -> None:
"""Return the live TTS controls (speed/word_gap/sentence_gap/pitch)
and their slider ranges so the page can render the panel."""
if dash.tts is None:
self._send_json({"ok": False, "error": "TTS not enabled"}, 503)
return
self._send_json(dash.tts_settings())
def _handle_tts_settings_post(self) -> None:
"""Apply new TTS controls to the live bot voice (next reply uses them)."""
if dash.tts is None:
self._send_json({"ok": False, "error": "TTS not enabled"}, 503)
return
raw = self._read_body()
try:
data = json.loads(raw.decode("utf-8")) if raw else {}
except (ValueError, AttributeError):
self._send_json({"ok": False, "error": "invalid JSON"}, 400)
return
res = dash.set_tts_settings(data)
monitor.log("info", "봇 목소리 파라미터 적용: "
+ json.dumps(res["settings"], ensure_ascii=False))
self._send_json(res)
def _handle_tts_preview(self) -> None:
"""Synthesize a short sample with the supplied controls WITHOUT
changing the live bot voice, and return the wav for playback."""
if dash.tts is None:
self._send_json({"ok": False, "error": "TTS not enabled"}, 503)
return
raw = self._read_body()
try:
data = json.loads(raw.decode("utf-8")) if raw else {}
text = (data.get("text") or "").strip()
except (ValueError, AttributeError):
self._send_json({"ok": False, "error": "invalid JSON"}, 400)
return
if not text:
self._send_json({"ok": False, "error": "text required"}, 400)
return
try:
wav = dash.tts_preview(text, data)
except Exception as exc: # noqa: BLE001 — surface the reason to the page
log.exception("TTS preview failed")
self._send_json({"ok": False, "error": f"{type(exc).__name__}: {exc}"}, 500)
return
self._send(200, wav, "audio/wav")
def _handle_log_mutate(self, action: str) -> None: def _handle_log_mutate(self, action: str) -> None:
"""Per-line log delete/edit by event id.""" """Per-line log delete/edit by event id."""
raw = self._read_body() raw = self._read_body()
@@ -397,6 +451,60 @@ class Dashboard:
if self.tts is not None: if self.tts is not None:
self._submit(self.tts._ensure()) self._submit(self.tts._ensure())
# -- TTS voice controls (dashboard sliders) -------------------------- #
# (min, max, step) for each slider — mirrors the tts_site manual ranges.
_TTS_RANGES = {
"speed": (0.5, 2.0, 0.05),
"word_gap": (-0.2, 0.5, 0.01),
"sentence_gap": (-0.5, 1.5, 0.05),
"pitch": (-12.0, 12.0, 1.0),
}
def _tts_values(self) -> dict:
t = self.tts
return {
"speed": float(getattr(t, "speed", 1.25)),
"word_gap": float(getattr(t, "word_gap", 0.25)),
"sentence_gap": float(getattr(t, "sentence_gap", 0.75)),
"pitch": float(getattr(t, "pitch", 0.0)),
}
def tts_settings(self) -> dict:
return {"ok": True, "settings": self._tts_values(),
"ranges": {k: list(v) for k, v in self._TTS_RANGES.items()}}
def set_tts_settings(self, data: dict) -> dict:
"""Clamp and apply provided controls to the live TTS instance."""
for key, (lo, hi, _step) in self._TTS_RANGES.items():
v = data.get(key)
if v is not None:
setattr(self.tts, key, float(max(lo, min(hi, float(v)))))
return self.tts_settings()
def _tts_overrides(self, data: dict) -> dict:
out = {}
for key, (lo, hi, _step) in self._TTS_RANGES.items():
v = data.get(key)
if v is not None:
out[key] = float(max(lo, min(hi, float(v))))
return out
def tts_preview(self, text: str, data: dict | None = None) -> bytes:
"""Synthesize `text` with per-call control overrides (no change to the
live bot voice) and return the wav bytes."""
import os
if self.tts is None:
raise RuntimeError("TTS not enabled")
overrides = self._tts_overrides(data or {})
out_path = self._submit(self.tts.synth(text, **overrides))
with open(out_path, "rb") as f:
wav = f.read()
try:
os.remove(out_path)
except OSError:
pass
return wav
def voice_turn(self, audio_bytes: bytes, speaker: str = "", def voice_turn(self, audio_bytes: bytes, speaker: str = "",
guild: str = "", channel: str = "") -> dict: guild: str = "", channel: str = "") -> dict:
"""One Discord voice turn: decode the uploaded utterance, recognise it """One Discord voice turn: decode the uploaded utterance, recognise it
@@ -621,6 +729,11 @@ PAGE = r"""<!DOCTYPE html>
.btn.rec{background:#4a1f27;border-color:#7a2531;color:#ffb3bb} .btn.rec{background:#4a1f27;border-color:#7a2531;color:#ffb3bb}
.sttstat{color:var(--muted);font-size:12.5px} .sttstat{color:var(--muted);font-size:12.5px}
.sttres{margin-top:12px;font-size:15px;min-height:1px} .sttres{margin-top:12px;font-size:15px;min-height:1px}
.ttsgrid{display:grid;grid-template-columns:repeat(auto-fit,minmax(210px,1fr));gap:12px 18px;margin:2px 0 12px}
.ttsgrid label{display:flex;flex-direction:column;gap:6px;font-size:12.5px;color:var(--muted)}
.ttsgrid label b{color:var(--fg);font-weight:600}
.ttsgrid input[type=range]{width:100%;accent-color:#3aa0ff}
.ttstext{flex:1;min-width:180px;background:#0f2333;border:1px solid #234a63;color:var(--fg);border-radius:10px;padding:8px 12px;font-size:13px}
.sttres .txt{color:var(--heard);font-weight:600;line-height:1.5} .sttres .txt{color:var(--heard);font-weight:600;line-height:1.5}
.sttres .meta{color:var(--muted);font-size:12px;margin-top:5px} .sttres .meta{color:var(--muted);font-size:12px;margin-top:5px}
.hbtn{padding:6px 12px;font-size:12.5px} .hbtn{padding:6px 12px;font-size:12.5px}
@@ -751,6 +864,27 @@ PAGE = r"""<!DOCTYPE html>
</div> </div>
<div id="sttres" class="sttres"></div> <div id="sttres" class="sttres"></div>
</section> </section>
<section class="sttbox" id="ttsbox" style="display:none">
<h2>🔊 봇 목소리(TTS) 조절 · MeloTTS</h2>
<div class="ttsgrid">
<label>글자 속도 <b id="tvSpeed">1.25x</b>
<input type="range" id="tSpeed" min="0.5" max="2.0" step="0.05" value="1.25"></label>
<label>단어 간격 <b id="tvWord">250 ms</b>
<input type="range" id="tWord" min="-0.2" max="0.5" step="0.01" value="0.25"></label>
<label>문장 간격 <b id="tvSent">0.75 s</b>
<input type="range" id="tSent" min="-0.5" max="1.5" step="0.05" value="0.75"></label>
<label>피치 <b id="tvPitch">0 반음</b>
<input type="range" id="tPitch" min="-12" max="12" step="1" value="0"></label>
</div>
<div class="sttrow">
<input id="tText" class="ttstext" placeholder="미리들을 문장" value="안녕하세요, 목소리 미리듣기입니다.">
<button id="tPreview" class="btn">▶ 미리듣기</button>
<button id="tApply" class="btn">💾 봇에 적용</button>
<button id="tReset" class="btn hbtn">기본값</button>
<span id="tStat" class="sttstat">슬라이더로 조절 → 미리듣기로 확인 → 봇에 적용하면 다음 답변부터 반영됩니다.</span>
</div>
<audio id="tAudio" controls style="display:none;margin-top:10px;width:100%"></audio>
</section>
<div class="demobar" id="demobar" style="display:none"></div> <div class="demobar" id="demobar" style="display:none"></div>
<div class="comp" id="comp"></div> <div class="comp" id="comp"></div>
<div class="tfilter" id="tfilter"> <div class="tfilter" id="tfilter">
@@ -839,6 +973,9 @@ function renderStatus(s){
// replayed sample data, not a real conversation. Say so loudly. // replayed sample data, not a real conversation. Say so loudly.
const sttReal = comps.stt && comps.stt!=='mock' && comps.stt!=='none'; const sttReal = comps.stt && comps.stt!=='mock' && comps.stt!=='none';
$('sttbox').style.display = sttReal ? 'block' : 'none'; $('sttbox').style.display = sttReal ? 'block' : 'none';
const ttsReal = comps.tts && comps.tts!=='mock' && comps.tts!=='none';
$('ttsbox').style.display = ttsReal ? 'block' : 'none';
if(ttsReal) initTts();
const mockParts = ['stt','brain','tts'].filter(k => comps[k]==='mock'); const mockParts = ['stt','brain','tts'].filter(k => comps[k]==='mock');
const bar = $('demobar'); const bar = $('demobar');
if(mockParts.length){ if(mockParts.length){
@@ -1030,6 +1167,58 @@ async function sendBlob(blob){
}; };
})(); })();
// --- 봇 목소리(TTS) 조절: 슬라이더 → 미리듣기 → 봇 적용 -------------------- #
let ttsInited = false;
function ttsLabels(){
const sp=parseFloat($('tSpeed').value), wg=parseFloat($('tWord').value),
sg=parseFloat($('tSent').value), pt=parseInt($('tPitch').value,10);
$('tvSpeed').textContent = sp.toFixed(2)+'x';
$('tvWord').textContent = Math.round(wg*1000)+' ms';
$('tvSent').textContent = sg.toFixed(2)+' s';
$('tvPitch').textContent = (pt>0?'+'+pt:pt)+' 반음';
}
function ttsVals(){
return {speed:parseFloat($('tSpeed').value), word_gap:parseFloat($('tWord').value),
sentence_gap:parseFloat($('tSent').value), pitch:parseInt($('tPitch').value,10)};
}
function ttsSet(s){
if(!s) return;
$('tSpeed').value=s.speed; $('tWord').value=s.word_gap;
$('tSent').value=s.sentence_gap; $('tPitch').value=s.pitch; ttsLabels();
}
async function initTts(){
if(ttsInited) return; ttsInited = true;
['tSpeed','tWord','tSent','tPitch'].forEach(id => $(id).addEventListener('input', ttsLabels));
$('tPreview').onclick = async ()=>{
const text=($('tText').value||'').trim();
if(!text){ $('tStat').textContent='미리들을 문장을 입력하세요.'; return; }
$('tStat').textContent='합성 중…';
try{
const r=await fetch('/api/tts/preview',{method:'POST',headers:{'Content-Type':'application/json'},
body:JSON.stringify(Object.assign({text},ttsVals()))});
if(!r.ok){ $('tStat').textContent='미리듣기 실패: '+(await r.text()); return; }
const blob=await r.blob(); const a=$('tAudio');
a.src=URL.createObjectURL(blob); a.style.display='block'; a.play();
$('tStat').textContent='미리듣기 재생 중 (봇에는 아직 미적용).';
}catch(e){ $('tStat').textContent='미리듣기 오류: '+e; }
};
$('tApply').onclick = async ()=>{
try{
const r=await fetch('/api/tts/settings',{method:'POST',headers:{'Content-Type':'application/json'},
body:JSON.stringify(ttsVals())});
const j=await r.json();
if(j.ok){ ttsSet(j.settings); toast('봇 목소리에 적용됨 · 다음 답변부터 반영'); }
else { $('tStat').textContent='적용 실패: '+(j.error||''); }
}catch(e){ $('tStat').textContent='적용 오류: '+e; }
};
$('tReset').onclick = ()=>{ ttsSet({speed:1.25,word_gap:0.25,sentence_gap:0.75,pitch:0});
$('tStat').textContent='기본값으로 되돌림 (미리듣기/적용을 눌러 반영).'; };
try{
const j=await (await fetch('/api/tts/settings')).json();
if(j.ok) ttsSet(j.settings);
}catch(e){}
}
// --- Reusable popup/modal (프롬프트, 화이트/블랙리스트 등이 공유) ---------- # // --- Reusable popup/modal (프롬프트, 화이트/블랙리스트 등이 공유) ---------- #
function openModal(title, actionsHtml){ function openModal(title, actionsHtml){
$('modalTitle').textContent = title; $('modalTitle').textContent = title;