diff --git a/wsai/backends/melo.py b/wsai/backends/melo.py index 54caf38..f402dc3 100644 --- a/wsai/backends/melo.py +++ b/wsai/backends/melo.py @@ -198,14 +198,30 @@ class MeloTTS: self._ready = True log.info("melo worker ready in %s ms on %s", self.load_ms, info.get("device")) - async def synth(self, text: str) -> str: + async def synth( + self, + text: str, + *, + speed: float | None = None, + word_gap: float | None = None, + sentence_gap: float | None = None, + pitch: float | None = None, + ) -> str: """Synthesize `text` to a wav and return its path (no sink). Reusable by - callers that want the wav directly (e.g. the Discord voice bridge).""" + callers that want the wav directly (e.g. the Discord voice bridge). + + The four controls default to the instance settings but may be overridden + per call (used by the dashboard preview so tuning does not disturb the + live bot voice until explicitly applied).""" await self._ensure() + spd = self.speed if speed is None else float(speed) + wg = self.word_gap if word_gap is None else float(word_gap) + sg = self.sentence_gap if sentence_gap is None else float(sentence_gap) + pt = self.pitch if pitch is None else float(pitch) text = normalize_for_speech(text) # Split on [감정] tags: each tag steers pitch/speed for the text that # follows (and is itself not spoken); non-emotion brackets stay as words. - segments = parse_segments(text, self.speed) + segments = parse_segments(text, spd) self._n += 1 out = str(self.out_dir / f"tts-{self._n:06d}.wav") if segments: @@ -215,18 +231,18 @@ class MeloTTS: for s in segments ], "out": out, - "word_gap": self.word_gap, - "sentence_gap": self.sentence_gap, - "pitch": self.pitch, + "word_gap": wg, + "sentence_gap": sg, + "pitch": pt, } else: # empty/whitespace reply: keep legacy single-utterance behaviour payload = { "text": text, "out": out, - "speed": self.speed, - "word_gap": self.word_gap, - "sentence_gap": self.sentence_gap, - "pitch": self.pitch, + "speed": spd, + "word_gap": wg, + "sentence_gap": sg, + "pitch": pt, } req = json.dumps(payload) s = time.monotonic() diff --git a/wsai/dashboard.py b/wsai/dashboard.py index 59c816e..3915637 100644 --- a/wsai/dashboard.py +++ b/wsai/dashboard.py @@ -83,6 +83,8 @@ def _make_handler(dash: "Dashboard"): q = _up.parse_qs(self.path.split("?", 1)[1] if "?" in self.path else "") gid = (q.get("guildId", [""])[0]) self._send_json({"ok": True, "guildId": gid, "lists": dash.bot.get_lists(gid)}) + elif path == "/api/tts/settings": + self._handle_tts_settings_get() elif path == "/events": self._stream_events() else: @@ -109,6 +111,10 @@ def _make_handler(dash: "Dashboard"): self._handle_bot_select() elif path == "/api/bot/lists": self._handle_bot_lists() + elif path == "/api/tts/settings": + self._handle_tts_settings_post() + elif path == "/api/tts/preview": + self._handle_tts_preview() else: self._send(404, b"not found", "text/plain; charset=utf-8") @@ -276,6 +282,54 @@ def _make_handler(dash: "Dashboard"): monitor.log("info", f"청취 화이트/블랙리스트 업데이트 (guild={guild_id})") self._send_json({"ok": True, "guildId": guild_id, "lists": saved}) + def _handle_tts_settings_get(self) -> None: + """Return the live TTS controls (speed/word_gap/sentence_gap/pitch) + and their slider ranges so the page can render the panel.""" + if dash.tts is None: + self._send_json({"ok": False, "error": "TTS not enabled"}, 503) + return + self._send_json(dash.tts_settings()) + + def _handle_tts_settings_post(self) -> None: + """Apply new TTS controls to the live bot voice (next reply uses them).""" + if dash.tts is None: + self._send_json({"ok": False, "error": "TTS not enabled"}, 503) + return + raw = self._read_body() + try: + data = json.loads(raw.decode("utf-8")) if raw else {} + except (ValueError, AttributeError): + self._send_json({"ok": False, "error": "invalid JSON"}, 400) + return + res = dash.set_tts_settings(data) + monitor.log("info", "봇 목소리 파라미터 적용: " + + json.dumps(res["settings"], ensure_ascii=False)) + self._send_json(res) + + def _handle_tts_preview(self) -> None: + """Synthesize a short sample with the supplied controls WITHOUT + changing the live bot voice, and return the wav for playback.""" + if dash.tts is None: + self._send_json({"ok": False, "error": "TTS not enabled"}, 503) + return + raw = self._read_body() + try: + data = json.loads(raw.decode("utf-8")) if raw else {} + text = (data.get("text") or "").strip() + except (ValueError, AttributeError): + self._send_json({"ok": False, "error": "invalid JSON"}, 400) + return + if not text: + self._send_json({"ok": False, "error": "text required"}, 400) + return + try: + wav = dash.tts_preview(text, data) + except Exception as exc: # noqa: BLE001 — surface the reason to the page + log.exception("TTS preview failed") + self._send_json({"ok": False, "error": f"{type(exc).__name__}: {exc}"}, 500) + return + self._send(200, wav, "audio/wav") + def _handle_log_mutate(self, action: str) -> None: """Per-line log delete/edit by event id.""" raw = self._read_body() @@ -397,6 +451,60 @@ class Dashboard: if self.tts is not None: self._submit(self.tts._ensure()) + # -- TTS voice controls (dashboard sliders) -------------------------- # + # (min, max, step) for each slider — mirrors the tts_site manual ranges. + _TTS_RANGES = { + "speed": (0.5, 2.0, 0.05), + "word_gap": (-0.2, 0.5, 0.01), + "sentence_gap": (-0.5, 1.5, 0.05), + "pitch": (-12.0, 12.0, 1.0), + } + + def _tts_values(self) -> dict: + t = self.tts + return { + "speed": float(getattr(t, "speed", 1.25)), + "word_gap": float(getattr(t, "word_gap", 0.25)), + "sentence_gap": float(getattr(t, "sentence_gap", 0.75)), + "pitch": float(getattr(t, "pitch", 0.0)), + } + + def tts_settings(self) -> dict: + return {"ok": True, "settings": self._tts_values(), + "ranges": {k: list(v) for k, v in self._TTS_RANGES.items()}} + + def set_tts_settings(self, data: dict) -> dict: + """Clamp and apply provided controls to the live TTS instance.""" + for key, (lo, hi, _step) in self._TTS_RANGES.items(): + v = data.get(key) + if v is not None: + setattr(self.tts, key, float(max(lo, min(hi, float(v))))) + return self.tts_settings() + + def _tts_overrides(self, data: dict) -> dict: + out = {} + for key, (lo, hi, _step) in self._TTS_RANGES.items(): + v = data.get(key) + if v is not None: + out[key] = float(max(lo, min(hi, float(v)))) + return out + + def tts_preview(self, text: str, data: dict | None = None) -> bytes: + """Synthesize `text` with per-call control overrides (no change to the + live bot voice) and return the wav bytes.""" + import os + if self.tts is None: + raise RuntimeError("TTS not enabled") + overrides = self._tts_overrides(data or {}) + out_path = self._submit(self.tts.synth(text, **overrides)) + with open(out_path, "rb") as f: + wav = f.read() + try: + os.remove(out_path) + except OSError: + pass + return wav + def voice_turn(self, audio_bytes: bytes, speaker: str = "", guild: str = "", channel: str = "") -> dict: """One Discord voice turn: decode the uploaded utterance, recognise it @@ -621,6 +729,11 @@ PAGE = r""" .btn.rec{background:#4a1f27;border-color:#7a2531;color:#ffb3bb} .sttstat{color:var(--muted);font-size:12.5px} .sttres{margin-top:12px;font-size:15px;min-height:1px} + .ttsgrid{display:grid;grid-template-columns:repeat(auto-fit,minmax(210px,1fr));gap:12px 18px;margin:2px 0 12px} + .ttsgrid label{display:flex;flex-direction:column;gap:6px;font-size:12.5px;color:var(--muted)} + .ttsgrid label b{color:var(--fg);font-weight:600} + .ttsgrid input[type=range]{width:100%;accent-color:#3aa0ff} + .ttstext{flex:1;min-width:180px;background:#0f2333;border:1px solid #234a63;color:var(--fg);border-radius:10px;padding:8px 12px;font-size:13px} .sttres .txt{color:var(--heard);font-weight:600;line-height:1.5} .sttres .meta{color:var(--muted);font-size:12px;margin-top:5px} .hbtn{padding:6px 12px;font-size:12.5px} @@ -751,6 +864,27 @@ PAGE = r"""
+