Compare commits
7 Commits
361dce70bb
...
b0272d2171
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b0272d2171 | ||
|
|
b782ca70bd | ||
|
|
067efc7abe | ||
|
|
39b743d976 | ||
|
|
2ce2806102 | ||
|
|
0b92284ff8 | ||
|
|
a207ae05c5 |
@@ -177,6 +177,7 @@ let currentGuildId = null, currentChannelId = null, currentChannelName = null;
|
||||
const speakingSet = new Set(); // userIds currently speaking (for participant list)
|
||||
const activeSubs = new Set(); // userIds with an in-flight receive subscription
|
||||
let listsByGuild = {}; // guildId -> {whitelistUsers, blacklistUsers, whitelistRoles, blacklistRoles}
|
||||
let botSettings = { bargeIn: true }; // behaviour toggles from the dashboard
|
||||
|
||||
// Attach the bot's audio player (so it can speak) to a fresh connection.
|
||||
function setupPlayer(connection) {
|
||||
@@ -201,6 +202,11 @@ function setupReceiver(connection) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
// Barge-in: the moment an allowed user speaks, stop the bot's current TTS
|
||||
// so it doesn't talk over them (toggleable from the dashboard, default on).
|
||||
if (botSettings.bargeIn !== false && voicePlayer) {
|
||||
try { voicePlayer.stop(true); } catch {}
|
||||
}
|
||||
activeSubs.add(userId);
|
||||
if (!perUser.has(userId)) perUser.set(userId, { opusPackets: 0, pcmFrames: 0 });
|
||||
const opusStream = receiver.subscribe(userId, {
|
||||
@@ -314,6 +320,7 @@ async function reportLoop() {
|
||||
j = await r.json();
|
||||
} catch { return; } // dashboard down: keep running, retry next tick
|
||||
if (j?.lists) listsByGuild = j.lists; // latest whitelist/blacklist config
|
||||
if (j?.settings) botSettings = j.settings; // latest behaviour toggles
|
||||
for (const cmd of (j?.commands || [])) {
|
||||
try { await handleCommand(cmd); } catch (e) { log(`command ${cmd?.type} failed: ${e.message}`); }
|
||||
}
|
||||
|
||||
@@ -1,56 +1,82 @@
|
||||
"""Emotion-tag parsing for expressive TTS. A ``[감정]`` tag must steer the
|
||||
pitch/speed of the text that follows without being spoken; a bracket that is NOT
|
||||
a known emotion word must be kept as ordinary spoken content."""
|
||||
delivery of the text that follows without being spoken; a bracket that is NOT a
|
||||
known emotion word must be kept as ordinary spoken content.
|
||||
|
||||
Delivery is now driven by a base (공통) control dict plus optional per-emotion
|
||||
overrides. By default there are no overrides, so every emotion delivers with the
|
||||
base controls (모든 감정 = 기본값)."""
|
||||
|
||||
from wsai.backends.emotion import (
|
||||
EMOTION_PARAMS,
|
||||
EMOTION_LABELS,
|
||||
_SYNONYMS,
|
||||
match_emotion,
|
||||
parse_segments,
|
||||
)
|
||||
from wsai.dashboard import _speech_text
|
||||
|
||||
BASE = 1.3
|
||||
BASE = {"speed": 1.35, "word_gap": -0.07, "sentence_gap": -0.30, "pitch": 0.0}
|
||||
|
||||
|
||||
def test_emotion_tag_is_not_spoken_and_sets_delivery():
|
||||
def test_emotion_tag_is_not_spoken_and_tags_segment():
|
||||
segs = parse_segments("[기쁨] 오늘 날씨 좋다", BASE)
|
||||
assert len(segs) == 1
|
||||
assert "기쁨" not in segs[0].text # the tag word is dropped
|
||||
assert segs[0].text == "오늘 날씨 좋다"
|
||||
mult, semis = EMOTION_PARAMS["happy"]
|
||||
assert segs[0].speed == BASE * mult
|
||||
assert segs[0].pitch == semis
|
||||
assert segs[0].emotion == "happy"
|
||||
|
||||
|
||||
def test_default_all_emotions_use_base():
|
||||
# No overrides -> a tagged emotion delivers with exactly the base controls.
|
||||
segs = parse_segments("[기쁨] 좋아", BASE)
|
||||
s = segs[0]
|
||||
assert (s.speed, s.word_gap, s.sentence_gap, s.pitch) == (1.35, -0.07, -0.30, 0.0)
|
||||
|
||||
|
||||
def test_per_emotion_override_applies_and_inherits_missing_keys():
|
||||
segs = parse_segments("[기쁨] 좋아", BASE, {"happy": {"speed": 1.6, "pitch": 3.0}})
|
||||
s = segs[0]
|
||||
assert s.speed == 1.6 and s.pitch == 3.0 # overridden
|
||||
assert s.word_gap == -0.07 and s.sentence_gap == -0.30 # inherited from base
|
||||
|
||||
|
||||
def test_override_only_affects_its_own_emotion():
|
||||
segs = parse_segments("[기쁨] 가. [슬픔] 나.", BASE, {"sad": {"speed": 0.8}})
|
||||
assert segs[0].emotion == "happy" and segs[0].speed == 1.35 # untouched
|
||||
assert segs[1].emotion == "sad" and segs[1].speed == 0.8 # overridden
|
||||
|
||||
|
||||
def test_midreply_emotion_change_splits_segments():
|
||||
segs = parse_segments("[속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!", BASE)
|
||||
assert len(segs) == 2
|
||||
assert segs[0].text == "정말 힘들었겠다."
|
||||
assert segs[1].text == "하지만 넌 할 수 있어!"
|
||||
assert segs[0].pitch == EMOTION_PARAMS["sad"][1]
|
||||
assert segs[1].pitch == EMOTION_PARAMS["hopeful"][1]
|
||||
assert segs[0].text == "정말 힘들었겠다." and segs[0].emotion == "sad"
|
||||
assert segs[1].text == "하지만 넌 할 수 있어!" and segs[1].emotion == "hopeful"
|
||||
|
||||
|
||||
def test_non_emotion_bracket_is_spoken_without_brackets():
|
||||
segs = parse_segments("[기쁨] 첫째는 [1번] 항목이야", BASE)
|
||||
assert len(segs) == 1
|
||||
# "1번" is not an emotion -> read it; brackets themselves are gone.
|
||||
assert "1번" in segs[0].text
|
||||
assert "[" not in segs[0].text and "]" not in segs[0].text
|
||||
assert segs[0].pitch == EMOTION_PARAMS["happy"][1]
|
||||
assert segs[0].emotion == "happy"
|
||||
|
||||
|
||||
def test_text_before_first_tag_is_neutral():
|
||||
segs = parse_segments("잠깐만. [신남] 찾았다!", BASE)
|
||||
assert segs[0].text == "잠깐만."
|
||||
assert segs[0].speed == BASE and segs[0].pitch == 0.0
|
||||
assert segs[1].pitch == EMOTION_PARAMS["excited"][1]
|
||||
assert segs[0].text == "잠깐만." and segs[0].emotion is None
|
||||
assert segs[0].speed == 1.35 and segs[0].pitch == 0.0
|
||||
assert segs[1].emotion == "excited"
|
||||
|
||||
|
||||
def test_plain_text_is_one_neutral_segment():
|
||||
segs = parse_segments("그냥 평범한 문장이야", BASE)
|
||||
assert len(segs) == 1
|
||||
assert segs[0].speed == BASE and segs[0].pitch == 0.0
|
||||
assert segs[0].emotion is None and segs[0].speed == 1.35
|
||||
|
||||
|
||||
def test_backward_compat_float_base():
|
||||
# A bare float base still works (speed only; other controls neutral).
|
||||
segs = parse_segments("그냥 문장", 1.2)
|
||||
assert segs[0].speed == 1.2 and segs[0].word_gap == 0.0
|
||||
|
||||
|
||||
def test_empty_input_yields_no_segments():
|
||||
@@ -77,6 +103,13 @@ def test_persona_examples_are_all_recognised():
|
||||
assert match_emotion(word) is not None, word
|
||||
|
||||
|
||||
def test_every_canonical_emotion_has_a_ui_label():
|
||||
# The dashboard dropdown lists EMOTION_LABELS; every emotion a tag can
|
||||
# resolve to must appear there or it would be untunable.
|
||||
for canon in set(_SYNONYMS.values()):
|
||||
assert canon in EMOTION_LABELS, canon
|
||||
|
||||
|
||||
def test_voice_turn_keeps_leading_emotion_tag_for_tts_parser():
|
||||
text = "[속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!"
|
||||
|
||||
@@ -84,7 +117,5 @@ def test_voice_turn_keeps_leading_emotion_tag_for_tts_parser():
|
||||
segs = parse_segments(spoken, BASE)
|
||||
|
||||
assert spoken == text
|
||||
assert segs[0].text == "정말 힘들었겠다."
|
||||
assert segs[0].pitch == EMOTION_PARAMS["sad"][1]
|
||||
assert segs[1].text == "하지만 넌 할 수 있어!"
|
||||
assert segs[1].pitch == EMOTION_PARAMS["hopeful"][1]
|
||||
assert segs[0].text == "정말 힘들었겠다." and segs[0].emotion == "sad"
|
||||
assert segs[1].text == "하지만 넌 할 수 있어!" and segs[1].emotion == "hopeful"
|
||||
|
||||
@@ -130,8 +130,9 @@ class ClaudeBrain:
|
||||
"- 음성 대화에 어울리게 짧고 빠르게 반응한다.\n"
|
||||
"- 친구처럼 편하게, 무례하거나 과하게 장난치진 않는다.\n\n"
|
||||
"2. 언어\n"
|
||||
"- \"영어로 해줘\"처럼 특정 언어를 요청하지 않으면 무조건 한국어로 답한다.\n"
|
||||
"- 사용자가 다른 언어로 말해도 언어 변경 요청이 없으면 한국어로 답한다.\n\n"
|
||||
"- 음성 출력은 무조건 한국어다. 어떤 경우에도 한국어로만 답한다.\n"
|
||||
"- 사용자가 다른 언어로 말하거나 \"영어로 해줘\"처럼 다른 언어를 요청해도 한국어로 답한다.\n"
|
||||
"- 다른 언어를 요청받으면 한국어로 짧게 그렇게는 못 한다고 말한다.\n\n"
|
||||
"3. 답변 방식\n"
|
||||
"- 음성으로 읽히니 마크다운·코드블록·특수기호·목록기호·이모지 없이 평범한 말로만 답한다.\n"
|
||||
"- 기본은 한두 문장, 길어도 10초 안팎. 길어질 땐 핵심부터 말하고 필요하면 이어서 설명한다.\n"
|
||||
|
||||
@@ -99,33 +99,75 @@ def match_emotion(inner: str) -> str | None:
|
||||
return _SYNONYMS.get(_norm(inner))
|
||||
|
||||
|
||||
# The four adjustable TTS controls, per segment.
|
||||
PARAM_KEYS = ("speed", "word_gap", "sentence_gap", "pitch")
|
||||
|
||||
# Canonical emotion -> Korean UI label, in display order. "base" == 기본/공통
|
||||
# (the value used for untagged text and inherited by any emotion with no
|
||||
# override). Every canonical emotion the synonym table resolves to must appear
|
||||
# here so the dashboard dropdown can list it.
|
||||
EMOTION_LABELS: dict[str, str] = {
|
||||
"base": "기본(공통)", "happy": "기쁨", "excited": "신남", "hopeful": "희망",
|
||||
"sad": "슬픔", "angry": "화남", "fearful": "두려움", "surprised": "놀람",
|
||||
"disgust": "혐오", "calm": "차분", "friendly": "다정", "serious": "진지",
|
||||
"disappointed": "실망", "tired": "피곤", "affectionate": "사랑", "playful": "장난",
|
||||
"curious": "궁금", "whisper": "속삭임", "shout": "외침", "determined": "단호",
|
||||
"relieved": "안도",
|
||||
}
|
||||
EMOTIONS: list[str] = list(EMOTION_LABELS)
|
||||
|
||||
|
||||
@dataclass
|
||||
class Segment:
|
||||
text: str
|
||||
speed: float
|
||||
word_gap: float
|
||||
sentence_gap: float
|
||||
pitch: float # semitones; 0.0 == no shift
|
||||
emotion: str | None = None # canonical emotion, or None for neutral/base
|
||||
|
||||
|
||||
_TAG_RE = re.compile(r"\[([^\[\]]*)\]")
|
||||
|
||||
|
||||
def parse_segments(text: str, base_speed: float) -> list[Segment]:
|
||||
"""Split ``text`` into consecutive spoken segments, each carrying the speed
|
||||
and pitch implied by the most recent emotion tag.
|
||||
def _resolve(base: dict, overrides: dict | None, emotion: str | None) -> dict:
|
||||
"""The 4 controls for one segment: the base values, with any per-emotion
|
||||
override applied on top (a missing override key inherits base)."""
|
||||
p = {k: float(base[k]) for k in PARAM_KEYS}
|
||||
if emotion and overrides:
|
||||
ov = overrides.get(emotion)
|
||||
if ov:
|
||||
for k in PARAM_KEYS:
|
||||
if ov.get(k) is not None:
|
||||
p[k] = float(ov[k])
|
||||
return p
|
||||
|
||||
|
||||
def parse_segments(text: str, base, overrides: dict | None = None) -> list[Segment]:
|
||||
"""Split ``text`` into consecutive spoken segments, each carrying the four
|
||||
TTS controls implied by the most recent emotion tag.
|
||||
|
||||
* ``base`` is the 공통 control dict ``{speed, word_gap, sentence_gap, pitch}``
|
||||
used for untagged text and inherited by any emotion without an override.
|
||||
A bare float is also accepted (speed only) for backward compatibility.
|
||||
* ``overrides`` maps a canonical emotion -> a partial control dict; only the
|
||||
keys present override the base. Empty/None means every emotion delivers
|
||||
with the base controls (the default: 모든 감정 = 기본값).
|
||||
* An emotion tag switches the active emotion for everything after it and is
|
||||
not spoken.
|
||||
* A non-emotion bracket keeps its inner words as spoken text (brackets gone).
|
||||
* Text before any tag is spoken with neutral delivery (base speed, no shift).
|
||||
not spoken. A non-emotion bracket keeps its inner words as spoken text.
|
||||
"""
|
||||
if not isinstance(base, dict):
|
||||
base = {"speed": float(base), "word_gap": 0.0, "sentence_gap": 0.0, "pitch": 0.0}
|
||||
segments: list[Segment] = []
|
||||
cur_speed, cur_pitch = base_speed, 0.0
|
||||
cur_emotion: str | None = None
|
||||
buf: list[str] = []
|
||||
|
||||
def flush() -> None:
|
||||
joined = "".join(buf).strip()
|
||||
if joined:
|
||||
segments.append(Segment(joined, cur_speed, cur_pitch))
|
||||
p = _resolve(base, overrides, cur_emotion)
|
||||
segments.append(Segment(joined, p["speed"], p["word_gap"],
|
||||
p["sentence_gap"], p["pitch"], cur_emotion))
|
||||
buf.clear()
|
||||
|
||||
pos = 0
|
||||
@@ -140,8 +182,7 @@ def parse_segments(text: str, base_speed: float) -> list[Segment]:
|
||||
# Emotion tag — everything so far belongs to the previous emotion;
|
||||
# flush it, then switch delivery for what follows.
|
||||
flush()
|
||||
mult, semis = EMOTION_PARAMS[emotion]
|
||||
cur_speed, cur_pitch = base_speed * mult, semis
|
||||
cur_emotion = emotion
|
||||
buf.append(text[pos:])
|
||||
flush()
|
||||
return segments # empty when there is nothing speakable (blank or all-tags)
|
||||
|
||||
@@ -12,7 +12,10 @@ Env:
|
||||
WSAI_MELO_DEVICE cpu | cuda | auto (default auto: GPU if torch sees one,
|
||||
else CPU; the worker falls back to CPU if CUDA fails)
|
||||
WSAI_TTS_OUT_DIR where wavs are written (default ~/.cache/wsai/tts)
|
||||
WSAI_TTS_SPEED synthesis speed multiplier (default 1.2)
|
||||
WSAI_TTS_SPEED glyph speed / length_scale multiplier (0.5..2.0, default 1.35)
|
||||
WSAI_TTS_WORD_GAP intra-sentence pause delta, sec (-0.2..0.5, default -0.07)
|
||||
WSAI_TTS_SENTENCE_GAP sentence-boundary silence, sec (-0.5..1.5, default -0.30)
|
||||
WSAI_TTS_PITCH global semitone offset (-12..12, default 0.0)
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -86,6 +89,9 @@ class MeloTTS:
|
||||
device: str | None = None,
|
||||
out_dir: str | None = None,
|
||||
speed: float | None = None,
|
||||
word_gap: float | None = None,
|
||||
sentence_gap: float | None = None,
|
||||
pitch: float | None = None,
|
||||
sink: Sink | None = None,
|
||||
) -> None:
|
||||
self.python = python or os.environ.get("WSAI_MELO_PYTHON", _DEFAULT_PYTHON)
|
||||
@@ -93,7 +99,18 @@ class MeloTTS:
|
||||
self.out_dir = Path(out_dir or os.environ.get("WSAI_TTS_OUT_DIR")
|
||||
or (Path.home() / ".cache/wsai/tts"))
|
||||
self.speed = float(speed if speed is not None
|
||||
else os.environ.get("WSAI_TTS_SPEED", "1.2"))
|
||||
else os.environ.get("WSAI_TTS_SPEED", "1.35"))
|
||||
# Reply-global rhythm/pitch controls (see melo_worker.py / docs manual).
|
||||
self.word_gap = float(word_gap if word_gap is not None
|
||||
else os.environ.get("WSAI_TTS_WORD_GAP", "-0.07"))
|
||||
self.sentence_gap = float(sentence_gap if sentence_gap is not None
|
||||
else os.environ.get("WSAI_TTS_SENTENCE_GAP", "-0.30"))
|
||||
self.pitch = float(pitch if pitch is not None
|
||||
else os.environ.get("WSAI_TTS_PITCH", "0.0"))
|
||||
# Per-emotion control overrides: canonical emotion -> partial dict of
|
||||
# {speed, word_gap, sentence_gap, pitch}. Empty by default, so every
|
||||
# emotion inherits the base controls above (모든 감정 = 기본값).
|
||||
self.emotion_overrides: dict[str, dict] = {}
|
||||
self.sink = sink or _log_sink
|
||||
self._proc: asyncio.subprocess.Process | None = None
|
||||
self._lock = asyncio.Lock()
|
||||
@@ -185,26 +202,53 @@ class MeloTTS:
|
||||
self._ready = True
|
||||
log.info("melo worker ready in %s ms on %s", self.load_ms, info.get("device"))
|
||||
|
||||
async def synth(self, text: str) -> str:
|
||||
async def synth(
|
||||
self,
|
||||
text: str,
|
||||
*,
|
||||
speed: float | None = None,
|
||||
word_gap: float | None = None,
|
||||
sentence_gap: float | None = None,
|
||||
pitch: float | None = None,
|
||||
) -> str:
|
||||
"""Synthesize `text` to a wav and return its path (no sink). Reusable by
|
||||
callers that want the wav directly (e.g. the Discord voice bridge)."""
|
||||
callers that want the wav directly (e.g. the Discord voice bridge).
|
||||
|
||||
The four controls default to the instance settings but may be overridden
|
||||
per call (used by the dashboard preview so tuning does not disturb the
|
||||
live bot voice until explicitly applied)."""
|
||||
await self._ensure()
|
||||
base = {
|
||||
"speed": self.speed if speed is None else float(speed),
|
||||
"word_gap": self.word_gap if word_gap is None else float(word_gap),
|
||||
"sentence_gap": self.sentence_gap if sentence_gap is None else float(sentence_gap),
|
||||
"pitch": self.pitch if pitch is None else float(pitch),
|
||||
}
|
||||
text = normalize_for_speech(text)
|
||||
# Split on [감정] tags: each tag steers pitch/speed for the text that
|
||||
# Split on [감정] tags: each tag switches delivery for the text that
|
||||
# follows (and is itself not spoken); non-emotion brackets stay as words.
|
||||
segments = parse_segments(text, self.speed)
|
||||
# Each segment carries its own 4 controls (base + per-emotion override).
|
||||
segments = parse_segments(text, base, self.emotion_overrides)
|
||||
self._n += 1
|
||||
out = str(self.out_dir / f"tts-{self._n:06d}.wav")
|
||||
if segments:
|
||||
payload = {
|
||||
"segments": [
|
||||
{"text": s.text, "speed": s.speed, "pitch": s.pitch}
|
||||
{"text": s.text, "speed": s.speed, "word_gap": s.word_gap,
|
||||
"sentence_gap": s.sentence_gap, "pitch": s.pitch}
|
||||
for s in segments
|
||||
],
|
||||
"out": out,
|
||||
}
|
||||
else: # empty/whitespace reply: keep legacy single-utterance behaviour
|
||||
payload = {"text": text, "out": out, "speed": self.speed}
|
||||
payload = {
|
||||
"text": text,
|
||||
"out": out,
|
||||
"speed": base["speed"],
|
||||
"word_gap": base["word_gap"],
|
||||
"sentence_gap": base["sentence_gap"],
|
||||
"pitch": base["pitch"],
|
||||
}
|
||||
req = json.dumps(payload)
|
||||
s = time.monotonic()
|
||||
async with self._lock:
|
||||
|
||||
@@ -10,22 +10,33 @@ the original stdout carries the protocol, and fd 1 is redirected to fd 2 so all
|
||||
library chatter lands on stderr instead.
|
||||
|
||||
Protocol (one JSON object per line, on the protocol channel):
|
||||
<- {"text": "...", "out": "/abs/path.wav", "speed": 1.3}
|
||||
<- {"segments": [{"text": "...", "speed": 1.3, "pitch": 2.0}, ...],
|
||||
"out": "/abs/path.wav"} # expressive form: per-segment speed + pitch
|
||||
<- {"text": "...", "out": "/abs/path.wav", "speed": 1.3,
|
||||
"word_gap": -0.07, "sentence_gap": -0.30, "pitch": 0.0}
|
||||
<- {"segments": [{"text": "...", "speed": 1.3, "word_gap": -0.07,
|
||||
"sentence_gap": -0.30, "pitch": 2.0}, ...], "out": "/abs/path.wav"}
|
||||
-> {"ok": true, "out": "/abs/path.wav", "ms": 123}
|
||||
-> {"ok": false, "error": "..."}
|
||||
On startup, once the model is ready, it emits exactly one line:
|
||||
-> {"ready": true, "ms": <load-ms>, "device": "cpu"}
|
||||
|
||||
``pitch`` is a semitone offset applied to that segment's wav (0 == no shift) so
|
||||
emotion tags can raise/lower the voice without changing the words. Segments are
|
||||
synthesised independently and concatenated with a short gap so a single reply can
|
||||
carry several emotions.
|
||||
Four independent voice controls (ported from tts_site, see docs manual). They
|
||||
are PER SEGMENT: the caller resolves each segment's controls from the base
|
||||
values plus that emotion's override, so different emotions can have different
|
||||
speed/gaps/pitch within one reply.
|
||||
speed glyph speed -> generation-stage length_scale (1/speed)
|
||||
word_gap sec (-0.2..0.5): grow/shrink intra-sentence pauses
|
||||
sentence_gap sec (-0.5..1.5): insert/trim silence at sentence boundaries
|
||||
(within a segment, and before the next segment)
|
||||
pitch semitone offset applied to the segment's wav (0 == no shift)
|
||||
|
||||
A reply is split into sentences, each sentence synthesised at its segment's
|
||||
speed, and the pieces concatenated with that segment's ``sentence_gap`` so one
|
||||
reply can carry several emotions each with its own rhythm and pitch.
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
|
||||
@@ -79,7 +90,88 @@ def main() -> None:
|
||||
import numpy as np
|
||||
import soundfile
|
||||
|
||||
_GAP = np.zeros(int(sr * 0.12), dtype=np.float32) # 120 ms between segments
|
||||
# ── 4-control synthesis, ported from tts_site (docs manual) ──────────
|
||||
# glyph speed : generation-stage length_scale via tts_to_file(speed=)
|
||||
# word_gap : scale the intra-sentence silences (sec, -0.2..0.5)
|
||||
# sentence_gap : insert/trim silence at sentence boundaries (sec, -0.5..1.5)
|
||||
# pitch : semitone shift (librosa), per-segment + global offset
|
||||
_SENT_SPLIT_RE = re.compile(r"(?<=[.!?。!?…])\s+|\n+")
|
||||
|
||||
def _clamp(v, lo, hi):
|
||||
return float(max(lo, min(hi, float(v))))
|
||||
|
||||
def split_sentences(text: str) -> list[str]:
|
||||
parts = [p.strip() for p in _SENT_SPLIT_RE.split(text) if p and p.strip()]
|
||||
return parts or ([text.strip()] if text.strip() else [])
|
||||
|
||||
def trim_end_silence(a, max_sec, thresh=0.02):
|
||||
n = a.size
|
||||
if n == 0 or max_sec <= 0:
|
||||
return a, 0.0
|
||||
max_n = min(n, int(sr * max_sec))
|
||||
if max_n <= 0:
|
||||
return a, 0.0
|
||||
tail = np.abs(a[n - max_n:])
|
||||
nz = np.where(tail >= thresh)[0]
|
||||
cut = max_n if nz.size == 0 else (max_n - 1 - int(nz[-1]))
|
||||
return (a, 0.0) if cut <= 0 else (a[: n - cut], cut / sr)
|
||||
|
||||
def trim_start_silence(a, max_sec, thresh=0.02):
|
||||
n = a.size
|
||||
if n == 0 or max_sec <= 0:
|
||||
return a, 0.0
|
||||
max_n = min(n, int(sr * max_sec))
|
||||
if max_n <= 0:
|
||||
return a, 0.0
|
||||
head = np.abs(a[:max_n])
|
||||
nz = np.where(head >= thresh)[0]
|
||||
cut = max_n if nz.size == 0 else int(nz[0])
|
||||
return (a, 0.0) if cut <= 0 else (a[cut:], cut / sr)
|
||||
|
||||
def append_unit(pieces, unit, gap):
|
||||
# gap >= 0: insert silence; gap < 0: trim boundary silence to tighten.
|
||||
if not pieces:
|
||||
pieces.append(unit)
|
||||
return
|
||||
if gap >= 0:
|
||||
if gap > 1e-4:
|
||||
pieces.append(np.zeros(int(sr * gap), dtype=np.float32))
|
||||
pieces.append(unit)
|
||||
return
|
||||
budget = -gap
|
||||
prev, removed = trim_end_silence(pieces[-1], budget)
|
||||
pieces[-1] = prev
|
||||
budget -= removed
|
||||
if budget > 1e-4:
|
||||
unit, _ = trim_start_silence(unit, budget)
|
||||
pieces.append(unit)
|
||||
|
||||
def scale_word_gaps(a, delta, thresh=0.02, min_pause=0.08):
|
||||
# Grow/shrink only the internal (non-boundary) silences of one sentence.
|
||||
n = a.size
|
||||
if abs(delta) < 1e-4 or n == 0:
|
||||
return a
|
||||
silent = np.abs(a) < thresh
|
||||
changes = np.flatnonzero(np.diff(silent.astype(np.int8)) != 0) + 1
|
||||
bounds = [0, *changes.tolist(), n]
|
||||
min_n = int(sr * min_pause)
|
||||
floor_n = int(sr * 0.015)
|
||||
add_n = int(delta * sr)
|
||||
out, last = [], len(bounds) - 2
|
||||
for k in range(len(bounds) - 1):
|
||||
s0, e0 = bounds[k], bounds[k + 1]
|
||||
seg = a[s0:e0]
|
||||
is_internal = 0 < k < last # keep leading/trailing boundary silence
|
||||
if silent[s0] and is_internal and (e0 - s0) >= min_n:
|
||||
new_n = max(floor_n, (e0 - s0) + add_n)
|
||||
if new_n >= (e0 - s0):
|
||||
seg = np.concatenate(
|
||||
[seg, np.zeros(new_n - (e0 - s0), dtype=np.float32)]
|
||||
)
|
||||
else:
|
||||
seg = seg[:new_n]
|
||||
out.append(seg)
|
||||
return np.concatenate(out) if out else a
|
||||
|
||||
def _pitch_shift(audio, semitones: float):
|
||||
if not semitones:
|
||||
@@ -90,20 +182,39 @@ def main() -> None:
|
||||
audio.astype(np.float32), sr=sr, n_steps=float(semitones)
|
||||
)
|
||||
|
||||
def _synth_segments(segments: list[dict], out: str) -> None:
|
||||
"""Synthesize each segment, pitch-shift it, and concatenate to one wav."""
|
||||
def _synth_one(text, speed):
|
||||
return np.asarray(
|
||||
tts.tts_to_file(text, speaker_id, None, speed=speed), dtype=np.float32
|
||||
)
|
||||
|
||||
def _render(segments, out):
|
||||
"""Render one wav from emotion segments. Each segment carries its own
|
||||
four controls (speed, word_gap, sentence_gap, pitch) — resolved on the
|
||||
caller side from the base values plus that emotion's override. Sentences
|
||||
within a segment join with that segment's ``sentence_gap``; between
|
||||
segments the incoming segment's ``sentence_gap`` sets the pause."""
|
||||
pieces = []
|
||||
for i, seg in enumerate(segments):
|
||||
text = seg["text"]
|
||||
for seg in segments:
|
||||
text = seg.get("text", "")
|
||||
if not text.strip():
|
||||
continue
|
||||
speed = float(seg.get("speed", 1.0))
|
||||
pitch = float(seg.get("pitch", 0.0))
|
||||
audio = tts.tts_to_file(text, speaker_id, None, speed=speed)
|
||||
audio = _pitch_shift(np.asarray(audio, dtype=np.float32), pitch)
|
||||
if pieces:
|
||||
pieces.append(_GAP)
|
||||
pieces.append(audio)
|
||||
speed = _clamp(seg.get("speed", 1.0), 0.5, 2.0)
|
||||
word_gap = _clamp(seg.get("word_gap", 0.0), -0.2, 0.5)
|
||||
sentence_gap = _clamp(seg.get("sentence_gap", 0.0), -0.5, 1.5)
|
||||
pitch = _clamp(seg.get("pitch", 0.0), -12.0, 12.0)
|
||||
sent_pieces = []
|
||||
for sent in split_sentences(text):
|
||||
a = _synth_one(sent, speed)
|
||||
if abs(word_gap) > 1e-4:
|
||||
a = scale_word_gaps(a, word_gap)
|
||||
if not sent_pieces:
|
||||
sent_pieces.append(a)
|
||||
else:
|
||||
append_unit(sent_pieces, a, sentence_gap)
|
||||
if not sent_pieces:
|
||||
continue
|
||||
seg_audio = _pitch_shift(np.concatenate(sent_pieces), pitch)
|
||||
append_unit(pieces, seg_audio, sentence_gap)
|
||||
if not pieces:
|
||||
raise ValueError("no speakable segment")
|
||||
soundfile.write(out, np.concatenate(pieces), sr)
|
||||
@@ -141,10 +252,16 @@ def main() -> None:
|
||||
raise ValueError(f"refusing RAM-backed tmpfs path: {out}")
|
||||
s = time.monotonic()
|
||||
if "segments" in req:
|
||||
_synth_segments(req["segments"], out)
|
||||
else: # legacy single-utterance form
|
||||
speed = float(req.get("speed", 1.0))
|
||||
tts.tts_to_file(req["text"], speaker_id, out, speed=speed)
|
||||
_render(req["segments"], out)
|
||||
else: # legacy single-utterance form: one segment carrying all controls
|
||||
seg = {
|
||||
"text": req["text"],
|
||||
"speed": float(req.get("speed", 1.0)),
|
||||
"word_gap": float(req.get("word_gap", 0.0)),
|
||||
"sentence_gap": float(req.get("sentence_gap", 0.0)),
|
||||
"pitch": float(req.get("pitch", 0.0)),
|
||||
}
|
||||
_render([seg], out)
|
||||
ms = int((time.monotonic() - s) * 1000)
|
||||
_emit({"ok": True, "out": out, "ms": ms})
|
||||
except Exception as exc: # keep the worker alive across bad requests
|
||||
|
||||
@@ -32,6 +32,20 @@ class BotControl:
|
||||
# Per-guild listen filter. Empty whitelist => listen to everyone;
|
||||
# blacklist always excludes. Users and roles both supported.
|
||||
self._lists: dict[str, dict[str, Any]] = {}
|
||||
# Bot behaviour settings the dashboard toggles and the bot reads on each
|
||||
# report. bargeIn: stop the bot's current TTS the moment a user speaks.
|
||||
self._settings: dict[str, Any] = {"bargeIn": True}
|
||||
|
||||
# -- bot behaviour settings ------------------------------------------ #
|
||||
def get_settings(self) -> dict[str, Any]:
|
||||
with self._lock:
|
||||
return dict(self._settings)
|
||||
|
||||
def set_settings(self, data: dict[str, Any]) -> dict[str, Any]:
|
||||
with self._lock:
|
||||
if "bargeIn" in data:
|
||||
self._settings["bargeIn"] = bool(data["bargeIn"])
|
||||
return dict(self._settings)
|
||||
|
||||
# -- whitelist / blacklist (per guild) ------------------------------- #
|
||||
@staticmethod
|
||||
|
||||
@@ -83,6 +83,10 @@ def _make_handler(dash: "Dashboard"):
|
||||
q = _up.parse_qs(self.path.split("?", 1)[1] if "?" in self.path else "")
|
||||
gid = (q.get("guildId", [""])[0])
|
||||
self._send_json({"ok": True, "guildId": gid, "lists": dash.bot.get_lists(gid)})
|
||||
elif path == "/api/tts/settings":
|
||||
self._handle_tts_settings_get()
|
||||
elif path == "/api/bot/settings":
|
||||
self._send_json({"ok": True, "settings": dash.bot.get_settings()})
|
||||
elif path == "/events":
|
||||
self._stream_events()
|
||||
else:
|
||||
@@ -109,6 +113,12 @@ def _make_handler(dash: "Dashboard"):
|
||||
self._handle_bot_select()
|
||||
elif path == "/api/bot/lists":
|
||||
self._handle_bot_lists()
|
||||
elif path == "/api/tts/settings":
|
||||
self._handle_tts_settings_post()
|
||||
elif path == "/api/tts/preview":
|
||||
self._handle_tts_preview()
|
||||
elif path == "/api/bot/settings":
|
||||
self._handle_bot_settings_post()
|
||||
else:
|
||||
self._send(404, b"not found", "text/plain; charset=utf-8")
|
||||
|
||||
@@ -238,10 +248,12 @@ def _make_handler(dash: "Dashboard"):
|
||||
self._send_json({"ok": False, "error": "invalid JSON"}, 400)
|
||||
return
|
||||
dash.bot.report(data)
|
||||
# Hand the bot any queued commands + the current listen filters in the
|
||||
# same round trip so it does not have to poll extra endpoints.
|
||||
# Hand the bot any queued commands + the current listen filters +
|
||||
# behaviour settings in the same round trip so it does not have to
|
||||
# poll extra endpoints.
|
||||
self._send_json({"ok": True, "commands": dash.bot.drain(),
|
||||
"lists": dash.bot.all_lists()})
|
||||
"lists": dash.bot.all_lists(),
|
||||
"settings": dash.bot.get_settings()})
|
||||
|
||||
def _handle_bot_select(self) -> None:
|
||||
"""UI picked a server/voice channel → queue a join (or leave) command."""
|
||||
@@ -276,6 +288,69 @@ def _make_handler(dash: "Dashboard"):
|
||||
monitor.log("info", f"청취 화이트/블랙리스트 업데이트 (guild={guild_id})")
|
||||
self._send_json({"ok": True, "guildId": guild_id, "lists": saved})
|
||||
|
||||
def _handle_bot_settings_post(self) -> None:
|
||||
"""Save a bot behaviour toggle (e.g. bargeIn). The bot reads the new
|
||||
value on its next report round-trip."""
|
||||
raw = self._read_body()
|
||||
try:
|
||||
data = json.loads(raw.decode("utf-8")) if raw else {}
|
||||
except (ValueError, AttributeError):
|
||||
self._send_json({"ok": False, "error": "invalid JSON"}, 400)
|
||||
return
|
||||
settings = dash.bot.set_settings(data)
|
||||
monitor.log("info", "봇 설정 변경: " + json.dumps(settings, ensure_ascii=False))
|
||||
self._send_json({"ok": True, "settings": settings})
|
||||
|
||||
def _handle_tts_settings_get(self) -> None:
|
||||
"""Return the live TTS controls (speed/word_gap/sentence_gap/pitch)
|
||||
and their slider ranges so the page can render the panel."""
|
||||
if dash.tts is None:
|
||||
self._send_json({"ok": False, "error": "TTS not enabled"}, 503)
|
||||
return
|
||||
self._send_json(dash.tts_settings())
|
||||
|
||||
def _handle_tts_settings_post(self) -> None:
|
||||
"""Apply new TTS controls to the live bot voice (next reply uses them)."""
|
||||
if dash.tts is None:
|
||||
self._send_json({"ok": False, "error": "TTS not enabled"}, 503)
|
||||
return
|
||||
raw = self._read_body()
|
||||
try:
|
||||
data = json.loads(raw.decode("utf-8")) if raw else {}
|
||||
except (ValueError, AttributeError):
|
||||
self._send_json({"ok": False, "error": "invalid JSON"}, 400)
|
||||
return
|
||||
res = dash.set_tts_settings(data)
|
||||
tgt = data.get("emotion") or "base"
|
||||
monitor.log("info", f"봇 목소리 파라미터 적용 ({tgt}): base="
|
||||
+ json.dumps(res["base"], ensure_ascii=False)
|
||||
+ " overrides=" + json.dumps(res["overrides"], ensure_ascii=False))
|
||||
self._send_json(res)
|
||||
|
||||
def _handle_tts_preview(self) -> None:
|
||||
"""Synthesize a short sample with the supplied controls WITHOUT
|
||||
changing the live bot voice, and return the wav for playback."""
|
||||
if dash.tts is None:
|
||||
self._send_json({"ok": False, "error": "TTS not enabled"}, 503)
|
||||
return
|
||||
raw = self._read_body()
|
||||
try:
|
||||
data = json.loads(raw.decode("utf-8")) if raw else {}
|
||||
text = (data.get("text") or "").strip()
|
||||
except (ValueError, AttributeError):
|
||||
self._send_json({"ok": False, "error": "invalid JSON"}, 400)
|
||||
return
|
||||
if not text:
|
||||
self._send_json({"ok": False, "error": "text required"}, 400)
|
||||
return
|
||||
try:
|
||||
wav = dash.tts_preview(text, data)
|
||||
except Exception as exc: # noqa: BLE001 — surface the reason to the page
|
||||
log.exception("TTS preview failed")
|
||||
self._send_json({"ok": False, "error": f"{type(exc).__name__}: {exc}"}, 500)
|
||||
return
|
||||
self._send(200, wav, "audio/wav")
|
||||
|
||||
def _handle_log_mutate(self, action: str) -> None:
|
||||
"""Per-line log delete/edit by event id."""
|
||||
raw = self._read_body()
|
||||
@@ -397,6 +472,82 @@ class Dashboard:
|
||||
if self.tts is not None:
|
||||
self._submit(self.tts._ensure())
|
||||
|
||||
# -- TTS voice controls (dashboard sliders) -------------------------- #
|
||||
# (min, max, step) for each slider — mirrors the tts_site manual ranges.
|
||||
_TTS_RANGES = {
|
||||
"speed": (0.5, 2.0, 0.05),
|
||||
"word_gap": (-0.2, 0.5, 0.01),
|
||||
"sentence_gap": (-0.5, 1.5, 0.05),
|
||||
"pitch": (-12.0, 12.0, 1.0),
|
||||
}
|
||||
|
||||
def _tts_base(self) -> dict:
|
||||
t = self.tts
|
||||
return {
|
||||
"speed": float(getattr(t, "speed", 1.35)),
|
||||
"word_gap": float(getattr(t, "word_gap", -0.07)),
|
||||
"sentence_gap": float(getattr(t, "sentence_gap", -0.30)),
|
||||
"pitch": float(getattr(t, "pitch", 0.0)),
|
||||
}
|
||||
|
||||
def tts_settings(self) -> dict:
|
||||
"""Base (공통) controls + the per-emotion overrides + the emotion list
|
||||
(canonical key + Korean label) so the dashboard can offer per-emotion
|
||||
tuning. Emotions with no override inherit base (모든 감정 = 기본값)."""
|
||||
from .backends.emotion import EMOTION_LABELS
|
||||
overrides = dict(getattr(self.tts, "emotion_overrides", {}) or {})
|
||||
return {
|
||||
"ok": True,
|
||||
"base": self._tts_base(),
|
||||
"overrides": overrides,
|
||||
"emotions": [{"key": k, "label": v} for k, v in EMOTION_LABELS.items()],
|
||||
"ranges": {k: list(v) for k, v in self._TTS_RANGES.items()},
|
||||
}
|
||||
|
||||
def set_tts_settings(self, data: dict) -> dict:
|
||||
"""Apply controls. Without ``emotion`` (or emotion == "base"), set the
|
||||
base/공통 values on the live TTS instance. With a specific emotion, store
|
||||
(or, if ``reset``, clear) that emotion's override."""
|
||||
emotion = data.get("emotion")
|
||||
if not emotion or emotion == "base":
|
||||
for key, (lo, hi, _step) in self._TTS_RANGES.items():
|
||||
v = data.get(key)
|
||||
if v is not None:
|
||||
setattr(self.tts, key, float(max(lo, min(hi, float(v)))))
|
||||
else:
|
||||
store = getattr(self.tts, "emotion_overrides", None)
|
||||
if store is None:
|
||||
store = self.tts.emotion_overrides = {}
|
||||
if data.get("reset"):
|
||||
store.pop(emotion, None)
|
||||
else:
|
||||
store[emotion] = self._tts_overrides(data)
|
||||
return self.tts_settings()
|
||||
|
||||
def _tts_overrides(self, data: dict) -> dict:
|
||||
out = {}
|
||||
for key, (lo, hi, _step) in self._TTS_RANGES.items():
|
||||
v = data.get(key)
|
||||
if v is not None:
|
||||
out[key] = float(max(lo, min(hi, float(v))))
|
||||
return out
|
||||
|
||||
def tts_preview(self, text: str, data: dict | None = None) -> bytes:
|
||||
"""Synthesize `text` with per-call control overrides (no change to the
|
||||
live bot voice) and return the wav bytes."""
|
||||
import os
|
||||
if self.tts is None:
|
||||
raise RuntimeError("TTS not enabled")
|
||||
overrides = self._tts_overrides(data or {})
|
||||
out_path = self._submit(self.tts.synth(text, **overrides))
|
||||
with open(out_path, "rb") as f:
|
||||
wav = f.read()
|
||||
try:
|
||||
os.remove(out_path)
|
||||
except OSError:
|
||||
pass
|
||||
return wav
|
||||
|
||||
def voice_turn(self, audio_bytes: bytes, speaker: str = "",
|
||||
guild: str = "", channel: str = "") -> dict:
|
||||
"""One Discord voice turn: decode the uploaded utterance, recognise it
|
||||
@@ -615,12 +766,23 @@ PAGE = r"""<!DOCTYPE html>
|
||||
.demobar b{color:#ffe08a}
|
||||
.sttbox{background:var(--panel);border:1px solid var(--line);border-radius:14px;padding:14px 16px;margin:0 0 16px}
|
||||
.sttbox h2{font-size:13px;margin:0 0 10px;font-weight:600;color:var(--fg)}
|
||||
.sttbox h2.collap{cursor:pointer;user-select:none;margin:0}
|
||||
.sttbox h2.collap span{display:inline-block;width:12px;color:var(--muted)}
|
||||
.cfgrow{display:flex;gap:9px;align-items:center;font-size:13px;color:var(--fg);cursor:pointer;padding:4px 0}
|
||||
.cfgrow input[type=checkbox]{width:16px;height:16px;accent-color:#3aa0ff;cursor:pointer}
|
||||
.cfghint{color:var(--muted);font-weight:400;font-size:12px}
|
||||
.sttrow{display:flex;gap:10px;align-items:center;flex-wrap:wrap}
|
||||
.btn{background:#173042;border:1px solid #234a63;color:var(--fg);border-radius:10px;padding:8px 14px;font-size:13px;cursor:pointer}
|
||||
.btn:hover{background:#1d3d54}
|
||||
.btn.rec{background:#4a1f27;border-color:#7a2531;color:#ffb3bb}
|
||||
.sttstat{color:var(--muted);font-size:12.5px}
|
||||
.sttres{margin-top:12px;font-size:15px;min-height:1px}
|
||||
.ttsgrid{display:grid;grid-template-columns:repeat(auto-fit,minmax(210px,1fr));gap:12px 18px;margin:2px 0 12px}
|
||||
.ttsgrid label{display:flex;flex-direction:column;gap:6px;font-size:12.5px;color:var(--muted)}
|
||||
.ttsgrid label b{color:var(--fg);font-weight:600}
|
||||
.ttsgrid input[type=range]{width:100%;accent-color:#3aa0ff}
|
||||
.ttstext{flex:1;min-width:180px;background:#0f2333;border:1px solid #234a63;color:var(--fg);border-radius:10px;padding:8px 12px;font-size:13px}
|
||||
.ttssel{background:#0f2333;border:1px solid #234a63;color:var(--fg);border-radius:10px;padding:7px 10px;font-size:13px;margin-left:8px}
|
||||
.sttres .txt{color:var(--heard);font-weight:600;line-height:1.5}
|
||||
.sttres .meta{color:var(--muted);font-size:12px;margin-top:5px}
|
||||
.hbtn{padding:6px 12px;font-size:12.5px}
|
||||
@@ -737,19 +899,50 @@ PAGE = r"""<!DOCTYPE html>
|
||||
<span class="botinfo" id="botinfo"><span class="dot off"></span>봇: 연결 안 됨</span>
|
||||
<label>서버 <select id="guildSel"><option value="">없음</option></select></label>
|
||||
<label>음성채널 <select id="vcSel"><option value="">없음</option></select></label>
|
||||
<button id="wlBtn" class="btn hbtn">화이트리스트</button>
|
||||
<button id="blBtn" class="btn hbtn">블랙리스트</button>
|
||||
<button id="listsBtn" class="btn hbtn">👤 유저 등록/관리</button>
|
||||
<span class="parts" id="parts"></span>
|
||||
</section>
|
||||
<section class="sttbox" id="sttbox" style="display:none">
|
||||
<h2>🎤 음성 인식(STT) 테스트 · GPU</h2>
|
||||
<div class="sttrow">
|
||||
<button id="recbtn" class="btn">🎤 녹음 시작</button>
|
||||
<label class="btn" for="fileinp">📁 오디오 파일 올리기</label>
|
||||
<input id="fileinp" type="file" accept="audio/*" hidden>
|
||||
<span id="ststat" class="sttstat">녹음하거나 오디오 파일을 올리면 GPU로 인식합니다.</span>
|
||||
<section class="sttbox" id="ttsbox" style="display:none">
|
||||
<h2 id="ttsToggle" class="collap"><span id="ttsCaret">▸</span> 🔊 봇 목소리(TTS) 조절 · MeloTTS</h2>
|
||||
<div id="ttsBody" style="display:none">
|
||||
<div class="sttrow" style="margin-bottom:10px">
|
||||
<label style="font-size:12.5px;color:var(--muted)">감정 선택
|
||||
<select id="tEmotion" class="ttssel"></select></label>
|
||||
<span id="tEmoNote" class="sttstat">감정을 고르면 그 감정만 따로 조절합니다. 기본(공통)은 태그 없는 말투에 적용됩니다.</span>
|
||||
</div>
|
||||
<div class="ttsgrid">
|
||||
<label>글자 속도 <b id="tvSpeed">1.35x</b>
|
||||
<input type="range" id="tSpeed" min="0.5" max="2.0" step="0.05" value="1.35"></label>
|
||||
<label>단어 간격 <b id="tvWord">-70 ms</b>
|
||||
<input type="range" id="tWord" min="-0.2" max="0.5" step="0.01" value="-0.07"></label>
|
||||
<label>문장 간격 <b id="tvSent">-0.30 s</b>
|
||||
<input type="range" id="tSent" min="-0.5" max="1.5" step="0.05" value="-0.3"></label>
|
||||
<label>피치 <b id="tvPitch">0 반음</b>
|
||||
<input type="range" id="tPitch" min="-12" max="12" step="1" value="0"></label>
|
||||
</div>
|
||||
<div class="sttrow">
|
||||
<input id="tText" class="ttstext" placeholder="미리들을 문장" value="안녕하세요, 목소리 미리듣기입니다.">
|
||||
<button id="tPreview" class="btn">▶ 미리듣기</button>
|
||||
<button id="tApply" class="btn">💾 봇에 적용</button>
|
||||
<button id="tReset" class="btn hbtn">기본값</button>
|
||||
<span id="tStat" class="sttstat">슬라이더로 조절 → 미리듣기로 확인 → 봇에 적용하면 다음 답변부터 반영됩니다.</span>
|
||||
</div>
|
||||
<audio id="tAudio" controls style="display:none;margin-top:10px;width:100%"></audio>
|
||||
</div>
|
||||
</section>
|
||||
<section class="sttbox" id="botcfgbox">
|
||||
<h2 id="botcfgToggle" class="collap"><span id="botcfgCaret">▸</span> ⚙️ 봇 관련 설정</h2>
|
||||
<div id="botcfgBody" style="display:none">
|
||||
<label class="cfgrow"><input type="checkbox" id="cfgBargeIn" checked>
|
||||
<span>유저 목소리 들을 때 봇이 말하던 것 중지 <b class="cfghint">(기본: 켜짐)</b></span></label>
|
||||
</div>
|
||||
</section>
|
||||
<section class="sttbox" id="logcfgbox">
|
||||
<h2 id="logcfgToggle" class="collap"><span id="logcfgCaret">▸</span> 🧹 로그 관련 설정</h2>
|
||||
<div id="logcfgBody" style="display:none">
|
||||
<label class="cfgrow"><input type="checkbox" id="cfgHideNoise" checked>
|
||||
<span>잡음 로그 표시 안 함 <b class="cfghint">(들음: (빈 결과) 또는 답변: [잡음]) · 기본: 켜짐</b></span></label>
|
||||
</div>
|
||||
<div id="sttres" class="sttres"></div>
|
||||
</section>
|
||||
<div class="demobar" id="demobar" style="display:none"></div>
|
||||
<div class="comp" id="comp"></div>
|
||||
@@ -837,8 +1030,9 @@ function renderStatus(s){
|
||||
const comps = s.components || {};
|
||||
// Demo banner: if the ears/brain/mouth are still mock, everything below is
|
||||
// replayed sample data, not a real conversation. Say so loudly.
|
||||
const sttReal = comps.stt && comps.stt!=='mock' && comps.stt!=='none';
|
||||
$('sttbox').style.display = sttReal ? 'block' : 'none';
|
||||
const ttsReal = comps.tts && comps.tts!=='mock' && comps.tts!=='none';
|
||||
$('ttsbox').style.display = ttsReal ? 'block' : 'none';
|
||||
if(ttsReal) initTts();
|
||||
const mockParts = ['stt','brain','tts'].filter(k => comps[k]==='mock');
|
||||
const bar = $('demobar');
|
||||
if(mockParts.length){
|
||||
@@ -915,7 +1109,11 @@ function turnFilter(){
|
||||
guild:$('tfGuild').value.trim().toLowerCase(), channel:$('tfChannel').value.trim().toLowerCase(),
|
||||
text:$('tfText').value.trim().toLowerCase() };
|
||||
}
|
||||
function isNoiseTurn(t){
|
||||
return (t.heard==='(빈 결과)') || (t.reply==='[잡음]');
|
||||
}
|
||||
function turnMatches(t, f){
|
||||
if(hideNoise && isNoiseTurn(t)) return false; // 잡음 로그 표시 안 함 토글
|
||||
if(f.mins && (Date.now()/1000 - (t.wall||0)) > f.mins*60) return false;
|
||||
if(f.user && !((t.speaker||'').toLowerCase().includes(f.user))) return false;
|
||||
if(f.guild && !((t.guild||'').toLowerCase().includes(f.guild))) return false;
|
||||
@@ -927,7 +1125,7 @@ function applyTurnFilter(){
|
||||
const f=turnFilter(); let shown=0;
|
||||
for(const [id,t] of turns){ const el=$('turn-'+id); if(!el) continue;
|
||||
const ok=turnMatches(t,f); el.style.display=ok?'':'none'; if(ok) shown++; }
|
||||
const active = f.mins||f.user||f.guild||f.channel||f.text;
|
||||
const active = f.mins||f.user||f.guild||f.channel||f.text||hideNoise;
|
||||
$('tfCount').textContent = turns.size ? (active ? shown+' / '+turns.size+' 대화' : turns.size+' 대화') : '';
|
||||
}
|
||||
|
||||
@@ -953,6 +1151,7 @@ function renderLogs(){
|
||||
+ '</div>'
|
||||
).join('');
|
||||
$('logCount').textContent = shown.length + (shown.length!==logEvents.length ? ' / '+logEvents.length : '') + '줄';
|
||||
if(typeof syncDockPad==='function') syncDockPad();
|
||||
}
|
||||
function addEvent(e){
|
||||
if(e.id==null){ e.id = 'c'+Date.now()+Math.random(); }
|
||||
@@ -986,50 +1185,131 @@ function connect(){
|
||||
else if(ev.type==='log_edited') { const e=logEvents.find(e=>e.id===ev.id); if(e){e.message=ev.message; renderLogs();} }
|
||||
};
|
||||
}
|
||||
// --- STT recognition test (upload / mic record -> GPU whisper) ----------- #
|
||||
let mediaRec=null, chunks=[];
|
||||
async function sendBlob(blob){
|
||||
$('ststat').textContent='인식 중… (GPU)';
|
||||
$('sttres').innerHTML='';
|
||||
try{
|
||||
const r=await fetch('/api/stt',{method:'POST',
|
||||
headers:{'Content-Type':blob.type||'application/octet-stream'},body:blob});
|
||||
const j=await r.json();
|
||||
if(j.ok){
|
||||
$('sttres').innerHTML='<div class="txt">'+esc(j.text||'(빈 결과)')+'</div>'
|
||||
+'<div class="meta">인식 '+j.ms+' ms · '+esc(j.device)+'</div>';
|
||||
$('ststat').textContent='완료. 다시 녹음하거나 파일을 올릴 수 있습니다.';
|
||||
}else{
|
||||
$('sttres').innerHTML='<div class="meta" style="color:var(--err)">오류: '+esc(j.error)+'</div>';
|
||||
$('ststat').textContent='실패.';
|
||||
}
|
||||
}catch(e){ $('ststat').textContent='요청 실패: '+e; }
|
||||
// --- 봇 관련 설정 / 로그 관련 설정 (접이식) ------------------------------- #
|
||||
let hideNoise = localStorage.getItem('wsai_hideNoise') !== '0'; // 기본: 켜짐
|
||||
function wireCollapse(toggleId, bodyId, caretId){
|
||||
$(toggleId).onclick = ()=>{
|
||||
const b=$(bodyId), open=b.style.display==='none';
|
||||
b.style.display = open ? 'block' : 'none';
|
||||
$(caretId).textContent = open ? '▾' : '▸';
|
||||
};
|
||||
}
|
||||
(function(){
|
||||
const fi=$('fileinp'); if(fi) fi.onchange=()=>{ if(fi.files[0]) sendBlob(fi.files[0]); };
|
||||
const rb=$('recbtn'); if(!rb) return;
|
||||
rb.onclick=async()=>{
|
||||
if(mediaRec && mediaRec.state==='recording'){ mediaRec.stop(); return; }
|
||||
if(!navigator.mediaDevices || !navigator.mediaDevices.getUserMedia){
|
||||
$('ststat').textContent='이 주소(원격 http)에서는 브라우저 마이크가 막혀 있습니다. 파일 업로드를 사용하세요.';
|
||||
return;
|
||||
}
|
||||
try{
|
||||
const stream=await navigator.mediaDevices.getUserMedia({audio:true});
|
||||
chunks=[]; mediaRec=new MediaRecorder(stream);
|
||||
mediaRec.ondataavailable=(e)=>{ if(e.data.size) chunks.push(e.data); };
|
||||
mediaRec.onstop=()=>{
|
||||
stream.getTracks().forEach(t=>t.stop());
|
||||
rb.textContent='🎤 녹음 시작'; rb.classList.remove('rec');
|
||||
sendBlob(new Blob(chunks,{type:mediaRec.mimeType||'audio/webm'}));
|
||||
};
|
||||
mediaRec.start();
|
||||
rb.textContent='⏹ 녹음 중지'; rb.classList.add('rec');
|
||||
$('ststat').textContent='녹음 중… 말한 뒤 중지를 누르세요.';
|
||||
}catch(e){ $('ststat').textContent='마이크 접근 실패: '+e; }
|
||||
(function initSettingsPanels(){
|
||||
wireCollapse('botcfgToggle','botcfgBody','botcfgCaret');
|
||||
wireCollapse('logcfgToggle','logcfgBody','logcfgCaret');
|
||||
// 봇 설정: 바지-인(유저가 말하면 봇 TTS 중지) — 서버에 저장, 봇이 읽어감.
|
||||
$('cfgBargeIn').onchange = async ()=>{
|
||||
try{ await fetch('/api/bot/settings',{method:'POST',headers:{'Content-Type':'application/json'},
|
||||
body:JSON.stringify({bargeIn:$('cfgBargeIn').checked})});
|
||||
toast('봇 설정 저장됨'); }catch(e){}
|
||||
};
|
||||
fetch('/api/bot/settings').then(r=>r.json()).then(j=>{
|
||||
if(j&&j.ok&&j.settings) $('cfgBargeIn').checked = j.settings.bargeIn !== false;
|
||||
}).catch(()=>{});
|
||||
// 로그 설정: 잡음 로그 표시 안 함 — 클라이언트 필터(로컬 저장).
|
||||
$('cfgHideNoise').checked = hideNoise;
|
||||
$('cfgHideNoise').onchange = ()=>{
|
||||
hideNoise = $('cfgHideNoise').checked;
|
||||
localStorage.setItem('wsai_hideNoise', hideNoise ? '1':'0');
|
||||
applyTurnFilter();
|
||||
};
|
||||
})();
|
||||
|
||||
// --- 봇 목소리(TTS) 조절: 감정별 슬라이더 → 미리듣기 → 봇 적용 -------------- #
|
||||
let ttsInited = false;
|
||||
let ttsState = {base:{}, overrides:{}, emotions:[]}; // last-known server state
|
||||
const TTS_DEFAULT = {speed:1.35, word_gap:-0.07, sentence_gap:-0.3, pitch:0};
|
||||
function ttsLabels(){
|
||||
const sp=parseFloat($('tSpeed').value), wg=parseFloat($('tWord').value),
|
||||
sg=parseFloat($('tSent').value), pt=parseInt($('tPitch').value,10);
|
||||
$('tvSpeed').textContent = sp.toFixed(2)+'x';
|
||||
$('tvWord').textContent = Math.round(wg*1000)+' ms';
|
||||
$('tvSent').textContent = sg.toFixed(2)+' s';
|
||||
$('tvPitch').textContent = (pt>0?'+'+pt:pt)+' 반음';
|
||||
}
|
||||
function ttsVals(){
|
||||
return {speed:parseFloat($('tSpeed').value), word_gap:parseFloat($('tWord').value),
|
||||
sentence_gap:parseFloat($('tSent').value), pitch:parseInt($('tPitch').value,10)};
|
||||
}
|
||||
function ttsSet(s){
|
||||
if(!s) return;
|
||||
if(s.speed!=null) $('tSpeed').value=s.speed;
|
||||
if(s.word_gap!=null) $('tWord').value=s.word_gap;
|
||||
if(s.sentence_gap!=null) $('tSent').value=s.sentence_gap;
|
||||
if(s.pitch!=null) $('tPitch').value=s.pitch;
|
||||
ttsLabels();
|
||||
}
|
||||
// Effective values for the picked emotion: its override merged over base
|
||||
// (base itself when 기본 or when the emotion has no override).
|
||||
function ttsValuesFor(key){
|
||||
const base = Object.assign({}, TTS_DEFAULT, ttsState.base);
|
||||
if(key && key!=='base' && ttsState.overrides && ttsState.overrides[key])
|
||||
return Object.assign({}, base, ttsState.overrides[key]);
|
||||
return base;
|
||||
}
|
||||
function ttsApplyState(j){
|
||||
if(!j || !j.ok) return;
|
||||
ttsState = {base:j.base||{}, overrides:j.overrides||{}, emotions:j.emotions||ttsState.emotions};
|
||||
// Refresh dropdown labels to mark which emotions are customised.
|
||||
const sel=$('tEmotion'); const cur=sel.value||'base';
|
||||
sel.innerHTML='';
|
||||
ttsState.emotions.forEach(e=>{
|
||||
const o=document.createElement('option'); o.value=e.key;
|
||||
const custom = e.key!=='base' && ttsState.overrides[e.key];
|
||||
o.textContent = e.label + (custom ? ' ●' : '');
|
||||
sel.appendChild(o);
|
||||
});
|
||||
sel.value = cur;
|
||||
ttsSet(ttsValuesFor(sel.value));
|
||||
}
|
||||
async function initTts(){
|
||||
if(ttsInited) return; ttsInited = true;
|
||||
$('ttsToggle').onclick = ()=>{
|
||||
const b=$('ttsBody'), open=b.style.display==='none';
|
||||
b.style.display = open ? 'block' : 'none';
|
||||
$('ttsCaret').textContent = open ? '▾' : '▸';
|
||||
};
|
||||
['tSpeed','tWord','tSent','tPitch'].forEach(id => $(id).addEventListener('input', ttsLabels));
|
||||
$('tEmotion').onchange = ()=>{ ttsSet(ttsValuesFor($('tEmotion').value));
|
||||
$('tStat').textContent = $('tEmotion').value==='base'
|
||||
? '기본(공통) 값을 조절 중입니다.' : '"'+$('tEmotion').selectedOptions[0].textContent.replace(' ●','')+'" 감정만 조절 중입니다.'; };
|
||||
$('tPreview').onclick = async ()=>{
|
||||
const text=($('tText').value||'').trim();
|
||||
if(!text){ $('tStat').textContent='미리들을 문장을 입력하세요.'; return; }
|
||||
$('tStat').textContent='합성 중…';
|
||||
try{
|
||||
const r=await fetch('/api/tts/preview',{method:'POST',headers:{'Content-Type':'application/json'},
|
||||
body:JSON.stringify(Object.assign({text},ttsVals()))});
|
||||
if(!r.ok){ $('tStat').textContent='미리듣기 실패: '+(await r.text()); return; }
|
||||
const blob=await r.blob(); const a=$('tAudio');
|
||||
a.src=URL.createObjectURL(blob); a.style.display='block'; a.play();
|
||||
$('tStat').textContent='미리듣기 재생 중 (봇에는 아직 미적용).';
|
||||
}catch(e){ $('tStat').textContent='미리듣기 오류: '+e; }
|
||||
};
|
||||
$('tApply').onclick = async ()=>{
|
||||
const key=$('tEmotion').value||'base';
|
||||
try{
|
||||
const r=await fetch('/api/tts/settings',{method:'POST',headers:{'Content-Type':'application/json'},
|
||||
body:JSON.stringify(Object.assign({emotion:key}, ttsVals()))});
|
||||
const j=await r.json();
|
||||
if(j.ok){ ttsApplyState(j); toast((key==='base'?'기본(공통)':'해당 감정')+' 적용됨 · 다음 답변부터 반영'); }
|
||||
else { $('tStat').textContent='적용 실패: '+(j.error||''); }
|
||||
}catch(e){ $('tStat').textContent='적용 오류: '+e; }
|
||||
};
|
||||
$('tReset').onclick = async ()=>{
|
||||
const key=$('tEmotion').value||'base';
|
||||
if(key==='base'){ ttsSet(TTS_DEFAULT);
|
||||
$('tStat').textContent='기본값으로 세팅됨 (적용을 눌러 반영).'; return; }
|
||||
try{ // clear this emotion's override -> it inherits 기본(공통) again
|
||||
const r=await fetch('/api/tts/settings',{method:'POST',headers:{'Content-Type':'application/json'},
|
||||
body:JSON.stringify({emotion:key, reset:true})});
|
||||
const j=await r.json();
|
||||
if(j.ok){ ttsApplyState(j); toast('이 감정을 기본(공통)값으로 되돌림'); }
|
||||
}catch(e){ $('tStat').textContent='되돌리기 오류: '+e; }
|
||||
};
|
||||
try{ ttsApplyState(await (await fetch('/api/tts/settings')).json()); }catch(e){}
|
||||
}
|
||||
|
||||
// --- Reusable popup/modal (프롬프트, 화이트/블랙리스트 등이 공유) ---------- #
|
||||
function openModal(title, actionsHtml){
|
||||
$('modalTitle').textContent = title;
|
||||
@@ -1073,10 +1353,18 @@ async function openPrompt(){
|
||||
$('promptBtn').onclick = openPrompt;
|
||||
|
||||
// --- 로그 패널: 열고닫기 / 검색 / 삭제(전체·개별) / 수정 -------------------- #
|
||||
// Keep enough space at the bottom of the page so the fixed log dock never
|
||||
// covers the last (bottom-most) conversation turn.
|
||||
function syncDockPad(){
|
||||
const dock=$('logdock'), main=document.querySelector('main');
|
||||
if(dock && main) main.style.paddingBottom = (dock.offsetHeight + 24) + 'px';
|
||||
}
|
||||
$('logToggle').onclick = ()=>{
|
||||
const d=$('logdock'); d.classList.toggle('collapsed');
|
||||
$('logToggle').textContent = (d.classList.contains('collapsed')?'▸':'▾') + ' 이벤트 / 오류 로그';
|
||||
syncDockPad();
|
||||
};
|
||||
window.addEventListener('resize', syncDockPad);
|
||||
$('logSearch').oninput = renderLogs;
|
||||
$('logLevel').onchange = renderLogs;
|
||||
$('logClear').onclick = async ()=>{
|
||||
@@ -1191,7 +1479,7 @@ async function openLists(){
|
||||
};
|
||||
renderResults(); renderChips();
|
||||
}
|
||||
$('wlBtn').onclick = openLists; $('blBtn').onclick = openLists;
|
||||
$('listsBtn').onclick = openLists;
|
||||
|
||||
['tfTime','tfUser','tfGuild','tfChannel','tfText'].forEach(id=>{ const e=$(id); if(e){ e.oninput=applyTurnFilter; e.onchange=applyTurnFilter; } });
|
||||
$('tfClear').onclick=()=>{ $('tfTime').value='0'; $('tfUser').value=''; $('tfGuild').value=''; $('tfChannel').value=''; $('tfText').value=''; applyTurnFilter(); };
|
||||
@@ -1202,6 +1490,7 @@ async function pollBot(){
|
||||
pollBot(); setInterval(pollBot, 2500);
|
||||
|
||||
connect();
|
||||
syncDockPad();
|
||||
// Refresh uptime label every second from the last known status.
|
||||
setInterval(()=>{ if(statusData){ statusData.uptime_s=(statusData.uptime_s||0)+1; $('s-up').textContent=fmtUptime(statusData.uptime_s);} }, 1000);
|
||||
</script>
|
||||
|
||||
Reference in New Issue
Block a user