Compare commits
14 Commits
b0272d2171
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
545a984ca5 | ||
|
|
04664ce61a | ||
|
|
0e4e7c6bb2 | ||
|
|
d5501d0b8a | ||
|
|
5966f6dedb | ||
|
|
8539acd0f8 | ||
|
|
0e77f659d5 | ||
|
|
c39b47cfd3 | ||
|
|
0588ee21ac | ||
|
|
7c899d3b19 | ||
|
|
ed5f328889 | ||
|
|
1ac45214ce | ||
|
|
d164630bb8 | ||
|
|
6d8ba3ad3a |
29
dave/bot.mjs
29
dave/bot.mjs
@@ -178,6 +178,16 @@ const speakingSet = new Set(); // userIds currently speaking (for participant
|
||||
const activeSubs = new Set(); // userIds with an in-flight receive subscription
|
||||
let listsByGuild = {}; // guildId -> {whitelistUsers, blacklistUsers, whitelistRoles, blacklistRoles}
|
||||
let botSettings = { bargeIn: true }; // behaviour toggles from the dashboard
|
||||
// Barge-in only fires when the Discord "speaking" (green ring) stays on for at
|
||||
// least this long — a brief blip (keyboard click, cough) shouldn't cut the bot
|
||||
// off, but a genuine ~0.7s of speech should. The dashboard's "유저 음성 인식 시간"
|
||||
// (botSettings.bargeInMs) wins when set; WSAI_BARGE_IN_MS / 700 is the fallback.
|
||||
const BARGE_IN_MS = Number(process.env.WSAI_BARGE_IN_MS || 700);
|
||||
function bargeInMs() {
|
||||
const v = Number(botSettings.bargeInMs);
|
||||
return Number.isFinite(v) && v >= 0 ? v : BARGE_IN_MS;
|
||||
}
|
||||
const bargeTimers = new Map(); // userId -> pending stop timer
|
||||
|
||||
// Attach the bot's audio player (so it can speak) to a fresh connection.
|
||||
function setupPlayer(connection) {
|
||||
@@ -202,15 +212,20 @@ function setupReceiver(connection) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
// Barge-in: the moment an allowed user speaks, stop the bot's current TTS
|
||||
// so it doesn't talk over them (toggleable from the dashboard, default on).
|
||||
if (botSettings.bargeIn !== false && voicePlayer) {
|
||||
// Barge-in: stop the bot's current TTS only if this user keeps speaking for
|
||||
// BARGE_IN_MS (the green ring stays on) — ignores momentary noise blips. The
|
||||
// 'end' handler cancels the pending stop if speaking stops in time.
|
||||
if (botSettings.bargeIn !== false && voicePlayer && !bargeTimers.has(userId)) {
|
||||
bargeTimers.set(userId, setTimeout(() => {
|
||||
bargeTimers.delete(userId);
|
||||
try { voicePlayer.stop(true); } catch {}
|
||||
}, bargeInMs()));
|
||||
}
|
||||
activeSubs.add(userId);
|
||||
if (!perUser.has(userId)) perUser.set(userId, { opusPackets: 0, pcmFrames: 0 });
|
||||
const opusStream = receiver.subscribe(userId, {
|
||||
end: { behavior: EndBehaviorType.AfterSilence, duration: 800 },
|
||||
// Wait a touch longer after silence so a soft/trailing word isn't clipped.
|
||||
end: { behavior: EndBehaviorType.AfterSilence, duration: 1000 },
|
||||
});
|
||||
const decoder = new prism.opus.Decoder({ rate: 48000, channels: 2, frameSize: 960 });
|
||||
const chunks = [];
|
||||
@@ -226,7 +241,11 @@ function setupReceiver(connection) {
|
||||
handleUtterance(userId, pcm).catch((e) => log(`voice turn error: ${e.message}`));
|
||||
});
|
||||
});
|
||||
receiver.speaking.on('end', (userId) => speakingSet.delete(userId));
|
||||
receiver.speaking.on('end', (userId) => {
|
||||
speakingSet.delete(userId);
|
||||
const t = bargeTimers.get(userId); // spoke too briefly -> cancel barge-in
|
||||
if (t) { clearTimeout(t); bargeTimers.delete(userId); }
|
||||
});
|
||||
}
|
||||
|
||||
// Join (or switch to) a voice channel on command from the dashboard.
|
||||
|
||||
131
tests/latency_ab.py
Normal file
131
tests/latency_ab.py
Normal file
@@ -0,0 +1,131 @@
|
||||
"""Ad-hoc TTFT/total latency A/B across Sonnet versions (+ Haiku reference).
|
||||
|
||||
Mirrors the production brain call: Claude Code identity system block, cached
|
||||
persona, a short voice-style history, one short user turn. Streams to measure
|
||||
time-to-first-token. Interleaves models each round to cancel network drift.
|
||||
|
||||
Run: .venv/bin/python tests/latency_ab.py
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import statistics
|
||||
import time
|
||||
|
||||
import anthropic
|
||||
|
||||
from wsai.backends.claude import ClaudeBrain
|
||||
|
||||
CRED = os.environ.get("CLAUDE_CREDENTIALS_PATH")
|
||||
if not CRED:
|
||||
raise SystemExit(
|
||||
"Set CLAUDE_CREDENTIALS_PATH to your Claude credentials JSON to run this A/B script."
|
||||
)
|
||||
CLAUDE_CODE_ID = "You are Claude Code, Anthropic's official CLI for Claude."
|
||||
|
||||
# (label, model, extra_system_line) — extra line appended to persona to force brevity
|
||||
BREVITY = "지금부터 답은 무조건 한 문장, 12단어 이내로만. 부연·재확인·군더더기 금지."
|
||||
VARIANTS = [
|
||||
("sonnet-4-5", "claude-sonnet-4-5", None),
|
||||
("sonnet-5+brev", "claude-sonnet-5", BREVITY),
|
||||
("haiku-4-5(ref)", "claude-haiku-4-5", None),
|
||||
]
|
||||
ROUNDS = 16
|
||||
|
||||
HISTORY = [
|
||||
("안녕", "[반가움] 안녕! 뭐 하고 있었어?"),
|
||||
("그냥 코딩", "[다정] 오 무슨 코딩?"),
|
||||
("파이썬", "[신남] 좋네, 잘 되고 있어?"),
|
||||
("응 그럭저럭", "[차분] 다행이다."),
|
||||
]
|
||||
USER = "지금 몇 시야?"
|
||||
|
||||
|
||||
def token() -> str:
|
||||
with open(CRED) as f:
|
||||
return json.load(f)["claudeAiOauth"]["accessToken"]
|
||||
|
||||
|
||||
def build(extra: str | None = None):
|
||||
persona = ClaudeBrain.PERSONA
|
||||
if extra:
|
||||
persona = persona + "\n\n" + extra
|
||||
system = [
|
||||
{"type": "text", "text": CLAUDE_CODE_ID},
|
||||
{"type": "text", "text": persona, "cache_control": {"type": "ephemeral"}},
|
||||
]
|
||||
msgs = []
|
||||
for u, a in HISTORY:
|
||||
msgs.append({"role": "user", "content": u})
|
||||
msgs.append({"role": "assistant", "content": a})
|
||||
msgs.append({"role": "user", "content": "[지금 화면] (아직 못 읽음)\n\n" + USER})
|
||||
return system, msgs
|
||||
|
||||
|
||||
def measure(client, model, system, msgs):
|
||||
t0 = time.monotonic()
|
||||
ttft = None
|
||||
out_tokens = 0
|
||||
try:
|
||||
with client.messages.stream(
|
||||
model=model, max_tokens=150, system=system, messages=msgs
|
||||
) as stream:
|
||||
for ev in stream.text_stream:
|
||||
if ttft is None:
|
||||
ttft = time.monotonic() - t0
|
||||
total = time.monotonic() - t0
|
||||
final = stream.get_final_message()
|
||||
out_tokens = final.usage.output_tokens
|
||||
return ttft, total, out_tokens, None
|
||||
except Exception as e: # noqa: BLE001
|
||||
return None, None, None, f"{type(e).__name__}: {str(e)[:120]}"
|
||||
|
||||
|
||||
def main():
|
||||
client = anthropic.Anthropic(auth_token=token(), max_retries=1)
|
||||
prebuilt = {label: build(extra) for label, model, extra in VARIANTS}
|
||||
results = {label: {"ttft": [], "total": [], "out": []} for label, _, _ in VARIANTS}
|
||||
dead = set()
|
||||
# one warm-up per variant to prime connection + cache (excluded from stats)
|
||||
for label, model, _ in VARIANTS:
|
||||
system, msgs = prebuilt[label]
|
||||
_, _, _, err = measure(client, model, system, msgs)
|
||||
if err:
|
||||
print(f"[skip] {label} ({model}): {err}")
|
||||
dead.add(label)
|
||||
print(f"\nliving: {[l for l, _, _ in VARIANTS if l not in dead]}")
|
||||
print(f"rounds: {ROUNDS} (interleaved)\n")
|
||||
for r in range(ROUNDS):
|
||||
for label, model, _ in VARIANTS:
|
||||
if label in dead:
|
||||
continue
|
||||
system, msgs = prebuilt[label]
|
||||
ttft, total, out, err = measure(client, model, system, msgs)
|
||||
if err:
|
||||
print(f" r{r} {label}: ERR {err}")
|
||||
continue
|
||||
results[label]["ttft"].append(ttft)
|
||||
results[label]["total"].append(total)
|
||||
results[label]["out"].append(out)
|
||||
print(f"round {r+1}/{ROUNDS} done")
|
||||
|
||||
print("\n=== median (min–max) over", ROUNDS, "runs ===")
|
||||
print(f"{'variant':<18} {'TTFT s':<18} {'total s':<18} {'out tok'}")
|
||||
for label, _, _ in VARIANTS:
|
||||
d = results[label]
|
||||
if not d["ttft"]:
|
||||
print(f"{label:<18} (no data)")
|
||||
continue
|
||||
tt = d["ttft"]
|
||||
to = d["total"]
|
||||
print(
|
||||
f"{label:<18} "
|
||||
f"{statistics.median(tt):.2f} ({min(tt):.2f}-{max(tt):.2f}) "
|
||||
f"{statistics.median(to):.2f} ({min(to):.2f}-{max(to):.2f}) "
|
||||
f"{statistics.median(d['out']):.0f}"
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -37,7 +37,7 @@ def test_monitor_records_turn_with_timed_steps():
|
||||
assert turn["total_ms"] >= 0
|
||||
# step-by-step: every stage is named and timed
|
||||
names = [s["name"] for s in turn["steps"]]
|
||||
assert names == ["화면 맥락", "두뇌(생각)", "응답(TTS/전송)"]
|
||||
assert names == ["화면 맥락", "LLM(생각)", "응답(TTS/전송)"]
|
||||
assert all(s["ok"] is True for s in turn["steps"])
|
||||
assert all(s["ms"] >= 0 for s in turn["steps"])
|
||||
|
||||
@@ -66,7 +66,7 @@ def test_monitor_marks_errors():
|
||||
|
||||
turn = snap["turns"][0]
|
||||
assert turn["status"] == "error"
|
||||
brain_step = next(s for s in turn["steps"] if s["name"] == "두뇌(생각)")
|
||||
brain_step = next(s for s in turn["steps"] if s["name"] == "LLM(생각)")
|
||||
assert brain_step["ok"] is False
|
||||
assert "boom" in brain_step["error"]
|
||||
assert snap["status"]["errors_total"] >= 1
|
||||
|
||||
@@ -82,11 +82,11 @@ def _run_stt_test(host: str, port: int) -> None:
|
||||
monitor.set_components({"source": "none", "vision": "none", "stt": "whisper",
|
||||
"brain": "none", "tts": "none"})
|
||||
monitor.set_status(running=True, listening=False)
|
||||
monitor.log("info", "STT 인식 테스트 서버 시작 — GPU 워밍업 중…")
|
||||
monitor.log("info", "STT 인식 테스트 서버 시작 — GPU 워밍업 중…", cat="READY")
|
||||
print("\n STT 워밍업 중… (모델 로드 + CUDA 예열)")
|
||||
dash.warm() # load + warm the GPU worker so the first recognition is instant
|
||||
dev = getattr(stt, "resolved_device", None) or "?"
|
||||
monitor.log("info", f"STT 준비 완료 (device={dev}). 녹음/파일 업로드로 인식하세요.")
|
||||
monitor.log("info", f"STT 준비 완료 (device={dev}). 녹음/파일 업로드로 인식하세요.", cat="READY")
|
||||
|
||||
shown = host if host not in ("0.0.0.0", "") else _lan_ip()
|
||||
print(f"\n 음성 인식 테스트 사이트: http://{shown}:{port} (STT device: {dev})")
|
||||
@@ -100,6 +100,28 @@ def _run_stt_test(host: str, port: int) -> None:
|
||||
dash.stop()
|
||||
|
||||
|
||||
def _apply_persisted_state(stt, tts, brain) -> None:
|
||||
"""Apply dashboard settings saved to the state store (STT/LLM model + TTS
|
||||
controls) onto freshly-built backends, so they survive a restart."""
|
||||
from . import state_store
|
||||
st = state_store.load()
|
||||
models = st.get("models") or {}
|
||||
if models.get("stt") and stt is not None:
|
||||
stt.model = str(models["stt"])
|
||||
if models.get("llm") and brain is not None:
|
||||
brain.model = str(models["llm"])
|
||||
tj = st.get("tts") or {}
|
||||
base = tj.get("base") or {}
|
||||
for k in ("speed", "word_gap", "sentence_gap", "pitch"):
|
||||
if base.get(k) is not None and tts is not None:
|
||||
try:
|
||||
setattr(tts, k, float(base[k]))
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
if tts is not None and isinstance(tj.get("overrides"), dict):
|
||||
tts.emotion_overrides = {k: dict(v) for k, v in tj["overrides"].items() if isinstance(v, dict)}
|
||||
|
||||
|
||||
def _run_voice_server(host: str, port: int) -> None:
|
||||
"""Serve the STT+TTS voice-turn endpoint that the Discord bot (dave/bot.mjs)
|
||||
calls: it POSTs a captured utterance wav and gets back the reply wav to play
|
||||
@@ -120,7 +142,7 @@ def _run_voice_server(host: str, port: int) -> None:
|
||||
if os.environ.get("WSAI_BRAIN", "claude").lower() not in ("none", "echo"):
|
||||
try:
|
||||
from .backends.claude import ClaudeBrain
|
||||
model = os.environ.get("WSAI_BRAIN_MODEL", "claude-sonnet-4-5")
|
||||
model = os.environ.get("WSAI_BRAIN_MODEL", "claude-sonnet-5")
|
||||
brain = ClaudeBrain(model=model)
|
||||
brain_name = "claude"
|
||||
except Exception as exc: # noqa: BLE001
|
||||
@@ -129,20 +151,24 @@ def _run_voice_server(host: str, port: int) -> None:
|
||||
monitor = Monitor()
|
||||
stt = WhisperSTT()
|
||||
tts = MeloTTS()
|
||||
# Restore persisted dashboard settings (models + TTS controls) BEFORE warmup
|
||||
# so the worker loads the last-chosen model. Bot lists/toggles are restored
|
||||
# inside BotControl. Applied before dash.warm() so nothing reloads twice.
|
||||
_apply_persisted_state(stt, tts, brain)
|
||||
dash = Dashboard(monitor, host=host, port=port, stt=stt, tts=tts, brain=brain)
|
||||
dash.start()
|
||||
monitor.set_components({"source": "none", "vision": "none", "stt": "whisper",
|
||||
"brain": brain_name, "tts": "melo"})
|
||||
monitor.set_status(running=True, listening=False)
|
||||
monitor.log("info", "디스코드 음성 서버 시작 — STT+TTS GPU 워밍업 중…")
|
||||
monitor.log("info", "디스코드 음성 서버 시작 — STT+TTS GPU 워밍업 중…", cat="READY")
|
||||
print("\n STT+TTS 워밍업 중… (모델 로드 + CUDA 예열)")
|
||||
dash.warm()
|
||||
sdev = getattr(stt, "resolved_device", None) or "?"
|
||||
monitor.set_status(listening=True)
|
||||
monitor.log("info", f"음성 서버 준비 완료 (STT device={sdev}). 디스코드 봇 연결 대기.")
|
||||
monitor.log("info", f"음성 서버 준비 완료 (STT device={sdev}). 디스코드 봇 연결 대기.", cat="READY")
|
||||
|
||||
shown = host if host not in ("0.0.0.0", "") else _lan_ip()
|
||||
print(f"\n 음성 서버 준비 완료 (STT device: {sdev}, 두뇌: {brain_name})")
|
||||
print(f"\n 음성 서버 준비 완료 (STT device: {sdev}, LLM: {brain_name})")
|
||||
print(f" 대시보드/상태: http://{shown}:{port}")
|
||||
print(f" 봇 연결 엔드포인트: http://127.0.0.1:{port}/api/voice-turn\n")
|
||||
try:
|
||||
|
||||
@@ -37,6 +37,10 @@ _CLAUDE_CODE_ID = "You are Claude Code, Anthropic's official CLI for Claude."
|
||||
# still fails fast rather than leaving the bot silent for many seconds.
|
||||
_MAX_RETRIES = int(os.environ.get("WSAI_BRAIN_MAX_RETRIES", "4"))
|
||||
|
||||
# Cap the reply length. Voice replies must be short (1 sentence), and a smaller
|
||||
# cap also means fewer tokens to generate -> lower latency. Tunable via env.
|
||||
_MAX_TOKENS = int(os.environ.get("WSAI_BRAIN_MAX_TOKENS", "150"))
|
||||
|
||||
|
||||
def _load_oauth_token() -> str | None:
|
||||
path = os.environ.get("CLAUDE_CREDENTIALS_PATH")
|
||||
@@ -123,23 +127,36 @@ class ClaudeVision:
|
||||
|
||||
|
||||
class ClaudeBrain:
|
||||
# Injected as an always-on, cached system block on top of the (editable)
|
||||
# persona. Sonnet 5 answers correctly but more verbosely than 4.5 for the
|
||||
# same voice prompt (measured 46-54 vs 28 output tokens), which erased its
|
||||
# ~0.4s time-to-first-token advantage in total turn time. This hard brevity
|
||||
# rule pulls Sonnet 5 back to ~30 tokens, so the faster first token actually
|
||||
# translates into a faster (and lower-variance) whole reply. Kept separate
|
||||
# from PERSONA so a dashboard persona edit can never drop it.
|
||||
BREVITY = "지금부터 답은 무조건 한 문장, 12단어 이내로만. 부연·재확인·군더더기 금지."
|
||||
|
||||
PERSONA = (
|
||||
"너는 디스코드를 이용해 사용자와 실시간으로 대화하는 AI 인공지능이야.\n\n"
|
||||
"1. 역할\n"
|
||||
"- 사용자의 말을 듣고 자연스럽게 대답한다.\n"
|
||||
"- 음성 대화에 어울리게 짧고 빠르게 반응한다.\n"
|
||||
"- 친구처럼 편하게, 무례하거나 과하게 장난치진 않는다.\n\n"
|
||||
"- 음성 대화에 어울리게 아주 짧고 빠르게 반응한다.\n"
|
||||
"- 다정하고 친근하게 대하되 항상 존댓말로 답한다. 무례하거나 과하게 장난치진 않는다.\n\n"
|
||||
"2. 언어\n"
|
||||
"- 음성 출력은 무조건 한국어다. 어떤 경우에도 한국어로만 답한다.\n"
|
||||
"- 항상 존댓말로 답한다. 어떤 경우에도 반말을 쓰지 않는다.\n"
|
||||
"- 사용자가 다른 언어로 말하거나 \"영어로 해줘\"처럼 다른 언어를 요청해도 한국어로 답한다.\n"
|
||||
"- 다른 언어를 요청받으면 한국어로 짧게 그렇게는 못 한다고 말한다.\n\n"
|
||||
"3. 답변 방식\n"
|
||||
"3. 답변 길이 (가장 중요)\n"
|
||||
"- 기본은 딱 한 문장. 정말 필요할 때만 최대 두 문장. 절대 길게 말하지 않는다.\n"
|
||||
"- 인사엔 인사만 짧게 답한다. 부르면 짧게 대답만 한다.\n"
|
||||
"- 자기소개나 \"무엇을 도와드릴까요\", \"필요한 거 있으면 말해\" 같은 상투적인 말을 덧붙이지 않는다.\n"
|
||||
"- 질문엔 군더더기 없이 핵심 답만 바로 말한다.\n"
|
||||
"- 음성으로 읽히니 마크다운·코드블록·특수기호·목록기호·이모지 없이 평범한 말로만 답한다.\n"
|
||||
"- 기본은 한두 문장, 길어도 10초 안팎. 길어질 땐 핵심부터 말하고 필요하면 이어서 설명한다.\n"
|
||||
"- URL·긴 숫자·시간·단위·코드는 소리내 읽기 좋게 풀어서 말한다.\n\n"
|
||||
"4. 대화 태도\n"
|
||||
"- 사용자의 말투·분위기에 맞춰 반응한다.\n"
|
||||
"- 모르면 지어내지 말고 모른다고 하고, 애매하면 되묻는다(\"다시 말해줄래?\").\n"
|
||||
"- 모르면 지어내지 말고 모른다고 하고, 애매하면 되묻는다(\"다시 말씀해 주시겠어요?\").\n"
|
||||
"- 잡음·침묵·의미 없는 소리엔 억지로 대답하지 않는다.\n\n"
|
||||
"5. 안전·사실성\n"
|
||||
"- 위험하거나 불법적인 요청은 돕지 않는다.\n"
|
||||
@@ -148,7 +165,7 @@ class ClaudeBrain:
|
||||
"6. 감정 표현\n"
|
||||
"- 감정은 대괄호 태그로 표현한다. 태그 자체는 읽히지 않고 뒤 문장의 목소리 톤(피치·속도)만 바뀐다.\n"
|
||||
"- 답변 맨 앞에 감정 태그 하나로 시작하고, 도중에 감정이 바뀌면 그 지점에 새 태그를 넣는다.\n"
|
||||
" 예: [속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!\n"
|
||||
" 예: [속상함] 정말 힘드셨겠어요. [힘차게] 하지만 하실 수 있어요!\n"
|
||||
"- 쓸 수 있는 감정: 기쁨, 신남, 힘차게, 속상함, 화남, 두려움, 놀람, 차분, 다정, 진지, 실망, 피곤, "
|
||||
"사랑스럽게, 웃으며, 속삭임, 외침, 단호, 안도, 궁금, 반가움.\n"
|
||||
"- 감정 단어가 아닌 진짜 대괄호(예: [1번], [메모])는 그대로 읽으니 필요하면 그렇게 써도 된다.\n\n"
|
||||
@@ -156,7 +173,7 @@ class ClaudeBrain:
|
||||
"- 너는 디스코드에서 함께 대화하는 실시간 AI 인공지능이다."
|
||||
)
|
||||
|
||||
def __init__(self, *, model: str = "claude-sonnet-4-5", api_key: str | None = None) -> None:
|
||||
def __init__(self, *, model: str = "claude-sonnet-5", api_key: str | None = None) -> None:
|
||||
self.model = model
|
||||
self._auth = _Auth(api_key)
|
||||
|
||||
@@ -169,11 +186,18 @@ class ClaudeBrain:
|
||||
msgs.append({"role": "user", "content": screen_note + user_text})
|
||||
client = self._auth.client()
|
||||
# Read the persona live each turn so a dashboard edit applies immediately
|
||||
# (falls back to the built-in PERSONA when no override is saved).
|
||||
# (falls back to the built-in PERSONA when no override is saved). The
|
||||
# brevity rule is appended as its own block so it survives persona edits.
|
||||
system = self._auth.system(get_persona(self.PERSONA), self.BREVITY)
|
||||
if system:
|
||||
# Cache the (static) system prompt so repeat turns skip re-processing
|
||||
# it — lower time-to-first-token. No-op below the model's cache
|
||||
# minimum, so it's harmless when it doesn't engage.
|
||||
system[-1] = {**system[-1], "cache_control": {"type": "ephemeral"}}
|
||||
resp = await client.messages.create(
|
||||
model=self.model,
|
||||
max_tokens=400,
|
||||
system=self._auth.system(get_persona(self.PERSONA)),
|
||||
max_tokens=_MAX_TOKENS,
|
||||
system=system,
|
||||
messages=msgs,
|
||||
)
|
||||
text = "".join(b.text for b in resp.content if b.type == "text")
|
||||
|
||||
@@ -12,7 +12,7 @@ Env:
|
||||
WSAI_MELO_DEVICE cpu | cuda | auto (default auto: GPU if torch sees one,
|
||||
else CPU; the worker falls back to CPU if CUDA fails)
|
||||
WSAI_TTS_OUT_DIR where wavs are written (default ~/.cache/wsai/tts)
|
||||
WSAI_TTS_SPEED glyph speed / length_scale multiplier (0.5..2.0, default 1.35)
|
||||
WSAI_TTS_SPEED glyph speed / length_scale multiplier (0.5..2.0, default 1.4)
|
||||
WSAI_TTS_WORD_GAP intra-sentence pause delta, sec (-0.2..0.5, default -0.07)
|
||||
WSAI_TTS_SENTENCE_GAP sentence-boundary silence, sec (-0.5..1.5, default -0.30)
|
||||
WSAI_TTS_PITCH global semitone offset (-12..12, default 0.0)
|
||||
@@ -99,7 +99,7 @@ class MeloTTS:
|
||||
self.out_dir = Path(out_dir or os.environ.get("WSAI_TTS_OUT_DIR")
|
||||
or (Path.home() / ".cache/wsai/tts"))
|
||||
self.speed = float(speed if speed is not None
|
||||
else os.environ.get("WSAI_TTS_SPEED", "1.35"))
|
||||
else os.environ.get("WSAI_TTS_SPEED", "1.4"))
|
||||
# Reply-global rhythm/pitch controls (see melo_worker.py / docs manual).
|
||||
self.word_gap = float(word_gap if word_gap is not None
|
||||
else os.environ.get("WSAI_TTS_WORD_GAP", "-0.07"))
|
||||
|
||||
@@ -18,7 +18,7 @@ stays usable directly.
|
||||
Env:
|
||||
WSAI_WHISPER_PYTHON interpreter with faster-whisper installed
|
||||
(default: /home/claude/jarvis-stt/whisper312/bin/python)
|
||||
WSAI_WHISPER_MODEL model size/name (default: small)
|
||||
WSAI_WHISPER_MODEL model size/name (default: medium)
|
||||
WSAI_WHISPER_DEVICE cpu | cuda | auto (default auto: GPU if present,
|
||||
else CPU; the worker falls back to CPU if CUDA fails)
|
||||
WSAI_WHISPER_LANGUAGE forced language, e.g. ko (default ko; "" = autodetect)
|
||||
@@ -67,7 +67,7 @@ class WhisperSTT:
|
||||
audio_source: AsyncIterator[str] | None = None,
|
||||
) -> None:
|
||||
self.python = python or os.environ.get("WSAI_WHISPER_PYTHON", _DEFAULT_PYTHON)
|
||||
self.model = model or os.environ.get("WSAI_WHISPER_MODEL", "small")
|
||||
self.model = model or os.environ.get("WSAI_WHISPER_MODEL", "medium")
|
||||
self.device = device or os.environ.get("WSAI_WHISPER_DEVICE", "auto")
|
||||
# "" means autodetect; a real code like "ko" forces the language.
|
||||
env_lang = os.environ.get("WSAI_WHISPER_LANGUAGE", "ko")
|
||||
|
||||
@@ -23,6 +23,31 @@ import os
|
||||
import sys
|
||||
import time
|
||||
|
||||
# Whisper's classic Korean hallucinations on silence/noise/keyboard clatter —
|
||||
# it "hears" video-outro boilerplate. Drop these when the whole utterance is one
|
||||
# of them AND the segment looked like non-speech, so a real "감사합니다" survives.
|
||||
_HALLUCINATIONS = {
|
||||
"감사합니다", "고맙습니다", "감사합니다.", "고맙습니다.",
|
||||
"시청해주셔서 감사합니다", "시청해 주셔서 감사합니다", "끝까지 시청해주셔서 감사합니다",
|
||||
"구독과 좋아요 부탁드립니다", "구독 좋아요 부탁드립니다", "다음 영상에서 만나요",
|
||||
"다음 시간에 만나요", "안녕히 계세요",
|
||||
}
|
||||
|
||||
|
||||
def _norm(t: str) -> str:
|
||||
return t.strip().rstrip(" .!?…~").strip()
|
||||
|
||||
|
||||
def _guard_hallucination(text: str, worst_no_speech: float) -> str:
|
||||
"""Blank out a lone known-hallucination phrase when the audio was probably
|
||||
not speech (high no_speech_prob)."""
|
||||
n = _norm(text)
|
||||
if not n:
|
||||
return ""
|
||||
if n in {_norm(h) for h in _HALLUCINATIONS} and worst_no_speech > 0.5:
|
||||
return ""
|
||||
return text
|
||||
|
||||
# Split protocol from library noise BEFORE importing anything heavy.
|
||||
_proto = os.fdopen(os.dup(1), "w", buffering=1) # private copy of real stdout
|
||||
os.dup2(2, 1) # fd1 -> stderr, so stray library prints don't hit the protocol
|
||||
@@ -108,9 +133,31 @@ def main() -> None:
|
||||
wav,
|
||||
language=language,
|
||||
beam_size=int(req.get("beam_size", 5)),
|
||||
# VAD strips non-speech (keyboard clatter, room noise, silence)
|
||||
# before decoding, which both improves accuracy and kills most
|
||||
# hallucinations. speech_pad_ms keeps a little lead/trail so soft
|
||||
# first/last words aren't clipped.
|
||||
vad_filter=bool(req.get("vad_filter", True)),
|
||||
vad_parameters=dict(min_silence_duration_ms=300, speech_pad_ms=250),
|
||||
# Don't feed the previous text back in — that's what makes Whisper
|
||||
# loop/hallucinate. Temperature fallback + thresholds reject
|
||||
# low-confidence (noisy/quiet) decodes instead of inventing words.
|
||||
condition_on_previous_text=False,
|
||||
temperature=[0.0, 0.2, 0.4, 0.6],
|
||||
no_speech_threshold=0.6,
|
||||
log_prob_threshold=-1.0,
|
||||
compression_ratio_threshold=2.4,
|
||||
)
|
||||
text = "".join(seg.text for seg in segments).strip()
|
||||
parts, worst_ns = [], 0.0
|
||||
for seg in segments:
|
||||
nsp = float(getattr(seg, "no_speech_prob", 0.0) or 0.0)
|
||||
alp = float(getattr(seg, "avg_logprob", 0.0) or 0.0)
|
||||
# Drop a segment that is almost certainly non-speech noise.
|
||||
if nsp > 0.8 and alp < -0.4:
|
||||
continue
|
||||
worst_ns = max(worst_ns, nsp)
|
||||
parts.append(seg.text)
|
||||
text = _guard_hallucination("".join(parts).strip(), worst_ns)
|
||||
ms = int((time.monotonic() - s) * 1000)
|
||||
_emit({"ok": True, "text": text, "language": info.language, "ms": ms})
|
||||
except Exception as exc: # keep the worker alive across bad requests
|
||||
|
||||
@@ -31,10 +31,19 @@ class BotControl:
|
||||
self._cmd_id = 0
|
||||
# Per-guild listen filter. Empty whitelist => listen to everyone;
|
||||
# blacklist always excludes. Users and roles both supported.
|
||||
self._lists: dict[str, dict[str, Any]] = {}
|
||||
# Restored from the persistent state store so it survives a restart.
|
||||
from . import state_store
|
||||
st = state_store.load()
|
||||
self._lists: dict[str, dict[str, Any]] = st.get("lists") or {}
|
||||
# Bot behaviour settings the dashboard toggles and the bot reads on each
|
||||
# report. bargeIn: stop the bot's current TTS the moment a user speaks.
|
||||
self._settings: dict[str, Any] = {"bargeIn": True}
|
||||
# bargeInMs: how long (ms) a user must keep speaking before that barge-in
|
||||
# fires — the "유저 음성 인식 시간" the dashboard exposes (default 700).
|
||||
self._settings: dict[str, Any] = {
|
||||
"bargeIn": True,
|
||||
"bargeInMs": 700,
|
||||
**(st.get("botSettings") or {}),
|
||||
}
|
||||
|
||||
# -- bot behaviour settings ------------------------------------------ #
|
||||
def get_settings(self) -> dict[str, Any]:
|
||||
@@ -45,7 +54,17 @@ class BotControl:
|
||||
with self._lock:
|
||||
if "bargeIn" in data:
|
||||
self._settings["bargeIn"] = bool(data["bargeIn"])
|
||||
return dict(self._settings)
|
||||
if "bargeInMs" in data:
|
||||
try:
|
||||
ms = int(data["bargeInMs"])
|
||||
except (TypeError, ValueError):
|
||||
ms = 700
|
||||
# Keep it sane: 0ms = instant, cap at 5s so a typo can't wedge it.
|
||||
self._settings["bargeInMs"] = max(0, min(5000, ms))
|
||||
out = dict(self._settings)
|
||||
from . import state_store
|
||||
state_store.patch("botSettings", out)
|
||||
return out
|
||||
|
||||
# -- whitelist / blacklist (per guild) ------------------------------- #
|
||||
@staticmethod
|
||||
@@ -72,6 +91,9 @@ class BotControl:
|
||||
]
|
||||
with self._lock:
|
||||
self._lists[guild_id] = clean
|
||||
snapshot = {g: dict(v) for g, v in self._lists.items()}
|
||||
from . import state_store
|
||||
state_store.patch("lists", snapshot) # persist per-guild lists across restarts
|
||||
return clean
|
||||
|
||||
# -- bot -> dashboard (state push) ----------------------------------- #
|
||||
|
||||
@@ -24,7 +24,7 @@ class Settings:
|
||||
text: str | None = None # None | (discord)
|
||||
|
||||
capture_interval: float = 1.5
|
||||
anthropic_model: str = "claude-sonnet-4-5"
|
||||
anthropic_model: str = "claude-sonnet-5"
|
||||
|
||||
@classmethod
|
||||
def from_env(cls) -> "Settings":
|
||||
|
||||
@@ -17,8 +17,10 @@ from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import queue
|
||||
import threading
|
||||
import time
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
|
||||
from .monitor import Monitor
|
||||
@@ -49,6 +51,20 @@ def _thought_summary(reply: str) -> str:
|
||||
return "감정 태그 없음 · 기본 톤으로 답변"
|
||||
|
||||
|
||||
# Lone connective/fillers that clearly expect more to come ("그러면…"). On their
|
||||
# own they carry no answerable intent, so we discard them instead of blurting a
|
||||
# confused reply. (A fuller version would buffer and wait for the continuation.)
|
||||
_FRAGMENT_FILLERS = {
|
||||
"그러면", "그래서", "그런데", "근데", "그리고", "그러니까", "그럼", "그게",
|
||||
"저기", "있잖아", "그", "음", "어", "저", "그러", "그니까",
|
||||
}
|
||||
|
||||
|
||||
def _is_fragment(text: str) -> bool:
|
||||
t = (text or "").strip().rstrip(" .,!?…~").strip()
|
||||
return t in _FRAGMENT_FILLERS
|
||||
|
||||
|
||||
def _make_handler(dash: "Dashboard"):
|
||||
monitor = dash.monitor
|
||||
|
||||
@@ -87,6 +103,8 @@ def _make_handler(dash: "Dashboard"):
|
||||
self._handle_tts_settings_get()
|
||||
elif path == "/api/bot/settings":
|
||||
self._send_json({"ok": True, "settings": dash.bot.get_settings()})
|
||||
elif path == "/api/models":
|
||||
self._send_json(dash.models_settings())
|
||||
elif path == "/events":
|
||||
self._stream_events()
|
||||
else:
|
||||
@@ -103,6 +121,11 @@ def _make_handler(dash: "Dashboard"):
|
||||
elif path == "/api/logs/clear":
|
||||
monitor.clear_events()
|
||||
self._send(200, json.dumps({"ok": True}).encode(), "application/json; charset=utf-8")
|
||||
elif path == "/api/turns/clear":
|
||||
monitor.clear_turns()
|
||||
self._send(200, json.dumps({"ok": True}).encode(), "application/json; charset=utf-8")
|
||||
elif path == "/api/turns/delete":
|
||||
self._handle_turn_delete()
|
||||
elif path == "/api/logs/delete":
|
||||
self._handle_log_mutate("delete")
|
||||
elif path == "/api/logs/edit":
|
||||
@@ -119,6 +142,10 @@ def _make_handler(dash: "Dashboard"):
|
||||
self._handle_tts_preview()
|
||||
elif path == "/api/bot/settings":
|
||||
self._handle_bot_settings_post()
|
||||
elif path == "/api/models/stt":
|
||||
self._handle_model_switch("stt")
|
||||
elif path == "/api/models/llm":
|
||||
self._handle_model_switch("llm")
|
||||
else:
|
||||
self._send(404, b"not found", "text/plain; charset=utf-8")
|
||||
|
||||
@@ -226,7 +253,7 @@ def _make_handler(dash: "Dashboard"):
|
||||
return
|
||||
prompt_store.set_persona(prompt)
|
||||
monitor.log("info", "봇 프롬프트가 수정되었습니다" if prompt.strip()
|
||||
else "봇 프롬프트가 기본값으로 초기화되었습니다")
|
||||
else "봇 프롬프트가 기본값으로 초기화되었습니다", cat="PROMPT")
|
||||
body = json.dumps({
|
||||
"ok": True,
|
||||
"prompt": prompt_store.get_persona(self._default_persona()),
|
||||
@@ -247,7 +274,17 @@ def _make_handler(dash: "Dashboard"):
|
||||
except (ValueError, AttributeError):
|
||||
self._send_json({"ok": False, "error": "invalid JSON"}, 400)
|
||||
return
|
||||
was_connected = dash.bot.state().get("connected")
|
||||
dash.bot.report(data)
|
||||
if not was_connected: # transition disconnected -> connected
|
||||
# identity is reported as {id, username, tag}; pull a string only
|
||||
# (concatenating a dict here would TypeError and 500 the report).
|
||||
ident = data.get("identity")
|
||||
if isinstance(ident, dict):
|
||||
who = ident.get("tag") or ident.get("username") or ident.get("id") or ""
|
||||
else:
|
||||
who = ident if isinstance(ident, str) else ""
|
||||
monitor.log("info", f"디스코드 봇 연결됨{(' · ' + who) if who else ''}", cat="CONNECT")
|
||||
# Hand the bot any queued commands + the current listen filters +
|
||||
# behaviour settings in the same round trip so it does not have to
|
||||
# poll extra endpoints.
|
||||
@@ -267,10 +304,10 @@ def _make_handler(dash: "Dashboard"):
|
||||
channel_id = (data.get("channelId") or "").strip()
|
||||
if channel_id and guild_id:
|
||||
cid = dash.bot.enqueue({"type": "join", "guildId": guild_id, "channelId": channel_id})
|
||||
monitor.log("info", f"음성채널 참여 요청 (guild={guild_id} channel={channel_id})")
|
||||
monitor.log("info", f"음성채널 참여 요청 (guild={guild_id} channel={channel_id})", cat="VOICE")
|
||||
else:
|
||||
cid = dash.bot.enqueue({"type": "leave"})
|
||||
monitor.log("info", "음성채널 나가기 요청")
|
||||
monitor.log("info", "음성채널 나가기 요청", cat="VOICE")
|
||||
self._send_json({"ok": True, "commandId": cid})
|
||||
|
||||
def _handle_bot_lists(self) -> None:
|
||||
@@ -285,9 +322,28 @@ def _make_handler(dash: "Dashboard"):
|
||||
self._send_json({"ok": False, "error": str(exc)}, 400)
|
||||
return
|
||||
saved = dash.bot.set_lists(guild_id, data.get("lists") or {})
|
||||
monitor.log("info", f"청취 화이트/블랙리스트 업데이트 (guild={guild_id})")
|
||||
monitor.log("info", f"청취 화이트/블랙리스트 업데이트 (guild={guild_id})", cat="FILTER")
|
||||
self._send_json({"ok": True, "guildId": guild_id, "lists": saved})
|
||||
|
||||
def _handle_model_switch(self, which: str) -> None:
|
||||
"""Switch the STT size or the LLM model live."""
|
||||
raw = self._read_body()
|
||||
try:
|
||||
data = json.loads(raw.decode("utf-8")) if raw else {}
|
||||
model = data.get("model")
|
||||
except (ValueError, AttributeError):
|
||||
self._send_json({"ok": False, "error": "invalid JSON"}, 400)
|
||||
return
|
||||
try:
|
||||
if which == "stt":
|
||||
res = dash.set_stt_model(model)
|
||||
else:
|
||||
res = dash.set_llm_model(model)
|
||||
except Exception as exc: # noqa: BLE001 — surface the reason to the page
|
||||
self._send_json({"ok": False, "error": f"{type(exc).__name__}: {exc}"}, 400)
|
||||
return
|
||||
self._send_json({"ok": True, which: res})
|
||||
|
||||
def _handle_bot_settings_post(self) -> None:
|
||||
"""Save a bot behaviour toggle (e.g. bargeIn). The bot reads the new
|
||||
value on its next report round-trip."""
|
||||
@@ -298,7 +354,7 @@ def _make_handler(dash: "Dashboard"):
|
||||
self._send_json({"ok": False, "error": "invalid JSON"}, 400)
|
||||
return
|
||||
settings = dash.bot.set_settings(data)
|
||||
monitor.log("info", "봇 설정 변경: " + json.dumps(settings, ensure_ascii=False))
|
||||
monitor.log("info", "봇 설정 변경: " + json.dumps(settings, ensure_ascii=False), cat="SETTING")
|
||||
self._send_json({"ok": True, "settings": settings})
|
||||
|
||||
def _handle_tts_settings_get(self) -> None:
|
||||
@@ -324,7 +380,8 @@ def _make_handler(dash: "Dashboard"):
|
||||
tgt = data.get("emotion") or "base"
|
||||
monitor.log("info", f"봇 목소리 파라미터 적용 ({tgt}): base="
|
||||
+ json.dumps(res["base"], ensure_ascii=False)
|
||||
+ " overrides=" + json.dumps(res["overrides"], ensure_ascii=False))
|
||||
+ " overrides=" + json.dumps(res["overrides"], ensure_ascii=False),
|
||||
cat="TTS")
|
||||
self._send_json(res)
|
||||
|
||||
def _handle_tts_preview(self) -> None:
|
||||
@@ -351,6 +408,20 @@ def _make_handler(dash: "Dashboard"):
|
||||
return
|
||||
self._send(200, wav, "audio/wav")
|
||||
|
||||
def _handle_turn_delete(self) -> None:
|
||||
"""Delete one conversation turn by id."""
|
||||
raw = self._read_body()
|
||||
try:
|
||||
data = json.loads(raw.decode("utf-8")) if raw else {}
|
||||
turn_id = int(data.get("id"))
|
||||
except (ValueError, TypeError, AttributeError):
|
||||
self._send(400, json.dumps({"ok": False, "error": "id required"}).encode(),
|
||||
"application/json; charset=utf-8")
|
||||
return
|
||||
ok = monitor.delete_turn(turn_id)
|
||||
self._send(200 if ok else 404,
|
||||
json.dumps({"ok": ok}).encode(), "application/json; charset=utf-8")
|
||||
|
||||
def _handle_log_mutate(self, action: str) -> None:
|
||||
"""Per-line log delete/edit by event id."""
|
||||
raw = self._read_body()
|
||||
@@ -422,6 +493,10 @@ class Dashboard:
|
||||
self.bot = BotControl() # dashboard <-> Discord bot control plane
|
||||
self._history: list[tuple[str, str]] = []
|
||||
self._history_turns = history_turns
|
||||
# A lone connective ("그러면") is held here to see if the next utterance
|
||||
# continues it; merged if it arrives within the window, else dropped.
|
||||
self._pending_fragment: dict | None = None
|
||||
self._fragment_window = float(os.environ.get("WSAI_FRAGMENT_MERGE_MS", "5000")) / 1000.0
|
||||
self._server: ThreadingHTTPServer | None = None
|
||||
self._thread: threading.Thread | None = None
|
||||
self._loop = None
|
||||
@@ -484,12 +559,26 @@ class Dashboard:
|
||||
def _tts_base(self) -> dict:
|
||||
t = self.tts
|
||||
return {
|
||||
"speed": float(getattr(t, "speed", 1.35)),
|
||||
"speed": float(getattr(t, "speed", 1.4)),
|
||||
"word_gap": float(getattr(t, "word_gap", -0.07)),
|
||||
"sentence_gap": float(getattr(t, "sentence_gap", -0.30)),
|
||||
"pitch": float(getattr(t, "pitch", 0.0)),
|
||||
}
|
||||
|
||||
def _persist_tts(self) -> None:
|
||||
from . import state_store
|
||||
state_store.patch("tts", {
|
||||
"base": self._tts_base(),
|
||||
"overrides": dict(getattr(self.tts, "emotion_overrides", {}) or {}),
|
||||
})
|
||||
|
||||
def _persist_models(self) -> None:
|
||||
from . import state_store
|
||||
state_store.patch("models", {
|
||||
"stt": getattr(self.stt, "model", None),
|
||||
"llm": getattr(self.brain, "model", None),
|
||||
})
|
||||
|
||||
def tts_settings(self) -> dict:
|
||||
"""Base (공통) controls + the per-emotion overrides + the emotion list
|
||||
(canonical key + Korean label) so the dashboard can offer per-emotion
|
||||
@@ -505,11 +594,20 @@ class Dashboard:
|
||||
}
|
||||
|
||||
def set_tts_settings(self, data: dict) -> dict:
|
||||
"""Apply controls. Without ``emotion`` (or emotion == "base"), set the
|
||||
base/공통 values on the live TTS instance. With a specific emotion, store
|
||||
(or, if ``reset``, clear) that emotion's override."""
|
||||
"""Apply controls. ``emotion`` selects the target:
|
||||
- "all" → set base to the values AND clear every per-emotion override,
|
||||
so ALL emotions speak with these values (전체변경).
|
||||
- "" / "base" → set the base/공통 values on the live TTS instance.
|
||||
- a specific emotion → store (or, if ``reset``, clear) its override."""
|
||||
emotion = data.get("emotion")
|
||||
if not emotion or emotion == "base":
|
||||
if emotion == "all":
|
||||
for key, (lo, hi, _step) in self._TTS_RANGES.items():
|
||||
v = data.get(key)
|
||||
if v is not None:
|
||||
setattr(self.tts, key, float(max(lo, min(hi, float(v)))))
|
||||
if getattr(self.tts, "emotion_overrides", None):
|
||||
self.tts.emotion_overrides.clear() # all emotions inherit the new base
|
||||
elif not emotion or emotion == "base":
|
||||
for key, (lo, hi, _step) in self._TTS_RANGES.items():
|
||||
v = data.get(key)
|
||||
if v is not None:
|
||||
@@ -522,6 +620,7 @@ class Dashboard:
|
||||
store.pop(emotion, None)
|
||||
else:
|
||||
store[emotion] = self._tts_overrides(data)
|
||||
self._persist_tts()
|
||||
return self.tts_settings()
|
||||
|
||||
def _tts_overrides(self, data: dict) -> dict:
|
||||
@@ -548,6 +647,81 @@ class Dashboard:
|
||||
pass
|
||||
return wav
|
||||
|
||||
# -- live model switching (STT size / LLM model) --------------------- #
|
||||
STT_OPTIONS = ["tiny", "base", "small", "medium", "large-v3"]
|
||||
LLM_OPTIONS = ["claude-haiku-4-5", "claude-sonnet-4-5", "claude-sonnet-5"]
|
||||
|
||||
def models_settings(self) -> dict:
|
||||
stt = self.stt
|
||||
brain = self.brain
|
||||
return {
|
||||
"ok": True,
|
||||
"stt": {
|
||||
"enabled": stt is not None,
|
||||
"current": getattr(stt, "model", None),
|
||||
"ready": bool(getattr(stt, "_ready", False)),
|
||||
"device": getattr(stt, "resolved_device", None),
|
||||
"options": self.STT_OPTIONS,
|
||||
},
|
||||
"llm": {
|
||||
"enabled": brain is not None,
|
||||
"current": getattr(brain, "model", None),
|
||||
"options": self.LLM_OPTIONS,
|
||||
},
|
||||
}
|
||||
|
||||
def set_stt_model(self, model: str) -> dict:
|
||||
"""Switch the whisper model size live. Tears down the current worker and
|
||||
warms the new one in the BACKGROUND so the HTTP call returns fast (the
|
||||
first switch to a not-yet-downloaded size fetches it, which can take a
|
||||
while); the next utterance waits for the reload if it isn't warm yet."""
|
||||
import asyncio
|
||||
if self.stt is None:
|
||||
raise RuntimeError("STT not enabled")
|
||||
model = str(model).strip()
|
||||
if not model:
|
||||
raise ValueError("model required")
|
||||
if model not in self.STT_OPTIONS:
|
||||
raise ValueError(f"unknown STT model: {model}")
|
||||
changed = model != self.stt.model
|
||||
if changed:
|
||||
self.stt.model = model
|
||||
self._submit(self.stt.aclose()) # drop old worker (fast)
|
||||
# Reload+warm in the background; don't block the HTTP response.
|
||||
asyncio.run_coroutine_threadsafe(self._warm_stt_bg(), self._loop)
|
||||
self.monitor.log("info", f"STT 모델 전환 시작: {model} (로딩 중…)", cat="MODEL")
|
||||
self._persist_models()
|
||||
return {**self.models_settings()["stt"], "changed": changed}
|
||||
|
||||
async def _warm_stt_bg(self) -> None:
|
||||
try:
|
||||
await self.stt.warmup()
|
||||
self.monitor.log("info", f"STT 모델 로드 완료: {self.stt.model} "
|
||||
f"(device={getattr(self.stt, 'resolved_device', '?')})", cat="READY")
|
||||
except Exception as exc: # noqa: BLE001
|
||||
self.monitor.log("error", f"STT 모델 로드 실패({self.stt.model}): {exc}", cat="MODEL")
|
||||
|
||||
def set_llm_model(self, model: str) -> dict:
|
||||
"""Switch the Claude model live — applied on the next reply (no reload)."""
|
||||
if self.brain is None:
|
||||
raise RuntimeError("LLM(brain) not enabled — echo 모드입니다")
|
||||
model = str(model).strip()
|
||||
if model not in self.LLM_OPTIONS:
|
||||
raise ValueError(f"unknown LLM model: {model}")
|
||||
changed = model != self.brain.model
|
||||
self.brain.model = model
|
||||
if changed:
|
||||
self.monitor.log("info", f"LLM 모델 변경: {model} (다음 답변부터 적용)", cat="MODEL")
|
||||
self._persist_models()
|
||||
return {**self.models_settings()["llm"], "changed": changed}
|
||||
|
||||
@staticmethod
|
||||
def _step(turn, name: str, ms: float) -> None:
|
||||
"""Record one finished timing step (STT / LLM / TTS) on a turn."""
|
||||
st = turn.step(name)
|
||||
st.ok, st.ms = True, float(ms)
|
||||
turn._steps.append(st)
|
||||
|
||||
def voice_turn(self, audio_bytes: bytes, speaker: str = "",
|
||||
guild: str = "", channel: str = "") -> dict:
|
||||
"""One Discord voice turn: decode the uploaded utterance, recognise it
|
||||
@@ -578,6 +752,7 @@ class Dashboard:
|
||||
check=True, capture_output=True,
|
||||
)
|
||||
heard = (self._submit(self.stt.transcribe(wav)) or "").strip()
|
||||
self._step(turn, "STT", (time.monotonic() - t0) * 1000) # 인식(+디코드)
|
||||
turn.heard(heard or "(빈 결과)")
|
||||
if not heard:
|
||||
# Nothing recognised (silence/noise): mark it as [잡음] and skip
|
||||
@@ -586,16 +761,35 @@ class Dashboard:
|
||||
turn.replied("[잡음]")
|
||||
turn.finish()
|
||||
return {"heard": heard, "reply": "[잡음]", "wav": b""}
|
||||
now = time.monotonic()
|
||||
pending = self._pending_fragment
|
||||
if pending and (now - pending["ts"]) > self._fragment_window:
|
||||
pending = self._pending_fragment = None # too old — give up on it
|
||||
if _is_fragment(heard):
|
||||
# A lone connective ("그러면") — hold it and wait for the next
|
||||
# utterance to continue it. If nothing follows within the window
|
||||
# it's silently dropped (never answered).
|
||||
self._pending_fragment = {"text": heard, "ts": now}
|
||||
turn.thought("불완전한 말(연결어) — 이어질 말 대기 (없으면 폐기)")
|
||||
turn.replied("[대기]")
|
||||
turn.finish()
|
||||
return {"heard": heard, "reply": "[대기]", "wav": b""}
|
||||
if pending:
|
||||
# A real utterance arrived in time — merge the held fragment in
|
||||
# front and answer the whole thing ("그러면" + "뭐 먹지").
|
||||
heard = (pending["text"].rstrip(" .,!?…~") + " " + heard).strip()
|
||||
self._pending_fragment = None
|
||||
turn.heard(heard)
|
||||
t_llm = time.monotonic()
|
||||
reply_text = self._think(heard)
|
||||
self._step(turn, "LLM" if self.brain else "echo", (time.monotonic() - t_llm) * 1000)
|
||||
turn.thought(_thought_summary(reply_text))
|
||||
turn.replied(reply_text)
|
||||
t_tts = time.monotonic()
|
||||
out_path = self._submit(self.tts.synth(_speech_text(reply_text)))
|
||||
with open(out_path, "rb") as f:
|
||||
reply_wav = f.read()
|
||||
ms = int((time.monotonic() - t0) * 1000)
|
||||
step = turn.step("STT+두뇌+TTS" if self.brain else "STT+TTS(GPU)")
|
||||
step.ok, step.ms = True, float(ms)
|
||||
turn._steps.append(step)
|
||||
self._step(turn, "TTS", (time.monotonic() - t_tts) * 1000) # 합성
|
||||
turn.finish()
|
||||
try:
|
||||
os.remove(out_path)
|
||||
@@ -631,7 +825,7 @@ class Dashboard:
|
||||
self.monitor.add_claude_usage(u.get("input", 0), u.get("output", 0))
|
||||
except Exception as exc: # noqa: BLE001
|
||||
log.exception("brain failed")
|
||||
self.monitor.log("error", f"두뇌 응답 실패: {exc}")
|
||||
self.monitor.log("error", f"LLM 응답 실패: {exc}", cat="BRAIN")
|
||||
blob = f"{getattr(exc, 'status_code', '')} {exc}".lower()
|
||||
if "529" in blob or "overload" in blob:
|
||||
# Transient server overload survived the SDK retries.
|
||||
@@ -729,8 +923,11 @@ PAGE = r"""<!DOCTYPE html>
|
||||
.comp{display:flex;gap:8px;flex-wrap:wrap;margin:2px 0 18px}
|
||||
.comp .pill{font-size:11.5px}
|
||||
.turn{background:var(--panel);border:1px solid var(--line);border-radius:14px;padding:14px 16px;margin:12px 0;
|
||||
animation:rise .25s ease}
|
||||
animation:rise .25s ease;position:relative}
|
||||
@keyframes rise{from{opacity:0;transform:translateY(6px)}to{opacity:1;transform:none}}
|
||||
.tdel{margin-left:auto;background:none;border:none;color:var(--muted);cursor:pointer;font-size:14px;
|
||||
line-height:1;padding:2px 4px;border-radius:6px}
|
||||
.tdel:hover{color:var(--err);background:#2a1620}
|
||||
.turn.active{border-color:#2b5d86;box-shadow:0 0 0 1px #14324a inset}
|
||||
.turn.error{border-color:#5c2530}
|
||||
.trow{display:flex;align-items:baseline;gap:10px;margin-bottom:8px}
|
||||
@@ -771,6 +968,12 @@ PAGE = r"""<!DOCTYPE html>
|
||||
.cfgrow{display:flex;gap:9px;align-items:center;font-size:13px;color:var(--fg);cursor:pointer;padding:4px 0}
|
||||
.cfgrow input[type=checkbox]{width:16px;height:16px;accent-color:#3aa0ff;cursor:pointer}
|
||||
.cfghint{color:var(--muted);font-weight:400;font-size:12px}
|
||||
.mload{display:flex;align-items:center;gap:9px;margin-top:10px;padding:8px 12px;
|
||||
background:#0f2333;border:1px solid #234a63;border-radius:10px;font-size:13px;color:var(--fg)}
|
||||
.mload .bar{flex:1;height:6px;border-radius:4px;background:#12304a;overflow:hidden;position:relative}
|
||||
.spin{width:15px;height:15px;border:2px solid #2a4a63;border-top-color:#3aa0ff;border-radius:50%;
|
||||
display:inline-block;animation:spin 0.8s linear infinite}
|
||||
@keyframes spin{to{transform:rotate(360deg)}}
|
||||
.sttrow{display:flex;gap:10px;align-items:center;flex-wrap:wrap}
|
||||
.btn{background:#173042;border:1px solid #234a63;color:var(--fg);border-radius:10px;padding:8px 14px;font-size:13px;cursor:pointer}
|
||||
.btn:hover{background:#1d3d54}
|
||||
@@ -836,6 +1039,10 @@ PAGE = r"""<!DOCTYPE html>
|
||||
.lst-item:last-child{border-bottom:none}
|
||||
.lst-item .nm{flex:1}
|
||||
.lst-item .rl{color:var(--muted);font-size:11px}
|
||||
.role-h{cursor:pointer;user-select:none}
|
||||
.role-h .rcaret{display:inline-block;width:11px;color:var(--muted);font-weight:400}
|
||||
.role-members{background:#0c141d;border-top:1px solid #16202b}
|
||||
.role-members .lst-item{padding-left:26px}
|
||||
.mini{padding:3px 8px;font-size:11.5px;border-radius:7px;cursor:pointer;border:1px solid var(--line);background:#173042;color:var(--fg)}
|
||||
.mini.w{border-color:#1f5236;color:#9ff0bd}
|
||||
.mini.b{border-color:#5c2530;color:#ffb3bb}
|
||||
@@ -849,7 +1056,13 @@ PAGE = r"""<!DOCTYPE html>
|
||||
main{padding-bottom:46px}
|
||||
.logdock{position:fixed;left:0;right:0;bottom:0;z-index:15;background:#0a0e13;
|
||||
border-top:1px solid var(--line);display:flex;flex-direction:column;
|
||||
max-height:45vh;box-shadow:0 -8px 24px rgba(0,0,0,.35)}
|
||||
max-height:90vh;box-shadow:0 -8px 24px rgba(0,0,0,.35)}
|
||||
.logresize{height:7px;cursor:ns-resize;background:#0d141c;border-bottom:1px solid var(--line);
|
||||
flex:0 0 auto;touch-action:none}
|
||||
.logresize:hover{background:#1d3d54}
|
||||
.logresize::before{content:"";display:block;width:44px;height:3px;margin:2px auto 0;
|
||||
border-radius:2px;background:#2a3a49}
|
||||
.logdock.collapsed .logresize{display:none}
|
||||
.logbar{display:flex;align-items:center;gap:8px;padding:6px 12px;border-bottom:1px solid var(--line);
|
||||
background:#0d141c;flex-wrap:wrap}
|
||||
.logtoggle{background:none;border:none;color:var(--fg);font-size:12.5px;cursor:pointer;font-weight:600;padding:4px 6px}
|
||||
@@ -869,6 +1082,19 @@ PAGE = r"""<!DOCTYPE html>
|
||||
.logline.info .lv{color:var(--accent)}
|
||||
.logline.error .lv{color:var(--err)}
|
||||
.logline.warn .lv{color:var(--warn)}
|
||||
.lc-tag{flex:0 0 auto;min-width:62px;text-align:center;font-size:10px;font-weight:700;
|
||||
letter-spacing:.3px;padding:1px 7px;border-radius:9px;background:#1b2b3a;color:#8fb3d6;
|
||||
border:1px solid #24405a}
|
||||
.lc-READY{background:#12331f;color:#7fe0a0;border-color:#1f5a34}
|
||||
.lc-CONNECT{background:#12283f;color:#7fb8ff;border-color:#22507f}
|
||||
.lc-MODEL{background:#2a2340;color:#c3a6ff;border-color:#463a7a}
|
||||
.lc-TTS{background:#0f2f33;color:#6fe0d8;border-color:#1f5a5a}
|
||||
.lc-VOICE{background:#33280f;color:#e0c06f;border-color:#5a481f}
|
||||
.lc-FILTER{background:#301a2a;color:#e79fd0;border-color:#5a2545}
|
||||
.lc-SETTING{background:#22303a;color:#9fc7e0;border-color:#2f5568}
|
||||
.lc-TURN,.lc-ERROR{background:#3a1520;color:#ffb3bb;border-color:#7a2531}
|
||||
.lc-BRAIN,.lc-VISION{background:#2a2340;color:#c3a6ff;border-color:#463a7a}
|
||||
.lc-WARN{background:#332a10;color:#ffd98a;border-color:#5a4a1f}
|
||||
.logline .lm{flex:1;color:var(--fg);white-space:pre-wrap;word-break:break-word}
|
||||
.logline.error .lm{color:#ffb3bb}
|
||||
.logline .lacts{opacity:0;display:flex;gap:4px}
|
||||
@@ -881,7 +1107,7 @@ PAGE = r"""<!DOCTYPE html>
|
||||
<header>
|
||||
<div>
|
||||
<h1>watch_sceen_ai · 실시간 상태</h1>
|
||||
<div class="sub">STT → 두뇌 → TTS 음성 루프를 단계별로 관찰</div>
|
||||
<div class="sub">STT → LLM → TTS 음성 루프를 단계별로 관찰</div>
|
||||
</div>
|
||||
<div class="pill"><span id="dot" class="dot off"></span><span id="listen">연결 대기</span></div>
|
||||
<button id="promptBtn" class="btn hbtn">📝 프롬프트</button>
|
||||
@@ -895,6 +1121,7 @@ PAGE = r"""<!DOCTYPE html>
|
||||
</div>
|
||||
</header>
|
||||
<main>
|
||||
<div class="comp" id="comp"></div>
|
||||
<section class="botbar" id="botbar">
|
||||
<span class="botinfo" id="botinfo"><span class="dot off"></span>봇: 연결 안 됨</span>
|
||||
<label>서버 <select id="guildSel"><option value="">없음</option></select></label>
|
||||
@@ -911,8 +1138,8 @@ PAGE = r"""<!DOCTYPE html>
|
||||
<span id="tEmoNote" class="sttstat">감정을 고르면 그 감정만 따로 조절합니다. 기본(공통)은 태그 없는 말투에 적용됩니다.</span>
|
||||
</div>
|
||||
<div class="ttsgrid">
|
||||
<label>글자 속도 <b id="tvSpeed">1.35x</b>
|
||||
<input type="range" id="tSpeed" min="0.5" max="2.0" step="0.05" value="1.35"></label>
|
||||
<label>글자 속도 <b id="tvSpeed">1.40x</b>
|
||||
<input type="range" id="tSpeed" min="0.5" max="2.0" step="0.05" value="1.4"></label>
|
||||
<label>단어 간격 <b id="tvWord">-70 ms</b>
|
||||
<input type="range" id="tWord" min="-0.2" max="0.5" step="0.01" value="-0.07"></label>
|
||||
<label>문장 간격 <b id="tvSent">-0.30 s</b>
|
||||
@@ -935,6 +1162,10 @@ PAGE = r"""<!DOCTYPE html>
|
||||
<div id="botcfgBody" style="display:none">
|
||||
<label class="cfgrow"><input type="checkbox" id="cfgBargeIn" checked>
|
||||
<span>유저 목소리 들을 때 봇이 말하던 것 중지 <b class="cfghint">(기본: 켜짐)</b></span></label>
|
||||
<label class="cfgrow"><span>유저 음성 인식 시간
|
||||
<input type="number" id="cfgBargeInMs" min="0" max="5000" step="50" value="700"
|
||||
style="width:80px;margin:0 4px"> ms
|
||||
<b class="cfghint">(이 시간 이상 말해야 봇이 멈춤 · 기본: 700)</b></span></label>
|
||||
</div>
|
||||
</section>
|
||||
<section class="sttbox" id="logcfgbox">
|
||||
@@ -944,8 +1175,28 @@ PAGE = r"""<!DOCTYPE html>
|
||||
<span>잡음 로그 표시 안 함 <b class="cfghint">(들음: (빈 결과) 또는 답변: [잡음]) · 기본: 켜짐</b></span></label>
|
||||
</div>
|
||||
</section>
|
||||
<section class="sttbox" id="modelbox">
|
||||
<h2 id="modelToggle" class="collap"><span id="modelCaret">▸</span> 🧠 모델 (STT · LLM)</h2>
|
||||
<div id="modelBody" style="display:none">
|
||||
<div class="sttrow">
|
||||
<label style="font-size:12.5px;color:var(--muted)">STT(귀)
|
||||
<select id="mSTT" class="ttssel"></select></label>
|
||||
<button id="mSTTapply" class="btn">적용</button>
|
||||
<span id="mSTTstat" class="sttstat"></span>
|
||||
</div>
|
||||
<div class="sttrow" style="margin-top:8px">
|
||||
<label style="font-size:12.5px;color:var(--muted)">LLM
|
||||
<select id="mLLM" class="ttssel"></select></label>
|
||||
<button id="mLLMapply" class="btn">적용</button>
|
||||
<span id="mLLMstat" class="sttstat"></span>
|
||||
</div>
|
||||
<div id="mLoad" class="mload" style="display:none">
|
||||
<span class="spin"></span><span id="mLoadText">로딩 중…</span>
|
||||
</div>
|
||||
<p class="cfghint" style="margin:8px 0 0">STT는 전환 시 모델을 다시 로드합니다(처음 medium/large-v3는 다운로드로 수 분 걸릴 수 있어요). LLM은 다음 답변부터 즉시 적용됩니다. 마지막으로 고른 모델은 재시작 후에도 유지됩니다(기본값 Sonnet 5).</p>
|
||||
</div>
|
||||
</section>
|
||||
<div class="demobar" id="demobar" style="display:none"></div>
|
||||
<div class="comp" id="comp"></div>
|
||||
<div class="tfilter" id="tfilter">
|
||||
<span class="tf-label">대화 로그 검색</span>
|
||||
<select id="tfTime">
|
||||
@@ -960,20 +1211,30 @@ PAGE = r"""<!DOCTYPE html>
|
||||
<input id="tfChannel" placeholder="채널">
|
||||
<input id="tfText" placeholder="내용(들음/답변)">
|
||||
<button id="tfClear" class="btn hbtn">초기화</button>
|
||||
<button id="turnsClear" class="btn hbtn">🗑 전체 삭제</button>
|
||||
<span id="tfCount" class="tf-count"></span>
|
||||
</div>
|
||||
<div id="turns"></div>
|
||||
<div id="empty" class="empty">아직 대화가 없습니다. 사용자가 말하면 여기에 단계별로 나타납니다.</div>
|
||||
</main>
|
||||
<div id="logdock" class="logdock">
|
||||
<div id="logResize" class="logresize" title="드래그해서 로그 높이 조절"></div>
|
||||
<div class="logbar">
|
||||
<button id="logToggle" class="logtoggle">▾ 이벤트 / 오류 로그</button>
|
||||
<input id="logSearch" class="logsearch" placeholder="로그 검색 (텍스트)">
|
||||
<select id="logLevel" class="logsel">
|
||||
<option value="">전체</option>
|
||||
<option value="error">오류만</option>
|
||||
<option value="warn">경고만</option>
|
||||
<option value="info">정보만</option>
|
||||
<select id="logLevel" class="logsel" title="레벨(타입)">
|
||||
<option value="">레벨 전체</option>
|
||||
<option value="error">오류(ERROR)</option>
|
||||
<option value="warn">경고(WARN)</option>
|
||||
<option value="info">정보(INFO)</option>
|
||||
</select>
|
||||
<select id="logCat" class="logsel" title="이벤트 종류"><option value="">종류 전체</option></select>
|
||||
<select id="logTime" class="logsel" title="시간">
|
||||
<option value="0">전체 시간</option>
|
||||
<option value="5">최근 5분</option>
|
||||
<option value="30">최근 30분</option>
|
||||
<option value="60">최근 1시간</option>
|
||||
<option value="180">최근 3시간</option>
|
||||
</select>
|
||||
<span id="logCount" class="logcount"></span>
|
||||
<button id="logClear" class="btn logbtn">로그 삭제</button>
|
||||
@@ -1037,14 +1298,14 @@ function renderStatus(s){
|
||||
const bar = $('demobar');
|
||||
if(mockParts.length){
|
||||
bar.style.display='block';
|
||||
bar.innerHTML = '⚠ <b>데모 모드</b> — 실제 음성/STT/두뇌/TTS가 아직 연결되지 않아, 아래 대화는 '
|
||||
bar.innerHTML = '⚠ <b>데모 모드</b> — 실제 음성/STT/LLM/TTS가 아직 연결되지 않아, 아래 대화는 '
|
||||
+ '실제로 들은 내용이 아니라 <b>목(mock) 예시 스크립트</b>입니다. '
|
||||
+ '실제 엔진(faster-whisper·Claude·MeloTTS)을 붙이면 이 자리에 진짜 발화·지연·오류가 표시됩니다.';
|
||||
} else {
|
||||
bar.style.display='none';
|
||||
}
|
||||
const el = $('comp'); el.innerHTML = '';
|
||||
const names = {source:'눈(소스)', vision:'시각', stt:'귀(STT)', brain:'두뇌', tts:'입(TTS)', text:'텍스트'};
|
||||
const names = {source:'눈(소스)', vision:'시각', stt:'귀(STT)', brain:'LLM', tts:'입(TTS)', text:'텍스트'};
|
||||
for(const k of Object.keys(names)){
|
||||
if(!(k in comps)) continue;
|
||||
const v = comps[k];
|
||||
@@ -1082,7 +1343,8 @@ function turnEl(t){
|
||||
+'<span class="badge">#'+t.id+' · '+esc(t.source||'voice')+'</span>'
|
||||
+(t.speaker?'<span class="badge">🗣 '+esc(t.speaker)+'</span>':'')
|
||||
+(t.channel?'<span class="badge">🔊 '+esc((t.guild?t.guild+' / ':'')+t.channel)+'</span>':'')
|
||||
+'<span class="time">'+fmtTime(t.wall)+'</span></div>'
|
||||
+'<span class="time">'+fmtTime(t.wall)+'</span>'
|
||||
+'<button class="tdel" data-id="'+t.id+'" title="이 대화 삭제">✕</button></div>'
|
||||
+'<div class="line"><span class="tag">들음</span><span class="heard">'+(t.heard?esc(t.heard):'<i style="color:var(--muted)">(수신 대기)</i>')+'</span></div>'
|
||||
+'<div class="line"><span class="tag">생각</span><span class="thought">'+(t.thought?esc(t.thought):'<i style="color:var(--muted)">…</i>')+'</span></div>'
|
||||
+'<div class="line"><span class="tag">답변</span><span class="reply">'+(t.reply?esc(t.reply):'<i style="color:var(--muted)">…생각 중</i>')+'</span></div>'
|
||||
@@ -1102,6 +1364,12 @@ function upsertTurn(t){
|
||||
else { cont.prepend(fresh); }
|
||||
applyTurnFilter();
|
||||
}
|
||||
function removeTurn(id){
|
||||
turns.delete(id);
|
||||
const el=$('turn-'+id); if(el) el.remove();
|
||||
if(turns.size===0) $('empty').style.display='block';
|
||||
applyTurnFilter();
|
||||
}
|
||||
|
||||
// --- 대화 로그 검색: 시간·유저·서버·채널·내용 필터 ------------------------ #
|
||||
function turnFilter(){
|
||||
@@ -1110,7 +1378,7 @@ function turnFilter(){
|
||||
text:$('tfText').value.trim().toLowerCase() };
|
||||
}
|
||||
function isNoiseTurn(t){
|
||||
return (t.heard==='(빈 결과)') || (t.reply==='[잡음]');
|
||||
return (t.heard==='(빈 결과)') || (t.reply==='[잡음]') || (t.reply==='[대기]');
|
||||
}
|
||||
function turnMatches(t, f){
|
||||
if(hideNoise && isNoiseTurn(t)) return false; // 잡음 로그 표시 안 함 토글
|
||||
@@ -1131,20 +1399,34 @@ function applyTurnFilter(){
|
||||
|
||||
// --- Bottom terminal log panel: store all events, render filtered ---------- #
|
||||
let logEvents = []; // {id, level, message, wall}
|
||||
function logCat(e){ return (e.cat || (e.level||'info').toUpperCase()); }
|
||||
function logMatches(e){
|
||||
const lv = $('logLevel').value;
|
||||
if(lv && e.level!==lv) return false;
|
||||
const c = $('logCat').value;
|
||||
if(c && logCat(e)!==c) return false;
|
||||
const mins = +$('logTime').value;
|
||||
if(mins && (Date.now()/1000 - (e.wall||0)) > mins*60) return false;
|
||||
const q = $('logSearch').value.trim().toLowerCase();
|
||||
if(q && !((e.message||'').toLowerCase().includes(q) || fmtTime(e.wall).includes(q))) return false;
|
||||
if(q && !((e.message||'').toLowerCase().includes(q) || logCat(e).toLowerCase().includes(q) || fmtTime(e.wall).includes(q))) return false;
|
||||
return true;
|
||||
}
|
||||
// Keep the 종류(카테고리) dropdown in sync with whatever categories appear.
|
||||
function refreshCatOptions(){
|
||||
const sel=$('logCat'); const cur=sel.value;
|
||||
const cats=[...new Set(logEvents.map(logCat))].sort();
|
||||
sel.innerHTML='<option value="">종류 전체</option>'
|
||||
+ cats.map(c=>'<option value="'+c+'">'+esc(c)+'</option>').join('');
|
||||
sel.value = cats.includes(cur) ? cur : '';
|
||||
}
|
||||
function renderLogs(){
|
||||
const body = $('logbody');
|
||||
refreshCatOptions();
|
||||
const shown = logEvents.filter(logMatches);
|
||||
body.innerHTML = shown.map(e =>
|
||||
'<div class="logline '+(e.level||'info')+'" data-id="'+e.id+'">'
|
||||
+ '<span class="lt">'+fmtTime(e.wall)+'</span>'
|
||||
+ '<span class="lv">'+esc(e.level||'info')+'</span>'
|
||||
+ '<span class="lc-tag lc-'+esc(logCat(e))+'">'+esc(logCat(e))+'</span>'
|
||||
+ '<span class="lm">'+esc(e.message)+'</span>'
|
||||
+ '<span class="lacts"><button class="lact" data-act="edit" title="수정">✎</button>'
|
||||
+ '<button class="lact" data-act="del" title="삭제">✕</button></span>'
|
||||
@@ -1179,6 +1461,8 @@ function connect(){
|
||||
if(ev.type==='snapshot') applySnapshot(ev.snapshot);
|
||||
else if(ev.type==='status') renderStatus(ev.status);
|
||||
else if(ev.type==='turn') upsertTurn(ev.turn);
|
||||
else if(ev.type==='turn_deleted') removeTurn(ev.id);
|
||||
else if(ev.type==='turns_cleared') { turns.clear(); $('turns').innerHTML=''; $('empty').style.display='block'; applyTurnFilter(); }
|
||||
else if(ev.type==='log') { addEvent(ev); if(statusData){ statusData.errors_total=(statusData.errors_total||0)+(ev.level==='error'?1:0); $('s-errors').textContent=statusData.errors_total; } }
|
||||
else if(ev.type==='logs_cleared') { logEvents=[]; renderLogs(); }
|
||||
else if(ev.type==='log_deleted') { logEvents=logEvents.filter(e=>e.id!==ev.id); renderLogs(); }
|
||||
@@ -1203,8 +1487,23 @@ function wireCollapse(toggleId, bodyId, caretId){
|
||||
body:JSON.stringify({bargeIn:$('cfgBargeIn').checked})});
|
||||
toast('봇 설정 저장됨'); }catch(e){}
|
||||
};
|
||||
// 유저 음성 인식 시간(ms): 서버가 0~5000으로 클램프, 봇이 다음 report에서 읽어감.
|
||||
$('cfgBargeInMs').onchange = async ()=>{
|
||||
let ms = Math.round(Number($('cfgBargeInMs').value));
|
||||
if(!Number.isFinite(ms)) ms = 700;
|
||||
ms = Math.max(0, Math.min(5000, ms));
|
||||
$('cfgBargeInMs').value = ms;
|
||||
try{ const r=await fetch('/api/bot/settings',{method:'POST',headers:{'Content-Type':'application/json'},
|
||||
body:JSON.stringify({bargeInMs:ms})});
|
||||
const j=await r.json();
|
||||
if(j&&j.ok&&j.settings&&j.settings.bargeInMs!=null) $('cfgBargeInMs').value=j.settings.bargeInMs;
|
||||
toast('봇 설정 저장됨'); }catch(e){}
|
||||
};
|
||||
fetch('/api/bot/settings').then(r=>r.json()).then(j=>{
|
||||
if(j&&j.ok&&j.settings) $('cfgBargeIn').checked = j.settings.bargeIn !== false;
|
||||
if(j&&j.ok&&j.settings){
|
||||
$('cfgBargeIn').checked = j.settings.bargeIn !== false;
|
||||
if(j.settings.bargeInMs!=null) $('cfgBargeInMs').value = j.settings.bargeInMs;
|
||||
}
|
||||
}).catch(()=>{});
|
||||
// 로그 설정: 잡음 로그 표시 안 함 — 클라이언트 필터(로컬 저장).
|
||||
$('cfgHideNoise').checked = hideNoise;
|
||||
@@ -1213,12 +1512,84 @@ function wireCollapse(toggleId, bodyId, caretId){
|
||||
localStorage.setItem('wsai_hideNoise', hideNoise ? '1':'0');
|
||||
applyTurnFilter();
|
||||
};
|
||||
// 모델 설정: STT 크기 / LLM 모델 실시간 전환.
|
||||
wireCollapse('modelToggle','modelBody','modelCaret');
|
||||
initModels();
|
||||
})();
|
||||
const STT_LABEL = {tiny:'tiny (가장 빠름)', base:'base', small:'small (빠름)',
|
||||
medium:'medium (기본·정확)', 'large-v3':'large-v3 (최고 정확도)'};
|
||||
const LLM_LABEL = {'claude-haiku-4-5':'Haiku 4.5 (가장 빠름)',
|
||||
'claude-sonnet-4-5':'Sonnet 4.5 (고품질·조금 느림)',
|
||||
'claude-sonnet-5':'Sonnet 5 (기본 · 빠르고 고품질)'};
|
||||
function fillSel(sel, options, current, labels){
|
||||
sel.innerHTML='';
|
||||
options.forEach(o=>{ const el=document.createElement('option');
|
||||
el.value=o; el.textContent=(labels[o]||o); sel.appendChild(el); });
|
||||
if(current) sel.value=current;
|
||||
}
|
||||
async function initModels(){
|
||||
let j; try{ j=await (await fetch('/api/models')).json(); }catch(e){ return; }
|
||||
if(!j||!j.ok) return;
|
||||
if(j.stt.enabled){ fillSel($('mSTT'), j.stt.options, j.stt.current, STT_LABEL);
|
||||
$('mSTTstat').textContent='현재: '+(j.stt.current||'?')+(j.stt.device?(' · '+j.stt.device):''); }
|
||||
else { $('mSTT').disabled=$('mSTTapply').disabled=true; $('mSTTstat').textContent='STT 비활성'; }
|
||||
if(j.llm.enabled){ fillSel($('mLLM'), j.llm.options, j.llm.current, LLM_LABEL);
|
||||
$('mLLMstat').textContent='현재: '+(LLM_LABEL[j.llm.current]||j.llm.current||'?'); }
|
||||
else { $('mLLM').disabled=$('mLLMapply').disabled=true; $('mLLMstat').textContent='LLM 비활성(echo 모드)'; }
|
||||
$('mSTTapply').onclick = async ()=>{
|
||||
const m=$('mSTT').value;
|
||||
try{ const r=await fetch('/api/models/stt',{method:'POST',headers:{'Content-Type':'application/json'},
|
||||
body:JSON.stringify({model:m})}); const jj=await r.json();
|
||||
if(!jj.ok){ $('mSTTstat').textContent='실패: '+(jj.error||''); return; }
|
||||
if(!jj.stt.changed){ flash3($('mSTTstat'), '변경사항 없음', '현재: '+(jj.stt.current||m)+(jj.stt.device?(' · '+jj.stt.device):'')); return; }
|
||||
pollSTTLoad(m); // 실제 변경 → 아래 로딩창 + 완료 시 이벤트 로그
|
||||
}catch(e){ $('mSTTstat').textContent='오류: '+e; }
|
||||
};
|
||||
$('mLLMapply').onclick = async ()=>{
|
||||
const m=$('mLLM').value;
|
||||
try{ const r=await fetch('/api/models/llm',{method:'POST',headers:{'Content-Type':'application/json'},
|
||||
body:JSON.stringify({model:m})}); const jj=await r.json();
|
||||
if(!jj.ok){ $('mLLMstat').textContent='실패: '+(jj.error||''); return; }
|
||||
if(!jj.llm.changed){ flash3($('mLLMstat'), '변경사항 없음', '현재: '+(LLM_LABEL[jj.llm.current]||jj.llm.current)); return; }
|
||||
toast('LLM 모델 변경: '+(LLM_LABEL[m]||m));
|
||||
$('mLLMstat').textContent='현재: '+(LLM_LABEL[m]||m)+' · 다음 답변부터';
|
||||
}catch(e){ $('mLLMstat').textContent='오류: '+e; }
|
||||
};
|
||||
}
|
||||
// 3초 동안 임시 메시지를 보여준 뒤 원래(현재 모델) 텍스트로 복귀.
|
||||
function flash3(el, temp, revert){
|
||||
el.textContent = temp;
|
||||
clearTimeout(el._t);
|
||||
el._t = setTimeout(()=>{ el.textContent = revert; }, 3000);
|
||||
}
|
||||
// STT 실제 전환: 아래 로딩창을 띄우고 /api/models를 폴링해 ready가 될 때까지 표시.
|
||||
let sttPollTimer=null;
|
||||
async function pollSTTLoad(target){
|
||||
const load=$('mLoad'), txt=$('mLoadText');
|
||||
load.style.display='flex';
|
||||
$('mSTTstat').textContent='전환 중…';
|
||||
const t0=Date.now();
|
||||
if(sttPollTimer) clearInterval(sttPollTimer);
|
||||
const tick=async ()=>{
|
||||
const sec=Math.round((Date.now()-t0)/1000);
|
||||
txt.textContent = 'STT '+target+' 로딩 중… (경과 '+sec+'초, 처음이면 다운로드로 수 분 걸릴 수 있어요)';
|
||||
try{
|
||||
const j=await (await fetch('/api/models')).json();
|
||||
if(j&&j.ok&&j.stt.current===target&&j.stt.ready){
|
||||
clearInterval(sttPollTimer); sttPollTimer=null;
|
||||
load.style.display='none';
|
||||
$('mSTTstat').textContent='현재: '+target+(j.stt.device?(' · '+j.stt.device):'')+' · 로드 완료';
|
||||
toast('STT 모델 로드 완료: '+target);
|
||||
}
|
||||
}catch(e){}
|
||||
};
|
||||
tick(); sttPollTimer=setInterval(tick, 1500);
|
||||
}
|
||||
|
||||
// --- 봇 목소리(TTS) 조절: 감정별 슬라이더 → 미리듣기 → 봇 적용 -------------- #
|
||||
let ttsInited = false;
|
||||
let ttsState = {base:{}, overrides:{}, emotions:[]}; // last-known server state
|
||||
const TTS_DEFAULT = {speed:1.35, word_gap:-0.07, sentence_gap:-0.3, pitch:0};
|
||||
const TTS_DEFAULT = {speed:1.4, word_gap:-0.07, sentence_gap:-0.3, pitch:0};
|
||||
function ttsLabels(){
|
||||
const sp=parseFloat($('tSpeed').value), wg=parseFloat($('tWord').value),
|
||||
sg=parseFloat($('tSent').value), pt=parseInt($('tPitch').value,10);
|
||||
@@ -1243,9 +1614,9 @@ function ttsSet(s){
|
||||
// (base itself when 기본 or when the emotion has no override).
|
||||
function ttsValuesFor(key){
|
||||
const base = Object.assign({}, TTS_DEFAULT, ttsState.base);
|
||||
if(key && key!=='base' && ttsState.overrides && ttsState.overrides[key])
|
||||
if(key && key!=='base' && key!=='all' && ttsState.overrides && ttsState.overrides[key])
|
||||
return Object.assign({}, base, ttsState.overrides[key]);
|
||||
return base;
|
||||
return base; // 'all' and 'base' both edit the base values
|
||||
}
|
||||
function ttsApplyState(j){
|
||||
if(!j || !j.ok) return;
|
||||
@@ -1253,6 +1624,8 @@ function ttsApplyState(j){
|
||||
// Refresh dropdown labels to mark which emotions are customised.
|
||||
const sel=$('tEmotion'); const cur=sel.value||'base';
|
||||
sel.innerHTML='';
|
||||
const allOpt=document.createElement('option'); // 최상단 전체변경
|
||||
allOpt.value='all'; allOpt.textContent='🌐 전체변경 (모든 감정)'; sel.appendChild(allOpt);
|
||||
ttsState.emotions.forEach(e=>{
|
||||
const o=document.createElement('option'); o.value=e.key;
|
||||
const custom = e.key!=='base' && ttsState.overrides[e.key];
|
||||
@@ -1270,9 +1643,11 @@ async function initTts(){
|
||||
$('ttsCaret').textContent = open ? '▾' : '▸';
|
||||
};
|
||||
['tSpeed','tWord','tSent','tPitch'].forEach(id => $(id).addEventListener('input', ttsLabels));
|
||||
$('tEmotion').onchange = ()=>{ ttsSet(ttsValuesFor($('tEmotion').value));
|
||||
$('tStat').textContent = $('tEmotion').value==='base'
|
||||
? '기본(공통) 값을 조절 중입니다.' : '"'+$('tEmotion').selectedOptions[0].textContent.replace(' ●','')+'" 감정만 조절 중입니다.'; };
|
||||
$('tEmotion').onchange = ()=>{ const k=$('tEmotion').value; ttsSet(ttsValuesFor(k));
|
||||
$('tStat').textContent = k==='all'
|
||||
? '모든 감정을 한 번에 조절 중입니다 (적용 시 전체 반영).'
|
||||
: (k==='base' ? '기본(공통) 값을 조절 중입니다.'
|
||||
: '"'+$('tEmotion').selectedOptions[0].textContent.replace(' ●','')+'" 감정만 조절 중입니다.'); };
|
||||
$('tPreview').onclick = async ()=>{
|
||||
const text=($('tText').value||'').trim();
|
||||
if(!text){ $('tStat').textContent='미리들을 문장을 입력하세요.'; return; }
|
||||
@@ -1292,13 +1667,13 @@ async function initTts(){
|
||||
const r=await fetch('/api/tts/settings',{method:'POST',headers:{'Content-Type':'application/json'},
|
||||
body:JSON.stringify(Object.assign({emotion:key}, ttsVals()))});
|
||||
const j=await r.json();
|
||||
if(j.ok){ ttsApplyState(j); toast((key==='base'?'기본(공통)':'해당 감정')+' 적용됨 · 다음 답변부터 반영'); }
|
||||
if(j.ok){ ttsApplyState(j); toast((key==='all'?'모든 감정':(key==='base'?'기본(공통)':'해당 감정'))+' 적용됨 · 다음 답변부터 반영'); }
|
||||
else { $('tStat').textContent='적용 실패: '+(j.error||''); }
|
||||
}catch(e){ $('tStat').textContent='적용 오류: '+e; }
|
||||
};
|
||||
$('tReset').onclick = async ()=>{
|
||||
const key=$('tEmotion').value||'base';
|
||||
if(key==='base'){ ttsSet(TTS_DEFAULT);
|
||||
if(key==='base' || key==='all'){ ttsSet(TTS_DEFAULT);
|
||||
$('tStat').textContent='기본값으로 세팅됨 (적용을 눌러 반영).'; return; }
|
||||
try{ // clear this emotion's override -> it inherits 기본(공통) again
|
||||
const r=await fetch('/api/tts/settings',{method:'POST',headers:{'Content-Type':'application/json'},
|
||||
@@ -1365,8 +1740,30 @@ $('logToggle').onclick = ()=>{
|
||||
syncDockPad();
|
||||
};
|
||||
window.addEventListener('resize', syncDockPad);
|
||||
// Drag the top edge of the log dock to resize its height (persisted).
|
||||
(function(){
|
||||
const dock=$('logdock'), body=$('logbody'), handle=$('logResize');
|
||||
const KEY='wsai_logH';
|
||||
const maxH = ()=> Math.round(window.innerHeight*0.82);
|
||||
function setH(px){ const h=Math.max(80, Math.min(maxH(), Math.round(px)));
|
||||
body.style.height=h+'px'; localStorage.setItem(KEY, h); syncDockPad(); }
|
||||
setH(parseInt(localStorage.getItem(KEY)||'180', 10));
|
||||
let startY=0, startH=0, dragging=false;
|
||||
handle.addEventListener('pointerdown', (e)=>{
|
||||
if(dock.classList.contains('collapsed')) return;
|
||||
dragging=true; startY=e.clientY; startH=body.offsetHeight;
|
||||
try{ handle.setPointerCapture(e.pointerId); }catch(_){}
|
||||
e.preventDefault();
|
||||
});
|
||||
handle.addEventListener('pointermove', (e)=>{ if(dragging) setH(startH + (startY - e.clientY)); });
|
||||
const end=(e)=>{ if(!dragging) return; dragging=false; try{ handle.releasePointerCapture(e.pointerId); }catch(_){} };
|
||||
handle.addEventListener('pointerup', end);
|
||||
handle.addEventListener('pointercancel', end);
|
||||
})();
|
||||
$('logSearch').oninput = renderLogs;
|
||||
$('logLevel').onchange = renderLogs;
|
||||
$('logCat').onchange = renderLogs;
|
||||
$('logTime').onchange = renderLogs;
|
||||
$('logClear').onclick = async ()=>{
|
||||
if(!confirm('로그를 모두 삭제할까요?')) return;
|
||||
try{ await fetch('/api/logs/clear',{method:'POST'}); toast('로그를 삭제했습니다'); }
|
||||
@@ -1445,8 +1842,10 @@ async function openLists(){
|
||||
catch(e){ lists = {whitelistUsers:[],blacklistUsers:[],whitelistRoles:[],blacklistRoles:[]}; }
|
||||
openModal('청취 화이트/블랙리스트', '<button class="btn primary" id="lstSave">저장</button>');
|
||||
$('modalBody').innerHTML =
|
||||
'<p class="modal-note">화이트리스트에 넣으면 그 대상만 청취(비어있으면 전체 청취), 블랙리스트는 제외됩니다. 유저/역할별로 추가할 수 있어요.</p>'
|
||||
+'<div class="lst-row"><select id="lstType"><option value="user">유저</option><option value="role">역할</option></select>'
|
||||
'<p class="modal-note">화이트리스트에 넣으면 그 대상만 청취(비어있으면 전체 청취), 블랙리스트는 제외됩니다. 유저·봇·통화방·역할별로 추가할 수 있어요.</p>'
|
||||
+'<div class="lst-row"><select id="lstType">'
|
||||
+'<option value="user">유저</option><option value="bot">봇</option>'
|
||||
+'<option value="voice">통화방</option><option value="role">역할</option></select>'
|
||||
+'<input id="lstSearch" class="lst-search" placeholder="이름으로 검색"></div>'
|
||||
+'<div class="lst-results" id="lstResults"></div>'
|
||||
+'<div class="lst-h">화이트리스트 (그 대상만 청취)</div><div class="chips" id="chipsW"></div>'
|
||||
@@ -1454,15 +1853,39 @@ async function openLists(){
|
||||
const has=(arr,id)=>(arr||[]).some(x=>x.id===id);
|
||||
function add(kind,item){ const k=LKEY[kind]; if(!has(lists[k],item.id)) lists[k].push(item); renderChips(); }
|
||||
function rm(k,id){ lists[k]=(lists[k]||[]).filter(x=>x.id!==id); renderChips(); }
|
||||
const NO_RES='<div class="lst-item"><span class="rl">결과 없음 · 봇이 아는 멤버/역할만 검색됩니다</span></div>';
|
||||
function memberRow(m){
|
||||
return '<div class="lst-item"><span class="nm">'+esc(m.name)+(m.bot?' <span class="rl">(봇)</span>':'')+'</span>'
|
||||
+'<button class="mini w" data-k="wu" data-id="'+esc(m.id)+'" data-nm="'+esc(m.name)+'">+화이트</button>'
|
||||
+'<button class="mini b" data-k="bu" data-id="'+esc(m.id)+'" data-nm="'+esc(m.name)+'">+블랙</button></div>';
|
||||
}
|
||||
function renderResults(){
|
||||
const type=$('lstType').value, q=$('lstSearch').value.trim().toLowerCase();
|
||||
const src = type==='user' ? (g.members||[]) : (g.roles||[]);
|
||||
const rows = src.filter(x=>!q || (x.name||'').toLowerCase().includes(q)).slice(0,100);
|
||||
$('lstResults').innerHTML = rows.length ? rows.map(x=>
|
||||
'<div class="lst-item"><span class="nm">'+esc(x.name)+(x.bot?' <span class="rl">(봇)</span>':'')+'</span>'
|
||||
+'<button class="mini w" data-k="'+(type==='user'?'wu':'wr')+'" data-id="'+esc(x.id)+'" data-nm="'+esc(x.name)+'">+화이트</button>'
|
||||
+'<button class="mini b" data-k="'+(type==='user'?'bu':'br')+'" data-id="'+esc(x.id)+'" data-nm="'+esc(x.name)+'">+블랙</button></div>'
|
||||
).join('') : '<div class="lst-item"><span class="rl">결과 없음 · 봇이 아는 멤버/역할만 검색됩니다</span></div>';
|
||||
const members=(g.members||[]); const byId={}; members.forEach(m=>byId[m.id]=m);
|
||||
const match=(n)=>!q||((n||'').toLowerCase().includes(q));
|
||||
let html='';
|
||||
if(type==='user' || type==='bot'){
|
||||
const rows=members.filter(m=>(type==='bot'?m.bot:!m.bot)&&match(m.name)).slice(0,300);
|
||||
html = rows.length ? rows.map(memberRow).join('') : NO_RES;
|
||||
} else if(type==='voice'){
|
||||
const vm=((botState&&botState.members)||[]).map(m=>({id:m.id,name:m.name,bot:!!(byId[m.id]&&byId[m.id].bot)}));
|
||||
const rows=vm.filter(m=>match(m.name)).slice(0,300);
|
||||
html = rows.length ? rows.map(memberRow).join('')
|
||||
: '<div class="lst-item"><span class="rl">봇이 통화방에 없거나 참여자가 없습니다</span></div>';
|
||||
} else { // role: 역할 등록 + 펼치면 역할 아래 유저(봇 포함) 등록
|
||||
const roles=(g.roles||[]).filter(r=>match(r.name)).slice(0,300);
|
||||
html = roles.length ? roles.map(r=>{
|
||||
const rid=esc(r.id), rmem=members.filter(m=>(m.roleIds||[]).includes(r.id));
|
||||
return '<div class="lst-role">'
|
||||
+'<div class="lst-item"><span class="nm role-h" data-rid="'+rid+'"><b class="rcaret">▸</b> '+esc(r.name)+' <span class="rl">('+rmem.length+'명)</span></span>'
|
||||
+'<button class="mini w" data-k="wr" data-id="'+rid+'" data-nm="'+esc(r.name)+'">+화이트</button>'
|
||||
+'<button class="mini b" data-k="br" data-id="'+rid+'" data-nm="'+esc(r.name)+'">+블랙</button></div>'
|
||||
+'<div class="role-members" data-rid="'+rid+'" style="display:none">'
|
||||
+(rmem.length? rmem.map(memberRow).join('') : '<div class="lst-item"><span class="rl">이 역할의 멤버 없음</span></div>')
|
||||
+'</div></div>';
|
||||
}).join('') : NO_RES;
|
||||
}
|
||||
$('lstResults').innerHTML = html;
|
||||
}
|
||||
const chip=(k,cls,x)=>'<span class="chip '+cls+'">'+esc(x.name)+' <button data-k="'+k+'" data-id="'+esc(x.id)+'">✕</button></span>';
|
||||
const empty='<span class="rl" style="color:var(--muted);font-size:12px">비어있음</span>';
|
||||
@@ -1471,7 +1894,13 @@ async function openLists(){
|
||||
$('chipsB').innerHTML = [...lists.blacklistUsers.map(x=>chip('blacklistUsers','b',x)),...lists.blacklistRoles.map(x=>chip('blacklistRoles','b',x))].join('') || empty;
|
||||
}
|
||||
$('lstType').onchange=renderResults; $('lstSearch').oninput=renderResults;
|
||||
$('lstResults').onclick=(e)=>{ const b=e.target.closest('.mini'); if(!b)return; add(b.dataset.k,{id:b.dataset.id,name:b.dataset.nm}); };
|
||||
$('lstResults').onclick=(e)=>{
|
||||
const rh=e.target.closest('.role-h');
|
||||
if(rh){ const box=$('lstResults').querySelector('.role-members[data-rid="'+rh.dataset.rid+'"]');
|
||||
if(box){ const open=box.style.display==='none'; box.style.display=open?'block':'none';
|
||||
const c=rh.querySelector('.rcaret'); if(c) c.textContent=open?'▾':'▸'; } return; }
|
||||
const b=e.target.closest('.mini'); if(!b)return; add(b.dataset.k,{id:b.dataset.id,name:b.dataset.nm});
|
||||
};
|
||||
$('chipsW').onclick=$('chipsB').onclick=(e)=>{ const b=e.target.closest('button'); if(!b)return; rm(b.dataset.k,b.dataset.id); };
|
||||
$('lstSave').onclick=async()=>{
|
||||
try{ await fetch('/api/bot/lists',{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify({guildId,lists})}); toast('청취 필터 저장됨 · 봇에 곧 반영'); closeModal(); }
|
||||
@@ -1483,6 +1912,18 @@ $('listsBtn').onclick = openLists;
|
||||
|
||||
['tfTime','tfUser','tfGuild','tfChannel','tfText'].forEach(id=>{ const e=$(id); if(e){ e.oninput=applyTurnFilter; e.onchange=applyTurnFilter; } });
|
||||
$('tfClear').onclick=()=>{ $('tfTime').value='0'; $('tfUser').value=''; $('tfGuild').value=''; $('tfChannel').value=''; $('tfText').value=''; applyTurnFilter(); };
|
||||
// 대화 로그: 개별 ✕ 삭제 + 전체 삭제.
|
||||
$('turns').addEventListener('click', async (e)=>{
|
||||
const b=e.target.closest('.tdel'); if(!b) return;
|
||||
const id=Number(b.dataset.id); removeTurn(id); // optimistic
|
||||
try{ await fetch('/api/turns/delete',{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify({id})}); }catch(_){}
|
||||
});
|
||||
$('turnsClear').onclick=async ()=>{
|
||||
if(!turns.size){ toast('삭제할 대화가 없습니다'); return; }
|
||||
if(!confirm('대화 로그를 모두 삭제할까요?')) return;
|
||||
try{ await fetch('/api/turns/clear',{method:'POST'}); toast('대화 로그를 모두 삭제했습니다'); }
|
||||
catch(e){ toast('삭제 실패: '+e); }
|
||||
};
|
||||
|
||||
async function pollBot(){
|
||||
try{ renderBot(await (await fetch('/api/bot/state')).json()); }catch(e){}
|
||||
|
||||
@@ -129,7 +129,7 @@ class Turn:
|
||||
# Guarded so the repeated _touch()/finish() calls can't double-count.
|
||||
if self.status == "error" and not self._error_logged:
|
||||
self._error_logged = True
|
||||
self._monitor.log("error", f"대화 #{self.id} 실패: {self.error}")
|
||||
self._monitor.log("error", f"대화 #{self.id} 실패: {self.error}", cat="TURN")
|
||||
self._touch()
|
||||
|
||||
# -- internal --------------------------------------------------------- #
|
||||
@@ -209,12 +209,17 @@ class Monitor:
|
||||
self._status["claude_output_tokens"] += int(output_tokens or 0)
|
||||
self._broadcast({"type": "status", "status": self.status_snapshot()})
|
||||
|
||||
def log(self, level: str, message: str) -> None:
|
||||
"""A free-form lifecycle/error line (startup, disconnect, crash…)."""
|
||||
def log(self, level: str, message: str, cat: str | None = None) -> None:
|
||||
"""A free-form lifecycle/error line (startup, disconnect, crash…).
|
||||
|
||||
``cat`` is a short category tag (READY, CONNECT, MODEL, TTS, VOICE,
|
||||
FILTER, SETTING, TURN, BRAIN, PIPELINE, …) shown as a chip and filterable
|
||||
on the dashboard. Defaults to the upper-cased level when omitted."""
|
||||
with self._lock:
|
||||
self._event_id += 1
|
||||
evt = {"type": "log", "id": self._event_id, "level": level,
|
||||
"message": message, "wall": _now_wall()}
|
||||
"cat": (cat or level.upper()), "message": message,
|
||||
"wall": _now_wall()}
|
||||
self._events.append(evt)
|
||||
if level == "error":
|
||||
self._status["errors_total"] += 1
|
||||
@@ -253,6 +258,26 @@ class Monitor:
|
||||
self._events.clear()
|
||||
self._broadcast({"type": "logs_cleared"})
|
||||
|
||||
def delete_turn(self, turn_id: int) -> bool:
|
||||
"""Remove one conversation turn by id (dashboard per-turn '✕')."""
|
||||
found = False
|
||||
with self._lock:
|
||||
for t in list(self._turns):
|
||||
if t.id == turn_id:
|
||||
self._turns.remove(t)
|
||||
found = True
|
||||
break
|
||||
if found:
|
||||
self._broadcast({"type": "turn_deleted", "id": turn_id})
|
||||
return found
|
||||
|
||||
def clear_turns(self) -> None:
|
||||
"""Wipe all conversation turns (dashboard '대화 전체 삭제'). Broadcasts a
|
||||
reset so every connected page clears its turn list too."""
|
||||
with self._lock:
|
||||
self._turns.clear()
|
||||
self._broadcast({"type": "turns_cleared"})
|
||||
|
||||
# -- turns ------------------------------------------------------------ #
|
||||
def turn(self, source: str = "voice") -> Turn:
|
||||
with self._lock:
|
||||
|
||||
@@ -65,7 +65,7 @@ class Pipeline:
|
||||
except Exception as exc: # a single bad frame must not kill the loop
|
||||
log.exception("vision.describe failed")
|
||||
if self.monitor is not None:
|
||||
self.monitor.log("error", f"화면 이해 실패: {exc}")
|
||||
self.monitor.log("error", f"화면 이해 실패: {exc}", cat="VISION")
|
||||
continue
|
||||
await self.context.update(obs)
|
||||
log.debug("screen: %s", obs.text[:120])
|
||||
@@ -87,7 +87,7 @@ class Pipeline:
|
||||
try:
|
||||
async with turn.step("화면 맥락"):
|
||||
screen = await self.context.latest()
|
||||
async with turn.step("두뇌(생각)"):
|
||||
async with turn.step("LLM(생각)"):
|
||||
reply = await self.brain.respond(utt.text, screen, self._history)
|
||||
turn.replied(reply.text)
|
||||
self._remember(utt.text, reply.text)
|
||||
@@ -122,7 +122,7 @@ class Pipeline:
|
||||
return
|
||||
if self.monitor is not None:
|
||||
self.monitor.set_status(listening=True)
|
||||
self.monitor.log("info", "음성 수신 시작 — 발화 대기 중")
|
||||
self.monitor.log("info", "음성 수신 시작 — 발화 대기 중", cat="PIPELINE")
|
||||
try:
|
||||
async for utt in self.stt.utterances():
|
||||
await self._handle(utt)
|
||||
@@ -157,7 +157,7 @@ class Pipeline:
|
||||
except Exception as exc: # a warm failure must not abort startup
|
||||
log.warning("prewarm %s failed: %s", name, exc)
|
||||
if self.monitor is not None:
|
||||
self.monitor.log("error", f"{name} 예열 실패: {exc}")
|
||||
self.monitor.log("error", f"{name} 예열 실패: {exc}", cat="READY")
|
||||
|
||||
async def run(self) -> None:
|
||||
# A TaskGroup (not bare gather) so that if ONE loop raises, the others
|
||||
@@ -167,7 +167,7 @@ class Pipeline:
|
||||
# still-live loop (close-during-use).
|
||||
if self.monitor is not None:
|
||||
self.monitor.set_status(running=True)
|
||||
self.monitor.log("info", "파이프라인 시작")
|
||||
self.monitor.log("info", "파이프라인 시작", cat="PIPELINE")
|
||||
await self._prewarm()
|
||||
try:
|
||||
async with asyncio.TaskGroup() as tg:
|
||||
@@ -177,12 +177,12 @@ class Pipeline:
|
||||
except* Exception as eg:
|
||||
if self.monitor is not None:
|
||||
for exc in eg.exceptions:
|
||||
self.monitor.log("error", f"루프 예외: {type(exc).__name__}: {exc}")
|
||||
self.monitor.log("error", f"루프 예외: {type(exc).__name__}: {exc}", cat="PIPELINE")
|
||||
raise
|
||||
finally:
|
||||
if self.monitor is not None:
|
||||
self.monitor.set_status(running=False, listening=False)
|
||||
self.monitor.log("info", "파이프라인 종료")
|
||||
self.monitor.log("info", "파이프라인 종료", cat="PIPELINE")
|
||||
await self.aclose()
|
||||
|
||||
async def aclose(self) -> None:
|
||||
|
||||
61
wsai/state_store.py
Normal file
61
wsai/state_store.py
Normal file
@@ -0,0 +1,61 @@
|
||||
"""Tiny JSON state store so dashboard/bot settings survive a restart.
|
||||
|
||||
Persists things the user configures on the dashboard — per-guild listen lists,
|
||||
bot behaviour toggles, TTS controls, and the chosen STT/LLM models — to a single
|
||||
JSON file so a service or container restart keeps them.
|
||||
|
||||
Path: ``WSAI_STATE_FILE`` env, else ``~/.config/wsai/state.json``. For a
|
||||
container deployment, mount that path (or point the env at a mounted volume) to
|
||||
keep the file across ``docker restart``.
|
||||
|
||||
Pure stdlib, thread-safe, best-effort: a read/write failure never raises into
|
||||
the caller (the dashboard must keep working even if the disk is unwritable).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import threading
|
||||
from typing import Any
|
||||
|
||||
log = logging.getLogger("wsai.state")
|
||||
|
||||
_LOCK = threading.Lock()
|
||||
|
||||
|
||||
def path() -> str:
|
||||
return os.environ.get("WSAI_STATE_FILE") or os.path.expanduser("~/.config/wsai/state.json")
|
||||
|
||||
|
||||
def load() -> dict[str, Any]:
|
||||
try:
|
||||
with open(path(), encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
return data if isinstance(data, dict) else {}
|
||||
except FileNotFoundError:
|
||||
return {}
|
||||
except Exception as exc: # noqa: BLE001 — corrupt/unreadable state must not crash startup
|
||||
log.warning("state load failed (%s): %s", path(), exc)
|
||||
return {}
|
||||
|
||||
|
||||
def save(state: dict[str, Any]) -> None:
|
||||
p = path()
|
||||
try:
|
||||
with _LOCK:
|
||||
os.makedirs(os.path.dirname(p), exist_ok=True)
|
||||
tmp = p + ".tmp"
|
||||
with open(tmp, "w", encoding="utf-8") as f:
|
||||
json.dump(state, f, ensure_ascii=False, indent=2)
|
||||
os.replace(tmp, p) # atomic
|
||||
except Exception as exc: # noqa: BLE001
|
||||
log.warning("state save failed (%s): %s", p, exc)
|
||||
|
||||
|
||||
def patch(key: str, value: Any) -> None:
|
||||
"""Read-modify-write one top-level key."""
|
||||
s = load()
|
||||
s[key] = value
|
||||
save(s)
|
||||
Reference in New Issue
Block a user