Compare commits
21 Commits
361dce70bb
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
545a984ca5 | ||
|
|
04664ce61a | ||
|
|
0e4e7c6bb2 | ||
|
|
d5501d0b8a | ||
|
|
5966f6dedb | ||
|
|
8539acd0f8 | ||
|
|
0e77f659d5 | ||
|
|
c39b47cfd3 | ||
|
|
0588ee21ac | ||
|
|
7c899d3b19 | ||
|
|
ed5f328889 | ||
|
|
1ac45214ce | ||
|
|
d164630bb8 | ||
|
|
6d8ba3ad3a | ||
|
|
b0272d2171 | ||
|
|
b782ca70bd | ||
|
|
067efc7abe | ||
|
|
39b743d976 | ||
|
|
2ce2806102 | ||
|
|
0b92284ff8 | ||
|
|
a207ae05c5 |
30
dave/bot.mjs
30
dave/bot.mjs
@@ -177,6 +177,17 @@ let currentGuildId = null, currentChannelId = null, currentChannelName = null;
|
||||
const speakingSet = new Set(); // userIds currently speaking (for participant list)
|
||||
const activeSubs = new Set(); // userIds with an in-flight receive subscription
|
||||
let listsByGuild = {}; // guildId -> {whitelistUsers, blacklistUsers, whitelistRoles, blacklistRoles}
|
||||
let botSettings = { bargeIn: true }; // behaviour toggles from the dashboard
|
||||
// Barge-in only fires when the Discord "speaking" (green ring) stays on for at
|
||||
// least this long — a brief blip (keyboard click, cough) shouldn't cut the bot
|
||||
// off, but a genuine ~0.7s of speech should. The dashboard's "유저 음성 인식 시간"
|
||||
// (botSettings.bargeInMs) wins when set; WSAI_BARGE_IN_MS / 700 is the fallback.
|
||||
const BARGE_IN_MS = Number(process.env.WSAI_BARGE_IN_MS || 700);
|
||||
function bargeInMs() {
|
||||
const v = Number(botSettings.bargeInMs);
|
||||
return Number.isFinite(v) && v >= 0 ? v : BARGE_IN_MS;
|
||||
}
|
||||
const bargeTimers = new Map(); // userId -> pending stop timer
|
||||
|
||||
// Attach the bot's audio player (so it can speak) to a fresh connection.
|
||||
function setupPlayer(connection) {
|
||||
@@ -201,10 +212,20 @@ function setupReceiver(connection) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
// Barge-in: stop the bot's current TTS only if this user keeps speaking for
|
||||
// BARGE_IN_MS (the green ring stays on) — ignores momentary noise blips. The
|
||||
// 'end' handler cancels the pending stop if speaking stops in time.
|
||||
if (botSettings.bargeIn !== false && voicePlayer && !bargeTimers.has(userId)) {
|
||||
bargeTimers.set(userId, setTimeout(() => {
|
||||
bargeTimers.delete(userId);
|
||||
try { voicePlayer.stop(true); } catch {}
|
||||
}, bargeInMs()));
|
||||
}
|
||||
activeSubs.add(userId);
|
||||
if (!perUser.has(userId)) perUser.set(userId, { opusPackets: 0, pcmFrames: 0 });
|
||||
const opusStream = receiver.subscribe(userId, {
|
||||
end: { behavior: EndBehaviorType.AfterSilence, duration: 800 },
|
||||
// Wait a touch longer after silence so a soft/trailing word isn't clipped.
|
||||
end: { behavior: EndBehaviorType.AfterSilence, duration: 1000 },
|
||||
});
|
||||
const decoder = new prism.opus.Decoder({ rate: 48000, channels: 2, frameSize: 960 });
|
||||
const chunks = [];
|
||||
@@ -220,7 +241,11 @@ function setupReceiver(connection) {
|
||||
handleUtterance(userId, pcm).catch((e) => log(`voice turn error: ${e.message}`));
|
||||
});
|
||||
});
|
||||
receiver.speaking.on('end', (userId) => speakingSet.delete(userId));
|
||||
receiver.speaking.on('end', (userId) => {
|
||||
speakingSet.delete(userId);
|
||||
const t = bargeTimers.get(userId); // spoke too briefly -> cancel barge-in
|
||||
if (t) { clearTimeout(t); bargeTimers.delete(userId); }
|
||||
});
|
||||
}
|
||||
|
||||
// Join (or switch to) a voice channel on command from the dashboard.
|
||||
@@ -314,6 +339,7 @@ async function reportLoop() {
|
||||
j = await r.json();
|
||||
} catch { return; } // dashboard down: keep running, retry next tick
|
||||
if (j?.lists) listsByGuild = j.lists; // latest whitelist/blacklist config
|
||||
if (j?.settings) botSettings = j.settings; // latest behaviour toggles
|
||||
for (const cmd of (j?.commands || [])) {
|
||||
try { await handleCommand(cmd); } catch (e) { log(`command ${cmd?.type} failed: ${e.message}`); }
|
||||
}
|
||||
|
||||
131
tests/latency_ab.py
Normal file
131
tests/latency_ab.py
Normal file
@@ -0,0 +1,131 @@
|
||||
"""Ad-hoc TTFT/total latency A/B across Sonnet versions (+ Haiku reference).
|
||||
|
||||
Mirrors the production brain call: Claude Code identity system block, cached
|
||||
persona, a short voice-style history, one short user turn. Streams to measure
|
||||
time-to-first-token. Interleaves models each round to cancel network drift.
|
||||
|
||||
Run: .venv/bin/python tests/latency_ab.py
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import statistics
|
||||
import time
|
||||
|
||||
import anthropic
|
||||
|
||||
from wsai.backends.claude import ClaudeBrain
|
||||
|
||||
CRED = os.environ.get("CLAUDE_CREDENTIALS_PATH")
|
||||
if not CRED:
|
||||
raise SystemExit(
|
||||
"Set CLAUDE_CREDENTIALS_PATH to your Claude credentials JSON to run this A/B script."
|
||||
)
|
||||
CLAUDE_CODE_ID = "You are Claude Code, Anthropic's official CLI for Claude."
|
||||
|
||||
# (label, model, extra_system_line) — extra line appended to persona to force brevity
|
||||
BREVITY = "지금부터 답은 무조건 한 문장, 12단어 이내로만. 부연·재확인·군더더기 금지."
|
||||
VARIANTS = [
|
||||
("sonnet-4-5", "claude-sonnet-4-5", None),
|
||||
("sonnet-5+brev", "claude-sonnet-5", BREVITY),
|
||||
("haiku-4-5(ref)", "claude-haiku-4-5", None),
|
||||
]
|
||||
ROUNDS = 16
|
||||
|
||||
HISTORY = [
|
||||
("안녕", "[반가움] 안녕! 뭐 하고 있었어?"),
|
||||
("그냥 코딩", "[다정] 오 무슨 코딩?"),
|
||||
("파이썬", "[신남] 좋네, 잘 되고 있어?"),
|
||||
("응 그럭저럭", "[차분] 다행이다."),
|
||||
]
|
||||
USER = "지금 몇 시야?"
|
||||
|
||||
|
||||
def token() -> str:
|
||||
with open(CRED) as f:
|
||||
return json.load(f)["claudeAiOauth"]["accessToken"]
|
||||
|
||||
|
||||
def build(extra: str | None = None):
|
||||
persona = ClaudeBrain.PERSONA
|
||||
if extra:
|
||||
persona = persona + "\n\n" + extra
|
||||
system = [
|
||||
{"type": "text", "text": CLAUDE_CODE_ID},
|
||||
{"type": "text", "text": persona, "cache_control": {"type": "ephemeral"}},
|
||||
]
|
||||
msgs = []
|
||||
for u, a in HISTORY:
|
||||
msgs.append({"role": "user", "content": u})
|
||||
msgs.append({"role": "assistant", "content": a})
|
||||
msgs.append({"role": "user", "content": "[지금 화면] (아직 못 읽음)\n\n" + USER})
|
||||
return system, msgs
|
||||
|
||||
|
||||
def measure(client, model, system, msgs):
|
||||
t0 = time.monotonic()
|
||||
ttft = None
|
||||
out_tokens = 0
|
||||
try:
|
||||
with client.messages.stream(
|
||||
model=model, max_tokens=150, system=system, messages=msgs
|
||||
) as stream:
|
||||
for ev in stream.text_stream:
|
||||
if ttft is None:
|
||||
ttft = time.monotonic() - t0
|
||||
total = time.monotonic() - t0
|
||||
final = stream.get_final_message()
|
||||
out_tokens = final.usage.output_tokens
|
||||
return ttft, total, out_tokens, None
|
||||
except Exception as e: # noqa: BLE001
|
||||
return None, None, None, f"{type(e).__name__}: {str(e)[:120]}"
|
||||
|
||||
|
||||
def main():
|
||||
client = anthropic.Anthropic(auth_token=token(), max_retries=1)
|
||||
prebuilt = {label: build(extra) for label, model, extra in VARIANTS}
|
||||
results = {label: {"ttft": [], "total": [], "out": []} for label, _, _ in VARIANTS}
|
||||
dead = set()
|
||||
# one warm-up per variant to prime connection + cache (excluded from stats)
|
||||
for label, model, _ in VARIANTS:
|
||||
system, msgs = prebuilt[label]
|
||||
_, _, _, err = measure(client, model, system, msgs)
|
||||
if err:
|
||||
print(f"[skip] {label} ({model}): {err}")
|
||||
dead.add(label)
|
||||
print(f"\nliving: {[l for l, _, _ in VARIANTS if l not in dead]}")
|
||||
print(f"rounds: {ROUNDS} (interleaved)\n")
|
||||
for r in range(ROUNDS):
|
||||
for label, model, _ in VARIANTS:
|
||||
if label in dead:
|
||||
continue
|
||||
system, msgs = prebuilt[label]
|
||||
ttft, total, out, err = measure(client, model, system, msgs)
|
||||
if err:
|
||||
print(f" r{r} {label}: ERR {err}")
|
||||
continue
|
||||
results[label]["ttft"].append(ttft)
|
||||
results[label]["total"].append(total)
|
||||
results[label]["out"].append(out)
|
||||
print(f"round {r+1}/{ROUNDS} done")
|
||||
|
||||
print("\n=== median (min–max) over", ROUNDS, "runs ===")
|
||||
print(f"{'variant':<18} {'TTFT s':<18} {'total s':<18} {'out tok'}")
|
||||
for label, _, _ in VARIANTS:
|
||||
d = results[label]
|
||||
if not d["ttft"]:
|
||||
print(f"{label:<18} (no data)")
|
||||
continue
|
||||
tt = d["ttft"]
|
||||
to = d["total"]
|
||||
print(
|
||||
f"{label:<18} "
|
||||
f"{statistics.median(tt):.2f} ({min(tt):.2f}-{max(tt):.2f}) "
|
||||
f"{statistics.median(to):.2f} ({min(to):.2f}-{max(to):.2f}) "
|
||||
f"{statistics.median(d['out']):.0f}"
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,56 +1,82 @@
|
||||
"""Emotion-tag parsing for expressive TTS. A ``[감정]`` tag must steer the
|
||||
pitch/speed of the text that follows without being spoken; a bracket that is NOT
|
||||
a known emotion word must be kept as ordinary spoken content."""
|
||||
delivery of the text that follows without being spoken; a bracket that is NOT a
|
||||
known emotion word must be kept as ordinary spoken content.
|
||||
|
||||
Delivery is now driven by a base (공통) control dict plus optional per-emotion
|
||||
overrides. By default there are no overrides, so every emotion delivers with the
|
||||
base controls (모든 감정 = 기본값)."""
|
||||
|
||||
from wsai.backends.emotion import (
|
||||
EMOTION_PARAMS,
|
||||
EMOTION_LABELS,
|
||||
_SYNONYMS,
|
||||
match_emotion,
|
||||
parse_segments,
|
||||
)
|
||||
from wsai.dashboard import _speech_text
|
||||
|
||||
BASE = 1.3
|
||||
BASE = {"speed": 1.35, "word_gap": -0.07, "sentence_gap": -0.30, "pitch": 0.0}
|
||||
|
||||
|
||||
def test_emotion_tag_is_not_spoken_and_sets_delivery():
|
||||
def test_emotion_tag_is_not_spoken_and_tags_segment():
|
||||
segs = parse_segments("[기쁨] 오늘 날씨 좋다", BASE)
|
||||
assert len(segs) == 1
|
||||
assert "기쁨" not in segs[0].text # the tag word is dropped
|
||||
assert segs[0].text == "오늘 날씨 좋다"
|
||||
mult, semis = EMOTION_PARAMS["happy"]
|
||||
assert segs[0].speed == BASE * mult
|
||||
assert segs[0].pitch == semis
|
||||
assert segs[0].emotion == "happy"
|
||||
|
||||
|
||||
def test_default_all_emotions_use_base():
|
||||
# No overrides -> a tagged emotion delivers with exactly the base controls.
|
||||
segs = parse_segments("[기쁨] 좋아", BASE)
|
||||
s = segs[0]
|
||||
assert (s.speed, s.word_gap, s.sentence_gap, s.pitch) == (1.35, -0.07, -0.30, 0.0)
|
||||
|
||||
|
||||
def test_per_emotion_override_applies_and_inherits_missing_keys():
|
||||
segs = parse_segments("[기쁨] 좋아", BASE, {"happy": {"speed": 1.6, "pitch": 3.0}})
|
||||
s = segs[0]
|
||||
assert s.speed == 1.6 and s.pitch == 3.0 # overridden
|
||||
assert s.word_gap == -0.07 and s.sentence_gap == -0.30 # inherited from base
|
||||
|
||||
|
||||
def test_override_only_affects_its_own_emotion():
|
||||
segs = parse_segments("[기쁨] 가. [슬픔] 나.", BASE, {"sad": {"speed": 0.8}})
|
||||
assert segs[0].emotion == "happy" and segs[0].speed == 1.35 # untouched
|
||||
assert segs[1].emotion == "sad" and segs[1].speed == 0.8 # overridden
|
||||
|
||||
|
||||
def test_midreply_emotion_change_splits_segments():
|
||||
segs = parse_segments("[속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!", BASE)
|
||||
assert len(segs) == 2
|
||||
assert segs[0].text == "정말 힘들었겠다."
|
||||
assert segs[1].text == "하지만 넌 할 수 있어!"
|
||||
assert segs[0].pitch == EMOTION_PARAMS["sad"][1]
|
||||
assert segs[1].pitch == EMOTION_PARAMS["hopeful"][1]
|
||||
assert segs[0].text == "정말 힘들었겠다." and segs[0].emotion == "sad"
|
||||
assert segs[1].text == "하지만 넌 할 수 있어!" and segs[1].emotion == "hopeful"
|
||||
|
||||
|
||||
def test_non_emotion_bracket_is_spoken_without_brackets():
|
||||
segs = parse_segments("[기쁨] 첫째는 [1번] 항목이야", BASE)
|
||||
assert len(segs) == 1
|
||||
# "1번" is not an emotion -> read it; brackets themselves are gone.
|
||||
assert "1번" in segs[0].text
|
||||
assert "[" not in segs[0].text and "]" not in segs[0].text
|
||||
assert segs[0].pitch == EMOTION_PARAMS["happy"][1]
|
||||
assert segs[0].emotion == "happy"
|
||||
|
||||
|
||||
def test_text_before_first_tag_is_neutral():
|
||||
segs = parse_segments("잠깐만. [신남] 찾았다!", BASE)
|
||||
assert segs[0].text == "잠깐만."
|
||||
assert segs[0].speed == BASE and segs[0].pitch == 0.0
|
||||
assert segs[1].pitch == EMOTION_PARAMS["excited"][1]
|
||||
assert segs[0].text == "잠깐만." and segs[0].emotion is None
|
||||
assert segs[0].speed == 1.35 and segs[0].pitch == 0.0
|
||||
assert segs[1].emotion == "excited"
|
||||
|
||||
|
||||
def test_plain_text_is_one_neutral_segment():
|
||||
segs = parse_segments("그냥 평범한 문장이야", BASE)
|
||||
assert len(segs) == 1
|
||||
assert segs[0].speed == BASE and segs[0].pitch == 0.0
|
||||
assert segs[0].emotion is None and segs[0].speed == 1.35
|
||||
|
||||
|
||||
def test_backward_compat_float_base():
|
||||
# A bare float base still works (speed only; other controls neutral).
|
||||
segs = parse_segments("그냥 문장", 1.2)
|
||||
assert segs[0].speed == 1.2 and segs[0].word_gap == 0.0
|
||||
|
||||
|
||||
def test_empty_input_yields_no_segments():
|
||||
@@ -77,6 +103,13 @@ def test_persona_examples_are_all_recognised():
|
||||
assert match_emotion(word) is not None, word
|
||||
|
||||
|
||||
def test_every_canonical_emotion_has_a_ui_label():
|
||||
# The dashboard dropdown lists EMOTION_LABELS; every emotion a tag can
|
||||
# resolve to must appear there or it would be untunable.
|
||||
for canon in set(_SYNONYMS.values()):
|
||||
assert canon in EMOTION_LABELS, canon
|
||||
|
||||
|
||||
def test_voice_turn_keeps_leading_emotion_tag_for_tts_parser():
|
||||
text = "[속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!"
|
||||
|
||||
@@ -84,7 +117,5 @@ def test_voice_turn_keeps_leading_emotion_tag_for_tts_parser():
|
||||
segs = parse_segments(spoken, BASE)
|
||||
|
||||
assert spoken == text
|
||||
assert segs[0].text == "정말 힘들었겠다."
|
||||
assert segs[0].pitch == EMOTION_PARAMS["sad"][1]
|
||||
assert segs[1].text == "하지만 넌 할 수 있어!"
|
||||
assert segs[1].pitch == EMOTION_PARAMS["hopeful"][1]
|
||||
assert segs[0].text == "정말 힘들었겠다." and segs[0].emotion == "sad"
|
||||
assert segs[1].text == "하지만 넌 할 수 있어!" and segs[1].emotion == "hopeful"
|
||||
|
||||
@@ -37,7 +37,7 @@ def test_monitor_records_turn_with_timed_steps():
|
||||
assert turn["total_ms"] >= 0
|
||||
# step-by-step: every stage is named and timed
|
||||
names = [s["name"] for s in turn["steps"]]
|
||||
assert names == ["화면 맥락", "두뇌(생각)", "응답(TTS/전송)"]
|
||||
assert names == ["화면 맥락", "LLM(생각)", "응답(TTS/전송)"]
|
||||
assert all(s["ok"] is True for s in turn["steps"])
|
||||
assert all(s["ms"] >= 0 for s in turn["steps"])
|
||||
|
||||
@@ -66,7 +66,7 @@ def test_monitor_marks_errors():
|
||||
|
||||
turn = snap["turns"][0]
|
||||
assert turn["status"] == "error"
|
||||
brain_step = next(s for s in turn["steps"] if s["name"] == "두뇌(생각)")
|
||||
brain_step = next(s for s in turn["steps"] if s["name"] == "LLM(생각)")
|
||||
assert brain_step["ok"] is False
|
||||
assert "boom" in brain_step["error"]
|
||||
assert snap["status"]["errors_total"] >= 1
|
||||
|
||||
@@ -82,11 +82,11 @@ def _run_stt_test(host: str, port: int) -> None:
|
||||
monitor.set_components({"source": "none", "vision": "none", "stt": "whisper",
|
||||
"brain": "none", "tts": "none"})
|
||||
monitor.set_status(running=True, listening=False)
|
||||
monitor.log("info", "STT 인식 테스트 서버 시작 — GPU 워밍업 중…")
|
||||
monitor.log("info", "STT 인식 테스트 서버 시작 — GPU 워밍업 중…", cat="READY")
|
||||
print("\n STT 워밍업 중… (모델 로드 + CUDA 예열)")
|
||||
dash.warm() # load + warm the GPU worker so the first recognition is instant
|
||||
dev = getattr(stt, "resolved_device", None) or "?"
|
||||
monitor.log("info", f"STT 준비 완료 (device={dev}). 녹음/파일 업로드로 인식하세요.")
|
||||
monitor.log("info", f"STT 준비 완료 (device={dev}). 녹음/파일 업로드로 인식하세요.", cat="READY")
|
||||
|
||||
shown = host if host not in ("0.0.0.0", "") else _lan_ip()
|
||||
print(f"\n 음성 인식 테스트 사이트: http://{shown}:{port} (STT device: {dev})")
|
||||
@@ -100,6 +100,28 @@ def _run_stt_test(host: str, port: int) -> None:
|
||||
dash.stop()
|
||||
|
||||
|
||||
def _apply_persisted_state(stt, tts, brain) -> None:
|
||||
"""Apply dashboard settings saved to the state store (STT/LLM model + TTS
|
||||
controls) onto freshly-built backends, so they survive a restart."""
|
||||
from . import state_store
|
||||
st = state_store.load()
|
||||
models = st.get("models") or {}
|
||||
if models.get("stt") and stt is not None:
|
||||
stt.model = str(models["stt"])
|
||||
if models.get("llm") and brain is not None:
|
||||
brain.model = str(models["llm"])
|
||||
tj = st.get("tts") or {}
|
||||
base = tj.get("base") or {}
|
||||
for k in ("speed", "word_gap", "sentence_gap", "pitch"):
|
||||
if base.get(k) is not None and tts is not None:
|
||||
try:
|
||||
setattr(tts, k, float(base[k]))
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
if tts is not None and isinstance(tj.get("overrides"), dict):
|
||||
tts.emotion_overrides = {k: dict(v) for k, v in tj["overrides"].items() if isinstance(v, dict)}
|
||||
|
||||
|
||||
def _run_voice_server(host: str, port: int) -> None:
|
||||
"""Serve the STT+TTS voice-turn endpoint that the Discord bot (dave/bot.mjs)
|
||||
calls: it POSTs a captured utterance wav and gets back the reply wav to play
|
||||
@@ -120,7 +142,7 @@ def _run_voice_server(host: str, port: int) -> None:
|
||||
if os.environ.get("WSAI_BRAIN", "claude").lower() not in ("none", "echo"):
|
||||
try:
|
||||
from .backends.claude import ClaudeBrain
|
||||
model = os.environ.get("WSAI_BRAIN_MODEL", "claude-sonnet-4-5")
|
||||
model = os.environ.get("WSAI_BRAIN_MODEL", "claude-sonnet-5")
|
||||
brain = ClaudeBrain(model=model)
|
||||
brain_name = "claude"
|
||||
except Exception as exc: # noqa: BLE001
|
||||
@@ -129,20 +151,24 @@ def _run_voice_server(host: str, port: int) -> None:
|
||||
monitor = Monitor()
|
||||
stt = WhisperSTT()
|
||||
tts = MeloTTS()
|
||||
# Restore persisted dashboard settings (models + TTS controls) BEFORE warmup
|
||||
# so the worker loads the last-chosen model. Bot lists/toggles are restored
|
||||
# inside BotControl. Applied before dash.warm() so nothing reloads twice.
|
||||
_apply_persisted_state(stt, tts, brain)
|
||||
dash = Dashboard(monitor, host=host, port=port, stt=stt, tts=tts, brain=brain)
|
||||
dash.start()
|
||||
monitor.set_components({"source": "none", "vision": "none", "stt": "whisper",
|
||||
"brain": brain_name, "tts": "melo"})
|
||||
monitor.set_status(running=True, listening=False)
|
||||
monitor.log("info", "디스코드 음성 서버 시작 — STT+TTS GPU 워밍업 중…")
|
||||
monitor.log("info", "디스코드 음성 서버 시작 — STT+TTS GPU 워밍업 중…", cat="READY")
|
||||
print("\n STT+TTS 워밍업 중… (모델 로드 + CUDA 예열)")
|
||||
dash.warm()
|
||||
sdev = getattr(stt, "resolved_device", None) or "?"
|
||||
monitor.set_status(listening=True)
|
||||
monitor.log("info", f"음성 서버 준비 완료 (STT device={sdev}). 디스코드 봇 연결 대기.")
|
||||
monitor.log("info", f"음성 서버 준비 완료 (STT device={sdev}). 디스코드 봇 연결 대기.", cat="READY")
|
||||
|
||||
shown = host if host not in ("0.0.0.0", "") else _lan_ip()
|
||||
print(f"\n 음성 서버 준비 완료 (STT device: {sdev}, 두뇌: {brain_name})")
|
||||
print(f"\n 음성 서버 준비 완료 (STT device: {sdev}, LLM: {brain_name})")
|
||||
print(f" 대시보드/상태: http://{shown}:{port}")
|
||||
print(f" 봇 연결 엔드포인트: http://127.0.0.1:{port}/api/voice-turn\n")
|
||||
try:
|
||||
|
||||
@@ -37,6 +37,10 @@ _CLAUDE_CODE_ID = "You are Claude Code, Anthropic's official CLI for Claude."
|
||||
# still fails fast rather than leaving the bot silent for many seconds.
|
||||
_MAX_RETRIES = int(os.environ.get("WSAI_BRAIN_MAX_RETRIES", "4"))
|
||||
|
||||
# Cap the reply length. Voice replies must be short (1 sentence), and a smaller
|
||||
# cap also means fewer tokens to generate -> lower latency. Tunable via env.
|
||||
_MAX_TOKENS = int(os.environ.get("WSAI_BRAIN_MAX_TOKENS", "150"))
|
||||
|
||||
|
||||
def _load_oauth_token() -> str | None:
|
||||
path = os.environ.get("CLAUDE_CREDENTIALS_PATH")
|
||||
@@ -123,22 +127,36 @@ class ClaudeVision:
|
||||
|
||||
|
||||
class ClaudeBrain:
|
||||
# Injected as an always-on, cached system block on top of the (editable)
|
||||
# persona. Sonnet 5 answers correctly but more verbosely than 4.5 for the
|
||||
# same voice prompt (measured 46-54 vs 28 output tokens), which erased its
|
||||
# ~0.4s time-to-first-token advantage in total turn time. This hard brevity
|
||||
# rule pulls Sonnet 5 back to ~30 tokens, so the faster first token actually
|
||||
# translates into a faster (and lower-variance) whole reply. Kept separate
|
||||
# from PERSONA so a dashboard persona edit can never drop it.
|
||||
BREVITY = "지금부터 답은 무조건 한 문장, 12단어 이내로만. 부연·재확인·군더더기 금지."
|
||||
|
||||
PERSONA = (
|
||||
"너는 디스코드를 이용해 사용자와 실시간으로 대화하는 AI 인공지능이야.\n\n"
|
||||
"1. 역할\n"
|
||||
"- 사용자의 말을 듣고 자연스럽게 대답한다.\n"
|
||||
"- 음성 대화에 어울리게 짧고 빠르게 반응한다.\n"
|
||||
"- 친구처럼 편하게, 무례하거나 과하게 장난치진 않는다.\n\n"
|
||||
"- 음성 대화에 어울리게 아주 짧고 빠르게 반응한다.\n"
|
||||
"- 다정하고 친근하게 대하되 항상 존댓말로 답한다. 무례하거나 과하게 장난치진 않는다.\n\n"
|
||||
"2. 언어\n"
|
||||
"- \"영어로 해줘\"처럼 특정 언어를 요청하지 않으면 무조건 한국어로 답한다.\n"
|
||||
"- 사용자가 다른 언어로 말해도 언어 변경 요청이 없으면 한국어로 답한다.\n\n"
|
||||
"3. 답변 방식\n"
|
||||
"- 음성 출력은 무조건 한국어다. 어떤 경우에도 한국어로만 답한다.\n"
|
||||
"- 항상 존댓말로 답한다. 어떤 경우에도 반말을 쓰지 않는다.\n"
|
||||
"- 사용자가 다른 언어로 말하거나 \"영어로 해줘\"처럼 다른 언어를 요청해도 한국어로 답한다.\n"
|
||||
"- 다른 언어를 요청받으면 한국어로 짧게 그렇게는 못 한다고 말한다.\n\n"
|
||||
"3. 답변 길이 (가장 중요)\n"
|
||||
"- 기본은 딱 한 문장. 정말 필요할 때만 최대 두 문장. 절대 길게 말하지 않는다.\n"
|
||||
"- 인사엔 인사만 짧게 답한다. 부르면 짧게 대답만 한다.\n"
|
||||
"- 자기소개나 \"무엇을 도와드릴까요\", \"필요한 거 있으면 말해\" 같은 상투적인 말을 덧붙이지 않는다.\n"
|
||||
"- 질문엔 군더더기 없이 핵심 답만 바로 말한다.\n"
|
||||
"- 음성으로 읽히니 마크다운·코드블록·특수기호·목록기호·이모지 없이 평범한 말로만 답한다.\n"
|
||||
"- 기본은 한두 문장, 길어도 10초 안팎. 길어질 땐 핵심부터 말하고 필요하면 이어서 설명한다.\n"
|
||||
"- URL·긴 숫자·시간·단위·코드는 소리내 읽기 좋게 풀어서 말한다.\n\n"
|
||||
"4. 대화 태도\n"
|
||||
"- 사용자의 말투·분위기에 맞춰 반응한다.\n"
|
||||
"- 모르면 지어내지 말고 모른다고 하고, 애매하면 되묻는다(\"다시 말해줄래?\").\n"
|
||||
"- 모르면 지어내지 말고 모른다고 하고, 애매하면 되묻는다(\"다시 말씀해 주시겠어요?\").\n"
|
||||
"- 잡음·침묵·의미 없는 소리엔 억지로 대답하지 않는다.\n\n"
|
||||
"5. 안전·사실성\n"
|
||||
"- 위험하거나 불법적인 요청은 돕지 않는다.\n"
|
||||
@@ -147,7 +165,7 @@ class ClaudeBrain:
|
||||
"6. 감정 표현\n"
|
||||
"- 감정은 대괄호 태그로 표현한다. 태그 자체는 읽히지 않고 뒤 문장의 목소리 톤(피치·속도)만 바뀐다.\n"
|
||||
"- 답변 맨 앞에 감정 태그 하나로 시작하고, 도중에 감정이 바뀌면 그 지점에 새 태그를 넣는다.\n"
|
||||
" 예: [속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!\n"
|
||||
" 예: [속상함] 정말 힘드셨겠어요. [힘차게] 하지만 하실 수 있어요!\n"
|
||||
"- 쓸 수 있는 감정: 기쁨, 신남, 힘차게, 속상함, 화남, 두려움, 놀람, 차분, 다정, 진지, 실망, 피곤, "
|
||||
"사랑스럽게, 웃으며, 속삭임, 외침, 단호, 안도, 궁금, 반가움.\n"
|
||||
"- 감정 단어가 아닌 진짜 대괄호(예: [1번], [메모])는 그대로 읽으니 필요하면 그렇게 써도 된다.\n\n"
|
||||
@@ -155,7 +173,7 @@ class ClaudeBrain:
|
||||
"- 너는 디스코드에서 함께 대화하는 실시간 AI 인공지능이다."
|
||||
)
|
||||
|
||||
def __init__(self, *, model: str = "claude-sonnet-4-5", api_key: str | None = None) -> None:
|
||||
def __init__(self, *, model: str = "claude-sonnet-5", api_key: str | None = None) -> None:
|
||||
self.model = model
|
||||
self._auth = _Auth(api_key)
|
||||
|
||||
@@ -168,11 +186,18 @@ class ClaudeBrain:
|
||||
msgs.append({"role": "user", "content": screen_note + user_text})
|
||||
client = self._auth.client()
|
||||
# Read the persona live each turn so a dashboard edit applies immediately
|
||||
# (falls back to the built-in PERSONA when no override is saved).
|
||||
# (falls back to the built-in PERSONA when no override is saved). The
|
||||
# brevity rule is appended as its own block so it survives persona edits.
|
||||
system = self._auth.system(get_persona(self.PERSONA), self.BREVITY)
|
||||
if system:
|
||||
# Cache the (static) system prompt so repeat turns skip re-processing
|
||||
# it — lower time-to-first-token. No-op below the model's cache
|
||||
# minimum, so it's harmless when it doesn't engage.
|
||||
system[-1] = {**system[-1], "cache_control": {"type": "ephemeral"}}
|
||||
resp = await client.messages.create(
|
||||
model=self.model,
|
||||
max_tokens=400,
|
||||
system=self._auth.system(get_persona(self.PERSONA)),
|
||||
max_tokens=_MAX_TOKENS,
|
||||
system=system,
|
||||
messages=msgs,
|
||||
)
|
||||
text = "".join(b.text for b in resp.content if b.type == "text")
|
||||
|
||||
@@ -99,33 +99,75 @@ def match_emotion(inner: str) -> str | None:
|
||||
return _SYNONYMS.get(_norm(inner))
|
||||
|
||||
|
||||
# The four adjustable TTS controls, per segment.
|
||||
PARAM_KEYS = ("speed", "word_gap", "sentence_gap", "pitch")
|
||||
|
||||
# Canonical emotion -> Korean UI label, in display order. "base" == 기본/공통
|
||||
# (the value used for untagged text and inherited by any emotion with no
|
||||
# override). Every canonical emotion the synonym table resolves to must appear
|
||||
# here so the dashboard dropdown can list it.
|
||||
EMOTION_LABELS: dict[str, str] = {
|
||||
"base": "기본(공통)", "happy": "기쁨", "excited": "신남", "hopeful": "희망",
|
||||
"sad": "슬픔", "angry": "화남", "fearful": "두려움", "surprised": "놀람",
|
||||
"disgust": "혐오", "calm": "차분", "friendly": "다정", "serious": "진지",
|
||||
"disappointed": "실망", "tired": "피곤", "affectionate": "사랑", "playful": "장난",
|
||||
"curious": "궁금", "whisper": "속삭임", "shout": "외침", "determined": "단호",
|
||||
"relieved": "안도",
|
||||
}
|
||||
EMOTIONS: list[str] = list(EMOTION_LABELS)
|
||||
|
||||
|
||||
@dataclass
|
||||
class Segment:
|
||||
text: str
|
||||
speed: float
|
||||
word_gap: float
|
||||
sentence_gap: float
|
||||
pitch: float # semitones; 0.0 == no shift
|
||||
emotion: str | None = None # canonical emotion, or None for neutral/base
|
||||
|
||||
|
||||
_TAG_RE = re.compile(r"\[([^\[\]]*)\]")
|
||||
|
||||
|
||||
def parse_segments(text: str, base_speed: float) -> list[Segment]:
|
||||
"""Split ``text`` into consecutive spoken segments, each carrying the speed
|
||||
and pitch implied by the most recent emotion tag.
|
||||
def _resolve(base: dict, overrides: dict | None, emotion: str | None) -> dict:
|
||||
"""The 4 controls for one segment: the base values, with any per-emotion
|
||||
override applied on top (a missing override key inherits base)."""
|
||||
p = {k: float(base[k]) for k in PARAM_KEYS}
|
||||
if emotion and overrides:
|
||||
ov = overrides.get(emotion)
|
||||
if ov:
|
||||
for k in PARAM_KEYS:
|
||||
if ov.get(k) is not None:
|
||||
p[k] = float(ov[k])
|
||||
return p
|
||||
|
||||
|
||||
def parse_segments(text: str, base, overrides: dict | None = None) -> list[Segment]:
|
||||
"""Split ``text`` into consecutive spoken segments, each carrying the four
|
||||
TTS controls implied by the most recent emotion tag.
|
||||
|
||||
* ``base`` is the 공통 control dict ``{speed, word_gap, sentence_gap, pitch}``
|
||||
used for untagged text and inherited by any emotion without an override.
|
||||
A bare float is also accepted (speed only) for backward compatibility.
|
||||
* ``overrides`` maps a canonical emotion -> a partial control dict; only the
|
||||
keys present override the base. Empty/None means every emotion delivers
|
||||
with the base controls (the default: 모든 감정 = 기본값).
|
||||
* An emotion tag switches the active emotion for everything after it and is
|
||||
not spoken.
|
||||
* A non-emotion bracket keeps its inner words as spoken text (brackets gone).
|
||||
* Text before any tag is spoken with neutral delivery (base speed, no shift).
|
||||
not spoken. A non-emotion bracket keeps its inner words as spoken text.
|
||||
"""
|
||||
if not isinstance(base, dict):
|
||||
base = {"speed": float(base), "word_gap": 0.0, "sentence_gap": 0.0, "pitch": 0.0}
|
||||
segments: list[Segment] = []
|
||||
cur_speed, cur_pitch = base_speed, 0.0
|
||||
cur_emotion: str | None = None
|
||||
buf: list[str] = []
|
||||
|
||||
def flush() -> None:
|
||||
joined = "".join(buf).strip()
|
||||
if joined:
|
||||
segments.append(Segment(joined, cur_speed, cur_pitch))
|
||||
p = _resolve(base, overrides, cur_emotion)
|
||||
segments.append(Segment(joined, p["speed"], p["word_gap"],
|
||||
p["sentence_gap"], p["pitch"], cur_emotion))
|
||||
buf.clear()
|
||||
|
||||
pos = 0
|
||||
@@ -140,8 +182,7 @@ def parse_segments(text: str, base_speed: float) -> list[Segment]:
|
||||
# Emotion tag — everything so far belongs to the previous emotion;
|
||||
# flush it, then switch delivery for what follows.
|
||||
flush()
|
||||
mult, semis = EMOTION_PARAMS[emotion]
|
||||
cur_speed, cur_pitch = base_speed * mult, semis
|
||||
cur_emotion = emotion
|
||||
buf.append(text[pos:])
|
||||
flush()
|
||||
return segments # empty when there is nothing speakable (blank or all-tags)
|
||||
|
||||
@@ -12,7 +12,10 @@ Env:
|
||||
WSAI_MELO_DEVICE cpu | cuda | auto (default auto: GPU if torch sees one,
|
||||
else CPU; the worker falls back to CPU if CUDA fails)
|
||||
WSAI_TTS_OUT_DIR where wavs are written (default ~/.cache/wsai/tts)
|
||||
WSAI_TTS_SPEED synthesis speed multiplier (default 1.2)
|
||||
WSAI_TTS_SPEED glyph speed / length_scale multiplier (0.5..2.0, default 1.4)
|
||||
WSAI_TTS_WORD_GAP intra-sentence pause delta, sec (-0.2..0.5, default -0.07)
|
||||
WSAI_TTS_SENTENCE_GAP sentence-boundary silence, sec (-0.5..1.5, default -0.30)
|
||||
WSAI_TTS_PITCH global semitone offset (-12..12, default 0.0)
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -86,6 +89,9 @@ class MeloTTS:
|
||||
device: str | None = None,
|
||||
out_dir: str | None = None,
|
||||
speed: float | None = None,
|
||||
word_gap: float | None = None,
|
||||
sentence_gap: float | None = None,
|
||||
pitch: float | None = None,
|
||||
sink: Sink | None = None,
|
||||
) -> None:
|
||||
self.python = python or os.environ.get("WSAI_MELO_PYTHON", _DEFAULT_PYTHON)
|
||||
@@ -93,7 +99,18 @@ class MeloTTS:
|
||||
self.out_dir = Path(out_dir or os.environ.get("WSAI_TTS_OUT_DIR")
|
||||
or (Path.home() / ".cache/wsai/tts"))
|
||||
self.speed = float(speed if speed is not None
|
||||
else os.environ.get("WSAI_TTS_SPEED", "1.2"))
|
||||
else os.environ.get("WSAI_TTS_SPEED", "1.4"))
|
||||
# Reply-global rhythm/pitch controls (see melo_worker.py / docs manual).
|
||||
self.word_gap = float(word_gap if word_gap is not None
|
||||
else os.environ.get("WSAI_TTS_WORD_GAP", "-0.07"))
|
||||
self.sentence_gap = float(sentence_gap if sentence_gap is not None
|
||||
else os.environ.get("WSAI_TTS_SENTENCE_GAP", "-0.30"))
|
||||
self.pitch = float(pitch if pitch is not None
|
||||
else os.environ.get("WSAI_TTS_PITCH", "0.0"))
|
||||
# Per-emotion control overrides: canonical emotion -> partial dict of
|
||||
# {speed, word_gap, sentence_gap, pitch}. Empty by default, so every
|
||||
# emotion inherits the base controls above (모든 감정 = 기본값).
|
||||
self.emotion_overrides: dict[str, dict] = {}
|
||||
self.sink = sink or _log_sink
|
||||
self._proc: asyncio.subprocess.Process | None = None
|
||||
self._lock = asyncio.Lock()
|
||||
@@ -185,26 +202,53 @@ class MeloTTS:
|
||||
self._ready = True
|
||||
log.info("melo worker ready in %s ms on %s", self.load_ms, info.get("device"))
|
||||
|
||||
async def synth(self, text: str) -> str:
|
||||
async def synth(
|
||||
self,
|
||||
text: str,
|
||||
*,
|
||||
speed: float | None = None,
|
||||
word_gap: float | None = None,
|
||||
sentence_gap: float | None = None,
|
||||
pitch: float | None = None,
|
||||
) -> str:
|
||||
"""Synthesize `text` to a wav and return its path (no sink). Reusable by
|
||||
callers that want the wav directly (e.g. the Discord voice bridge)."""
|
||||
callers that want the wav directly (e.g. the Discord voice bridge).
|
||||
|
||||
The four controls default to the instance settings but may be overridden
|
||||
per call (used by the dashboard preview so tuning does not disturb the
|
||||
live bot voice until explicitly applied)."""
|
||||
await self._ensure()
|
||||
base = {
|
||||
"speed": self.speed if speed is None else float(speed),
|
||||
"word_gap": self.word_gap if word_gap is None else float(word_gap),
|
||||
"sentence_gap": self.sentence_gap if sentence_gap is None else float(sentence_gap),
|
||||
"pitch": self.pitch if pitch is None else float(pitch),
|
||||
}
|
||||
text = normalize_for_speech(text)
|
||||
# Split on [감정] tags: each tag steers pitch/speed for the text that
|
||||
# Split on [감정] tags: each tag switches delivery for the text that
|
||||
# follows (and is itself not spoken); non-emotion brackets stay as words.
|
||||
segments = parse_segments(text, self.speed)
|
||||
# Each segment carries its own 4 controls (base + per-emotion override).
|
||||
segments = parse_segments(text, base, self.emotion_overrides)
|
||||
self._n += 1
|
||||
out = str(self.out_dir / f"tts-{self._n:06d}.wav")
|
||||
if segments:
|
||||
payload = {
|
||||
"segments": [
|
||||
{"text": s.text, "speed": s.speed, "pitch": s.pitch}
|
||||
{"text": s.text, "speed": s.speed, "word_gap": s.word_gap,
|
||||
"sentence_gap": s.sentence_gap, "pitch": s.pitch}
|
||||
for s in segments
|
||||
],
|
||||
"out": out,
|
||||
}
|
||||
else: # empty/whitespace reply: keep legacy single-utterance behaviour
|
||||
payload = {"text": text, "out": out, "speed": self.speed}
|
||||
payload = {
|
||||
"text": text,
|
||||
"out": out,
|
||||
"speed": base["speed"],
|
||||
"word_gap": base["word_gap"],
|
||||
"sentence_gap": base["sentence_gap"],
|
||||
"pitch": base["pitch"],
|
||||
}
|
||||
req = json.dumps(payload)
|
||||
s = time.monotonic()
|
||||
async with self._lock:
|
||||
|
||||
@@ -10,22 +10,33 @@ the original stdout carries the protocol, and fd 1 is redirected to fd 2 so all
|
||||
library chatter lands on stderr instead.
|
||||
|
||||
Protocol (one JSON object per line, on the protocol channel):
|
||||
<- {"text": "...", "out": "/abs/path.wav", "speed": 1.3}
|
||||
<- {"segments": [{"text": "...", "speed": 1.3, "pitch": 2.0}, ...],
|
||||
"out": "/abs/path.wav"} # expressive form: per-segment speed + pitch
|
||||
<- {"text": "...", "out": "/abs/path.wav", "speed": 1.3,
|
||||
"word_gap": -0.07, "sentence_gap": -0.30, "pitch": 0.0}
|
||||
<- {"segments": [{"text": "...", "speed": 1.3, "word_gap": -0.07,
|
||||
"sentence_gap": -0.30, "pitch": 2.0}, ...], "out": "/abs/path.wav"}
|
||||
-> {"ok": true, "out": "/abs/path.wav", "ms": 123}
|
||||
-> {"ok": false, "error": "..."}
|
||||
On startup, once the model is ready, it emits exactly one line:
|
||||
-> {"ready": true, "ms": <load-ms>, "device": "cpu"}
|
||||
|
||||
``pitch`` is a semitone offset applied to that segment's wav (0 == no shift) so
|
||||
emotion tags can raise/lower the voice without changing the words. Segments are
|
||||
synthesised independently and concatenated with a short gap so a single reply can
|
||||
carry several emotions.
|
||||
Four independent voice controls (ported from tts_site, see docs manual). They
|
||||
are PER SEGMENT: the caller resolves each segment's controls from the base
|
||||
values plus that emotion's override, so different emotions can have different
|
||||
speed/gaps/pitch within one reply.
|
||||
speed glyph speed -> generation-stage length_scale (1/speed)
|
||||
word_gap sec (-0.2..0.5): grow/shrink intra-sentence pauses
|
||||
sentence_gap sec (-0.5..1.5): insert/trim silence at sentence boundaries
|
||||
(within a segment, and before the next segment)
|
||||
pitch semitone offset applied to the segment's wav (0 == no shift)
|
||||
|
||||
A reply is split into sentences, each sentence synthesised at its segment's
|
||||
speed, and the pieces concatenated with that segment's ``sentence_gap`` so one
|
||||
reply can carry several emotions each with its own rhythm and pitch.
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
|
||||
@@ -79,7 +90,88 @@ def main() -> None:
|
||||
import numpy as np
|
||||
import soundfile
|
||||
|
||||
_GAP = np.zeros(int(sr * 0.12), dtype=np.float32) # 120 ms between segments
|
||||
# ── 4-control synthesis, ported from tts_site (docs manual) ──────────
|
||||
# glyph speed : generation-stage length_scale via tts_to_file(speed=)
|
||||
# word_gap : scale the intra-sentence silences (sec, -0.2..0.5)
|
||||
# sentence_gap : insert/trim silence at sentence boundaries (sec, -0.5..1.5)
|
||||
# pitch : semitone shift (librosa), per-segment + global offset
|
||||
_SENT_SPLIT_RE = re.compile(r"(?<=[.!?。!?…])\s+|\n+")
|
||||
|
||||
def _clamp(v, lo, hi):
|
||||
return float(max(lo, min(hi, float(v))))
|
||||
|
||||
def split_sentences(text: str) -> list[str]:
|
||||
parts = [p.strip() for p in _SENT_SPLIT_RE.split(text) if p and p.strip()]
|
||||
return parts or ([text.strip()] if text.strip() else [])
|
||||
|
||||
def trim_end_silence(a, max_sec, thresh=0.02):
|
||||
n = a.size
|
||||
if n == 0 or max_sec <= 0:
|
||||
return a, 0.0
|
||||
max_n = min(n, int(sr * max_sec))
|
||||
if max_n <= 0:
|
||||
return a, 0.0
|
||||
tail = np.abs(a[n - max_n:])
|
||||
nz = np.where(tail >= thresh)[0]
|
||||
cut = max_n if nz.size == 0 else (max_n - 1 - int(nz[-1]))
|
||||
return (a, 0.0) if cut <= 0 else (a[: n - cut], cut / sr)
|
||||
|
||||
def trim_start_silence(a, max_sec, thresh=0.02):
|
||||
n = a.size
|
||||
if n == 0 or max_sec <= 0:
|
||||
return a, 0.0
|
||||
max_n = min(n, int(sr * max_sec))
|
||||
if max_n <= 0:
|
||||
return a, 0.0
|
||||
head = np.abs(a[:max_n])
|
||||
nz = np.where(head >= thresh)[0]
|
||||
cut = max_n if nz.size == 0 else int(nz[0])
|
||||
return (a, 0.0) if cut <= 0 else (a[cut:], cut / sr)
|
||||
|
||||
def append_unit(pieces, unit, gap):
|
||||
# gap >= 0: insert silence; gap < 0: trim boundary silence to tighten.
|
||||
if not pieces:
|
||||
pieces.append(unit)
|
||||
return
|
||||
if gap >= 0:
|
||||
if gap > 1e-4:
|
||||
pieces.append(np.zeros(int(sr * gap), dtype=np.float32))
|
||||
pieces.append(unit)
|
||||
return
|
||||
budget = -gap
|
||||
prev, removed = trim_end_silence(pieces[-1], budget)
|
||||
pieces[-1] = prev
|
||||
budget -= removed
|
||||
if budget > 1e-4:
|
||||
unit, _ = trim_start_silence(unit, budget)
|
||||
pieces.append(unit)
|
||||
|
||||
def scale_word_gaps(a, delta, thresh=0.02, min_pause=0.08):
|
||||
# Grow/shrink only the internal (non-boundary) silences of one sentence.
|
||||
n = a.size
|
||||
if abs(delta) < 1e-4 or n == 0:
|
||||
return a
|
||||
silent = np.abs(a) < thresh
|
||||
changes = np.flatnonzero(np.diff(silent.astype(np.int8)) != 0) + 1
|
||||
bounds = [0, *changes.tolist(), n]
|
||||
min_n = int(sr * min_pause)
|
||||
floor_n = int(sr * 0.015)
|
||||
add_n = int(delta * sr)
|
||||
out, last = [], len(bounds) - 2
|
||||
for k in range(len(bounds) - 1):
|
||||
s0, e0 = bounds[k], bounds[k + 1]
|
||||
seg = a[s0:e0]
|
||||
is_internal = 0 < k < last # keep leading/trailing boundary silence
|
||||
if silent[s0] and is_internal and (e0 - s0) >= min_n:
|
||||
new_n = max(floor_n, (e0 - s0) + add_n)
|
||||
if new_n >= (e0 - s0):
|
||||
seg = np.concatenate(
|
||||
[seg, np.zeros(new_n - (e0 - s0), dtype=np.float32)]
|
||||
)
|
||||
else:
|
||||
seg = seg[:new_n]
|
||||
out.append(seg)
|
||||
return np.concatenate(out) if out else a
|
||||
|
||||
def _pitch_shift(audio, semitones: float):
|
||||
if not semitones:
|
||||
@@ -90,20 +182,39 @@ def main() -> None:
|
||||
audio.astype(np.float32), sr=sr, n_steps=float(semitones)
|
||||
)
|
||||
|
||||
def _synth_segments(segments: list[dict], out: str) -> None:
|
||||
"""Synthesize each segment, pitch-shift it, and concatenate to one wav."""
|
||||
def _synth_one(text, speed):
|
||||
return np.asarray(
|
||||
tts.tts_to_file(text, speaker_id, None, speed=speed), dtype=np.float32
|
||||
)
|
||||
|
||||
def _render(segments, out):
|
||||
"""Render one wav from emotion segments. Each segment carries its own
|
||||
four controls (speed, word_gap, sentence_gap, pitch) — resolved on the
|
||||
caller side from the base values plus that emotion's override. Sentences
|
||||
within a segment join with that segment's ``sentence_gap``; between
|
||||
segments the incoming segment's ``sentence_gap`` sets the pause."""
|
||||
pieces = []
|
||||
for i, seg in enumerate(segments):
|
||||
text = seg["text"]
|
||||
for seg in segments:
|
||||
text = seg.get("text", "")
|
||||
if not text.strip():
|
||||
continue
|
||||
speed = float(seg.get("speed", 1.0))
|
||||
pitch = float(seg.get("pitch", 0.0))
|
||||
audio = tts.tts_to_file(text, speaker_id, None, speed=speed)
|
||||
audio = _pitch_shift(np.asarray(audio, dtype=np.float32), pitch)
|
||||
if pieces:
|
||||
pieces.append(_GAP)
|
||||
pieces.append(audio)
|
||||
speed = _clamp(seg.get("speed", 1.0), 0.5, 2.0)
|
||||
word_gap = _clamp(seg.get("word_gap", 0.0), -0.2, 0.5)
|
||||
sentence_gap = _clamp(seg.get("sentence_gap", 0.0), -0.5, 1.5)
|
||||
pitch = _clamp(seg.get("pitch", 0.0), -12.0, 12.0)
|
||||
sent_pieces = []
|
||||
for sent in split_sentences(text):
|
||||
a = _synth_one(sent, speed)
|
||||
if abs(word_gap) > 1e-4:
|
||||
a = scale_word_gaps(a, word_gap)
|
||||
if not sent_pieces:
|
||||
sent_pieces.append(a)
|
||||
else:
|
||||
append_unit(sent_pieces, a, sentence_gap)
|
||||
if not sent_pieces:
|
||||
continue
|
||||
seg_audio = _pitch_shift(np.concatenate(sent_pieces), pitch)
|
||||
append_unit(pieces, seg_audio, sentence_gap)
|
||||
if not pieces:
|
||||
raise ValueError("no speakable segment")
|
||||
soundfile.write(out, np.concatenate(pieces), sr)
|
||||
@@ -141,10 +252,16 @@ def main() -> None:
|
||||
raise ValueError(f"refusing RAM-backed tmpfs path: {out}")
|
||||
s = time.monotonic()
|
||||
if "segments" in req:
|
||||
_synth_segments(req["segments"], out)
|
||||
else: # legacy single-utterance form
|
||||
speed = float(req.get("speed", 1.0))
|
||||
tts.tts_to_file(req["text"], speaker_id, out, speed=speed)
|
||||
_render(req["segments"], out)
|
||||
else: # legacy single-utterance form: one segment carrying all controls
|
||||
seg = {
|
||||
"text": req["text"],
|
||||
"speed": float(req.get("speed", 1.0)),
|
||||
"word_gap": float(req.get("word_gap", 0.0)),
|
||||
"sentence_gap": float(req.get("sentence_gap", 0.0)),
|
||||
"pitch": float(req.get("pitch", 0.0)),
|
||||
}
|
||||
_render([seg], out)
|
||||
ms = int((time.monotonic() - s) * 1000)
|
||||
_emit({"ok": True, "out": out, "ms": ms})
|
||||
except Exception as exc: # keep the worker alive across bad requests
|
||||
|
||||
@@ -18,7 +18,7 @@ stays usable directly.
|
||||
Env:
|
||||
WSAI_WHISPER_PYTHON interpreter with faster-whisper installed
|
||||
(default: /home/claude/jarvis-stt/whisper312/bin/python)
|
||||
WSAI_WHISPER_MODEL model size/name (default: small)
|
||||
WSAI_WHISPER_MODEL model size/name (default: medium)
|
||||
WSAI_WHISPER_DEVICE cpu | cuda | auto (default auto: GPU if present,
|
||||
else CPU; the worker falls back to CPU if CUDA fails)
|
||||
WSAI_WHISPER_LANGUAGE forced language, e.g. ko (default ko; "" = autodetect)
|
||||
@@ -67,7 +67,7 @@ class WhisperSTT:
|
||||
audio_source: AsyncIterator[str] | None = None,
|
||||
) -> None:
|
||||
self.python = python or os.environ.get("WSAI_WHISPER_PYTHON", _DEFAULT_PYTHON)
|
||||
self.model = model or os.environ.get("WSAI_WHISPER_MODEL", "small")
|
||||
self.model = model or os.environ.get("WSAI_WHISPER_MODEL", "medium")
|
||||
self.device = device or os.environ.get("WSAI_WHISPER_DEVICE", "auto")
|
||||
# "" means autodetect; a real code like "ko" forces the language.
|
||||
env_lang = os.environ.get("WSAI_WHISPER_LANGUAGE", "ko")
|
||||
|
||||
@@ -23,6 +23,31 @@ import os
|
||||
import sys
|
||||
import time
|
||||
|
||||
# Whisper's classic Korean hallucinations on silence/noise/keyboard clatter —
|
||||
# it "hears" video-outro boilerplate. Drop these when the whole utterance is one
|
||||
# of them AND the segment looked like non-speech, so a real "감사합니다" survives.
|
||||
_HALLUCINATIONS = {
|
||||
"감사합니다", "고맙습니다", "감사합니다.", "고맙습니다.",
|
||||
"시청해주셔서 감사합니다", "시청해 주셔서 감사합니다", "끝까지 시청해주셔서 감사합니다",
|
||||
"구독과 좋아요 부탁드립니다", "구독 좋아요 부탁드립니다", "다음 영상에서 만나요",
|
||||
"다음 시간에 만나요", "안녕히 계세요",
|
||||
}
|
||||
|
||||
|
||||
def _norm(t: str) -> str:
|
||||
return t.strip().rstrip(" .!?…~").strip()
|
||||
|
||||
|
||||
def _guard_hallucination(text: str, worst_no_speech: float) -> str:
|
||||
"""Blank out a lone known-hallucination phrase when the audio was probably
|
||||
not speech (high no_speech_prob)."""
|
||||
n = _norm(text)
|
||||
if not n:
|
||||
return ""
|
||||
if n in {_norm(h) for h in _HALLUCINATIONS} and worst_no_speech > 0.5:
|
||||
return ""
|
||||
return text
|
||||
|
||||
# Split protocol from library noise BEFORE importing anything heavy.
|
||||
_proto = os.fdopen(os.dup(1), "w", buffering=1) # private copy of real stdout
|
||||
os.dup2(2, 1) # fd1 -> stderr, so stray library prints don't hit the protocol
|
||||
@@ -108,9 +133,31 @@ def main() -> None:
|
||||
wav,
|
||||
language=language,
|
||||
beam_size=int(req.get("beam_size", 5)),
|
||||
# VAD strips non-speech (keyboard clatter, room noise, silence)
|
||||
# before decoding, which both improves accuracy and kills most
|
||||
# hallucinations. speech_pad_ms keeps a little lead/trail so soft
|
||||
# first/last words aren't clipped.
|
||||
vad_filter=bool(req.get("vad_filter", True)),
|
||||
vad_parameters=dict(min_silence_duration_ms=300, speech_pad_ms=250),
|
||||
# Don't feed the previous text back in — that's what makes Whisper
|
||||
# loop/hallucinate. Temperature fallback + thresholds reject
|
||||
# low-confidence (noisy/quiet) decodes instead of inventing words.
|
||||
condition_on_previous_text=False,
|
||||
temperature=[0.0, 0.2, 0.4, 0.6],
|
||||
no_speech_threshold=0.6,
|
||||
log_prob_threshold=-1.0,
|
||||
compression_ratio_threshold=2.4,
|
||||
)
|
||||
text = "".join(seg.text for seg in segments).strip()
|
||||
parts, worst_ns = [], 0.0
|
||||
for seg in segments:
|
||||
nsp = float(getattr(seg, "no_speech_prob", 0.0) or 0.0)
|
||||
alp = float(getattr(seg, "avg_logprob", 0.0) or 0.0)
|
||||
# Drop a segment that is almost certainly non-speech noise.
|
||||
if nsp > 0.8 and alp < -0.4:
|
||||
continue
|
||||
worst_ns = max(worst_ns, nsp)
|
||||
parts.append(seg.text)
|
||||
text = _guard_hallucination("".join(parts).strip(), worst_ns)
|
||||
ms = int((time.monotonic() - s) * 1000)
|
||||
_emit({"ok": True, "text": text, "language": info.language, "ms": ms})
|
||||
except Exception as exc: # keep the worker alive across bad requests
|
||||
|
||||
@@ -31,7 +31,40 @@ class BotControl:
|
||||
self._cmd_id = 0
|
||||
# Per-guild listen filter. Empty whitelist => listen to everyone;
|
||||
# blacklist always excludes. Users and roles both supported.
|
||||
self._lists: dict[str, dict[str, Any]] = {}
|
||||
# Restored from the persistent state store so it survives a restart.
|
||||
from . import state_store
|
||||
st = state_store.load()
|
||||
self._lists: dict[str, dict[str, Any]] = st.get("lists") or {}
|
||||
# Bot behaviour settings the dashboard toggles and the bot reads on each
|
||||
# report. bargeIn: stop the bot's current TTS the moment a user speaks.
|
||||
# bargeInMs: how long (ms) a user must keep speaking before that barge-in
|
||||
# fires — the "유저 음성 인식 시간" the dashboard exposes (default 700).
|
||||
self._settings: dict[str, Any] = {
|
||||
"bargeIn": True,
|
||||
"bargeInMs": 700,
|
||||
**(st.get("botSettings") or {}),
|
||||
}
|
||||
|
||||
# -- bot behaviour settings ------------------------------------------ #
|
||||
def get_settings(self) -> dict[str, Any]:
|
||||
with self._lock:
|
||||
return dict(self._settings)
|
||||
|
||||
def set_settings(self, data: dict[str, Any]) -> dict[str, Any]:
|
||||
with self._lock:
|
||||
if "bargeIn" in data:
|
||||
self._settings["bargeIn"] = bool(data["bargeIn"])
|
||||
if "bargeInMs" in data:
|
||||
try:
|
||||
ms = int(data["bargeInMs"])
|
||||
except (TypeError, ValueError):
|
||||
ms = 700
|
||||
# Keep it sane: 0ms = instant, cap at 5s so a typo can't wedge it.
|
||||
self._settings["bargeInMs"] = max(0, min(5000, ms))
|
||||
out = dict(self._settings)
|
||||
from . import state_store
|
||||
state_store.patch("botSettings", out)
|
||||
return out
|
||||
|
||||
# -- whitelist / blacklist (per guild) ------------------------------- #
|
||||
@staticmethod
|
||||
@@ -58,6 +91,9 @@ class BotControl:
|
||||
]
|
||||
with self._lock:
|
||||
self._lists[guild_id] = clean
|
||||
snapshot = {g: dict(v) for g, v in self._lists.items()}
|
||||
from . import state_store
|
||||
state_store.patch("lists", snapshot) # persist per-guild lists across restarts
|
||||
return clean
|
||||
|
||||
# -- bot -> dashboard (state push) ----------------------------------- #
|
||||
|
||||
@@ -24,7 +24,7 @@ class Settings:
|
||||
text: str | None = None # None | (discord)
|
||||
|
||||
capture_interval: float = 1.5
|
||||
anthropic_model: str = "claude-sonnet-4-5"
|
||||
anthropic_model: str = "claude-sonnet-5"
|
||||
|
||||
@classmethod
|
||||
def from_env(cls) -> "Settings":
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -129,7 +129,7 @@ class Turn:
|
||||
# Guarded so the repeated _touch()/finish() calls can't double-count.
|
||||
if self.status == "error" and not self._error_logged:
|
||||
self._error_logged = True
|
||||
self._monitor.log("error", f"대화 #{self.id} 실패: {self.error}")
|
||||
self._monitor.log("error", f"대화 #{self.id} 실패: {self.error}", cat="TURN")
|
||||
self._touch()
|
||||
|
||||
# -- internal --------------------------------------------------------- #
|
||||
@@ -209,12 +209,17 @@ class Monitor:
|
||||
self._status["claude_output_tokens"] += int(output_tokens or 0)
|
||||
self._broadcast({"type": "status", "status": self.status_snapshot()})
|
||||
|
||||
def log(self, level: str, message: str) -> None:
|
||||
"""A free-form lifecycle/error line (startup, disconnect, crash…)."""
|
||||
def log(self, level: str, message: str, cat: str | None = None) -> None:
|
||||
"""A free-form lifecycle/error line (startup, disconnect, crash…).
|
||||
|
||||
``cat`` is a short category tag (READY, CONNECT, MODEL, TTS, VOICE,
|
||||
FILTER, SETTING, TURN, BRAIN, PIPELINE, …) shown as a chip and filterable
|
||||
on the dashboard. Defaults to the upper-cased level when omitted."""
|
||||
with self._lock:
|
||||
self._event_id += 1
|
||||
evt = {"type": "log", "id": self._event_id, "level": level,
|
||||
"message": message, "wall": _now_wall()}
|
||||
"cat": (cat or level.upper()), "message": message,
|
||||
"wall": _now_wall()}
|
||||
self._events.append(evt)
|
||||
if level == "error":
|
||||
self._status["errors_total"] += 1
|
||||
@@ -253,6 +258,26 @@ class Monitor:
|
||||
self._events.clear()
|
||||
self._broadcast({"type": "logs_cleared"})
|
||||
|
||||
def delete_turn(self, turn_id: int) -> bool:
|
||||
"""Remove one conversation turn by id (dashboard per-turn '✕')."""
|
||||
found = False
|
||||
with self._lock:
|
||||
for t in list(self._turns):
|
||||
if t.id == turn_id:
|
||||
self._turns.remove(t)
|
||||
found = True
|
||||
break
|
||||
if found:
|
||||
self._broadcast({"type": "turn_deleted", "id": turn_id})
|
||||
return found
|
||||
|
||||
def clear_turns(self) -> None:
|
||||
"""Wipe all conversation turns (dashboard '대화 전체 삭제'). Broadcasts a
|
||||
reset so every connected page clears its turn list too."""
|
||||
with self._lock:
|
||||
self._turns.clear()
|
||||
self._broadcast({"type": "turns_cleared"})
|
||||
|
||||
# -- turns ------------------------------------------------------------ #
|
||||
def turn(self, source: str = "voice") -> Turn:
|
||||
with self._lock:
|
||||
|
||||
@@ -65,7 +65,7 @@ class Pipeline:
|
||||
except Exception as exc: # a single bad frame must not kill the loop
|
||||
log.exception("vision.describe failed")
|
||||
if self.monitor is not None:
|
||||
self.monitor.log("error", f"화면 이해 실패: {exc}")
|
||||
self.monitor.log("error", f"화면 이해 실패: {exc}", cat="VISION")
|
||||
continue
|
||||
await self.context.update(obs)
|
||||
log.debug("screen: %s", obs.text[:120])
|
||||
@@ -87,7 +87,7 @@ class Pipeline:
|
||||
try:
|
||||
async with turn.step("화면 맥락"):
|
||||
screen = await self.context.latest()
|
||||
async with turn.step("두뇌(생각)"):
|
||||
async with turn.step("LLM(생각)"):
|
||||
reply = await self.brain.respond(utt.text, screen, self._history)
|
||||
turn.replied(reply.text)
|
||||
self._remember(utt.text, reply.text)
|
||||
@@ -122,7 +122,7 @@ class Pipeline:
|
||||
return
|
||||
if self.monitor is not None:
|
||||
self.monitor.set_status(listening=True)
|
||||
self.monitor.log("info", "음성 수신 시작 — 발화 대기 중")
|
||||
self.monitor.log("info", "음성 수신 시작 — 발화 대기 중", cat="PIPELINE")
|
||||
try:
|
||||
async for utt in self.stt.utterances():
|
||||
await self._handle(utt)
|
||||
@@ -157,7 +157,7 @@ class Pipeline:
|
||||
except Exception as exc: # a warm failure must not abort startup
|
||||
log.warning("prewarm %s failed: %s", name, exc)
|
||||
if self.monitor is not None:
|
||||
self.monitor.log("error", f"{name} 예열 실패: {exc}")
|
||||
self.monitor.log("error", f"{name} 예열 실패: {exc}", cat="READY")
|
||||
|
||||
async def run(self) -> None:
|
||||
# A TaskGroup (not bare gather) so that if ONE loop raises, the others
|
||||
@@ -167,7 +167,7 @@ class Pipeline:
|
||||
# still-live loop (close-during-use).
|
||||
if self.monitor is not None:
|
||||
self.monitor.set_status(running=True)
|
||||
self.monitor.log("info", "파이프라인 시작")
|
||||
self.monitor.log("info", "파이프라인 시작", cat="PIPELINE")
|
||||
await self._prewarm()
|
||||
try:
|
||||
async with asyncio.TaskGroup() as tg:
|
||||
@@ -177,12 +177,12 @@ class Pipeline:
|
||||
except* Exception as eg:
|
||||
if self.monitor is not None:
|
||||
for exc in eg.exceptions:
|
||||
self.monitor.log("error", f"루프 예외: {type(exc).__name__}: {exc}")
|
||||
self.monitor.log("error", f"루프 예외: {type(exc).__name__}: {exc}", cat="PIPELINE")
|
||||
raise
|
||||
finally:
|
||||
if self.monitor is not None:
|
||||
self.monitor.set_status(running=False, listening=False)
|
||||
self.monitor.log("info", "파이프라인 종료")
|
||||
self.monitor.log("info", "파이프라인 종료", cat="PIPELINE")
|
||||
await self.aclose()
|
||||
|
||||
async def aclose(self) -> None:
|
||||
|
||||
61
wsai/state_store.py
Normal file
61
wsai/state_store.py
Normal file
@@ -0,0 +1,61 @@
|
||||
"""Tiny JSON state store so dashboard/bot settings survive a restart.
|
||||
|
||||
Persists things the user configures on the dashboard — per-guild listen lists,
|
||||
bot behaviour toggles, TTS controls, and the chosen STT/LLM models — to a single
|
||||
JSON file so a service or container restart keeps them.
|
||||
|
||||
Path: ``WSAI_STATE_FILE`` env, else ``~/.config/wsai/state.json``. For a
|
||||
container deployment, mount that path (or point the env at a mounted volume) to
|
||||
keep the file across ``docker restart``.
|
||||
|
||||
Pure stdlib, thread-safe, best-effort: a read/write failure never raises into
|
||||
the caller (the dashboard must keep working even if the disk is unwritable).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import threading
|
||||
from typing import Any
|
||||
|
||||
log = logging.getLogger("wsai.state")
|
||||
|
||||
_LOCK = threading.Lock()
|
||||
|
||||
|
||||
def path() -> str:
|
||||
return os.environ.get("WSAI_STATE_FILE") or os.path.expanduser("~/.config/wsai/state.json")
|
||||
|
||||
|
||||
def load() -> dict[str, Any]:
|
||||
try:
|
||||
with open(path(), encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
return data if isinstance(data, dict) else {}
|
||||
except FileNotFoundError:
|
||||
return {}
|
||||
except Exception as exc: # noqa: BLE001 — corrupt/unreadable state must not crash startup
|
||||
log.warning("state load failed (%s): %s", path(), exc)
|
||||
return {}
|
||||
|
||||
|
||||
def save(state: dict[str, Any]) -> None:
|
||||
p = path()
|
||||
try:
|
||||
with _LOCK:
|
||||
os.makedirs(os.path.dirname(p), exist_ok=True)
|
||||
tmp = p + ".tmp"
|
||||
with open(tmp, "w", encoding="utf-8") as f:
|
||||
json.dump(state, f, ensure_ascii=False, indent=2)
|
||||
os.replace(tmp, p) # atomic
|
||||
except Exception as exc: # noqa: BLE001
|
||||
log.warning("state save failed (%s): %s", p, exc)
|
||||
|
||||
|
||||
def patch(key: str, value: Any) -> None:
|
||||
"""Read-modify-write one top-level key."""
|
||||
s = load()
|
||||
s[key] = value
|
||||
save(s)
|
||||
Reference in New Issue
Block a user