From 0588ee21ac6ca597667408e410dbc84d44289ad0 Mon Sep 17 00:00:00 2001 From: EJClaw Date: Thu, 27 Aug 2026 00:03:34 +0900 Subject: [PATCH] =?UTF-8?q?feat(dashboard):=20per-stage=20STT/LLM/TTS=20ti?= =?UTF-8?q?ming=20+=20rename=20=EB=91=90=EB=87=8C=20->=20LLM?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - voice_turn now records three separate timed steps (STT, LLM, TTS) instead of one combined "STT+두뇌+TTS" step, so each stage's latency shows in the turn. - Rename 두뇌 -> LLM everywhere user-facing: component chip, sub header, demo banner, model label, brain-failure log, startup print, pipeline step name. Co-Authored-By: Claude Opus 4.7 --- wsai/__main__.py | 2 +- wsai/dashboard.py | 26 +++++++++++++++++--------- wsai/pipeline.py | 2 +- 3 files changed, 19 insertions(+), 11 deletions(-) diff --git a/wsai/__main__.py b/wsai/__main__.py index 8258d3e..6f402e2 100644 --- a/wsai/__main__.py +++ b/wsai/__main__.py @@ -142,7 +142,7 @@ def _run_voice_server(host: str, port: int) -> None: monitor.log("info", f"음성 서버 준비 완료 (STT device={sdev}). 디스코드 봇 연결 대기.", cat="READY") shown = host if host not in ("0.0.0.0", "") else _lan_ip() - print(f"\n 음성 서버 준비 완료 (STT device: {sdev}, 두뇌: {brain_name})") + print(f"\n 음성 서버 준비 완료 (STT device: {sdev}, LLM: {brain_name})") print(f" 대시보드/상태: http://{shown}:{port}") print(f" 봇 연결 엔드포인트: http://127.0.0.1:{port}/api/voice-turn\n") try: diff --git a/wsai/dashboard.py b/wsai/dashboard.py index 3ff1cfc..a96d3ed 100644 --- a/wsai/dashboard.py +++ b/wsai/dashboard.py @@ -650,6 +650,13 @@ class Dashboard: self.monitor.log("info", f"LLM 모델 변경: {model} (다음 답변부터 적용)", cat="MODEL") return {**self.models_settings()["llm"], "changed": changed} + @staticmethod + def _step(turn, name: str, ms: float) -> None: + """Record one finished timing step (STT / LLM / TTS) on a turn.""" + st = turn.step(name) + st.ok, st.ms = True, float(ms) + turn._steps.append(st) + def voice_turn(self, audio_bytes: bytes, speaker: str = "", guild: str = "", channel: str = "") -> dict: """One Discord voice turn: decode the uploaded utterance, recognise it @@ -680,6 +687,7 @@ class Dashboard: check=True, capture_output=True, ) heard = (self._submit(self.stt.transcribe(wav)) or "").strip() + self._step(turn, "STT", (time.monotonic() - t0) * 1000) # 인식(+디코드) turn.heard(heard or "(빈 결과)") if not heard: # Nothing recognised (silence/noise): mark it as [잡음] and skip @@ -688,16 +696,16 @@ class Dashboard: turn.replied("[잡음]") turn.finish() return {"heard": heard, "reply": "[잡음]", "wav": b""} + t_llm = time.monotonic() reply_text = self._think(heard) + self._step(turn, "LLM" if self.brain else "echo", (time.monotonic() - t_llm) * 1000) turn.thought(_thought_summary(reply_text)) turn.replied(reply_text) + t_tts = time.monotonic() out_path = self._submit(self.tts.synth(_speech_text(reply_text))) with open(out_path, "rb") as f: reply_wav = f.read() - ms = int((time.monotonic() - t0) * 1000) - step = turn.step("STT+두뇌+TTS" if self.brain else "STT+TTS(GPU)") - step.ok, step.ms = True, float(ms) - turn._steps.append(step) + self._step(turn, "TTS", (time.monotonic() - t_tts) * 1000) # 합성 turn.finish() try: os.remove(out_path) @@ -733,7 +741,7 @@ class Dashboard: self.monitor.add_claude_usage(u.get("input", 0), u.get("output", 0)) except Exception as exc: # noqa: BLE001 log.exception("brain failed") - self.monitor.log("error", f"두뇌 응답 실패: {exc}", cat="BRAIN") + self.monitor.log("error", f"LLM 응답 실패: {exc}", cat="BRAIN") blob = f"{getattr(exc, 'status_code', '')} {exc}".lower() if "529" in blob or "overload" in blob: # Transient server overload survived the SDK retries. @@ -1008,7 +1016,7 @@ PAGE = r"""

watch_sceen_ai · 실시간 상태

-
STT → 두뇌 → TTS 음성 루프를 단계별로 관찰
+
STT → LLM → TTS 음성 루프를 단계별로 관찰
연결 대기
@@ -1081,7 +1089,7 @@ PAGE = r"""
-