feat(dashboard): per-stage STT/LLM/TTS timing + rename 두뇌 -> LLM
- voice_turn now records three separate timed steps (STT, LLM, TTS) instead of one combined "STT+두뇌+TTS" step, so each stage's latency shows in the turn. - Rename 두뇌 -> LLM everywhere user-facing: component chip, sub header, demo banner, model label, brain-failure log, startup print, pipeline step name. Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
@@ -142,7 +142,7 @@ def _run_voice_server(host: str, port: int) -> None:
|
|||||||
monitor.log("info", f"음성 서버 준비 완료 (STT device={sdev}). 디스코드 봇 연결 대기.", cat="READY")
|
monitor.log("info", f"음성 서버 준비 완료 (STT device={sdev}). 디스코드 봇 연결 대기.", cat="READY")
|
||||||
|
|
||||||
shown = host if host not in ("0.0.0.0", "") else _lan_ip()
|
shown = host if host not in ("0.0.0.0", "") else _lan_ip()
|
||||||
print(f"\n 음성 서버 준비 완료 (STT device: {sdev}, 두뇌: {brain_name})")
|
print(f"\n 음성 서버 준비 완료 (STT device: {sdev}, LLM: {brain_name})")
|
||||||
print(f" 대시보드/상태: http://{shown}:{port}")
|
print(f" 대시보드/상태: http://{shown}:{port}")
|
||||||
print(f" 봇 연결 엔드포인트: http://127.0.0.1:{port}/api/voice-turn\n")
|
print(f" 봇 연결 엔드포인트: http://127.0.0.1:{port}/api/voice-turn\n")
|
||||||
try:
|
try:
|
||||||
|
|||||||
@@ -650,6 +650,13 @@ class Dashboard:
|
|||||||
self.monitor.log("info", f"LLM 모델 변경: {model} (다음 답변부터 적용)", cat="MODEL")
|
self.monitor.log("info", f"LLM 모델 변경: {model} (다음 답변부터 적용)", cat="MODEL")
|
||||||
return {**self.models_settings()["llm"], "changed": changed}
|
return {**self.models_settings()["llm"], "changed": changed}
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _step(turn, name: str, ms: float) -> None:
|
||||||
|
"""Record one finished timing step (STT / LLM / TTS) on a turn."""
|
||||||
|
st = turn.step(name)
|
||||||
|
st.ok, st.ms = True, float(ms)
|
||||||
|
turn._steps.append(st)
|
||||||
|
|
||||||
def voice_turn(self, audio_bytes: bytes, speaker: str = "",
|
def voice_turn(self, audio_bytes: bytes, speaker: str = "",
|
||||||
guild: str = "", channel: str = "") -> dict:
|
guild: str = "", channel: str = "") -> dict:
|
||||||
"""One Discord voice turn: decode the uploaded utterance, recognise it
|
"""One Discord voice turn: decode the uploaded utterance, recognise it
|
||||||
@@ -680,6 +687,7 @@ class Dashboard:
|
|||||||
check=True, capture_output=True,
|
check=True, capture_output=True,
|
||||||
)
|
)
|
||||||
heard = (self._submit(self.stt.transcribe(wav)) or "").strip()
|
heard = (self._submit(self.stt.transcribe(wav)) or "").strip()
|
||||||
|
self._step(turn, "STT", (time.monotonic() - t0) * 1000) # 인식(+디코드)
|
||||||
turn.heard(heard or "(빈 결과)")
|
turn.heard(heard or "(빈 결과)")
|
||||||
if not heard:
|
if not heard:
|
||||||
# Nothing recognised (silence/noise): mark it as [잡음] and skip
|
# Nothing recognised (silence/noise): mark it as [잡음] and skip
|
||||||
@@ -688,16 +696,16 @@ class Dashboard:
|
|||||||
turn.replied("[잡음]")
|
turn.replied("[잡음]")
|
||||||
turn.finish()
|
turn.finish()
|
||||||
return {"heard": heard, "reply": "[잡음]", "wav": b""}
|
return {"heard": heard, "reply": "[잡음]", "wav": b""}
|
||||||
|
t_llm = time.monotonic()
|
||||||
reply_text = self._think(heard)
|
reply_text = self._think(heard)
|
||||||
|
self._step(turn, "LLM" if self.brain else "echo", (time.monotonic() - t_llm) * 1000)
|
||||||
turn.thought(_thought_summary(reply_text))
|
turn.thought(_thought_summary(reply_text))
|
||||||
turn.replied(reply_text)
|
turn.replied(reply_text)
|
||||||
|
t_tts = time.monotonic()
|
||||||
out_path = self._submit(self.tts.synth(_speech_text(reply_text)))
|
out_path = self._submit(self.tts.synth(_speech_text(reply_text)))
|
||||||
with open(out_path, "rb") as f:
|
with open(out_path, "rb") as f:
|
||||||
reply_wav = f.read()
|
reply_wav = f.read()
|
||||||
ms = int((time.monotonic() - t0) * 1000)
|
self._step(turn, "TTS", (time.monotonic() - t_tts) * 1000) # 합성
|
||||||
step = turn.step("STT+두뇌+TTS" if self.brain else "STT+TTS(GPU)")
|
|
||||||
step.ok, step.ms = True, float(ms)
|
|
||||||
turn._steps.append(step)
|
|
||||||
turn.finish()
|
turn.finish()
|
||||||
try:
|
try:
|
||||||
os.remove(out_path)
|
os.remove(out_path)
|
||||||
@@ -733,7 +741,7 @@ class Dashboard:
|
|||||||
self.monitor.add_claude_usage(u.get("input", 0), u.get("output", 0))
|
self.monitor.add_claude_usage(u.get("input", 0), u.get("output", 0))
|
||||||
except Exception as exc: # noqa: BLE001
|
except Exception as exc: # noqa: BLE001
|
||||||
log.exception("brain failed")
|
log.exception("brain failed")
|
||||||
self.monitor.log("error", f"두뇌 응답 실패: {exc}", cat="BRAIN")
|
self.monitor.log("error", f"LLM 응답 실패: {exc}", cat="BRAIN")
|
||||||
blob = f"{getattr(exc, 'status_code', '')} {exc}".lower()
|
blob = f"{getattr(exc, 'status_code', '')} {exc}".lower()
|
||||||
if "529" in blob or "overload" in blob:
|
if "529" in blob or "overload" in blob:
|
||||||
# Transient server overload survived the SDK retries.
|
# Transient server overload survived the SDK retries.
|
||||||
@@ -1008,7 +1016,7 @@ PAGE = r"""<!DOCTYPE html>
|
|||||||
<header>
|
<header>
|
||||||
<div>
|
<div>
|
||||||
<h1>watch_sceen_ai · 실시간 상태</h1>
|
<h1>watch_sceen_ai · 실시간 상태</h1>
|
||||||
<div class="sub">STT → 두뇌 → TTS 음성 루프를 단계별로 관찰</div>
|
<div class="sub">STT → LLM → TTS 음성 루프를 단계별로 관찰</div>
|
||||||
</div>
|
</div>
|
||||||
<div class="pill"><span id="dot" class="dot off"></span><span id="listen">연결 대기</span></div>
|
<div class="pill"><span id="dot" class="dot off"></span><span id="listen">연결 대기</span></div>
|
||||||
<button id="promptBtn" class="btn hbtn">📝 프롬프트</button>
|
<button id="promptBtn" class="btn hbtn">📝 프롬프트</button>
|
||||||
@@ -1081,7 +1089,7 @@ PAGE = r"""<!DOCTYPE html>
|
|||||||
<span id="mSTTstat" class="sttstat"></span>
|
<span id="mSTTstat" class="sttstat"></span>
|
||||||
</div>
|
</div>
|
||||||
<div class="sttrow" style="margin-top:8px">
|
<div class="sttrow" style="margin-top:8px">
|
||||||
<label style="font-size:12.5px;color:var(--muted)">LLM(두뇌)
|
<label style="font-size:12.5px;color:var(--muted)">LLM
|
||||||
<select id="mLLM" class="ttssel"></select></label>
|
<select id="mLLM" class="ttssel"></select></label>
|
||||||
<button id="mLLMapply" class="btn">적용</button>
|
<button id="mLLMapply" class="btn">적용</button>
|
||||||
<span id="mLLMstat" class="sttstat"></span>
|
<span id="mLLMstat" class="sttstat"></span>
|
||||||
@@ -1194,14 +1202,14 @@ function renderStatus(s){
|
|||||||
const bar = $('demobar');
|
const bar = $('demobar');
|
||||||
if(mockParts.length){
|
if(mockParts.length){
|
||||||
bar.style.display='block';
|
bar.style.display='block';
|
||||||
bar.innerHTML = '⚠ <b>데모 모드</b> — 실제 음성/STT/두뇌/TTS가 아직 연결되지 않아, 아래 대화는 '
|
bar.innerHTML = '⚠ <b>데모 모드</b> — 실제 음성/STT/LLM/TTS가 아직 연결되지 않아, 아래 대화는 '
|
||||||
+ '실제로 들은 내용이 아니라 <b>목(mock) 예시 스크립트</b>입니다. '
|
+ '실제로 들은 내용이 아니라 <b>목(mock) 예시 스크립트</b>입니다. '
|
||||||
+ '실제 엔진(faster-whisper·Claude·MeloTTS)을 붙이면 이 자리에 진짜 발화·지연·오류가 표시됩니다.';
|
+ '실제 엔진(faster-whisper·Claude·MeloTTS)을 붙이면 이 자리에 진짜 발화·지연·오류가 표시됩니다.';
|
||||||
} else {
|
} else {
|
||||||
bar.style.display='none';
|
bar.style.display='none';
|
||||||
}
|
}
|
||||||
const el = $('comp'); el.innerHTML = '';
|
const el = $('comp'); el.innerHTML = '';
|
||||||
const names = {source:'눈(소스)', vision:'시각', stt:'귀(STT)', brain:'두뇌', tts:'입(TTS)', text:'텍스트'};
|
const names = {source:'눈(소스)', vision:'시각', stt:'귀(STT)', brain:'LLM', tts:'입(TTS)', text:'텍스트'};
|
||||||
for(const k of Object.keys(names)){
|
for(const k of Object.keys(names)){
|
||||||
if(!(k in comps)) continue;
|
if(!(k in comps)) continue;
|
||||||
const v = comps[k];
|
const v = comps[k];
|
||||||
|
|||||||
@@ -87,7 +87,7 @@ class Pipeline:
|
|||||||
try:
|
try:
|
||||||
async with turn.step("화면 맥락"):
|
async with turn.step("화면 맥락"):
|
||||||
screen = await self.context.latest()
|
screen = await self.context.latest()
|
||||||
async with turn.step("두뇌(생각)"):
|
async with turn.step("LLM(생각)"):
|
||||||
reply = await self.brain.respond(utt.text, screen, self._history)
|
reply = await self.brain.respond(utt.text, screen, self._history)
|
||||||
turn.replied(reply.text)
|
turn.replied(reply.text)
|
||||||
self._remember(utt.text, reply.text)
|
self._remember(utt.text, reply.text)
|
||||||
|
|||||||
Reference in New Issue
Block a user