feat(voice): buffer a lone connective and merge with the next utterance

Instead of immediately discarding a fragment like "그러면", hold it briefly
(WSAI_FRAGMENT_MERGE_MS, default 5s). If the next utterance arrives in time,
merge it in front and answer the whole thing ("그러면" + "뭐 먹지"); if nothing
follows within the window, it's silently dropped. Also redeployed the wsai-bot
container so the sustained (700ms) barge-in bot.mjs is now live.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
EJClaw
2026-08-27 02:27:35 +09:00
parent 5966f6dedb
commit d5501d0b8a

View File

@@ -17,8 +17,10 @@ from __future__ import annotations
import json
import logging
import os
import queue
import threading
import time
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from .monitor import Monitor
@@ -491,6 +493,10 @@ class Dashboard:
self.bot = BotControl() # dashboard <-> Discord bot control plane
self._history: list[tuple[str, str]] = []
self._history_turns = history_turns
# A lone connective ("그러면") is held here to see if the next utterance
# continues it; merged if it arrives within the window, else dropped.
self._pending_fragment: dict | None = None
self._fragment_window = float(os.environ.get("WSAI_FRAGMENT_MERGE_MS", "5000")) / 1000.0
self._server: ThreadingHTTPServer | None = None
self._thread: threading.Thread | None = None
self._loop = None
@@ -755,13 +761,25 @@ class Dashboard:
turn.replied("[잡음]")
turn.finish()
return {"heard": heard, "reply": "[잡음]", "wav": b""}
now = time.monotonic()
pending = self._pending_fragment
if pending and (now - pending["ts"]) > self._fragment_window:
pending = self._pending_fragment = None # too old — give up on it
if _is_fragment(heard):
# A lone connective ("그러면") — no answerable intent yet. Discard
# rather than reply to a fragment.
turn.thought("불완전한 말(연결어) — 대답하지 않고 폐기")
# A lone connective ("그러면") — hold it and wait for the next
# utterance to continue it. If nothing follows within the window
# it's silently dropped (never answered).
self._pending_fragment = {"text": heard, "ts": now}
turn.thought("불완전한 말(연결어) — 이어질 말 대기 (없으면 폐기)")
turn.replied("[대기]")
turn.finish()
return {"heard": heard, "reply": "[대기]", "wav": b""}
if pending:
# A real utterance arrived in time — merge the held fragment in
# front and answer the whole thing ("그러면" + "뭐 먹지").
heard = (pending["text"].rstrip(" .,!?…~") + " " + heard).strip()
self._pending_fragment = None
turn.heard(heard)
t_llm = time.monotonic()
reply_text = self._think(heard)
self._step(turn, "LLM" if self.brain else "echo", (time.monotonic() - t_llm) * 1000)