feat(voice): buffer a lone connective and merge with the next utterance

Instead of immediately discarding a fragment like "그러면", hold it briefly
(WSAI_FRAGMENT_MERGE_MS, default 5s). If the next utterance arrives in time,
merge it in front and answer the whole thing ("그러면" + "뭐 먹지"); if nothing
follows within the window, it's silently dropped. Also redeployed the wsai-bot
container so the sustained (700ms) barge-in bot.mjs is now live.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
EJClaw
2026-08-27 02:27:35 +09:00
parent 5966f6dedb
commit d5501d0b8a

View File

@@ -17,8 +17,10 @@ from __future__ import annotations
import json import json
import logging import logging
import os
import queue import queue
import threading import threading
import time
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from .monitor import Monitor from .monitor import Monitor
@@ -491,6 +493,10 @@ class Dashboard:
self.bot = BotControl() # dashboard <-> Discord bot control plane self.bot = BotControl() # dashboard <-> Discord bot control plane
self._history: list[tuple[str, str]] = [] self._history: list[tuple[str, str]] = []
self._history_turns = history_turns self._history_turns = history_turns
# A lone connective ("그러면") is held here to see if the next utterance
# continues it; merged if it arrives within the window, else dropped.
self._pending_fragment: dict | None = None
self._fragment_window = float(os.environ.get("WSAI_FRAGMENT_MERGE_MS", "5000")) / 1000.0
self._server: ThreadingHTTPServer | None = None self._server: ThreadingHTTPServer | None = None
self._thread: threading.Thread | None = None self._thread: threading.Thread | None = None
self._loop = None self._loop = None
@@ -755,13 +761,25 @@ class Dashboard:
turn.replied("[잡음]") turn.replied("[잡음]")
turn.finish() turn.finish()
return {"heard": heard, "reply": "[잡음]", "wav": b""} return {"heard": heard, "reply": "[잡음]", "wav": b""}
now = time.monotonic()
pending = self._pending_fragment
if pending and (now - pending["ts"]) > self._fragment_window:
pending = self._pending_fragment = None # too old — give up on it
if _is_fragment(heard): if _is_fragment(heard):
# A lone connective ("그러면") — no answerable intent yet. Discard # A lone connective ("그러면") — hold it and wait for the next
# rather than reply to a fragment. # utterance to continue it. If nothing follows within the window
turn.thought("불완전한 말(연결어) — 대답하지 않고 폐기") # it's silently dropped (never answered).
self._pending_fragment = {"text": heard, "ts": now}
turn.thought("불완전한 말(연결어) — 이어질 말 대기 (없으면 폐기)")
turn.replied("[대기]") turn.replied("[대기]")
turn.finish() turn.finish()
return {"heard": heard, "reply": "[대기]", "wav": b""} return {"heard": heard, "reply": "[대기]", "wav": b""}
if pending:
# A real utterance arrived in time — merge the held fragment in
# front and answer the whole thing ("그러면" + "뭐 먹지").
heard = (pending["text"].rstrip(" .,!?…~") + " " + heard).strip()
self._pending_fragment = None
turn.heard(heard)
t_llm = time.monotonic() t_llm = time.monotonic()
reply_text = self._think(heard) reply_text = self._think(heard)
self._step(turn, "LLM" if self.brain else "echo", (time.monotonic() - t_llm) * 1000) self._step(turn, "LLM" if self.brain else "echo", (time.monotonic() - t_llm) * 1000)