From 8539acd0f882e7295d06f8289946e3a647bf2727 Mon Sep 17 00:00:00 2001 From: EJClaw Date: Thu, 27 Aug 2026 02:13:50 +0900 Subject: [PATCH] feat(stt/voice): better recognition, sustained barge-in, fragment discard MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit STT (whisper_worker): add VAD tuning (speech_pad_ms so soft first/last words aren't clipped), condition_on_previous_text=False + temperature fallback + no_speech/logprob/compression thresholds to reject noisy/quiet decodes, drop per-segment non-speech, and a hallucination guard that blanks Whisper's classic Korean silence/noise boilerplate ("감사합니다" 등) when no_speech_prob is high. Barge-in (bot.mjs): stop the bot's TTS only when the speaking (green ring) stays on for >= WSAI_BARGE_IN_MS (default 700ms), not on the first blip — cancelled if speaking stops in time. AfterSilence 800->1000ms so trailing soft words finish. Fragment gate (dashboard): a lone connective filler ("그러면") carries no answerable intent -> discard as [대기] instead of replying; hidden by the noise filter like [잡음]. Co-Authored-By: Claude Opus 4.7 --- dave/bot.mjs | 26 +++++++++++++---- wsai/backends/whisper_worker.py | 49 ++++++++++++++++++++++++++++++++- wsai/dashboard.py | 23 +++++++++++++++- 3 files changed, 90 insertions(+), 8 deletions(-) diff --git a/dave/bot.mjs b/dave/bot.mjs index 6d7f994..42493ce 100644 --- a/dave/bot.mjs +++ b/dave/bot.mjs @@ -178,6 +178,11 @@ const speakingSet = new Set(); // userIds currently speaking (for participant const activeSubs = new Set(); // userIds with an in-flight receive subscription let listsByGuild = {}; // guildId -> {whitelistUsers, blacklistUsers, whitelistRoles, blacklistRoles} let botSettings = { bargeIn: true }; // behaviour toggles from the dashboard +// Barge-in only fires when the Discord "speaking" (green ring) stays on for at +// least this long — a brief blip (keyboard click, cough) shouldn't cut the bot +// off, but a genuine ~0.7s of speech should. Tunable via WSAI_BARGE_IN_MS. +const BARGE_IN_MS = Number(process.env.WSAI_BARGE_IN_MS || 700); +const bargeTimers = new Map(); // userId -> pending stop timer // Attach the bot's audio player (so it can speak) to a fresh connection. function setupPlayer(connection) { @@ -202,15 +207,20 @@ function setupReceiver(connection) { return; } } - // Barge-in: the moment an allowed user speaks, stop the bot's current TTS - // so it doesn't talk over them (toggleable from the dashboard, default on). - if (botSettings.bargeIn !== false && voicePlayer) { - try { voicePlayer.stop(true); } catch {} + // Barge-in: stop the bot's current TTS only if this user keeps speaking for + // BARGE_IN_MS (the green ring stays on) — ignores momentary noise blips. The + // 'end' handler cancels the pending stop if speaking stops in time. + if (botSettings.bargeIn !== false && voicePlayer && !bargeTimers.has(userId)) { + bargeTimers.set(userId, setTimeout(() => { + bargeTimers.delete(userId); + try { voicePlayer.stop(true); } catch {} + }, BARGE_IN_MS)); } activeSubs.add(userId); if (!perUser.has(userId)) perUser.set(userId, { opusPackets: 0, pcmFrames: 0 }); const opusStream = receiver.subscribe(userId, { - end: { behavior: EndBehaviorType.AfterSilence, duration: 800 }, + // Wait a touch longer after silence so a soft/trailing word isn't clipped. + end: { behavior: EndBehaviorType.AfterSilence, duration: 1000 }, }); const decoder = new prism.opus.Decoder({ rate: 48000, channels: 2, frameSize: 960 }); const chunks = []; @@ -226,7 +236,11 @@ function setupReceiver(connection) { handleUtterance(userId, pcm).catch((e) => log(`voice turn error: ${e.message}`)); }); }); - receiver.speaking.on('end', (userId) => speakingSet.delete(userId)); + receiver.speaking.on('end', (userId) => { + speakingSet.delete(userId); + const t = bargeTimers.get(userId); // spoke too briefly -> cancel barge-in + if (t) { clearTimeout(t); bargeTimers.delete(userId); } + }); } // Join (or switch to) a voice channel on command from the dashboard. diff --git a/wsai/backends/whisper_worker.py b/wsai/backends/whisper_worker.py index 16ea832..bf6958e 100644 --- a/wsai/backends/whisper_worker.py +++ b/wsai/backends/whisper_worker.py @@ -23,6 +23,31 @@ import os import sys import time +# Whisper's classic Korean hallucinations on silence/noise/keyboard clatter — +# it "hears" video-outro boilerplate. Drop these when the whole utterance is one +# of them AND the segment looked like non-speech, so a real "감사합니다" survives. +_HALLUCINATIONS = { + "감사합니다", "고맙습니다", "감사합니다.", "고맙습니다.", + "시청해주셔서 감사합니다", "시청해 주셔서 감사합니다", "끝까지 시청해주셔서 감사합니다", + "구독과 좋아요 부탁드립니다", "구독 좋아요 부탁드립니다", "다음 영상에서 만나요", + "다음 시간에 만나요", "안녕히 계세요", +} + + +def _norm(t: str) -> str: + return t.strip().rstrip(" .!?…~").strip() + + +def _guard_hallucination(text: str, worst_no_speech: float) -> str: + """Blank out a lone known-hallucination phrase when the audio was probably + not speech (high no_speech_prob).""" + n = _norm(text) + if not n: + return "" + if n in {_norm(h) for h in _HALLUCINATIONS} and worst_no_speech > 0.5: + return "" + return text + # Split protocol from library noise BEFORE importing anything heavy. _proto = os.fdopen(os.dup(1), "w", buffering=1) # private copy of real stdout os.dup2(2, 1) # fd1 -> stderr, so stray library prints don't hit the protocol @@ -108,9 +133,31 @@ def main() -> None: wav, language=language, beam_size=int(req.get("beam_size", 5)), + # VAD strips non-speech (keyboard clatter, room noise, silence) + # before decoding, which both improves accuracy and kills most + # hallucinations. speech_pad_ms keeps a little lead/trail so soft + # first/last words aren't clipped. vad_filter=bool(req.get("vad_filter", True)), + vad_parameters=dict(min_silence_duration_ms=300, speech_pad_ms=250), + # Don't feed the previous text back in — that's what makes Whisper + # loop/hallucinate. Temperature fallback + thresholds reject + # low-confidence (noisy/quiet) decodes instead of inventing words. + condition_on_previous_text=False, + temperature=[0.0, 0.2, 0.4, 0.6], + no_speech_threshold=0.6, + log_prob_threshold=-1.0, + compression_ratio_threshold=2.4, ) - text = "".join(seg.text for seg in segments).strip() + parts, worst_ns = [], 0.0 + for seg in segments: + nsp = float(getattr(seg, "no_speech_prob", 0.0) or 0.0) + alp = float(getattr(seg, "avg_logprob", 0.0) or 0.0) + # Drop a segment that is almost certainly non-speech noise. + if nsp > 0.8 and alp < -0.4: + continue + worst_ns = max(worst_ns, nsp) + parts.append(seg.text) + text = _guard_hallucination("".join(parts).strip(), worst_ns) ms = int((time.monotonic() - s) * 1000) _emit({"ok": True, "text": text, "language": info.language, "ms": ms}) except Exception as exc: # keep the worker alive across bad requests diff --git a/wsai/dashboard.py b/wsai/dashboard.py index f14203a..0994598 100644 --- a/wsai/dashboard.py +++ b/wsai/dashboard.py @@ -49,6 +49,20 @@ def _thought_summary(reply: str) -> str: return "감정 태그 없음 · 기본 톤으로 답변" +# Lone connective/fillers that clearly expect more to come ("그러면…"). On their +# own they carry no answerable intent, so we discard them instead of blurting a +# confused reply. (A fuller version would buffer and wait for the continuation.) +_FRAGMENT_FILLERS = { + "그러면", "그래서", "그런데", "근데", "그리고", "그러니까", "그럼", "그게", + "저기", "있잖아", "그", "음", "어", "저", "그러", "그니까", +} + + +def _is_fragment(text: str) -> bool: + t = (text or "").strip().rstrip(" .,!?…~").strip() + return t in _FRAGMENT_FILLERS + + def _make_handler(dash: "Dashboard"): monitor = dash.monitor @@ -722,6 +736,13 @@ class Dashboard: turn.replied("[잡음]") turn.finish() return {"heard": heard, "reply": "[잡음]", "wav": b""} + if _is_fragment(heard): + # A lone connective ("그러면") — no answerable intent yet. Discard + # rather than reply to a fragment. + turn.thought("불완전한 말(연결어) — 대답하지 않고 폐기") + turn.replied("[대기]") + turn.finish() + return {"heard": heard, "reply": "[대기]", "wav": b""} t_llm = time.monotonic() reply_text = self._think(heard) self._step(turn, "LLM" if self.brain else "echo", (time.monotonic() - t_llm) * 1000) @@ -1305,7 +1326,7 @@ function turnFilter(){ text:$('tfText').value.trim().toLowerCase() }; } function isNoiseTurn(t){ - return (t.heard==='(빈 결과)') || (t.reply==='[잡음]'); + return (t.heard==='(빈 결과)') || (t.reply==='[잡음]') || (t.reply==='[대기]'); } function turnMatches(t, f){ if(hideNoise && isNoiseTurn(t)) return false; // 잡음 로그 표시 안 함 토글