feat(stt/voice): better recognition, sustained barge-in, fragment discard
STT (whisper_worker): add VAD tuning (speech_pad_ms so soft first/last words
aren't clipped), condition_on_previous_text=False + temperature fallback +
no_speech/logprob/compression thresholds to reject noisy/quiet decodes, drop
per-segment non-speech, and a hallucination guard that blanks Whisper's classic
Korean silence/noise boilerplate ("감사합니다" 등) when no_speech_prob is high.
Barge-in (bot.mjs): stop the bot's TTS only when the speaking (green ring) stays
on for >= WSAI_BARGE_IN_MS (default 700ms), not on the first blip — cancelled if
speaking stops in time. AfterSilence 800->1000ms so trailing soft words finish.
Fragment gate (dashboard): a lone connective filler ("그러면") carries no
answerable intent -> discard as [대기] instead of replying; hidden by the noise
filter like [잡음].
Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
24
dave/bot.mjs
24
dave/bot.mjs
@@ -178,6 +178,11 @@ const speakingSet = new Set(); // userIds currently speaking (for participant
|
|||||||
const activeSubs = new Set(); // userIds with an in-flight receive subscription
|
const activeSubs = new Set(); // userIds with an in-flight receive subscription
|
||||||
let listsByGuild = {}; // guildId -> {whitelistUsers, blacklistUsers, whitelistRoles, blacklistRoles}
|
let listsByGuild = {}; // guildId -> {whitelistUsers, blacklistUsers, whitelistRoles, blacklistRoles}
|
||||||
let botSettings = { bargeIn: true }; // behaviour toggles from the dashboard
|
let botSettings = { bargeIn: true }; // behaviour toggles from the dashboard
|
||||||
|
// Barge-in only fires when the Discord "speaking" (green ring) stays on for at
|
||||||
|
// least this long — a brief blip (keyboard click, cough) shouldn't cut the bot
|
||||||
|
// off, but a genuine ~0.7s of speech should. Tunable via WSAI_BARGE_IN_MS.
|
||||||
|
const BARGE_IN_MS = Number(process.env.WSAI_BARGE_IN_MS || 700);
|
||||||
|
const bargeTimers = new Map(); // userId -> pending stop timer
|
||||||
|
|
||||||
// Attach the bot's audio player (so it can speak) to a fresh connection.
|
// Attach the bot's audio player (so it can speak) to a fresh connection.
|
||||||
function setupPlayer(connection) {
|
function setupPlayer(connection) {
|
||||||
@@ -202,15 +207,20 @@ function setupReceiver(connection) {
|
|||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// Barge-in: the moment an allowed user speaks, stop the bot's current TTS
|
// Barge-in: stop the bot's current TTS only if this user keeps speaking for
|
||||||
// so it doesn't talk over them (toggleable from the dashboard, default on).
|
// BARGE_IN_MS (the green ring stays on) — ignores momentary noise blips. The
|
||||||
if (botSettings.bargeIn !== false && voicePlayer) {
|
// 'end' handler cancels the pending stop if speaking stops in time.
|
||||||
|
if (botSettings.bargeIn !== false && voicePlayer && !bargeTimers.has(userId)) {
|
||||||
|
bargeTimers.set(userId, setTimeout(() => {
|
||||||
|
bargeTimers.delete(userId);
|
||||||
try { voicePlayer.stop(true); } catch {}
|
try { voicePlayer.stop(true); } catch {}
|
||||||
|
}, BARGE_IN_MS));
|
||||||
}
|
}
|
||||||
activeSubs.add(userId);
|
activeSubs.add(userId);
|
||||||
if (!perUser.has(userId)) perUser.set(userId, { opusPackets: 0, pcmFrames: 0 });
|
if (!perUser.has(userId)) perUser.set(userId, { opusPackets: 0, pcmFrames: 0 });
|
||||||
const opusStream = receiver.subscribe(userId, {
|
const opusStream = receiver.subscribe(userId, {
|
||||||
end: { behavior: EndBehaviorType.AfterSilence, duration: 800 },
|
// Wait a touch longer after silence so a soft/trailing word isn't clipped.
|
||||||
|
end: { behavior: EndBehaviorType.AfterSilence, duration: 1000 },
|
||||||
});
|
});
|
||||||
const decoder = new prism.opus.Decoder({ rate: 48000, channels: 2, frameSize: 960 });
|
const decoder = new prism.opus.Decoder({ rate: 48000, channels: 2, frameSize: 960 });
|
||||||
const chunks = [];
|
const chunks = [];
|
||||||
@@ -226,7 +236,11 @@ function setupReceiver(connection) {
|
|||||||
handleUtterance(userId, pcm).catch((e) => log(`voice turn error: ${e.message}`));
|
handleUtterance(userId, pcm).catch((e) => log(`voice turn error: ${e.message}`));
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
receiver.speaking.on('end', (userId) => speakingSet.delete(userId));
|
receiver.speaking.on('end', (userId) => {
|
||||||
|
speakingSet.delete(userId);
|
||||||
|
const t = bargeTimers.get(userId); // spoke too briefly -> cancel barge-in
|
||||||
|
if (t) { clearTimeout(t); bargeTimers.delete(userId); }
|
||||||
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
// Join (or switch to) a voice channel on command from the dashboard.
|
// Join (or switch to) a voice channel on command from the dashboard.
|
||||||
|
|||||||
@@ -23,6 +23,31 @@ import os
|
|||||||
import sys
|
import sys
|
||||||
import time
|
import time
|
||||||
|
|
||||||
|
# Whisper's classic Korean hallucinations on silence/noise/keyboard clatter —
|
||||||
|
# it "hears" video-outro boilerplate. Drop these when the whole utterance is one
|
||||||
|
# of them AND the segment looked like non-speech, so a real "감사합니다" survives.
|
||||||
|
_HALLUCINATIONS = {
|
||||||
|
"감사합니다", "고맙습니다", "감사합니다.", "고맙습니다.",
|
||||||
|
"시청해주셔서 감사합니다", "시청해 주셔서 감사합니다", "끝까지 시청해주셔서 감사합니다",
|
||||||
|
"구독과 좋아요 부탁드립니다", "구독 좋아요 부탁드립니다", "다음 영상에서 만나요",
|
||||||
|
"다음 시간에 만나요", "안녕히 계세요",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _norm(t: str) -> str:
|
||||||
|
return t.strip().rstrip(" .!?…~").strip()
|
||||||
|
|
||||||
|
|
||||||
|
def _guard_hallucination(text: str, worst_no_speech: float) -> str:
|
||||||
|
"""Blank out a lone known-hallucination phrase when the audio was probably
|
||||||
|
not speech (high no_speech_prob)."""
|
||||||
|
n = _norm(text)
|
||||||
|
if not n:
|
||||||
|
return ""
|
||||||
|
if n in {_norm(h) for h in _HALLUCINATIONS} and worst_no_speech > 0.5:
|
||||||
|
return ""
|
||||||
|
return text
|
||||||
|
|
||||||
# Split protocol from library noise BEFORE importing anything heavy.
|
# Split protocol from library noise BEFORE importing anything heavy.
|
||||||
_proto = os.fdopen(os.dup(1), "w", buffering=1) # private copy of real stdout
|
_proto = os.fdopen(os.dup(1), "w", buffering=1) # private copy of real stdout
|
||||||
os.dup2(2, 1) # fd1 -> stderr, so stray library prints don't hit the protocol
|
os.dup2(2, 1) # fd1 -> stderr, so stray library prints don't hit the protocol
|
||||||
@@ -108,9 +133,31 @@ def main() -> None:
|
|||||||
wav,
|
wav,
|
||||||
language=language,
|
language=language,
|
||||||
beam_size=int(req.get("beam_size", 5)),
|
beam_size=int(req.get("beam_size", 5)),
|
||||||
|
# VAD strips non-speech (keyboard clatter, room noise, silence)
|
||||||
|
# before decoding, which both improves accuracy and kills most
|
||||||
|
# hallucinations. speech_pad_ms keeps a little lead/trail so soft
|
||||||
|
# first/last words aren't clipped.
|
||||||
vad_filter=bool(req.get("vad_filter", True)),
|
vad_filter=bool(req.get("vad_filter", True)),
|
||||||
|
vad_parameters=dict(min_silence_duration_ms=300, speech_pad_ms=250),
|
||||||
|
# Don't feed the previous text back in — that's what makes Whisper
|
||||||
|
# loop/hallucinate. Temperature fallback + thresholds reject
|
||||||
|
# low-confidence (noisy/quiet) decodes instead of inventing words.
|
||||||
|
condition_on_previous_text=False,
|
||||||
|
temperature=[0.0, 0.2, 0.4, 0.6],
|
||||||
|
no_speech_threshold=0.6,
|
||||||
|
log_prob_threshold=-1.0,
|
||||||
|
compression_ratio_threshold=2.4,
|
||||||
)
|
)
|
||||||
text = "".join(seg.text for seg in segments).strip()
|
parts, worst_ns = [], 0.0
|
||||||
|
for seg in segments:
|
||||||
|
nsp = float(getattr(seg, "no_speech_prob", 0.0) or 0.0)
|
||||||
|
alp = float(getattr(seg, "avg_logprob", 0.0) or 0.0)
|
||||||
|
# Drop a segment that is almost certainly non-speech noise.
|
||||||
|
if nsp > 0.8 and alp < -0.4:
|
||||||
|
continue
|
||||||
|
worst_ns = max(worst_ns, nsp)
|
||||||
|
parts.append(seg.text)
|
||||||
|
text = _guard_hallucination("".join(parts).strip(), worst_ns)
|
||||||
ms = int((time.monotonic() - s) * 1000)
|
ms = int((time.monotonic() - s) * 1000)
|
||||||
_emit({"ok": True, "text": text, "language": info.language, "ms": ms})
|
_emit({"ok": True, "text": text, "language": info.language, "ms": ms})
|
||||||
except Exception as exc: # keep the worker alive across bad requests
|
except Exception as exc: # keep the worker alive across bad requests
|
||||||
|
|||||||
@@ -49,6 +49,20 @@ def _thought_summary(reply: str) -> str:
|
|||||||
return "감정 태그 없음 · 기본 톤으로 답변"
|
return "감정 태그 없음 · 기본 톤으로 답변"
|
||||||
|
|
||||||
|
|
||||||
|
# Lone connective/fillers that clearly expect more to come ("그러면…"). On their
|
||||||
|
# own they carry no answerable intent, so we discard them instead of blurting a
|
||||||
|
# confused reply. (A fuller version would buffer and wait for the continuation.)
|
||||||
|
_FRAGMENT_FILLERS = {
|
||||||
|
"그러면", "그래서", "그런데", "근데", "그리고", "그러니까", "그럼", "그게",
|
||||||
|
"저기", "있잖아", "그", "음", "어", "저", "그러", "그니까",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _is_fragment(text: str) -> bool:
|
||||||
|
t = (text or "").strip().rstrip(" .,!?…~").strip()
|
||||||
|
return t in _FRAGMENT_FILLERS
|
||||||
|
|
||||||
|
|
||||||
def _make_handler(dash: "Dashboard"):
|
def _make_handler(dash: "Dashboard"):
|
||||||
monitor = dash.monitor
|
monitor = dash.monitor
|
||||||
|
|
||||||
@@ -722,6 +736,13 @@ class Dashboard:
|
|||||||
turn.replied("[잡음]")
|
turn.replied("[잡음]")
|
||||||
turn.finish()
|
turn.finish()
|
||||||
return {"heard": heard, "reply": "[잡음]", "wav": b""}
|
return {"heard": heard, "reply": "[잡음]", "wav": b""}
|
||||||
|
if _is_fragment(heard):
|
||||||
|
# A lone connective ("그러면") — no answerable intent yet. Discard
|
||||||
|
# rather than reply to a fragment.
|
||||||
|
turn.thought("불완전한 말(연결어) — 대답하지 않고 폐기")
|
||||||
|
turn.replied("[대기]")
|
||||||
|
turn.finish()
|
||||||
|
return {"heard": heard, "reply": "[대기]", "wav": b""}
|
||||||
t_llm = time.monotonic()
|
t_llm = time.monotonic()
|
||||||
reply_text = self._think(heard)
|
reply_text = self._think(heard)
|
||||||
self._step(turn, "LLM" if self.brain else "echo", (time.monotonic() - t_llm) * 1000)
|
self._step(turn, "LLM" if self.brain else "echo", (time.monotonic() - t_llm) * 1000)
|
||||||
@@ -1305,7 +1326,7 @@ function turnFilter(){
|
|||||||
text:$('tfText').value.trim().toLowerCase() };
|
text:$('tfText').value.trim().toLowerCase() };
|
||||||
}
|
}
|
||||||
function isNoiseTurn(t){
|
function isNoiseTurn(t){
|
||||||
return (t.heard==='(빈 결과)') || (t.reply==='[잡음]');
|
return (t.heard==='(빈 결과)') || (t.reply==='[잡음]') || (t.reply==='[대기]');
|
||||||
}
|
}
|
||||||
function turnMatches(t, f){
|
function turnMatches(t, f){
|
||||||
if(hideNoise && isNoiseTurn(t)) return false; // 잡음 로그 표시 안 함 토글
|
if(hideNoise && isNoiseTurn(t)) return false; // 잡음 로그 표시 안 함 토글
|
||||||
|
|||||||
Reference in New Issue
Block a user