feat(stt/voice): better recognition, sustained barge-in, fragment discard
STT (whisper_worker): add VAD tuning (speech_pad_ms so soft first/last words
aren't clipped), condition_on_previous_text=False + temperature fallback +
no_speech/logprob/compression thresholds to reject noisy/quiet decodes, drop
per-segment non-speech, and a hallucination guard that blanks Whisper's classic
Korean silence/noise boilerplate ("감사합니다" 등) when no_speech_prob is high.
Barge-in (bot.mjs): stop the bot's TTS only when the speaking (green ring) stays
on for >= WSAI_BARGE_IN_MS (default 700ms), not on the first blip — cancelled if
speaking stops in time. AfterSilence 800->1000ms so trailing soft words finish.
Fragment gate (dashboard): a lone connective filler ("그러면") carries no
answerable intent -> discard as [대기] instead of replying; hidden by the noise
filter like [잡음].
Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
26
dave/bot.mjs
26
dave/bot.mjs
@@ -178,6 +178,11 @@ const speakingSet = new Set(); // userIds currently speaking (for participant
|
||||
const activeSubs = new Set(); // userIds with an in-flight receive subscription
|
||||
let listsByGuild = {}; // guildId -> {whitelistUsers, blacklistUsers, whitelistRoles, blacklistRoles}
|
||||
let botSettings = { bargeIn: true }; // behaviour toggles from the dashboard
|
||||
// Barge-in only fires when the Discord "speaking" (green ring) stays on for at
|
||||
// least this long — a brief blip (keyboard click, cough) shouldn't cut the bot
|
||||
// off, but a genuine ~0.7s of speech should. Tunable via WSAI_BARGE_IN_MS.
|
||||
const BARGE_IN_MS = Number(process.env.WSAI_BARGE_IN_MS || 700);
|
||||
const bargeTimers = new Map(); // userId -> pending stop timer
|
||||
|
||||
// Attach the bot's audio player (so it can speak) to a fresh connection.
|
||||
function setupPlayer(connection) {
|
||||
@@ -202,15 +207,20 @@ function setupReceiver(connection) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
// Barge-in: the moment an allowed user speaks, stop the bot's current TTS
|
||||
// so it doesn't talk over them (toggleable from the dashboard, default on).
|
||||
if (botSettings.bargeIn !== false && voicePlayer) {
|
||||
try { voicePlayer.stop(true); } catch {}
|
||||
// Barge-in: stop the bot's current TTS only if this user keeps speaking for
|
||||
// BARGE_IN_MS (the green ring stays on) — ignores momentary noise blips. The
|
||||
// 'end' handler cancels the pending stop if speaking stops in time.
|
||||
if (botSettings.bargeIn !== false && voicePlayer && !bargeTimers.has(userId)) {
|
||||
bargeTimers.set(userId, setTimeout(() => {
|
||||
bargeTimers.delete(userId);
|
||||
try { voicePlayer.stop(true); } catch {}
|
||||
}, BARGE_IN_MS));
|
||||
}
|
||||
activeSubs.add(userId);
|
||||
if (!perUser.has(userId)) perUser.set(userId, { opusPackets: 0, pcmFrames: 0 });
|
||||
const opusStream = receiver.subscribe(userId, {
|
||||
end: { behavior: EndBehaviorType.AfterSilence, duration: 800 },
|
||||
// Wait a touch longer after silence so a soft/trailing word isn't clipped.
|
||||
end: { behavior: EndBehaviorType.AfterSilence, duration: 1000 },
|
||||
});
|
||||
const decoder = new prism.opus.Decoder({ rate: 48000, channels: 2, frameSize: 960 });
|
||||
const chunks = [];
|
||||
@@ -226,7 +236,11 @@ function setupReceiver(connection) {
|
||||
handleUtterance(userId, pcm).catch((e) => log(`voice turn error: ${e.message}`));
|
||||
});
|
||||
});
|
||||
receiver.speaking.on('end', (userId) => speakingSet.delete(userId));
|
||||
receiver.speaking.on('end', (userId) => {
|
||||
speakingSet.delete(userId);
|
||||
const t = bargeTimers.get(userId); // spoke too briefly -> cancel barge-in
|
||||
if (t) { clearTimeout(t); bargeTimers.delete(userId); }
|
||||
});
|
||||
}
|
||||
|
||||
// Join (or switch to) a voice channel on command from the dashboard.
|
||||
|
||||
Reference in New Issue
Block a user