- Log voice connection errors instead of letting them surface silently. - Collapse bursty repeated receive-stream errors (DAVE E2EE group-transition decrypt failures) into one line + a suppressed-count summary, so a member joining/leaving no longer floods the log. Deploy: voice-server now runs as the wsai-voice.service user unit (STT+Claude Haiku brain+TTS on GPU); the bot container reaches it via host.docker.internal:8787. Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
250 lines
11 KiB
JavaScript
250 lines
11 KiB
JavaScript
// Official Discord BOT voice joiner (replaces the selfbot join.mjs).
|
|
//
|
|
// Why this exists: the project migrated off the user-token selfbot path onto an
|
|
// official bot application. For the voice loop (STT input) this is the fully
|
|
// supported, ToS-safe path — @discordjs/voice lets a bot JOIN a voice channel,
|
|
// pass the DAVE/MLS E2EE handshake (via @snazzah/davey, handled internally), and
|
|
// RECEIVE per-user Opus audio. Only screenshare VIDEO receive still requires a
|
|
// selfbot, and video ("눈") is deferred, so nothing here needs a user token.
|
|
//
|
|
// This replicates M1 (join target channel, stay, detect speakers) and, for free,
|
|
// gives partial M2: it decodes each speaker's Opus to PCM and counts frames — the
|
|
// exact stream the STT stage will consume.
|
|
//
|
|
// Usage:
|
|
// node bot.mjs # join and stay until killed
|
|
// RUN_MS=15000 node bot.mjs # join, hold 15s, then leave (for verification)
|
|
//
|
|
// Requires: the bot must be INVITED to the target guild with the "Connect" and
|
|
// "Speak" voice permissions. See dave/README or run `node bot.mjs --invite`.
|
|
|
|
import fs from 'node:fs';
|
|
import { Readable } from 'node:stream';
|
|
import {
|
|
Client,
|
|
GatewayIntentBits,
|
|
} from 'discord.js';
|
|
import {
|
|
joinVoiceChannel,
|
|
getVoiceConnection,
|
|
entersState,
|
|
VoiceConnectionStatus,
|
|
EndBehaviorType,
|
|
createAudioPlayer,
|
|
createAudioResource,
|
|
StreamType,
|
|
NoSubscriberBehavior,
|
|
} from '@discordjs/voice';
|
|
import prism from 'prism-media';
|
|
|
|
// ---------- config ----------
|
|
function loadEnvFile() {
|
|
try {
|
|
return Object.fromEntries(
|
|
fs.readFileSync(new URL('../.env', import.meta.url), 'utf8')
|
|
.split('\n').filter(l => l && !l.startsWith('#') && l.includes('='))
|
|
.map(l => { const i = l.indexOf('='); return [l.slice(0, i).trim(), l.slice(i + 1).trim()]; })
|
|
);
|
|
} catch { return {}; }
|
|
}
|
|
const env = loadEnvFile();
|
|
const TOKEN = process.env.DISCORD_BOT_TOKEN || env.DISCORD_BOT_TOKEN;
|
|
const GUILD_ID = process.env.GUILD_ID || env.GUILD_ID || '1352269198297923648';
|
|
const CHANNEL_ID = process.env.CHANNEL_ID || env.CHANNEL_ID || '1352269198914621465';
|
|
const RUN_MS = process.env.RUN_MS != null ? Number(process.env.RUN_MS) : 0; // 0 = stay forever
|
|
// Python STT+TTS voice-turn endpoint (run `python -m wsai --voice-server`).
|
|
const VOICE_ENDPOINT = process.env.WSAI_VOICE_ENDPOINT || env.WSAI_VOICE_ENDPOINT
|
|
|| 'http://127.0.0.1:8787/api/voice-turn';
|
|
// Ignore utterances shorter than this many PCM bytes (48kHz*2ch*2B = 192000 B/s),
|
|
// so key clicks / brief noise don't trigger a turn. ~0.35s.
|
|
const MIN_UTTERANCE_BYTES = Number(process.env.WSAI_MIN_UTTERANCE_BYTES || 67000);
|
|
|
|
const t0 = Date.now();
|
|
const log = (...a) => console.log(`[+${String(Date.now() - t0).padStart(6)}ms]`, ...a);
|
|
|
|
// Throttle bursty repeated logs. DAVE (E2EE) group transitions — someone joins
|
|
// or leaves the voice channel — briefly deliver undecryptable packets, so the
|
|
// same "recv stream error" can fire many times in a second. Log the first
|
|
// occurrence of a given message immediately, then collapse repeats within a
|
|
// window into one summary line instead of flooding the log.
|
|
const _throttle = new Map(); // key -> { count, timer }
|
|
function logThrottled(key, msg, windowMs = 10_000) {
|
|
const e = _throttle.get(key);
|
|
if (e) { e.count++; return; }
|
|
log(msg);
|
|
const timer = setTimeout(() => {
|
|
const cur = _throttle.get(key);
|
|
_throttle.delete(key);
|
|
if (cur && cur.count > 0) log(`${msg} (+${cur.count} more in ${Math.round(windowMs / 1000)}s)`);
|
|
}, windowMs);
|
|
if (typeof timer.unref === 'function') timer.unref();
|
|
_throttle.set(key, { count: 0, timer });
|
|
}
|
|
|
|
// PCM s16le -> WAV container (so the Python side can ffmpeg-decode it).
|
|
function wavHeader(dataLen, sampleRate = 48000, channels = 2, bits = 16) {
|
|
const blockAlign = channels * bits / 8;
|
|
const b = Buffer.alloc(44);
|
|
b.write('RIFF', 0); b.writeUInt32LE(36 + dataLen, 4); b.write('WAVE', 8);
|
|
b.write('fmt ', 12); b.writeUInt32LE(16, 16); b.writeUInt16LE(1, 20);
|
|
b.writeUInt16LE(channels, 22); b.writeUInt32LE(sampleRate, 24);
|
|
b.writeUInt32LE(sampleRate * blockAlign, 28); b.writeUInt16LE(blockAlign, 32);
|
|
b.writeUInt16LE(bits, 34); b.write('data', 36); b.writeUInt32LE(dataLen, 40);
|
|
return b;
|
|
}
|
|
|
|
// Speak audio into the voice channel. Feed the reply wav bytes through ffmpeg
|
|
// (StreamType.Arbitrary) so @discordjs/voice re-encodes to Opus.
|
|
let voicePlayer = null;
|
|
function playReply(wavBytes) {
|
|
if (!voicePlayer || !wavBytes || wavBytes.length === 0) return;
|
|
const resource = createAudioResource(Readable.from(wavBytes), { inputType: StreamType.Arbitrary });
|
|
voicePlayer.play(resource);
|
|
}
|
|
|
|
// One utterance: PCM -> WAV -> POST to Python -> play the reply back.
|
|
async function handleUtterance(userId, pcm) {
|
|
if (pcm.length < MIN_UTTERANCE_BYTES) {
|
|
log(`utterance too short user=${userId} bytes=${pcm.length} — skip`);
|
|
return;
|
|
}
|
|
const wav = Buffer.concat([wavHeader(pcm.length), pcm]);
|
|
let resp;
|
|
try {
|
|
resp = await fetch(VOICE_ENDPOINT, {
|
|
method: 'POST', headers: { 'Content-Type': 'audio/wav' }, body: wav,
|
|
});
|
|
} catch (e) {
|
|
log(`voice-turn POST failed (is \`python -m wsai --voice-server\` running?): ${e.message}`);
|
|
return;
|
|
}
|
|
if (!resp.ok) { log(`voice-turn HTTP ${resp.status}`); return; }
|
|
const heard = decodeURIComponent(resp.headers.get('X-Heard') || '');
|
|
const reply = decodeURIComponent(resp.headers.get('X-Reply') || '');
|
|
const buf = Buffer.from(await resp.arrayBuffer());
|
|
log(`heard="${heard}" reply="${reply}" replyWav=${buf.length}B`);
|
|
playReply(buf);
|
|
}
|
|
|
|
// ---------- invite URL helper ----------
|
|
// A bot cannot add itself to a server; a server admin must click an OAuth2 invite
|
|
// URL once. Print it so the user can authorise the bot with voice permissions.
|
|
if (process.argv.includes('--invite')) {
|
|
const APP_ID = process.env.DISCORD_APP_ID || env.DISCORD_APP_ID || '';
|
|
// permissions: Connect(1<<20) | Speak(1<<21) | UseVAD(1<<25) | ViewChannel(1<<10)
|
|
const perms = (1n << 20n) | (1n << 21n) | (1n << 25n) | (1n << 10n);
|
|
if (!APP_ID) { console.error('set DISCORD_APP_ID (application/client id) in .env to build the invite URL'); process.exit(2); }
|
|
console.log(`https://discord.com/oauth2/authorize?client_id=${APP_ID}&scope=bot&permissions=${perms}`);
|
|
process.exit(0);
|
|
}
|
|
|
|
if (!TOKEN) { console.error('no DISCORD_BOT_TOKEN in env or ../.env'); process.exit(2); }
|
|
|
|
// ---------- client ----------
|
|
const client = new Client({
|
|
intents: [GatewayIntentBits.Guilds, GatewayIntentBits.GuildVoiceStates],
|
|
});
|
|
|
|
const perUser = new Map(); // userId -> { opusPackets, pcmFrames }
|
|
|
|
let leaving = false;
|
|
function leaveAndExit(code = 0) {
|
|
if (leaving) return;
|
|
leaving = true;
|
|
try { getVoiceConnection(GUILD_ID)?.destroy(); } catch {}
|
|
const summary = [...perUser.entries()].map(([u, s]) => `${u}:opus=${s.opusPackets},pcm=${s.pcmFrames}`);
|
|
log(`leaving. speakers heard: ${summary.length ? summary.join(' ') : '(none)'}`);
|
|
try { client.destroy(); } catch {}
|
|
setTimeout(() => process.exit(code), 300);
|
|
}
|
|
process.on('SIGINT', () => { log('SIGINT'); leaveAndExit(0); });
|
|
process.on('SIGTERM', () => { log('SIGTERM'); leaveAndExit(0); });
|
|
|
|
// Hard time-box ceiling, armed at startup regardless of handshake state.
|
|
if (RUN_MS > 0) setTimeout(() => { log(`RUN_MS=${RUN_MS} hard ceiling elapsed — leaving`); leaveAndExit(0); }, RUN_MS);
|
|
|
|
client.once('clientReady', async () => {
|
|
log(`logged in as ${client.user.tag} (${client.user.id})`);
|
|
let guild, channel;
|
|
try {
|
|
guild = await client.guilds.fetch(GUILD_ID);
|
|
channel = await guild.channels.fetch(CHANNEL_ID);
|
|
} catch (e) {
|
|
log(`FATAL: cannot access guild/channel — is the bot invited to guild ${GUILD_ID}? (${e.message})`);
|
|
log('run `node bot.mjs --invite` and have a server admin authorise the bot, then retry.');
|
|
return leaveAndExit(1);
|
|
}
|
|
if (!channel || !channel.isVoiceBased()) { log(`FATAL: channel ${CHANNEL_ID} is not a voice channel`); return leaveAndExit(1); }
|
|
|
|
log(`joining voice: guild="${guild.name}" channel="${channel.name}"`);
|
|
const connection = joinVoiceChannel({
|
|
channelId: CHANNEL_ID,
|
|
guildId: GUILD_ID,
|
|
adapterCreator: guild.voiceAdapterCreator,
|
|
selfDeaf: false, // MUST be false to receive audio (the STT input path)
|
|
selfMute: false, // false so we can also speak later (M5 TTS)
|
|
});
|
|
connection.on('error', (e) => log(`voice connection error: ${e.message}`));
|
|
|
|
try {
|
|
// The DAVE/MLS handshake here cycles signalling<->connecting several times
|
|
// and can take ~25s, so give it a generous ceiling before declaring failure.
|
|
await entersState(connection, VoiceConnectionStatus.Ready, 40_000);
|
|
} catch (e) {
|
|
log(`FATAL: voice connection did not become Ready in 40s (${e.message})`);
|
|
return leaveAndExit(1);
|
|
}
|
|
log(`✅ JOINED & READY. channel=${CHANNEL_ID} — staying connected, listening for speakers…`);
|
|
|
|
// Playback path (bot speaks): one player, subscribed to the connection.
|
|
voicePlayer = createAudioPlayer({ behaviors: { noSubscriber: NoSubscriberBehavior.Play } });
|
|
voicePlayer.on('error', (e) => log(`player error: ${e.message}`));
|
|
connection.subscribe(voicePlayer);
|
|
log(`voice endpoint: ${VOICE_ENDPOINT}`);
|
|
|
|
// ---------- receive path: capture each utterance and run the voice turn ----
|
|
const receiver = connection.receiver;
|
|
const active = new Set(); // userIds with an in-flight subscription (avoid dupes)
|
|
receiver.speaking.on('start', (userId) => {
|
|
if (userId === client.user.id || active.has(userId)) return; // skip self / dupes
|
|
active.add(userId);
|
|
if (!perUser.has(userId)) perUser.set(userId, { opusPackets: 0, pcmFrames: 0 });
|
|
log(`SPEAKING start user=${userId}`);
|
|
const opusStream = receiver.subscribe(userId, {
|
|
// End the utterance after a short silence so natural pauses don't cut words.
|
|
end: { behavior: EndBehaviorType.AfterSilence, duration: 800 },
|
|
});
|
|
// Decode Opus -> 48kHz stereo s16le PCM, buffered until the utterance ends.
|
|
const decoder = new prism.opus.Decoder({ rate: 48000, channels: 2, frameSize: 960 });
|
|
const chunks = [];
|
|
opusStream.on('data', () => { perUser.get(userId).opusPackets++; });
|
|
// A receive-stream error (e.g. a DAVE decrypt/UDP GenericFailure on one
|
|
// packet) must NOT crash the process — log it and free the slot so the
|
|
// next utterance still works.
|
|
opusStream.on('error', (e) => { logThrottled(`recv:${e.message}`, `recv stream error user=${userId}: ${e.message}`); active.delete(userId); });
|
|
opusStream.pipe(decoder);
|
|
decoder.on('data', (d) => {
|
|
chunks.push(d);
|
|
const s = perUser.get(userId); s.pcmFrames++;
|
|
});
|
|
decoder.on('error', (e) => log(`decode error user=${userId}: ${e.message}`));
|
|
decoder.on('end', () => {
|
|
active.delete(userId);
|
|
const pcm = Buffer.concat(chunks);
|
|
log(`utterance end user=${userId} pcm=${pcm.length}B — running voice turn`);
|
|
handleUtterance(userId, pcm).catch((e) => log(`voice turn error: ${e.message}`));
|
|
});
|
|
});
|
|
|
|
connection.on(VoiceConnectionStatus.Disconnected, () => {
|
|
log('voice: disconnected — attempting to resume…');
|
|
Promise.race([
|
|
entersState(connection, VoiceConnectionStatus.Signalling, 5_000),
|
|
entersState(connection, VoiceConnectionStatus.Connecting, 5_000),
|
|
]).catch(() => { log('voice: could not resume, leaving'); leaveAndExit(0); });
|
|
});
|
|
});
|
|
|
|
client.on('error', (e) => log('client error', e.message));
|
|
client.login(TOKEN).catch((e) => { log(`FATAL: login failed — ${e.message}`); process.exit(1); });
|