Files
EJClaw/src/router.ts
Codex c016b9c2fa port: apply 7 upstream security/robustness patches
Ports from isolated upstream-port branch (base b3c5a4b), verified in
isolation via baseline-vs-port failure-set diff and re-verified live
(195 pass / 0 fail on affected tests):
- redact Discord bot tokens in outbound (router.ts SECRET_PATTERNS)
- block SSRF to private hosts in MoA base URL (moa.ts)
- refuse public dashboard bind without auth token (web-dashboard-server.ts)
- merge upstream .gitignore rules for python/build/secret noise
- real CPU utilization from /proc/stat instead of load avg (unified-dashboard.ts)
- width-safe placeholder for missing usage window on mobile (unified-dashboard.ts)
- bump direct deps to patch known vulnerabilities (discord.js/yaml/cron-parser)

Risky upstream commits (d5a94af phantom reset-time, patch 6 Codex usage)
intentionally skipped to avoid touching the credential-isolation tree.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-08-24 19:35:59 +09:00

277 lines
9.3 KiB
TypeScript

import { Channel, NewMessage } from './types.js';
import { formatLocalTime } from './timezone.js';
export function escapeXml(s: string): string {
if (!s) return '';
return s
.replace(/&/g, '&amp;')
.replace(/</g, '&lt;')
.replace(/>/g, '&gt;')
.replace(/"/g, '&quot;');
}
export function formatMessages(
messages: NewMessage[],
timezone: string,
): string {
const lines = messages.map((m) => {
const displayTime = formatLocalTime(m.timestamp, timezone);
return `<message sender="${escapeXml(m.sender_name)}" time="${escapeXml(displayTime)}">${escapeXml(m.content)}</message>`;
});
const header = `<context timezone="${escapeXml(timezone)}" />\n`;
return `${header}<messages>\n${lines.join('\n')}\n</messages>`;
}
export function stripInternalTags(text: string): string {
return text.replace(/<internal>[\s\S]*?<\/internal>/g, '').trim();
}
/**
* Patterns that match common API keys / tokens.
* Matched strings are replaced with `[REDACTED]`.
*/
const SECRET_PATTERNS: RegExp[] = [
/sk-ant-[A-Za-z0-9_-]{20,}/g, // Anthropic
/sk-[A-Za-z0-9_-]{20,}/g, // OpenAI
/gsk_[A-Za-z0-9_-]{20,}/g, // Groq
/xai-[A-Za-z0-9_-]{20,}/g, // xAI
/ghp_[A-Za-z0-9_]{36,}/g, // GitHub PAT classic
/github_pat_[A-Za-z0-9_]{20,}/g, // GitHub PAT fine-grained
/glpat-[A-Za-z0-9_-]{20,}/g, // GitLab PAT
/AKIA[A-Z0-9]{16}/g, // AWS Access Key
/Bearer\s+eyJ[A-Za-z0-9_-]{40,}/g, // Bearer JWT
// Discord bot token: base64 id "." 6-char ts "." 27+ char hmac
/\b[MN][A-Za-z0-9_-]{23,}\.[A-Za-z0-9_-]{6}\.[A-Za-z0-9_-]{27,}\b/g,
];
function redactSecrets(text: string): string {
let result = text;
for (const pattern of SECRET_PATTERNS) {
pattern.lastIndex = 0;
result = result.replace(pattern, '[REDACTED]');
}
return result;
}
/**
* Strip leaked tool-call serialization text.
*
* When a model (especially Codex) enters a degenerate loop, it emits
* tool-call intent as plaintext instead of actual tool calls. The format is:
* to=functions.<name> <arbitrary tokens> {<json>}
* e.g. `to=functions.exec_command code {"cmd":"git status","yield_time_ms":1000}`
*
* The tokens between the function name and JSON body can include non-ASCII
* characters (CJK, etc.) when the model hallucinates. The regex allows one or
* more non-whitespace descriptor tokens before the JSON brace.
*
* This function removes such fragments so they never reach Discord.
*/
export function stripToolCallLeaks(text: string): string {
// Match tool-call serialization: to=functions.<name> <descriptor tokens> {<json>}
// Handles up to one level of nested braces in the JSON body.
const stripped = text.replace(
/to=functions\.\w+(?:\s+[^\s{}]+)+\s+\{[^{}]*(?:\{[^{}]*\}[^{}]*)*\}/g,
'',
);
// Collapse excessive blank lines left after stripping
return stripped.replace(/\n{3,}/g, '\n\n').trim();
}
/**
* Escape Discord markdown delimiters in a prose segment so the source
* characters render literally instead of triggering formatting. Used on
* non-code segments only — callers must preserve fenced code blocks
* separately.
*
* Discord rendering reference:
* *italic*, _italic_, **bold**, ***bold-italic***
* __underline__ ~~strike~~ ||spoiler||
* `inline` # H1 ## H2 ### H3 (at line start)
* > quote, >>> multi-line quote (at line start)
*
* `<@id>`/`<#id>`/`<:emoji:id>` mentions and `[text](url)` links contain no
* markdown delimiters and pass through unchanged.
*/
function escapeMarkdownInProse(segment: string): string {
return (
segment
// Escape backslashes first so we don't double-escape the backslashes
// we are about to introduce for the other markers.
.replace(/\\/g, '\\\\')
// Inline markdown delimiters — escape positionally so Discord prints
// them as literal characters.
.replace(/`/g, '\\`')
.replace(/\*/g, '\\*')
.replace(/_/g, '\\_')
.replace(/~/g, '\\~')
.replace(/\|/g, '\\|')
// Heading hashes — only meaningful at the start of a line (after
// optional indent) and followed by a space. Escape only the first
// hash; the rest are now harmless literal characters.
.replace(/^([ \t]*)(#)(?=#{0,2}[ \t])/gm, '$1\\$2')
// Block-quote markers — same line-start constraint.
.replace(/^([ \t]*)(>)(?=>{0,2}([ \t]|$))/gm, '$1\\$2')
);
}
/**
* Escape stray Discord markdown delimiters in prose while preserving
* well-formed triple-backtick fenced code blocks (legit code snippets).
*/
export function neutralizeStrayMarkdown(text: string): string {
if (!text) return text;
const parts: string[] = [];
let i = 0;
while (i < text.length) {
const fenceStart = text.indexOf('```', i);
if (fenceStart === -1) {
parts.push(escapeMarkdownInProse(text.slice(i)));
break;
}
parts.push(escapeMarkdownInProse(text.slice(i, fenceStart)));
const fenceEnd = text.indexOf('```', fenceStart + 3);
if (fenceEnd === -1) {
// Unterminated fence — not a real code block; treat as prose.
parts.push(escapeMarkdownInProse(text.slice(fenceStart)));
break;
}
// Preserve the entire fenced block including delimiters.
parts.push(text.slice(fenceStart, fenceEnd + 3));
i = fenceEnd + 3;
}
return parts.join('');
}
/** @deprecated Kept for back-compat; use neutralizeStrayMarkdown. */
export const neutralizeStrayBackticks = neutralizeStrayMarkdown;
/**
* Prevent accidental mass pings. Agent-authored text that literally contains
* `@everyone` / `@here` must never notify the whole channel — that is only ever
* intended via a dedicated broadcast path (e.g. the disk-usage alert), not via a
* normal reply/progress/edit message. A zero-width space (U+200B) after the `@`
* keeps the text readable ("@everyone" still reads the same) while stopping
* Discord from parsing it as a mention. Idempotent, and leaves `<@id>` user/role
* mentions and ordinary emails (`me@example.com`) untouched.
*/
export function neutralizeMassMentions(text: string): string {
return text.replace(/@(everyone|here)\b/g, '@\u200b$1');
}
/**
* Sanitize raw agent output for internal use (storage, IPC, intermediate
* channel buffers). Strips internal tags + tool-call leaks, redacts secrets,
* and neutralizes accidental @everyone/@here pings, but does NOT touch markdown
* delimiters.
*
* Use this when the text will pass through another `formatOutbound` call
* downstream (e.g., the Discord channel boundary). Applying the markdown
* escape twice would double-escape backslashes and produce visible garbage
* in Discord.
*/
export function sanitizeForOutbound(rawText: string): string {
let text = stripInternalTags(rawText);
if (!text) return '';
text = stripToolCallLeaks(text);
if (!text) return '';
text = redactSecrets(text);
return neutralizeMassMentions(text);
}
/**
* Full outbound formatting for the final Discord-send boundary: sanitize
* + escape Discord markdown delimiters. Call this exactly once per
* outbound message, at the channel boundary.
*/
export function formatOutbound(rawText: string): string {
const sanitized = sanitizeForOutbound(rawText);
if (!sanitized) return '';
return neutralizeStrayMarkdown(sanitized);
}
export function findChannel(
channels: Channel[],
jid: string,
): Channel | undefined {
return channels.find((c) => c.ownsJid(jid));
}
export interface DeliveryRouteResolution {
channel?: Channel;
requestedRoleChannelName: string | null;
selectedChannelName: string | null;
usedRoleChannel: boolean;
fallbackUsed: boolean;
}
function resolveRequestedRoleChannelName(senderRole?: string): string | null {
if (senderRole === 'reviewer') return 'discord-review';
if (senderRole === 'arbiter') return 'discord-arbiter';
return null;
}
export function resolveChannelForDeliveryRole(
channels: Channel[],
jid: string,
senderRole?: string,
): DeliveryRouteResolution {
const requestedRoleChannelName = resolveRequestedRoleChannelName(senderRole);
if (!requestedRoleChannelName) {
const channel = findChannel(channels, jid);
return {
channel,
requestedRoleChannelName,
selectedChannelName: channel?.name ?? null,
usedRoleChannel: false,
fallbackUsed: false,
};
}
const roleChannel = findChannelByName(channels, requestedRoleChannelName);
if (roleChannel) {
return {
channel: roleChannel,
requestedRoleChannelName,
selectedChannelName: roleChannel.name,
usedRoleChannel: true,
fallbackUsed: false,
};
}
const fallbackChannel = findChannel(channels, jid);
return {
channel: fallbackChannel,
requestedRoleChannelName,
selectedChannelName: fallbackChannel?.name ?? null,
usedRoleChannel: false,
fallbackUsed: true,
};
}
export function findChannelForDeliveryRole(
channels: Channel[],
jid: string,
senderRole?: string,
): Channel | undefined {
return resolveChannelForDeliveryRole(channels, jid, senderRole).channel;
}
export function findChannelByName(
channels: Channel[],
name: string,
): Channel | undefined {
return channels.find((c) => c.name === name);
}
/**
* Normalize message text for deduplication comparison.
* - Trim leading/trailing whitespace
* - Collapse consecutive whitespace/newlines into single space
*/
export function normalizeMessageForDedupe(text: string): string {
return text.trim().replace(/\s+/g, ' ').toLowerCase();
}