Ports from isolated upstream-port branch (base b3c5a4b), verified in
isolation via baseline-vs-port failure-set diff and re-verified live
(195 pass / 0 fail on affected tests):
- redact Discord bot tokens in outbound (router.ts SECRET_PATTERNS)
- block SSRF to private hosts in MoA base URL (moa.ts)
- refuse public dashboard bind without auth token (web-dashboard-server.ts)
- merge upstream .gitignore rules for python/build/secret noise
- real CPU utilization from /proc/stat instead of load avg (unified-dashboard.ts)
- width-safe placeholder for missing usage window on mobile (unified-dashboard.ts)
- bump direct deps to patch known vulnerabilities (discord.js/yaml/cron-parser)
Risky upstream commits (d5a94af phantom reset-time, patch 6 Codex usage)
intentionally skipped to avoid touching the credential-isolation tree.
Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
277 lines
9.3 KiB
TypeScript
277 lines
9.3 KiB
TypeScript
import { Channel, NewMessage } from './types.js';
|
|
import { formatLocalTime } from './timezone.js';
|
|
|
|
export function escapeXml(s: string): string {
|
|
if (!s) return '';
|
|
return s
|
|
.replace(/&/g, '&')
|
|
.replace(/</g, '<')
|
|
.replace(/>/g, '>')
|
|
.replace(/"/g, '"');
|
|
}
|
|
|
|
export function formatMessages(
|
|
messages: NewMessage[],
|
|
timezone: string,
|
|
): string {
|
|
const lines = messages.map((m) => {
|
|
const displayTime = formatLocalTime(m.timestamp, timezone);
|
|
return `<message sender="${escapeXml(m.sender_name)}" time="${escapeXml(displayTime)}">${escapeXml(m.content)}</message>`;
|
|
});
|
|
|
|
const header = `<context timezone="${escapeXml(timezone)}" />\n`;
|
|
|
|
return `${header}<messages>\n${lines.join('\n')}\n</messages>`;
|
|
}
|
|
|
|
export function stripInternalTags(text: string): string {
|
|
return text.replace(/<internal>[\s\S]*?<\/internal>/g, '').trim();
|
|
}
|
|
|
|
/**
|
|
* Patterns that match common API keys / tokens.
|
|
* Matched strings are replaced with `[REDACTED]`.
|
|
*/
|
|
const SECRET_PATTERNS: RegExp[] = [
|
|
/sk-ant-[A-Za-z0-9_-]{20,}/g, // Anthropic
|
|
/sk-[A-Za-z0-9_-]{20,}/g, // OpenAI
|
|
/gsk_[A-Za-z0-9_-]{20,}/g, // Groq
|
|
/xai-[A-Za-z0-9_-]{20,}/g, // xAI
|
|
/ghp_[A-Za-z0-9_]{36,}/g, // GitHub PAT classic
|
|
/github_pat_[A-Za-z0-9_]{20,}/g, // GitHub PAT fine-grained
|
|
/glpat-[A-Za-z0-9_-]{20,}/g, // GitLab PAT
|
|
/AKIA[A-Z0-9]{16}/g, // AWS Access Key
|
|
/Bearer\s+eyJ[A-Za-z0-9_-]{40,}/g, // Bearer JWT
|
|
// Discord bot token: base64 id "." 6-char ts "." 27+ char hmac
|
|
/\b[MN][A-Za-z0-9_-]{23,}\.[A-Za-z0-9_-]{6}\.[A-Za-z0-9_-]{27,}\b/g,
|
|
];
|
|
|
|
function redactSecrets(text: string): string {
|
|
let result = text;
|
|
for (const pattern of SECRET_PATTERNS) {
|
|
pattern.lastIndex = 0;
|
|
result = result.replace(pattern, '[REDACTED]');
|
|
}
|
|
return result;
|
|
}
|
|
|
|
/**
|
|
* Strip leaked tool-call serialization text.
|
|
*
|
|
* When a model (especially Codex) enters a degenerate loop, it emits
|
|
* tool-call intent as plaintext instead of actual tool calls. The format is:
|
|
* to=functions.<name> <arbitrary tokens> {<json>}
|
|
* e.g. `to=functions.exec_command code {"cmd":"git status","yield_time_ms":1000}`
|
|
*
|
|
* The tokens between the function name and JSON body can include non-ASCII
|
|
* characters (CJK, etc.) when the model hallucinates. The regex allows one or
|
|
* more non-whitespace descriptor tokens before the JSON brace.
|
|
*
|
|
* This function removes such fragments so they never reach Discord.
|
|
*/
|
|
export function stripToolCallLeaks(text: string): string {
|
|
// Match tool-call serialization: to=functions.<name> <descriptor tokens> {<json>}
|
|
// Handles up to one level of nested braces in the JSON body.
|
|
const stripped = text.replace(
|
|
/to=functions\.\w+(?:\s+[^\s{}]+)+\s+\{[^{}]*(?:\{[^{}]*\}[^{}]*)*\}/g,
|
|
'',
|
|
);
|
|
// Collapse excessive blank lines left after stripping
|
|
return stripped.replace(/\n{3,}/g, '\n\n').trim();
|
|
}
|
|
|
|
/**
|
|
* Escape Discord markdown delimiters in a prose segment so the source
|
|
* characters render literally instead of triggering formatting. Used on
|
|
* non-code segments only — callers must preserve fenced code blocks
|
|
* separately.
|
|
*
|
|
* Discord rendering reference:
|
|
* *italic*, _italic_, **bold**, ***bold-italic***
|
|
* __underline__ ~~strike~~ ||spoiler||
|
|
* `inline` # H1 ## H2 ### H3 (at line start)
|
|
* > quote, >>> multi-line quote (at line start)
|
|
*
|
|
* `<@id>`/`<#id>`/`<:emoji:id>` mentions and `[text](url)` links contain no
|
|
* markdown delimiters and pass through unchanged.
|
|
*/
|
|
function escapeMarkdownInProse(segment: string): string {
|
|
return (
|
|
segment
|
|
// Escape backslashes first so we don't double-escape the backslashes
|
|
// we are about to introduce for the other markers.
|
|
.replace(/\\/g, '\\\\')
|
|
// Inline markdown delimiters — escape positionally so Discord prints
|
|
// them as literal characters.
|
|
.replace(/`/g, '\\`')
|
|
.replace(/\*/g, '\\*')
|
|
.replace(/_/g, '\\_')
|
|
.replace(/~/g, '\\~')
|
|
.replace(/\|/g, '\\|')
|
|
// Heading hashes — only meaningful at the start of a line (after
|
|
// optional indent) and followed by a space. Escape only the first
|
|
// hash; the rest are now harmless literal characters.
|
|
.replace(/^([ \t]*)(#)(?=#{0,2}[ \t])/gm, '$1\\$2')
|
|
// Block-quote markers — same line-start constraint.
|
|
.replace(/^([ \t]*)(>)(?=>{0,2}([ \t]|$))/gm, '$1\\$2')
|
|
);
|
|
}
|
|
|
|
/**
|
|
* Escape stray Discord markdown delimiters in prose while preserving
|
|
* well-formed triple-backtick fenced code blocks (legit code snippets).
|
|
*/
|
|
export function neutralizeStrayMarkdown(text: string): string {
|
|
if (!text) return text;
|
|
const parts: string[] = [];
|
|
let i = 0;
|
|
while (i < text.length) {
|
|
const fenceStart = text.indexOf('```', i);
|
|
if (fenceStart === -1) {
|
|
parts.push(escapeMarkdownInProse(text.slice(i)));
|
|
break;
|
|
}
|
|
parts.push(escapeMarkdownInProse(text.slice(i, fenceStart)));
|
|
const fenceEnd = text.indexOf('```', fenceStart + 3);
|
|
if (fenceEnd === -1) {
|
|
// Unterminated fence — not a real code block; treat as prose.
|
|
parts.push(escapeMarkdownInProse(text.slice(fenceStart)));
|
|
break;
|
|
}
|
|
// Preserve the entire fenced block including delimiters.
|
|
parts.push(text.slice(fenceStart, fenceEnd + 3));
|
|
i = fenceEnd + 3;
|
|
}
|
|
return parts.join('');
|
|
}
|
|
|
|
/** @deprecated Kept for back-compat; use neutralizeStrayMarkdown. */
|
|
export const neutralizeStrayBackticks = neutralizeStrayMarkdown;
|
|
|
|
/**
|
|
* Prevent accidental mass pings. Agent-authored text that literally contains
|
|
* `@everyone` / `@here` must never notify the whole channel — that is only ever
|
|
* intended via a dedicated broadcast path (e.g. the disk-usage alert), not via a
|
|
* normal reply/progress/edit message. A zero-width space (U+200B) after the `@`
|
|
* keeps the text readable ("@everyone" still reads the same) while stopping
|
|
* Discord from parsing it as a mention. Idempotent, and leaves `<@id>` user/role
|
|
* mentions and ordinary emails (`me@example.com`) untouched.
|
|
*/
|
|
export function neutralizeMassMentions(text: string): string {
|
|
return text.replace(/@(everyone|here)\b/g, '@\u200b$1');
|
|
}
|
|
|
|
/**
|
|
* Sanitize raw agent output for internal use (storage, IPC, intermediate
|
|
* channel buffers). Strips internal tags + tool-call leaks, redacts secrets,
|
|
* and neutralizes accidental @everyone/@here pings, but does NOT touch markdown
|
|
* delimiters.
|
|
*
|
|
* Use this when the text will pass through another `formatOutbound` call
|
|
* downstream (e.g., the Discord channel boundary). Applying the markdown
|
|
* escape twice would double-escape backslashes and produce visible garbage
|
|
* in Discord.
|
|
*/
|
|
export function sanitizeForOutbound(rawText: string): string {
|
|
let text = stripInternalTags(rawText);
|
|
if (!text) return '';
|
|
text = stripToolCallLeaks(text);
|
|
if (!text) return '';
|
|
text = redactSecrets(text);
|
|
return neutralizeMassMentions(text);
|
|
}
|
|
|
|
/**
|
|
* Full outbound formatting for the final Discord-send boundary: sanitize
|
|
* + escape Discord markdown delimiters. Call this exactly once per
|
|
* outbound message, at the channel boundary.
|
|
*/
|
|
export function formatOutbound(rawText: string): string {
|
|
const sanitized = sanitizeForOutbound(rawText);
|
|
if (!sanitized) return '';
|
|
return neutralizeStrayMarkdown(sanitized);
|
|
}
|
|
|
|
export function findChannel(
|
|
channels: Channel[],
|
|
jid: string,
|
|
): Channel | undefined {
|
|
return channels.find((c) => c.ownsJid(jid));
|
|
}
|
|
|
|
export interface DeliveryRouteResolution {
|
|
channel?: Channel;
|
|
requestedRoleChannelName: string | null;
|
|
selectedChannelName: string | null;
|
|
usedRoleChannel: boolean;
|
|
fallbackUsed: boolean;
|
|
}
|
|
|
|
function resolveRequestedRoleChannelName(senderRole?: string): string | null {
|
|
if (senderRole === 'reviewer') return 'discord-review';
|
|
if (senderRole === 'arbiter') return 'discord-arbiter';
|
|
return null;
|
|
}
|
|
|
|
export function resolveChannelForDeliveryRole(
|
|
channels: Channel[],
|
|
jid: string,
|
|
senderRole?: string,
|
|
): DeliveryRouteResolution {
|
|
const requestedRoleChannelName = resolveRequestedRoleChannelName(senderRole);
|
|
if (!requestedRoleChannelName) {
|
|
const channel = findChannel(channels, jid);
|
|
return {
|
|
channel,
|
|
requestedRoleChannelName,
|
|
selectedChannelName: channel?.name ?? null,
|
|
usedRoleChannel: false,
|
|
fallbackUsed: false,
|
|
};
|
|
}
|
|
|
|
const roleChannel = findChannelByName(channels, requestedRoleChannelName);
|
|
if (roleChannel) {
|
|
return {
|
|
channel: roleChannel,
|
|
requestedRoleChannelName,
|
|
selectedChannelName: roleChannel.name,
|
|
usedRoleChannel: true,
|
|
fallbackUsed: false,
|
|
};
|
|
}
|
|
|
|
const fallbackChannel = findChannel(channels, jid);
|
|
return {
|
|
channel: fallbackChannel,
|
|
requestedRoleChannelName,
|
|
selectedChannelName: fallbackChannel?.name ?? null,
|
|
usedRoleChannel: false,
|
|
fallbackUsed: true,
|
|
};
|
|
}
|
|
|
|
export function findChannelForDeliveryRole(
|
|
channels: Channel[],
|
|
jid: string,
|
|
senderRole?: string,
|
|
): Channel | undefined {
|
|
return resolveChannelForDeliveryRole(channels, jid, senderRole).channel;
|
|
}
|
|
|
|
export function findChannelByName(
|
|
channels: Channel[],
|
|
name: string,
|
|
): Channel | undefined {
|
|
return channels.find((c) => c.name === name);
|
|
}
|
|
|
|
/**
|
|
* Normalize message text for deduplication comparison.
|
|
* - Trim leading/trailing whitespace
|
|
* - Collapse consecutive whitespace/newlines into single space
|
|
*/
|
|
export function normalizeMessageForDedupe(text: string): string {
|
|
return text.trim().replace(/\s+/g, ' ').toLowerCase();
|
|
}
|