Compare commits
10 Commits
712664ca00
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
84672590d7 | ||
|
|
41b3b4e0e2 | ||
|
|
2259471b66 | ||
|
|
5376df5389 | ||
|
|
007e95fe40 | ||
|
|
bc9ba1c2d2 | ||
|
|
984922884c | ||
|
|
f5d1ec302d | ||
|
|
03e6f8bcc6 | ||
|
|
6fad6dcfab |
4
.gitignore
vendored
4
.gitignore
vendored
@@ -16,6 +16,10 @@ data/
|
||||
data-*/
|
||||
logs/
|
||||
.ejclaw-reviewer-runtime/
|
||||
# Scratch dirs created in CWD by outbound-attachments tests; leak into the repo
|
||||
# root only when a test run is interrupted mid-flight.
|
||||
.ejclaw-attachment-*/
|
||||
.ejclaw-*images-*/
|
||||
|
||||
# Groups - only track base structure and specific CLAUDE.md files
|
||||
groups-*/
|
||||
|
||||
@@ -22,6 +22,11 @@ You have been summoned because the owner and reviewer reached a deadlock after m
|
||||
- The situation cannot be resolved without user input, regardless of technical agreement
|
||||
- The same NEEDS_CONTEXT or BLOCKED is repeated after a prior PROCEED — this means your PROCEED did not resolve the issue
|
||||
|
||||
When you ESCALATE, your message is the ONLY thing the user reads to make their choice — they will not scroll back through the owner↔reviewer conversation. So make it fully self-contained:
|
||||
- Restate the actual decision the user must make, in one plain sentence.
|
||||
- List the concrete options as a short numbered list (1., 2., …) with a one-line consequence for each, so the user can reply with just a number or a short phrase.
|
||||
- Include only the minimum context needed to decide — do not assume the user remembers the prior discussion.
|
||||
|
||||
## MoA (Mixture of Agents) Reference Opinions
|
||||
|
||||
You may receive reference opinions from external models appended to your prompt. When present:
|
||||
|
||||
@@ -96,6 +96,28 @@ describe('message-runtime-prompts carry-forward guidance', () => {
|
||||
);
|
||||
});
|
||||
|
||||
it('prepends the carry-forward warning for a carried arbiter verdict too', () => {
|
||||
const prompt = buildPairedTurnPrompt({
|
||||
taskId: 'task-1',
|
||||
chatJid: 'group@test',
|
||||
timezone: 'UTC',
|
||||
missedMessages: [makeHumanMessage('1번 선택할게')],
|
||||
labeledFallbackMessages: [makeHumanMessage('1번 선택할게')],
|
||||
turnOutputs: [
|
||||
makeTurnOutput(
|
||||
'[Carried forward context from the previous task: latest arbiter verdict]\nESCALATE\n1) A 2) B',
|
||||
),
|
||||
],
|
||||
});
|
||||
|
||||
expect(
|
||||
prompt.startsWith('System note:\nIf you see a message beginning with'),
|
||||
).toBe(true);
|
||||
expect(prompt).toContain(
|
||||
'Respond only to the latest human request and the current task.',
|
||||
);
|
||||
});
|
||||
|
||||
it('prepends a carry-forward warning to reviewer pending prompts', () => {
|
||||
const prompt = buildReviewerPendingPrompt({
|
||||
chatJid: 'group@test',
|
||||
|
||||
@@ -10,11 +10,15 @@ import type {
|
||||
PairedTurnOutput,
|
||||
} from './types.js';
|
||||
|
||||
const CARRIED_FORWARD_OWNER_FINAL_MARKER =
|
||||
'[Carried forward context from the previous task: latest owner final]';
|
||||
// Common prefix for every carried-forward context block. Covers both the owner
|
||||
// final ("… latest owner final]") and the arbiter verdict ("… latest arbiter
|
||||
// verdict]"), so the guidance triggers whichever the previous task ended on —
|
||||
// e.g. an arbiter ESCALATE the user is now replying to.
|
||||
const CARRIED_FORWARD_MARKER_PREFIX =
|
||||
'[Carried forward context from the previous task:';
|
||||
|
||||
const CARRIED_FORWARD_OWNER_FINAL_GUIDANCE = `System note:
|
||||
If you see a message beginning with "${CARRIED_FORWARD_OWNER_FINAL_MARKER}", treat it as background only. Do not repeat, continue, or answer that carried-forward final directly. Respond only to the latest human request and the current task.`;
|
||||
If you see a message beginning with "${CARRIED_FORWARD_MARKER_PREFIX}", treat it as background only. Do not repeat, continue, or answer that carried-forward final directly. Respond only to the latest human request and the current task.`;
|
||||
|
||||
const ARBITER_TURN_OUTPUT_CONTEXT_LIMIT = 6;
|
||||
|
||||
@@ -159,9 +163,9 @@ function currentTaskHumanMessages(
|
||||
});
|
||||
}
|
||||
|
||||
function hasCarriedForwardOwnerFinal(outputs: PairedTurnOutput[]): boolean {
|
||||
function hasCarriedForwardFinal(outputs: PairedTurnOutput[]): boolean {
|
||||
return outputs.some((output) =>
|
||||
output.output_text.startsWith(CARRIED_FORWARD_OWNER_FINAL_MARKER),
|
||||
output.output_text.startsWith(CARRIED_FORWARD_MARKER_PREFIX),
|
||||
);
|
||||
}
|
||||
|
||||
@@ -169,7 +173,7 @@ function prependCarriedForwardGuidance(
|
||||
prompt: string,
|
||||
turnOutputs: PairedTurnOutput[],
|
||||
): string {
|
||||
if (!hasCarriedForwardOwnerFinal(turnOutputs)) {
|
||||
if (!hasCarriedForwardFinal(turnOutputs)) {
|
||||
return prompt;
|
||||
}
|
||||
return `${CARRIED_FORWARD_OWNER_FINAL_GUIDANCE}\n\n${prompt}`;
|
||||
|
||||
@@ -1,5 +1,13 @@
|
||||
import { type AgentOutput } from './agent-runner.js';
|
||||
import { getLastBotFinalMessage } from './db.js';
|
||||
import {
|
||||
getLastBotFinalMessage,
|
||||
hasActiveCiWatcherForChat,
|
||||
storeMessage,
|
||||
} from './db.js';
|
||||
import {
|
||||
resolvePhantomFinalOutcome,
|
||||
scheduleAutoContinueForPhantomFinal,
|
||||
} from './phantom-notification-guard.js';
|
||||
import { runAgentForGroup } from './message-agent-executor.js';
|
||||
import { MessageTurnController } from './message-turn-controller.js';
|
||||
import {
|
||||
@@ -207,9 +215,25 @@ export function createExecuteTurn(deps: CreateExecuteTurnDeps): ExecuteTurnFn {
|
||||
getCloseReason: () =>
|
||||
deps.queue.getCloseReasonForRun?.(chatJid, runId) ?? null,
|
||||
deliverFinalText: async (text, options) => {
|
||||
// Guard against phantom "I'll wait … I'll be notified" dead-end finals
|
||||
// with no registered watcher. Paired flows already continue on their own
|
||||
// (owner → reviewer → …), so this only applies to non-paired turns:
|
||||
// re-invoke the agent once to actually finish (capped, no loop), and if
|
||||
// it dead-ends again, fall back to an explicit follow-up notice so the
|
||||
// user is never left staring at silence.
|
||||
const outcome =
|
||||
resolvedDeliveryRole == null
|
||||
? resolvePhantomFinalOutcome({
|
||||
chatJid,
|
||||
text,
|
||||
hasActiveWatcher: hasActiveCiWatcherForChat(chatJid),
|
||||
canAutoContinue: true,
|
||||
})
|
||||
: { kind: 'clean' as const, text };
|
||||
let delivered = false;
|
||||
try {
|
||||
return await deps.deliverFinalText({
|
||||
text,
|
||||
delivered = await deps.deliverFinalText({
|
||||
text: outcome.text,
|
||||
...(options?.attachments?.length
|
||||
? { attachments: options.attachments }
|
||||
: {}),
|
||||
@@ -231,6 +255,30 @@ export function createExecuteTurn(deps: CreateExecuteTurnDeps): ExecuteTurnFn {
|
||||
);
|
||||
return false;
|
||||
}
|
||||
try {
|
||||
const scheduled = scheduleAutoContinueForPhantomFinal({
|
||||
outcomeKind: outcome.kind,
|
||||
delivered,
|
||||
chatJid,
|
||||
groupFolder: group.folder,
|
||||
runId,
|
||||
storeMessage,
|
||||
enqueueMessageCheck: (jid, folder) =>
|
||||
deps.queue.enqueueMessageCheck(jid, folder),
|
||||
});
|
||||
if (scheduled) {
|
||||
logger.info(
|
||||
{ group: group.name, chatJid, runId },
|
||||
'Auto-continuing after phantom-notification dead-end final (no watcher registered)',
|
||||
);
|
||||
}
|
||||
} catch (err) {
|
||||
logger.warn(
|
||||
{ group: group.name, chatJid, runId, err },
|
||||
'Failed to schedule phantom-notification auto-continue',
|
||||
);
|
||||
}
|
||||
return delivered;
|
||||
},
|
||||
});
|
||||
|
||||
|
||||
@@ -246,4 +246,66 @@ describe('paired execution carry-forward attachments', () => {
|
||||
expect.objectContaining({ createdAt: '2026-03-28T00:01:00.000Z' }),
|
||||
);
|
||||
});
|
||||
|
||||
it('carries the arbiter verdict forward when the previous task ended on an arbiter escalation', () => {
|
||||
vi.clearAllMocks();
|
||||
const previousTask = buildTask({
|
||||
id: 'task-escalated',
|
||||
status: 'completed',
|
||||
completion_reason: 'arbiter_escalated',
|
||||
});
|
||||
vi.mocked(db.getLatestOpenPairedTaskForChat).mockReturnValue(undefined);
|
||||
vi.mocked(db.getLatestPairedTaskForChat).mockReturnValue(previousTask);
|
||||
// The arbiter's ESCALATE is the last thing the user saw — they are now
|
||||
// replying to it, so it (not the earlier owner turn) must carry forward.
|
||||
vi.mocked(db.getPairedTurnOutputs).mockReturnValue([
|
||||
{
|
||||
id: 1,
|
||||
task_id: previousTask.id,
|
||||
turn_number: 1,
|
||||
role: 'owner',
|
||||
output_text: 'STEP_DONE\n방향 A로 진행했습니다.',
|
||||
created_at: '2026-03-28T00:01:00.000Z',
|
||||
},
|
||||
{
|
||||
id: 2,
|
||||
task_id: previousTask.id,
|
||||
turn_number: 2,
|
||||
role: 'reviewer',
|
||||
output_text: 'REVISE\n방향 B가 맞습니다.',
|
||||
created_at: '2026-03-28T00:02:00.000Z',
|
||||
},
|
||||
{
|
||||
id: 3,
|
||||
task_id: previousTask.id,
|
||||
turn_number: 3,
|
||||
role: 'arbiter',
|
||||
output_text:
|
||||
'ESCALATE\n두 방향(A/B) 중 사용자가 선택해야 합니다. 1) A 유지 2) B 전환',
|
||||
created_at: '2026-03-28T00:03:00.000Z',
|
||||
},
|
||||
]);
|
||||
|
||||
resolveOwnerTaskForHumanMessage({
|
||||
group,
|
||||
chatJid: 'dc:test',
|
||||
roomRoleContext: ownerContext,
|
||||
});
|
||||
|
||||
expect(db.insertPairedTurnOutput).toHaveBeenCalledWith(
|
||||
expect.any(String),
|
||||
0,
|
||||
'owner',
|
||||
expect.stringContaining('latest arbiter verdict'),
|
||||
expect.objectContaining({ createdAt: '2026-03-28T00:03:00.000Z' }),
|
||||
);
|
||||
// And it must carry the arbiter's escalation text, not the earlier owner turn.
|
||||
expect(db.insertPairedTurnOutput).toHaveBeenCalledWith(
|
||||
expect.any(String),
|
||||
0,
|
||||
'owner',
|
||||
expect.stringContaining('두 방향(A/B) 중 사용자가 선택'),
|
||||
expect.anything(),
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -14,6 +14,7 @@ vi.mock('./db.js', () => {
|
||||
return {
|
||||
cancelPairedTurn: vi.fn(),
|
||||
createPairedTask: vi.fn(),
|
||||
getEffectiveRuntimeRoomMode: vi.fn(() => 'tribunal'),
|
||||
getLatestPairedTaskForChat: vi.fn(),
|
||||
getLatestOpenPairedTaskForChat: vi.fn(),
|
||||
getPairedTaskById: vi.fn(),
|
||||
@@ -33,6 +34,7 @@ vi.mock('./db.js', () => {
|
||||
});
|
||||
|
||||
vi.mock('./paired-workspace-manager.js', () => ({
|
||||
ensurePairedWorkspaceProvisioned: vi.fn(() => '/repo/self-healed-canonical'),
|
||||
isOwnerWorkspaceRepairNeededError: vi.fn(() => false),
|
||||
markPairedTaskReviewReady: vi.fn(),
|
||||
prepareReviewerWorkspaceForExecution: vi.fn(),
|
||||
@@ -48,6 +50,19 @@ vi.mock('./logger.js', () => ({
|
||||
},
|
||||
}));
|
||||
|
||||
// This suite verifies the flag-OFF ("by default") carry-forward behavior. Pin
|
||||
// the switch to false so the result is deterministic regardless of the
|
||||
// deployment's .env override (the flag-ON path is covered separately in
|
||||
// paired-execution-context-carry-forward.test.ts).
|
||||
vi.mock('./config.js', async () => {
|
||||
const actual =
|
||||
await vi.importActual<typeof import('./config.js')>('./config.js');
|
||||
return {
|
||||
...actual,
|
||||
PAIRED_CARRY_FORWARD_LATEST_OWNER_FINAL: false,
|
||||
};
|
||||
});
|
||||
|
||||
import * as db from './db.js';
|
||||
import * as config from './config.js';
|
||||
import {
|
||||
@@ -299,6 +314,54 @@ describe('paired execution context', () => {
|
||||
expect(db.insertPairedTurnOutput).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('self-heals a tribunal room with no work_dir by provisioning the canonical workspace', () => {
|
||||
// Reproduces the silent-room incident: a tribunal room whose work_dir was
|
||||
// never provisioned would return a null task and never run the owner.
|
||||
const noWorkDirGroup: RegisteredGroup = { ...group, workDir: undefined };
|
||||
vi.mocked(db.getEffectiveRuntimeRoomMode).mockReturnValue('tribunal');
|
||||
vi.mocked(
|
||||
pairedWorkspaceManager.ensurePairedWorkspaceProvisioned,
|
||||
).mockReturnValue('/repo/self-healed-canonical');
|
||||
vi.mocked(db.getLatestOpenPairedTaskForChat).mockReturnValue(undefined);
|
||||
vi.mocked(db.getLatestPairedTaskForChat).mockReturnValue(undefined);
|
||||
|
||||
const result = resolveOwnerTaskForHumanMessage({
|
||||
group: noWorkDirGroup,
|
||||
chatJid: 'dc:test',
|
||||
roomRoleContext: ownerContext,
|
||||
existingTask: null,
|
||||
});
|
||||
|
||||
expect(
|
||||
pairedWorkspaceManager.ensurePairedWorkspaceProvisioned,
|
||||
).toHaveBeenCalledWith({
|
||||
chatJid: 'dc:test',
|
||||
groupFolder: noWorkDirGroup.folder,
|
||||
});
|
||||
expect(result.task).not.toBeNull();
|
||||
expect(db.createPairedTask).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('does not provision a workspace for a single-mode room with no work_dir', () => {
|
||||
const noWorkDirGroup: RegisteredGroup = { ...group, workDir: undefined };
|
||||
vi.mocked(db.getEffectiveRuntimeRoomMode).mockReturnValue('single');
|
||||
vi.mocked(db.getLatestOpenPairedTaskForChat).mockReturnValue(undefined);
|
||||
vi.mocked(db.getLatestPairedTaskForChat).mockReturnValue(undefined);
|
||||
|
||||
const result = resolveOwnerTaskForHumanMessage({
|
||||
group: noWorkDirGroup,
|
||||
chatJid: 'dc:test',
|
||||
roomRoleContext: ownerContext,
|
||||
existingTask: null,
|
||||
});
|
||||
|
||||
expect(
|
||||
pairedWorkspaceManager.ensurePairedWorkspaceProvisioned,
|
||||
).not.toHaveBeenCalled();
|
||||
expect(result.task).toBeNull();
|
||||
expect(db.createPairedTask).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('records a quick reopen when a new owner task starts shortly after TASK_DONE completion', () => {
|
||||
const previousTask = buildPairedTask({
|
||||
id: 'task-completed',
|
||||
|
||||
@@ -19,6 +19,7 @@ import {
|
||||
import {
|
||||
cancelPairedTurn,
|
||||
createPairedTask,
|
||||
getEffectiveRuntimeRoomMode,
|
||||
getLatestPairedTaskForChat,
|
||||
getLatestOpenPairedTaskForChat,
|
||||
getPairedTaskById,
|
||||
@@ -48,6 +49,7 @@ import {
|
||||
} from './paired-task-status.js';
|
||||
import { resolveCanonicalSourceRef } from './paired-source-ref.js';
|
||||
import {
|
||||
ensurePairedWorkspaceProvisioned,
|
||||
isOwnerWorkspaceRepairNeededError,
|
||||
prepareReviewerWorkspaceForExecution,
|
||||
provisionOwnerWorkspaceForPairedTask,
|
||||
@@ -74,7 +76,21 @@ function ensurePairedProject(
|
||||
chatJid: string,
|
||||
): string | null {
|
||||
if (!group.workDir) {
|
||||
return null;
|
||||
// Self-heal: a tribunal room whose work_dir was never provisioned (null)
|
||||
// makes resolveOwnerTaskForHumanMessage return a null task, so the owner
|
||||
// never runs and the room goes completely silent — the user's messages get
|
||||
// no reply at all. Provision the canonical workspace on demand so the owner
|
||||
// can always start. Guarded to tribunal rooms so a single-mode room never
|
||||
// gets a spurious paired workspace. Idempotent: ensurePairedWorkspace-
|
||||
// Provisioned skips git init/commit when they already exist and upserts the
|
||||
// paired project itself.
|
||||
if (getEffectiveRuntimeRoomMode(chatJid) !== 'tribunal') {
|
||||
return null;
|
||||
}
|
||||
return ensurePairedWorkspaceProvisioned({
|
||||
chatJid,
|
||||
groupFolder: group.folder,
|
||||
});
|
||||
}
|
||||
|
||||
const now = new Date().toISOString();
|
||||
@@ -235,20 +251,26 @@ function isIntermediateStepOutput(outputText: string): boolean {
|
||||
return /^\s*STEP_DONE\b/.test(outputText);
|
||||
}
|
||||
|
||||
function getLatestOwnerFinalOutput(taskId: string): PairedTurnOutput | null {
|
||||
const ownerOutputs = [...getPairedTurnOutputs(taskId)]
|
||||
function getLatestUserFacingFinalOutput(
|
||||
taskId: string,
|
||||
): PairedTurnOutput | null {
|
||||
// The user only ever sees owner finals and arbiter verdicts (reviewer turns
|
||||
// are internal). Whichever of those the previous task ended on is what the
|
||||
// user is replying to next — e.g. an arbiter ESCALATE asking them to choose.
|
||||
// So carry the chronologically-latest owner/arbiter final forward.
|
||||
const userFacingOutputs = [...getPairedTurnOutputs(taskId)]
|
||||
.reverse()
|
||||
.filter((output) => output.role === 'owner');
|
||||
.filter((output) => output.role === 'owner' || output.role === 'arbiter');
|
||||
return (
|
||||
ownerOutputs.find(
|
||||
userFacingOutputs.find(
|
||||
(output) => !isIntermediateStepOutput(output.output_text),
|
||||
) ??
|
||||
ownerOutputs[0] ??
|
||||
userFacingOutputs[0] ??
|
||||
null
|
||||
);
|
||||
}
|
||||
|
||||
function carryForwardLatestOwnerFinal(args: {
|
||||
function carryForwardLatestFinal(args: {
|
||||
sourceTask: PairedTask;
|
||||
targetTask: PairedTask;
|
||||
}): void {
|
||||
@@ -256,29 +278,34 @@ function carryForwardLatestOwnerFinal(args: {
|
||||
return;
|
||||
}
|
||||
|
||||
const latestOwnerFinal = getLatestOwnerFinalOutput(args.sourceTask.id);
|
||||
if (!latestOwnerFinal) {
|
||||
const latestFinal = getLatestUserFacingFinalOutput(args.sourceTask.id);
|
||||
if (!latestFinal) {
|
||||
return;
|
||||
}
|
||||
|
||||
const label =
|
||||
latestFinal.role === 'arbiter'
|
||||
? 'latest arbiter verdict'
|
||||
: 'latest owner final';
|
||||
insertPairedTurnOutput(
|
||||
args.targetTask.id,
|
||||
0,
|
||||
'owner',
|
||||
`[Carried forward context from the previous task: latest owner final]\n${latestOwnerFinal.output_text}`,
|
||||
`[Carried forward context from the previous task: ${label}]\n${latestFinal.output_text}`,
|
||||
{
|
||||
createdAt: latestOwnerFinal.created_at,
|
||||
attachments: latestOwnerFinal.attachments,
|
||||
createdAt: latestFinal.created_at,
|
||||
attachments: latestFinal.attachments,
|
||||
},
|
||||
);
|
||||
logger.info(
|
||||
{
|
||||
sourceTaskId: args.sourceTask.id,
|
||||
targetTaskId: args.targetTask.id,
|
||||
carriedChars: latestOwnerFinal.output_text.length,
|
||||
attachmentCount: latestOwnerFinal.attachments?.length ?? 0,
|
||||
carriedRole: latestFinal.role,
|
||||
carriedChars: latestFinal.output_text.length,
|
||||
attachmentCount: latestFinal.attachments?.length ?? 0,
|
||||
},
|
||||
'Carried forward latest owner final into superseding paired task',
|
||||
'Carried forward latest user-facing final into superseding paired task',
|
||||
);
|
||||
}
|
||||
|
||||
@@ -312,7 +339,7 @@ export function resolveOwnerTaskForHumanMessage(args: {
|
||||
// fresh task with the previous task's latest owner final so the owner keeps
|
||||
// continuity across sessions instead of answering with "no context".
|
||||
if (newTask && previousTask) {
|
||||
carryForwardLatestOwnerFinal({
|
||||
carryForwardLatestFinal({
|
||||
sourceTask: previousTask,
|
||||
targetTask: newTask,
|
||||
});
|
||||
@@ -357,7 +384,7 @@ export function resolveOwnerTaskForHumanMessage(args: {
|
||||
canonicalWorkDir,
|
||||
roomRoleContext: args.roomRoleContext,
|
||||
});
|
||||
carryForwardLatestOwnerFinal({
|
||||
carryForwardLatestFinal({
|
||||
sourceTask: existing,
|
||||
targetTask: newTask,
|
||||
});
|
||||
|
||||
233
src/phantom-notification-guard.test.ts
Normal file
233
src/phantom-notification-guard.test.ts
Normal file
@@ -0,0 +1,233 @@
|
||||
import { beforeEach, describe, expect, it, vi } from 'vitest';
|
||||
|
||||
import {
|
||||
applyPhantomNotificationGuard,
|
||||
isPhantomNotificationDeadEnd,
|
||||
PHANTOM_CONTINUE_NUDGE,
|
||||
PHANTOM_NOTIFICATION_NOTICE,
|
||||
resolvePhantomFinalOutcome,
|
||||
scheduleAutoContinueForPhantomFinal,
|
||||
_resetPhantomAutoContinueForTests,
|
||||
} from './phantom-notification-guard.js';
|
||||
|
||||
describe('isPhantomNotificationDeadEnd', () => {
|
||||
it('detects the reported "I\'ll wait for the build … I\'ll be notified" final', () => {
|
||||
expect(
|
||||
isPhantomNotificationDeadEnd(
|
||||
"I'll wait for the build to complete — I'll be notified.",
|
||||
),
|
||||
).toBe(true);
|
||||
});
|
||||
|
||||
it('detects close variants of the notification promise', () => {
|
||||
expect(
|
||||
isPhantomNotificationDeadEnd('I will be notified when it finishes.'),
|
||||
).toBe(true);
|
||||
expect(
|
||||
isPhantomNotificationDeadEnd('Waiting for the deploy to finish now.'),
|
||||
).toBe(true);
|
||||
expect(
|
||||
isPhantomNotificationDeadEnd("I'll be notified once CI is green."),
|
||||
).toBe(true);
|
||||
});
|
||||
|
||||
it('does not flag ordinary status finals', () => {
|
||||
expect(
|
||||
isPhantomNotificationDeadEnd(
|
||||
'TASK_DONE\n빌드를 끝냈고 테스트 1597개가 모두 통과했습니다.',
|
||||
),
|
||||
).toBe(false);
|
||||
expect(
|
||||
isPhantomNotificationDeadEnd('The build completed and I pushed the fix.'),
|
||||
).toBe(false);
|
||||
expect(isPhantomNotificationDeadEnd('')).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('applyPhantomNotificationGuard', () => {
|
||||
const deadEnd = "I'll wait for the build to complete — I'll be notified.";
|
||||
|
||||
it('appends the follow-up notice for a dead-end final with no active watcher', () => {
|
||||
const result = applyPhantomNotificationGuard(deadEnd, {
|
||||
hasActiveWatcher: false,
|
||||
});
|
||||
expect(result.startsWith(deadEnd)).toBe(true);
|
||||
expect(result).toContain(PHANTOM_NOTIFICATION_NOTICE.trim());
|
||||
});
|
||||
|
||||
it('leaves the final untouched when a watcher is actually active', () => {
|
||||
expect(
|
||||
applyPhantomNotificationGuard(deadEnd, { hasActiveWatcher: true }),
|
||||
).toBe(deadEnd);
|
||||
});
|
||||
|
||||
it('leaves ordinary finals untouched', () => {
|
||||
const ok = 'TASK_DONE\n작업을 마쳤습니다.';
|
||||
expect(applyPhantomNotificationGuard(ok, { hasActiveWatcher: false })).toBe(
|
||||
ok,
|
||||
);
|
||||
});
|
||||
|
||||
it('is idempotent — never appends the notice twice', () => {
|
||||
const once = applyPhantomNotificationGuard(deadEnd, {
|
||||
hasActiveWatcher: false,
|
||||
});
|
||||
const twice = applyPhantomNotificationGuard(once, {
|
||||
hasActiveWatcher: false,
|
||||
});
|
||||
expect(twice).toBe(once);
|
||||
});
|
||||
});
|
||||
|
||||
describe('resolvePhantomFinalOutcome', () => {
|
||||
const deadEnd = "I'll wait for the build to complete — I'll be notified.";
|
||||
const chatJid = 'dc:test-room';
|
||||
|
||||
beforeEach(() => {
|
||||
_resetPhantomAutoContinueForTests();
|
||||
});
|
||||
|
||||
it('auto-continues once, then falls back to the notice on a repeated dead-end', () => {
|
||||
const first = resolvePhantomFinalOutcome({
|
||||
chatJid,
|
||||
text: deadEnd,
|
||||
hasActiveWatcher: false,
|
||||
canAutoContinue: true,
|
||||
});
|
||||
expect(first.kind).toBe('auto-continue');
|
||||
|
||||
// Second consecutive dead-end for the same chat hits the cap → notice, no
|
||||
// further auto-continue (this is the loop guard).
|
||||
const second = resolvePhantomFinalOutcome({
|
||||
chatJid,
|
||||
text: deadEnd,
|
||||
hasActiveWatcher: false,
|
||||
canAutoContinue: true,
|
||||
});
|
||||
expect(second.kind).toBe('notice');
|
||||
expect(second.text).toContain(PHANTOM_NOTIFICATION_NOTICE.trim());
|
||||
});
|
||||
|
||||
it('resets the streak after a clean final so a later dead-end can auto-continue again', () => {
|
||||
expect(
|
||||
resolvePhantomFinalOutcome({
|
||||
chatJid,
|
||||
text: deadEnd,
|
||||
hasActiveWatcher: false,
|
||||
canAutoContinue: true,
|
||||
}).kind,
|
||||
).toBe('auto-continue');
|
||||
|
||||
// A clean final resets the streak.
|
||||
expect(
|
||||
resolvePhantomFinalOutcome({
|
||||
chatJid,
|
||||
text: 'TASK_DONE\n작업을 마쳤습니다.',
|
||||
hasActiveWatcher: false,
|
||||
canAutoContinue: true,
|
||||
}).kind,
|
||||
).toBe('clean');
|
||||
|
||||
// Streak reset → next dead-end auto-continues again.
|
||||
expect(
|
||||
resolvePhantomFinalOutcome({
|
||||
chatJid,
|
||||
text: deadEnd,
|
||||
hasActiveWatcher: false,
|
||||
canAutoContinue: true,
|
||||
}).kind,
|
||||
).toBe('auto-continue');
|
||||
});
|
||||
|
||||
it('never auto-continues when a watcher is active (legitimate wait)', () => {
|
||||
expect(
|
||||
resolvePhantomFinalOutcome({
|
||||
chatJid,
|
||||
text: deadEnd,
|
||||
hasActiveWatcher: true,
|
||||
canAutoContinue: true,
|
||||
}).kind,
|
||||
).toBe('clean');
|
||||
});
|
||||
|
||||
it('emits the notice (never auto-continue) when auto-continue is disabled', () => {
|
||||
const outcome = resolvePhantomFinalOutcome({
|
||||
chatJid,
|
||||
text: deadEnd,
|
||||
hasActiveWatcher: false,
|
||||
canAutoContinue: false,
|
||||
});
|
||||
expect(outcome.kind).toBe('notice');
|
||||
});
|
||||
});
|
||||
|
||||
describe('scheduleAutoContinueForPhantomFinal', () => {
|
||||
it('injects the corrective nudge and re-enqueues when the dead-end final delivered', () => {
|
||||
const storeMessage = vi.fn();
|
||||
const enqueueMessageCheck = vi.fn();
|
||||
|
||||
const scheduled = scheduleAutoContinueForPhantomFinal({
|
||||
outcomeKind: 'auto-continue',
|
||||
delivered: true,
|
||||
chatJid: 'dc:test-room',
|
||||
groupFolder: 'test-room',
|
||||
runId: 'run-xyz',
|
||||
storeMessage,
|
||||
enqueueMessageCheck,
|
||||
});
|
||||
|
||||
expect(scheduled).toBe(true);
|
||||
expect(storeMessage).toHaveBeenCalledTimes(1);
|
||||
expect(storeMessage).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
chat_jid: 'dc:test-room',
|
||||
content: PHANTOM_CONTINUE_NUDGE,
|
||||
is_from_me: false,
|
||||
is_bot_message: false,
|
||||
message_source_kind: 'ipc_injected_human',
|
||||
}),
|
||||
);
|
||||
expect(enqueueMessageCheck).toHaveBeenCalledWith(
|
||||
'dc:test-room',
|
||||
'test-room',
|
||||
);
|
||||
});
|
||||
|
||||
it('does nothing when the outcome is not auto-continue', () => {
|
||||
const storeMessage = vi.fn();
|
||||
const enqueueMessageCheck = vi.fn();
|
||||
|
||||
const scheduled = scheduleAutoContinueForPhantomFinal({
|
||||
outcomeKind: 'notice',
|
||||
delivered: true,
|
||||
chatJid: 'dc:test-room',
|
||||
groupFolder: 'test-room',
|
||||
runId: 'run-xyz',
|
||||
storeMessage,
|
||||
enqueueMessageCheck,
|
||||
});
|
||||
|
||||
expect(scheduled).toBe(false);
|
||||
expect(storeMessage).not.toHaveBeenCalled();
|
||||
expect(enqueueMessageCheck).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('does nothing when the final was not delivered', () => {
|
||||
const storeMessage = vi.fn();
|
||||
const enqueueMessageCheck = vi.fn();
|
||||
|
||||
const scheduled = scheduleAutoContinueForPhantomFinal({
|
||||
outcomeKind: 'auto-continue',
|
||||
delivered: false,
|
||||
chatJid: 'dc:test-room',
|
||||
groupFolder: 'test-room',
|
||||
runId: 'run-xyz',
|
||||
storeMessage,
|
||||
enqueueMessageCheck,
|
||||
});
|
||||
|
||||
expect(scheduled).toBe(false);
|
||||
expect(storeMessage).not.toHaveBeenCalled();
|
||||
expect(enqueueMessageCheck).not.toHaveBeenCalled();
|
||||
});
|
||||
});
|
||||
149
src/phantom-notification-guard.ts
Normal file
149
src/phantom-notification-guard.ts
Normal file
@@ -0,0 +1,149 @@
|
||||
import type { NewMessage } from './types.js';
|
||||
|
||||
// Guard against "phantom notification" dead-end finals.
|
||||
//
|
||||
// Agents sometimes end a turn with a sentence like "I'll wait for the build to
|
||||
// complete — I'll be notified", expecting a background-completion callback that
|
||||
// this environment never delivers. The turn then simply ends and the user is
|
||||
// left staring at that message forever (no watcher was registered, so nothing
|
||||
// ever resumes). The claude-platform prompt already forbids this, but a prompt
|
||||
// rule cannot reliably stop the model, so we also enforce it at delivery time:
|
||||
// when a final looks like such a dead-end and no watcher is actually active, we
|
||||
// append an explicit notice telling the user no notification is coming and to
|
||||
// send a follow-up — turning a silent hang into an actionable prompt.
|
||||
|
||||
const PHANTOM_NOTIFICATION_PATTERNS: readonly RegExp[] = [
|
||||
// First-person promise of a callback that never arrives.
|
||||
/\bi['’]?\s*ll\s+be\s+notified\b/i,
|
||||
/\bi\s+will\s+be\s+notified\b/i,
|
||||
/\bi['’]?\s*m\s+(?:going to|gonna)\s+(?:wait|be notified)\b/i,
|
||||
// "wait for the build/compile/deploy to complete/finish"
|
||||
/\bwait(?:ing)?\s+for\s+the\s+[\w\s-]{0,40}?\bto\s+(?:complete|finish|be\s+done)\b/i,
|
||||
// "notified when it finishes/completes/is done"
|
||||
/\bnotified\s+when\s+[\w\s-]{0,40}?(?:finishes|completes|is\s+done)\b/i,
|
||||
];
|
||||
|
||||
/**
|
||||
* True when `text` reads like a background-completion "I'll be notified"
|
||||
* dead-end. Conservative: matches the distinctive first-person phrasing rather
|
||||
* than any mention of waiting, to avoid annotating legitimate status reports.
|
||||
*/
|
||||
export function isPhantomNotificationDeadEnd(text: string): boolean {
|
||||
if (!text) return false;
|
||||
return PHANTOM_NOTIFICATION_PATTERNS.some((pattern) => pattern.test(text));
|
||||
}
|
||||
|
||||
export const PHANTOM_NOTIFICATION_NOTICE =
|
||||
'\n\n[자동 안내] 이 환경에는 백그라운드 완료 알림이 없어서, 위처럼 "기다렸다가 알림을 받겠다"고 해도 실제로는 아무것도 다시 시작되지 않습니다. 감시 작업(watch_ci 등)도 등록되지 않았습니다. 계속 진행하려면 이 방에 메시지를 한 번 더 보내 주세요.';
|
||||
|
||||
const NOTICE_MARKER = '[자동 안내]';
|
||||
|
||||
/**
|
||||
* Returns `text` unchanged unless it is a phantom-notification dead-end AND no
|
||||
* watcher is active for the chat, in which case an explicit follow-up notice is
|
||||
* appended. Idempotent: never appends the notice twice. When a watcher IS
|
||||
* active the "I'll be notified" claim is legitimate, so the text is untouched.
|
||||
*/
|
||||
export function applyPhantomNotificationGuard(
|
||||
text: string,
|
||||
opts: { hasActiveWatcher: boolean },
|
||||
): string {
|
||||
if (opts.hasActiveWatcher) return text;
|
||||
if (!isPhantomNotificationDeadEnd(text)) return text;
|
||||
if (text.includes(NOTICE_MARKER)) return text;
|
||||
return `${text}${PHANTOM_NOTIFICATION_NOTICE}`;
|
||||
}
|
||||
|
||||
// ── Auto-continue after a phantom dead-end ────────────────────────────────
|
||||
//
|
||||
// When a non-paired turn dead-ends on a phantom "I'll be notified" final with no
|
||||
// watcher, instead of just leaving a notice we re-invoke the agent once so it
|
||||
// actually finishes the work in the foreground. A hard per-chat cap makes this
|
||||
// safe: at most MAX_PHANTOM_AUTO_CONTINUES re-runs per phantom streak, so a model
|
||||
// that keeps producing the dead-end can never spin the bot in a loop. The streak
|
||||
// resets as soon as any clean (non-phantom) final is produced.
|
||||
|
||||
const MAX_PHANTOM_AUTO_CONTINUES = 1;
|
||||
const phantomAutoContinueStreak = new Map<string, number>();
|
||||
|
||||
/** Corrective nudge injected (as an inbound instruction) before the re-run. */
|
||||
export const PHANTOM_CONTINUE_NUDGE =
|
||||
'[시스템 자동 안내] 방금 실제 감시 작업(watch_ci 등)을 등록하지 않은 채 "기다렸다가 알림을 받겠다"는 식으로 응답을 끝냈습니다. 이 환경에는 백그라운드 완료 알림이 없어, 그대로 두면 아무것도 다시 시작되지 않습니다. 지금 바로 포그라운드에서 작업을 끝까지 진행해 실제 결과를 보고하세요. 정말 오래 걸리는 작업이면 watch_ci/schedule_task로 감시를 등록한 뒤 그 사실을 알려주세요. 다시는 "알림을 받겠다"며 끝내지 마세요.';
|
||||
|
||||
export type PhantomFinalOutcome =
|
||||
| { kind: 'clean'; text: string }
|
||||
| { kind: 'auto-continue'; text: string }
|
||||
| { kind: 'notice'; text: string };
|
||||
|
||||
/**
|
||||
* Decide what to do with a final that may be a phantom dead-end. Pure except for
|
||||
* the per-chat streak counter, which is the loop guard.
|
||||
*
|
||||
* - `clean`: not a dead-end (or a watcher is active) — deliver as-is; resets streak.
|
||||
* - `auto-continue`: dead-end, under the cap — deliver as-is, and the caller must
|
||||
* re-invoke the agent once with PHANTOM_CONTINUE_NUDGE.
|
||||
* - `notice`: dead-end but the cap is exhausted (or auto-continue disabled) —
|
||||
* deliver with the follow-up notice appended so the user isn't left hanging.
|
||||
*/
|
||||
export function resolvePhantomFinalOutcome(args: {
|
||||
chatJid: string;
|
||||
text: string;
|
||||
hasActiveWatcher: boolean;
|
||||
canAutoContinue: boolean;
|
||||
}): PhantomFinalOutcome {
|
||||
const { chatJid, text, hasActiveWatcher, canAutoContinue } = args;
|
||||
if (hasActiveWatcher || !isPhantomNotificationDeadEnd(text)) {
|
||||
phantomAutoContinueStreak.delete(chatJid);
|
||||
return { kind: 'clean', text };
|
||||
}
|
||||
const prior = phantomAutoContinueStreak.get(chatJid) ?? 0;
|
||||
if (canAutoContinue && prior < MAX_PHANTOM_AUTO_CONTINUES) {
|
||||
phantomAutoContinueStreak.set(chatJid, prior + 1);
|
||||
return { kind: 'auto-continue', text };
|
||||
}
|
||||
return {
|
||||
kind: 'notice',
|
||||
text: applyPhantomNotificationGuard(text, { hasActiveWatcher: false }),
|
||||
};
|
||||
}
|
||||
|
||||
/** Test-only: clear the per-chat auto-continue streak counters. */
|
||||
export function _resetPhantomAutoContinueForTests(): void {
|
||||
phantomAutoContinueStreak.clear();
|
||||
}
|
||||
|
||||
/**
|
||||
* Perform the auto-continue side effect after a final was delivered: inject the
|
||||
* corrective nudge as an inbound instruction and re-enqueue a turn so the agent
|
||||
* runs again and finishes the work. No-op unless the outcome was `auto-continue`
|
||||
* AND the final actually delivered. Returns true when it scheduled a re-run.
|
||||
*
|
||||
* Extracted from the delivery callback so the storeMessage → enqueueMessageCheck
|
||||
* path is unit-testable without standing up a full turn.
|
||||
*/
|
||||
export function scheduleAutoContinueForPhantomFinal(args: {
|
||||
outcomeKind: PhantomFinalOutcome['kind'];
|
||||
delivered: boolean;
|
||||
chatJid: string;
|
||||
groupFolder: string;
|
||||
runId: string;
|
||||
storeMessage: (message: NewMessage) => void;
|
||||
enqueueMessageCheck: (chatJid: string, groupFolder: string) => void;
|
||||
}): boolean {
|
||||
if (args.outcomeKind !== 'auto-continue' || !args.delivered) {
|
||||
return false;
|
||||
}
|
||||
args.storeMessage({
|
||||
id: `phantom-continue-${args.runId}-${Date.now().toString(36)}`,
|
||||
chat_jid: args.chatJid,
|
||||
sender: 'ejclaw-system',
|
||||
sender_name: 'EJClaw',
|
||||
content: PHANTOM_CONTINUE_NUDGE,
|
||||
timestamp: new Date().toISOString(),
|
||||
is_from_me: false,
|
||||
is_bot_message: false,
|
||||
message_source_kind: 'ipc_injected_human',
|
||||
});
|
||||
args.enqueueMessageCheck(args.chatJid, args.groupFolder);
|
||||
return true;
|
||||
}
|
||||
Reference in New Issue
Block a user