repositories / pi-ext
pi-ext
bugabingas pi extensions
owned by admin
extensions/angel/__tests__/harness.test.ts
Rawimport { createRequire } from "node:module";
import { pathToFileURL } from "node:url";
import { stripVTControlCharacters } from "node:util";
import {
type AssistantMessage,
fauxAssistantMessage,
fauxToolCall,
type ToolResultMessage,
} from "@earendil-works/pi-ai";
import {
type BoundaryContextPreview,
convertToLlm,
type ExtensionRunner,
initTheme,
} from "@earendil-works/pi-coding-agent";
import type { KeybindingsManager } from "@earendil-works/pi-tui";
import {
afterAll,
afterEach,
beforeAll,
describe,
expect,
it,
vi,
} from "vitest";
import {
calls,
createTestSession,
says,
type TestSession,
when,
} from "../../../test/harness";
import type { ConsultationOrigin, ConsultationResult } from "../advisor.ts";
import { type AngelDependencies, createAngelExtension } from "../index.ts";
function result(origin: ConsultationOrigin): ConsultationResult {
return {
advice: `**${origin} advice**\n\nInspect the evidence first.`,
metadata: {
origin,
model: "openai/gpt-4o",
thinkingLevel: "max",
runtimeMs: 250,
tokens: {
input: 10,
output: 5,
cacheRead: 0,
cacheWrite: 0,
total: 15,
},
cost: 0.01,
childSessionId: `child-${origin}`,
childSessionFile: `/sessions/child-${origin}.jsonl`,
},
};
}
function assistantCall(
toolName: string,
toolCallId: string,
arguments_: Record<string, unknown> = {},
) {
return fauxAssistantMessage(
fauxToolCall(toolName, arguments_, { id: toolCallId }),
);
}
function toolResult(
toolName: string,
toolCallId: string,
isError = true,
text = isError ? "failure" : "success",
) {
return {
role: "toolResult" as const,
toolCallId,
toolName,
content: [{ type: "text" as const, text }],
isError,
timestamp: Date.now(),
};
}
const BOUNDARY_CONTEXT: BoundaryContextPreview = {
contextEntries: [],
contextMessages: [],
llmMessages: [],
pendingMessages: [],
canContinue: false,
};
function emitTurnEnd(
runner: ExtensionRunner,
event: {
turnIndex: number;
message: AssistantMessage;
toolResults: ToolResultMessage[];
},
) {
return runner.emitBoundary(
{
type: "turn_end",
...event,
messageEntryId: `message-${event.turnIndex}`,
toolResultEntryIds: event.toolResults.map(
(_, index) => `tool-result-${event.turnIndex}-${index}`,
),
outcome: "completed",
},
() => BOUNDARY_CONTEXT,
);
}
function theme() {
return {
fg: (_style: string, text: string) => text,
bg: (_style: string, text: string) => text,
bold: (text: string) => text,
italic: (text: string) => text,
strikethrough: (text: string) => text,
} as any;
}
function rendered(
component: { render(width: number): string[] } | undefined,
): string {
if (!component) throw new Error("expected rendered component");
return stripVTControlCharacters(component.render(120).join("\n"))
.replaceAll(/\s+/gu, " ")
.trim();
}
function dependencies() {
const runConsultation = vi.fn<
NonNullable<AngelDependencies["runConsultation"]>
>(async (_ctx, request, options) => {
options.onProgress?.({
stage: "investigating",
message: "Inspecting evidence",
});
return result(request.origin);
});
return {
runConsultation,
loadSettings: () => ({
pairs: [{ executor: "openai/gpt-4o", advisor: "openai/gpt-4o" }],
thinkingLevel: "max" as const,
}),
};
}
describe("angel extension runtime", () => {
let testSession: TestSession | undefined;
let testKeybindings: KeybindingsManager;
let restoreKeybindings = () => {};
beforeAll(async () => {
initTheme("default", false);
const codingAgentEntry = import.meta.resolve(
"@earendil-works/pi-coding-agent",
);
const piTuiEntry = createRequire(codingAgentEntry).resolve(
"@earendil-works/pi-tui",
);
const piTui = (await import(
pathToFileURL(piTuiEntry).href
)) as typeof import("@earendil-works/pi-tui");
const previousKeybindings = piTui.getKeybindings();
testKeybindings = new piTui.KeybindingsManager({
"app.tools.expand": {
defaultKeys: "ctrl+o",
description: "Toggle tool output",
},
});
piTui.setKeybindings(testKeybindings);
restoreKeybindings = () => piTui.setKeybindings(previousKeybindings);
});
afterAll(() => restoreKeybindings());
afterEach(() => {
testSession?.dispose();
testSession = undefined;
vi.restoreAllMocks();
});
it("loads with a strong tool contract and no bash gate", async () => {
const deps = dependencies();
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
const tool = runner.extensions[0]?.tools.get("angel")?.definition;
expect(tool?.description).toContain("independent investigation");
expect(tool?.description).toContain("evidence-backed advice");
expect(tool?.promptGuidelines).toHaveLength(4);
expect(runner.hasHandlers("tool_call")).toBe(false);
});
it("uses the brain icon and native tool expansion across consultation states", async () => {
testSession = await createTestSession({
extensionFactories: [createAngelExtension(dependencies())],
});
const tool =
testSession.session.extensionRunner.extensions[0]?.tools.get(
"angel",
)?.definition;
expect(tool).toBeDefined();
const question = `Which explanation fits ${"the available evidence ".repeat(6)}?`;
const context = "Only the integration environment reproduces the failure.";
const collapsedCall = rendered(
tool?.renderCall?.({ question, context }, theme(), {
expanded: false,
} as any),
);
const expandedCall = rendered(
tool?.renderCall?.({ question, context }, theme(), {
expanded: true,
} as any),
);
expect(collapsedCall).toContain(" angel");
expect(collapsedCall).toContain("ctrl+o to expand");
testKeybindings.setUserBindings({ "app.tools.expand": "alt+o" });
expect(
rendered(
tool?.renderCall?.({ question, context }, theme(), {
expanded: false,
} as any),
),
).toContain("alt+o to expand");
testKeybindings.setUserBindings({});
expect(collapsedCall).not.toContain(context);
expect(expandedCall).toContain(question);
expect(expandedCall).toContain(context);
expect(expandedCall).not.toContain("to expand");
expect(() =>
tool?.renderCall?.({}, theme(), { expanded: false } as any),
).not.toThrow();
const preview = "Provisional investigation text";
const progress = {
content: [{ type: "text", text: "Inspecting" }],
details: {
kind: "progress",
progress: { stage: "investigating", message: "Inspecting", preview },
},
};
const collapsedProgress = rendered(
tool?.renderResult?.(
progress,
{ expanded: false, isPartial: true },
theme(),
{} as any,
),
);
const expandedProgress = rendered(
tool?.renderResult?.(
progress,
{ expanded: true, isPartial: true },
theme(),
{} as any,
),
);
expect(collapsedProgress).toContain(" Inspecting");
expect(collapsedProgress).not.toContain(preview);
expect(expandedProgress).toContain(preview);
expect(expandedProgress).not.toContain("👼");
const consultation = result("executor");
const complete = {
content: [{ type: "text", text: consultation.advice }],
details: { kind: "complete", result: consultation },
};
const collapsedComplete = rendered(
tool?.renderResult?.(
complete,
{ expanded: false, isPartial: false },
theme(),
{} as any,
),
);
const expandedComplete = rendered(
tool?.renderResult?.(
complete,
{ expanded: true, isPartial: false },
theme(),
{} as any,
),
);
expect(collapsedComplete).toContain(" Advice ready");
expect(
`${collapsedCall}\n${collapsedComplete}`.match(/to expand/gu),
).toHaveLength(1);
expect(expandedComplete).toContain(consultation.metadata.childSessionFile);
expect(expandedComplete).not.toContain(consultation.advice);
const failure = {
content: [{ type: "text", text: "Angel process failed\nEND-DIAGNOSTIC" }],
details: {},
};
const collapsedFailure = rendered(
tool?.renderResult?.(
failure,
{ expanded: false, isPartial: false },
theme(),
{} as any,
),
);
const expandedFailure = rendered(
tool?.renderResult?.(
failure,
{ expanded: true, isPartial: false },
theme(),
{} as any,
),
);
expect(collapsedFailure).toBe("Angel failed");
expect(expandedFailure).toContain("END-DIAGNOSTIC");
});
it("publishes and clears its fallback availability status", async () => {
testSession = await createTestSession({
extensionFactories: [createAngelExtension(dependencies())],
});
expect(testSession.events.uiCallsFor("setStatus").at(-1)?.args[1]).toMatch(
/^ angel: gpt-4o:/u,
);
const runner = testSession.session.extensionRunner;
await runner
.getCommand("angel")
?.handler("off", runner.createCommandContext());
expect(
testSession.events.uiCallsFor("setStatus").at(-1)?.args[1],
).toBeUndefined();
});
it("returns tool advice once and appends one UI-only advice entry", async () => {
const deps = dependencies();
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
await testSession.run(
when("Need a second opinion", [
calls("angel", { question: "Which approach is supported?" }),
says("I will use the evidence."),
]),
);
expect(deps.runConsultation).toHaveBeenCalledOnce();
expect(deps.runConsultation.mock.calls[0]?.[1]).toMatchObject({
origin: "executor",
question: "Which approach is supported?",
});
const toolResult = testSession.events.toolResultsFor("angel")[0];
expect(toolResult?.text).toContain("executor advice");
const adviceEntries = testSession.session.sessionManager
.getEntries()
.filter(
(entry: { type: string; customType?: string }) =>
entry.type === "custom" && entry.customType === "angel-advice",
);
expect(adviceEntries).toHaveLength(1);
expect(
testSession.session.sessionManager
.buildSessionContext()
.messages.filter(
(message: { role: string; customType?: string }) =>
message.role === "custom" && message.customType === "angel-advice",
),
).toHaveLength(0);
});
it.each(["steer", "followUp"] as const)(
"keeps executor advice alive across queued %s input",
async (delivery) => {
const deps = dependencies();
let release: (() => void) | undefined;
let signal: AbortSignal | undefined;
let markStarted: (() => void) | undefined;
const started = new Promise<void>((resolve) => {
markStarted = resolve;
});
deps.runConsultation.mockImplementation(
(_ctx, request, options) =>
new Promise((resolve) => {
signal = options.signal;
release = () => resolve(result(request.origin));
markStarted?.();
}),
);
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const running = testSession.run(
when("Investigate", [
calls("angel", { question: "What is the cause?" }),
says("First task finished."),
...(delivery === "followUp" ? [says("Next task finished.")] : []),
]),
);
await started;
await testSession.session[delivery]("Next task");
expect(signal?.aborted).toBe(false);
release?.();
await running;
expect(testSession.events.toolResultsFor("angel")[0]?.text).toContain(
"executor advice",
);
expect(
testSession.session.sessionManager
.getEntries()
.filter(
(entry: { type: string; customType?: string }) =>
entry.type === "custom" && entry.customType === "angel-advice",
),
).toHaveLength(1);
expect(
testSession.session.messages.some(
(message) =>
message.role === "user" &&
JSON.stringify(message).includes("Next task"),
),
).toBe(true);
expect(
testSession.session.sessionManager
.buildSessionContext()
.messages.filter(
(message: { role: string; customType?: string }) =>
message.role === "custom" &&
message.customType === "angel-advice",
),
).toHaveLength(0);
},
);
it("starts executor work after ordinary input arrives during authentication", async () => {
const deps = dependencies();
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
const context = runner.createCommandContext();
const model = context.model;
if (!model) throw new Error("test model unavailable");
const auth = await context.modelRegistry.getApiKeyAndHeaders(model);
let releaseAuth: ((value: typeof auth) => void) | undefined;
vi.spyOn(context.modelRegistry, "getApiKeyAndHeaders").mockImplementation(
() =>
new Promise((resolve) => {
releaseAuth = resolve;
}),
);
const running = testSession.run(
when("Investigate", [
calls("angel", { question: "What is the cause?" }),
says("Done."),
]),
);
await vi.waitFor(() => expect(releaseAuth).toBeDefined());
await runner.emit({
type: "input",
text: "Follow up",
source: "interactive",
});
releaseAuth?.(auth);
await running;
expect(deps.runConsultation).toHaveBeenCalledOnce();
expect(testSession.events.toolResultsFor("angel")[0]?.text).toContain(
"executor advice",
);
});
it("injects advice only after the same opaque operation fails again", async () => {
const deps = dependencies();
const contexts: unknown[][] = [];
testSession = await createTestSession({
propagateErrors: false,
extensionFactories: [
(pi) => {
pi.registerTool({
name: "failing_tool",
label: "Failing Tool",
description: "Fail deterministically",
parameters: {
type: "object",
properties: {},
additionalProperties: false,
},
async execute() {
throw new Error("deterministic failure");
},
});
pi.on("tool_call", (event) => {
if (event.toolName === "failing_tool")
return { block: true, reason: "deterministic failure" };
});
pi.on("context", (event) => {
contexts.push(event.messages);
});
},
createAngelExtension(deps),
],
});
await testSession.run(
when("Recover from this", [
calls("failing_tool"),
calls("failing_tool"),
says("Recovered after advice."),
]),
);
expect(deps.runConsultation).toHaveBeenCalledOnce();
expect(contexts).toHaveLength(3);
expect(JSON.stringify(convertToLlm(contexts[1] as never))).not.toContain(
"error advice",
);
const recoveryContext = JSON.stringify(convertToLlm(contexts[2] as never));
expect(recoveryContext).toContain("error advice");
expect(recoveryContext).not.toContain("child-error");
});
it("references the completed batch without duplicating result bodies", async () => {
const deps = dependencies();
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 0,
message: assistantCall("bash", "private-first", { command: "check" }),
toolResults: [toolResult("bash", "private-first")],
});
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 1,
message: fauxAssistantMessage([
fauxToolCall("bash", { command: "check" }, { id: "private-repeat" }),
fauxToolCall(
"opaque_tool",
{ target: "sibling" },
{ id: "private-sibling" },
),
]),
toolResults: [
{
...toolResult("bash", "private-repeat", true, "SECRET RESULT BODY"),
details: { privateValue: "SECRET DETAILS" },
},
toolResult("opaque_tool", "private-sibling", false, "SECRET SIBLING"),
],
});
const request = deps.runConsultation.mock.calls[0]?.[1];
expect(request?.extraContext).toContain("private-repeat");
expect(request?.extraContext).toContain("private-sibling");
expect(request?.extraContext).not.toContain("SECRET RESULT BODY");
expect(request?.extraContext).not.toContain("SECRET DETAILS");
expect(request?.extraContext).not.toContain("SECRET SIBLING");
});
it("consults for further distinct stalled operations without a session quota", async () => {
const deps = dependencies();
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
for (let index = 0; index < 6; index++) {
const firstA = `${index}-first-a`;
const firstB = `${index}-first-b`;
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: index * 2,
message: fauxAssistantMessage([
fauxToolCall("bash", { command: `check-${index}-a` }, { id: firstA }),
fauxToolCall("bash", { command: `check-${index}-b` }, { id: firstB }),
]),
toolResults: [
toolResult("bash", firstA, true, `failure ${firstA}`),
toolResult("bash", firstB, true, `failure ${firstB}`),
],
});
const repeatA = `${index}-repeat-a`;
const repeatB = `${index}-repeat-b`;
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: index * 2 + 1,
message: fauxAssistantMessage([
fauxToolCall(
"bash",
{ command: `check-${index}-a` },
{ id: repeatA },
),
fauxToolCall(
"bash",
{ command: `check-${index}-b` },
{ id: repeatB },
),
]),
toolResults: [
toolResult("bash", repeatA, true, `failure ${repeatA}`),
toolResult("bash", repeatB, true, `failure ${repeatB}`),
],
});
}
expect(deps.runConsultation).toHaveBeenCalledTimes(6);
for (const [index, call] of deps.runConsultation.mock.calls.entries()) {
expect(call[1]).toMatchObject({ origin: "error" });
expect(call[1].question).toContain("2 operations");
expect(call[1].extraContext).toContain(`${index}-repeat-a`);
expect(call[1].extraContext).toContain(`${index}-repeat-b`);
expect(call[1].extraContext).not.toContain(`failure ${index}-repeat-a`);
expect(call[1].extraContext).not.toContain(`failure ${index}-repeat-b`);
}
const messages = testSession.session.sessionManager
.getEntries()
.filter(
(entry: { type: string; customType?: string }) =>
entry.type === "custom_message" &&
entry.customType === "angel-advice",
);
expect(messages).toHaveLength(6);
});
it("does not consult for successful results, explicit cancellation, or Angel errors", async () => {
const deps = dependencies();
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 0,
message: fauxAssistantMessage([
fauxToolCall("read", { path: "README.md" }, { id: "ok" }),
fauxToolCall("bash", { command: "long task" }, { id: "cancel" }),
fauxToolCall("angel", { question: "why?" }, { id: "angel-error" }),
]),
toolResults: [
toolResult("read", "ok", false, "ok"),
toolResult("bash", "cancel", true, "partial output\n\nCommand aborted"),
toolResult("angel", "angel-error"),
],
});
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 1,
message: assistantCall("bash", "cancel-again", { command: "long task" }),
toolResults: [
toolResult(
"bash",
"cancel-again",
true,
"partial output\n\nOperation aborted",
),
],
});
expect(deps.runConsultation).not.toHaveBeenCalled();
});
it("excludes only known low-signal tools registered as Pi built-ins", async () => {
const deps = dependencies();
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
for (const toolName of ["read", "edit", "write", "grep", "find", "ls"]) {
const arguments_ = { target: `same-${toolName}` };
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 0,
message: assistantCall(toolName, `${toolName}-first`, arguments_),
toolResults: [toolResult(toolName, `${toolName}-first`)],
});
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 1,
message: assistantCall(toolName, `${toolName}-repeat`, arguments_),
toolResults: [toolResult(toolName, `${toolName}-repeat`)],
});
}
expect(deps.runConsultation).not.toHaveBeenCalled();
});
it("does not exclude an extension override that uses a built-in name", async () => {
const deps = dependencies();
testSession = await createTestSession({
extensionFactories: [
(pi) => {
pi.registerTool({
name: "read",
label: "Replacement Read",
description: "Override used to verify provenance",
parameters: { type: "object", properties: {} },
execute: async () => ({
content: [{ type: "text", text: "unused" }],
details: {},
}),
});
},
createAngelExtension(deps),
],
});
const runner = testSession.session.extensionRunner;
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 0,
message: assistantCall("read", "override-first", { target: "same" }),
toolResults: [toolResult("read", "override-first")],
});
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 1,
message: assistantCall("read", "override-repeat", { target: "same" }),
toolResults: [toolResult("read", "override-repeat")],
});
expect(deps.runConsultation).toHaveBeenCalledOnce();
});
it("handles unknown extension tools opaquely with canonical arguments", async () => {
const deps = dependencies();
testSession = await createTestSession({
extensionFactories: [
(pi) => {
pi.registerTool({
name: "opaque_tool",
label: "Opaque Tool",
description: "An unrelated extension tool",
parameters: { type: "object", properties: {} },
execute: async () => ({
content: [{ type: "text", text: "unused" }],
details: {},
}),
});
},
createAngelExtension(deps),
],
});
const runner = testSession.session.extensionRunner;
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 0,
message: assistantCall("opaque_tool", "opaque-first", {
alpha: 1,
nested: { beta: 2, gamma: 3 },
}),
toolResults: [toolResult("opaque_tool", "opaque-first")],
});
expect(deps.runConsultation).not.toHaveBeenCalled();
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 1,
message: assistantCall("opaque_tool", "opaque-repeat", {
nested: { gamma: 3, beta: 2 },
alpha: 1,
}),
toolResults: [toolResult("opaque_tool", "opaque-repeat")],
});
expect(deps.runConsultation).toHaveBeenCalledOnce();
expect(deps.runConsultation.mock.calls[0]?.[1].question).toContain(
"1 operation that failed again",
);
});
it("keys opaque operations from arguments after tool-call mutation", async () => {
const deps = dependencies();
const executedValues: string[] = [];
testSession = await createTestSession({
propagateErrors: true,
extensionFactories: [
(pi) => {
pi.registerTool({
name: "mutated_tool",
label: "Mutated Tool",
description: "Receives normalized arguments",
parameters: {
type: "object",
properties: { value: { type: "string" } },
required: ["value"],
},
execute: async (_id, params) => {
executedValues.push(params.value);
throw new Error("deterministic failure");
},
});
pi.on("tool_call", (event) => {
if (event.toolName === "mutated_tool")
event.input.value = "normalized";
});
},
createAngelExtension(deps),
],
});
await testSession.run(
when("Exercise normalized arguments", [
calls("mutated_tool", { value: "first" }),
calls("mutated_tool", { value: "second" }),
says("Recovered after advice."),
]),
);
expect(executedValues).toEqual(["normalized", "normalized"]);
const finalized = testSession.session.messages.filter(
(message) =>
message.role === "toolResult" && message.toolName === "mutated_tool",
);
expect(finalized).toHaveLength(2);
expect(finalized.every((message) => message.isError)).toBe(true);
expect(deps.runConsultation).toHaveBeenCalledOnce();
});
it("treats changed arguments and matching success as new incidents", async () => {
const deps = dependencies();
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
const emit = async (
turnIndex: number,
id: string,
command: string,
isError: boolean,
) =>
emitTurnEnd(runner, {
type: "turn_end",
turnIndex,
message: assistantCall("bash", id, { command }),
toolResults: [toolResult("bash", id, isError)],
});
await emit(0, "one-fails", "check one", true);
await emit(1, "two-fails", "check two", true);
await emit(2, "one-succeeds", "check one", false);
await emit(3, "one-fails-new", "check one", true);
expect(deps.runConsultation).not.toHaveBeenCalled();
await emit(4, "one-repeats", "check one", true);
expect(deps.runConsultation).toHaveBeenCalledOnce();
await emit(5, "one-still-fails", "check one", true);
expect(deps.runConsultation).toHaveBeenCalledOnce();
await emit(6, "one-recovers", "check one", false);
await emit(7, "one-new-incident", "check one", true);
await emit(8, "one-new-repeat", "check one", true);
expect(deps.runConsultation).toHaveBeenCalledTimes(2);
});
it("preserves unrelated pending failures after consultation", async () => {
const deps = dependencies();
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
const emitFailure = async (
turnIndex: number,
id: string,
command: string,
) =>
emitTurnEnd(runner, {
type: "turn_end",
turnIndex,
message: assistantCall("bash", id, { command }),
toolResults: [toolResult("bash", id)],
});
await emitFailure(0, "a-first", "check a");
await emitFailure(1, "b-first", "check b");
await emitFailure(2, "b-repeat", "check b");
expect(deps.runConsultation).toHaveBeenCalledOnce();
await emitFailure(3, "a-repeat", "check a");
expect(deps.runConsultation).toHaveBeenCalledTimes(2);
});
it("lets a matching success win over a same-batch failure", async () => {
const deps = dependencies();
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 0,
message: assistantCall("bash", "first", { command: "check" }),
toolResults: [toolResult("bash", "first")],
});
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 1,
message: fauxAssistantMessage([
fauxToolCall("bash", { command: "check" }, { id: "success" }),
fauxToolCall("bash", { command: "check" }, { id: "failure" }),
]),
toolResults: [
toolResult("bash", "success", false),
toolResult("bash", "failure"),
],
});
expect(deps.runConsultation).not.toHaveBeenCalled();
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 2,
message: assistantCall("bash", "new-first", { command: "check" }),
toolResults: [toolResult("bash", "new-first")],
});
expect(deps.runConsultation).not.toHaveBeenCalled();
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 3,
message: assistantCall("bash", "new-repeat", { command: "check" }),
toolResults: [toolResult("bash", "new-repeat")],
});
expect(deps.runConsultation).toHaveBeenCalledOnce();
});
it("clears remembered failures when a new human task starts", async () => {
const deps = dependencies();
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 0,
message: assistantCall("bash", "old-task", { command: "check" }),
toolResults: [toolResult("bash", "old-task")],
});
await runner.emit({
type: "input",
text: "A different task",
source: "interactive",
});
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 0,
message: assistantCall("bash", "new-task", { command: "check" }),
toolResults: [toolResult("bash", "new-task")],
});
expect(deps.runConsultation).not.toHaveBeenCalled();
});
it.each([
["steer", false],
["followUp", true],
] as const)(
"treats queued %s delivery as a new task",
async (delivery, endsBeforeDelivery) => {
const deps = dependencies();
let releaseFirst: (() => void) | undefined;
let markStarted: (() => void) | undefined;
const started = new Promise<void>((resolve) => {
markStarted = resolve;
});
let execution = 0;
testSession = await createTestSession({
propagateErrors: true,
extensionFactories: [
(pi) => {
pi.registerTool({
name: "queued_failure",
label: "Queued Failure",
description: "Fails around queued user-message delivery",
parameters: {
type: "object",
properties: { value: { type: "string" } },
required: ["value"],
},
execute: async () => {
execution += 1;
if (execution === 1) {
markStarted?.();
await new Promise<void>((resolve) => {
releaseFirst = resolve;
});
}
throw new Error("deterministic failure");
},
});
},
createAngelExtension(deps),
],
});
const actions = endsBeforeDelivery
? [
calls("queued_failure", { value: "same" }),
says("First task stopped."),
calls("queued_failure", { value: "same" }),
says("Second task stopped."),
]
: [
calls("queued_failure", { value: "same" }),
calls("queued_failure", { value: "same" }),
says("Second task stopped."),
];
const running = testSession.run(when("Initial task", actions));
await started;
await testSession.session[delivery]("Replacement task");
releaseFirst?.();
await running;
expect(deps.runConsultation).not.toHaveBeenCalled();
},
);
it("rejects an old turn that ends after new input arrives", async () => {
const deps = dependencies();
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
await runner.emit({
type: "turn_start",
turnIndex: 0,
timestamp: Date.now(),
});
await runner.emit({
type: "input",
text: "Replace the old task",
source: "interactive",
});
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 0,
message: assistantCall("bash", "stale", { command: "check" }),
toolResults: [toolResult("bash", "stale")],
});
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 0,
message: assistantCall("bash", "current", { command: "check" }),
toolResults: [toolResult("bash", "current")],
});
expect(deps.runConsultation).not.toHaveBeenCalled();
});
it("skips results whose originating structured call is unavailable", async () => {
const deps = dependencies();
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
for (let turnIndex = 0; turnIndex < 2; turnIndex++) {
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex,
message: fauxAssistantMessage("no structured call"),
toolResults: [toolResult("opaque_tool", `missing-${turnIndex}`)],
});
}
expect(deps.runConsultation).not.toHaveBeenCalled();
});
it("does not track failures while the advisor is unavailable", async () => {
const deps = dependencies();
deps.loadSettings = () => ({ pairs: [] });
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
for (let turnIndex = 0; turnIndex < 2; turnIndex++) {
const id = `unavailable-${turnIndex}`;
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex,
message: assistantCall("bash", id, { command: "check" }),
toolResults: [toolResult("bash", id)],
});
}
expect(testSession.session.getActiveToolNames()).not.toContain("angel");
expect(deps.runConsultation).not.toHaveBeenCalled();
expect(
testSession.session.messages.some(
(message) =>
message.role === "custom" && message.customType === "angel-status",
),
).toBe(false);
});
it("does not count duplicate failures from one completed turn", async () => {
const deps = dependencies();
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 0,
message: fauxAssistantMessage([
fauxToolCall("bash", { command: "same" }, { id: "same-a" }),
fauxToolCall("bash", { command: "same" }, { id: "same-b" }),
]),
toolResults: [toolResult("bash", "same-a"), toolResult("bash", "same-b")],
});
expect(deps.runConsultation).not.toHaveBeenCalled();
});
it("lets session-local on and off override the configured initial state", async () => {
const deps = dependencies();
deps.loadSettings = () => ({
pairs: [{ executor: "openai/gpt-4o", advisor: "openai/gpt-4o" }],
enabled: false,
});
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
const command = runner.getCommand("angel");
const context = {
...runner.createCommandContext(),
mode: "print" as const,
hasUI: false,
};
expect(testSession.session.getActiveToolNames()).not.toContain("angel");
await command?.handler("on", context);
expect(testSession.session.getActiveToolNames()).toContain("angel");
await command?.handler("off", context);
expect(testSession.session.getActiveToolNames()).not.toContain("angel");
});
it("clears remembered failures when session control changes", async () => {
const deps = dependencies();
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
const command = runner.getCommand("angel");
const context = {
...runner.createCommandContext(),
mode: "print" as const,
hasUI: false,
};
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 0,
message: assistantCall("bash", "before-toggle", { command: "check" }),
toolResults: [toolResult("bash", "before-toggle")],
});
await command?.handler("off", context);
await command?.handler("on", context);
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 1,
message: assistantCall("bash", "after-toggle", { command: "check" }),
toolResults: [toolResult("bash", "after-toggle")],
});
expect(deps.runConsultation).not.toHaveBeenCalled();
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 2,
message: assistantCall("bash", "after-toggle-repeat", {
command: "check",
}),
toolResults: [toolResult("bash", "after-toggle-repeat")],
});
expect(deps.runConsultation).toHaveBeenCalledOnce();
});
it("reports advisor authentication failure without starting child work", async () => {
const deps = dependencies();
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
const command = runner.getCommand("angel");
const context = {
...runner.createCommandContext(),
mode: "print" as const,
hasUI: false,
};
vi.spyOn(context.modelRegistry, "getApiKeyAndHeaders").mockResolvedValue({
ok: false,
error: "missing advisor credentials",
});
const write = vi
.spyOn(process.stdout, "write")
.mockImplementation(() => true);
await command?.handler("Can Angel authenticate?", context);
expect(write).toHaveBeenCalledWith(
expect.stringContaining("missing advisor credentials"),
);
expect(deps.runConsultation).not.toHaveBeenCalled();
const failure = testSession.session.messages.find(
(message: { role: string; customType?: string }) =>
message.role === "custom" && message.customType === "angel-status",
) as { content?: string } | undefined;
expect(failure?.content).toContain("missing advisor credentials");
});
it("cancels a non-TUI human consultation without disabling Angel", async () => {
const deps = dependencies();
let finish: ((value: ConsultationResult) => void) | undefined;
deps.runConsultation.mockImplementation(
() =>
new Promise((resolve) => {
finish = resolve;
}),
);
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
const command = runner.getCommand("angel");
const context = {
...runner.createCommandContext(),
mode: "rpc" as const,
hasUI: false,
};
const pending = command?.handler("Investigate this", context);
await vi.waitFor(() => expect(finish).toBeDefined());
await command?.handler("cancel", context);
finish?.(result("human"));
await pending;
expect(testSession.session.getActiveToolNames()).toContain("angel");
expect(
testSession.session.messages.some(
(message) =>
message.role === "custom" && message.customType === "angel-advice",
),
).toBe(false);
});
it("does not deliver in-flight error advice after Angel is disabled", async () => {
const deps = dependencies();
let finish: ((value: ConsultationResult) => void) | undefined;
deps.runConsultation.mockImplementation(
() =>
new Promise((resolve) => {
finish = resolve;
}),
);
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 0,
message: assistantCall("bash", "disable-first", { command: "check" }),
toolResults: [toolResult("bash", "disable-first")],
});
const pending = emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 1,
message: assistantCall("bash", "disable-repeat", { command: "check" }),
toolResults: [toolResult("bash", "disable-repeat")],
});
await vi.waitFor(() => expect(finish).toBeDefined());
const command = runner.getCommand("angel");
await command?.handler("off", {
...runner.createCommandContext(),
mode: "print",
hasUI: false,
});
finish?.(result("error"));
await pending;
expect(
testSession.session.sessionManager
.getEntries()
.filter(
(entry: { type: string; customType?: string }) =>
entry.type === "custom_message" &&
entry.customType === "angel-advice",
),
).toHaveLength(0);
});
it("discards stale automatic advice without aborting it on queued input", async () => {
const deps = dependencies();
let finish: ((value: ConsultationResult) => void) | undefined;
let signal: AbortSignal | undefined;
deps.runConsultation.mockImplementation(
(_ctx, _request, options) =>
new Promise((resolve) => {
signal = options.signal;
finish = resolve;
}),
);
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 0,
message: assistantCall("bash", "first", { command: "check" }),
toolResults: [toolResult("bash", "first")],
});
const pending = emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 1,
message: assistantCall("bash", "repeat", { command: "check" }),
toolResults: [toolResult("bash", "repeat")],
});
await vi.waitFor(() => expect(finish).toBeDefined());
await runner.emit({
type: "input",
text: "A different task",
source: "interactive",
});
expect(signal?.aborted).toBe(false);
finish?.(result("error"));
await pending;
expect(
testSession.session.messages.filter(
(message) =>
message.role === "custom" && message.customType === "angel-advice",
),
).toHaveLength(0);
});
it("reports automatic cancellation without steering failure", async () => {
const deps = dependencies();
deps.runConsultation.mockRejectedValue(
new DOMException("cancelled", "AbortError"),
);
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 0,
message: assistantCall("bash", "cancel-first", { command: "check" }),
toolResults: [toolResult("bash", "cancel-first")],
});
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 1,
message: assistantCall("bash", "cancel-repeat", { command: "check" }),
toolResults: [toolResult("bash", "cancel-repeat")],
});
const status = testSession.session.messages.find(
(message) =>
message.role === "custom" && message.customType === "angel-status",
);
expect(status?.content).toBe("Angel consultation cancelled.");
expect(status?.details).toMatchObject({ kind: "cancelled" });
});
it("does not start child work when shutdown aborts authentication", async () => {
const deps = dependencies();
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
const context = runner.createCommandContext();
const model = context.model;
if (!model) throw new Error("test model unavailable");
const successfulAuth =
await context.modelRegistry.getApiKeyAndHeaders(model);
let finishAuth: ((value: typeof successfulAuth) => void) | undefined;
vi.spyOn(context.modelRegistry, "getApiKeyAndHeaders").mockImplementation(
() =>
new Promise((resolve) => {
finishAuth = resolve;
}),
);
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 0,
message: assistantCall("bash", "failed-first", { command: "check" }),
toolResults: [toolResult("bash", "failed-first")],
});
const pending = emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 1,
message: assistantCall("bash", "failed-repeat", { command: "check" }),
toolResults: [toolResult("bash", "failed-repeat")],
});
await vi.waitFor(() => expect(finishAuth).toBeDefined());
await runner.emit({ type: "session_shutdown", reason: "switch" });
await runner.emit({ type: "session_start", reason: "switch" });
finishAuth?.(successfulAuth);
await pending;
expect(deps.runConsultation).not.toHaveBeenCalled();
});
it("does not deliver a stale consultation after session replacement", async () => {
const deps = dependencies();
let finish: ((value: ConsultationResult) => void) | undefined;
deps.runConsultation.mockImplementation(
() =>
new Promise((resolve) => {
finish = resolve;
}),
);
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 0,
message: assistantCall("bash", "failed-first", { command: "check" }),
toolResults: [toolResult("bash", "failed-first")],
});
const pending = emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 1,
message: assistantCall("bash", "failed-repeat", { command: "check" }),
toolResults: [toolResult("bash", "failed-repeat")],
});
await vi.waitFor(() => expect(finish).toBeDefined());
await runner.emit({ type: "session_shutdown", reason: "switch" });
await runner.emit({ type: "session_start", reason: "switch" });
finish?.(result("error"));
await pending;
expect(
testSession.session.sessionManager
.getEntries()
.filter(
(entry: { type: string; customType?: string }) =>
entry.type === "custom_message" &&
entry.customType === "angel-advice",
),
).toHaveLength(0);
});
it("does not deliver advice after parent tree navigation", async () => {
const deps = dependencies();
let finish: ((value: ConsultationResult) => void) | undefined;
deps.runConsultation.mockImplementation(
() =>
new Promise((resolve) => {
finish = resolve;
}),
);
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
await emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 0,
message: assistantCall("bash", "failed-first", { command: "check" }),
toolResults: [toolResult("bash", "failed-first")],
});
const pending = emitTurnEnd(runner, {
type: "turn_end",
turnIndex: 1,
message: assistantCall("bash", "failed-repeat", { command: "check" }),
toolResults: [toolResult("bash", "failed-repeat")],
});
await vi.waitFor(() => expect(finish).toBeDefined());
await runner.emit({
type: "session_before_tree",
preparation: {
targetId: "target",
oldLeafId: null,
commonAncestorId: null,
entriesToSummarize: [],
userWantsSummary: false,
},
signal: new AbortController().signal,
});
finish?.(result("error"));
await pending;
expect(
testSession.session.sessionManager
.getEntries()
.filter(
(entry: { type: string; customType?: string }) =>
entry.type === "custom_message" &&
entry.customType === "angel-advice",
),
).toHaveLength(0);
});
it("runs a human question without starting an executor turn", async () => {
const deps = dependencies();
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const command = testSession.session.extensionRunner.getCommand("angel");
const context = {
...testSession.session.extensionRunner.createCommandContext(),
mode: "print" as const,
hasUI: false,
};
const assistantCount = testSession.session.messages.filter(
(message: { role: string }) => message.role === "assistant",
).length;
const write = vi
.spyOn(process.stdout, "write")
.mockImplementation(() => true);
await command?.handler("What evidence decides this?", context);
expect(write).toHaveBeenCalledWith(`${result("human").advice}\n`);
expect(deps.runConsultation).toHaveBeenCalledOnce();
expect(deps.runConsultation.mock.calls[0]?.[1]).toMatchObject({
origin: "human",
question: "What evidence decides this?",
});
const consultationOptions = deps.runConsultation.mock.calls[0]?.[2];
expect(consultationOptions).toMatchObject({
loadExtensions: false,
additionalExtensionPaths: [],
});
expect(consultationOptions).not.toHaveProperty("expectedToolNames");
expect(
testSession.session.messages.filter(
(message: { role: string }) => message.role === "assistant",
),
).toHaveLength(assistantCount);
expect(
testSession.session.messages.find(
(message: { role: string; customType?: string }) =>
message.role === "custom" && message.customType === "angel-advice",
),
).toBeDefined();
});
it("passes the configured extension whitelist to the child", async () => {
const deps = dependencies();
deps.loadSettings = () => ({
pairs: [{ executor: "openai/gpt-4o", advisor: "openai/gpt-4o" }],
subagentExtensions: ["web"],
});
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
const command = runner.getCommand("angel");
const context = {
...runner.createCommandContext(),
mode: "print" as const,
hasUI: false,
};
vi.spyOn(process.stdout, "write").mockImplementation(() => true);
await command?.handler("Use web evidence", context);
const options = deps.runConsultation.mock.calls[0]?.[2];
expect(options).toMatchObject({ loadExtensions: false });
expect(options?.additionalExtensionPaths).toHaveLength(1);
expect(
options?.additionalExtensionPaths?.[0]?.replaceAll("\\", "/"),
).toMatch(/extensions\/web\/index\.ts$/);
});
it.each([
["provider failure", new Error("provider unavailable")],
["cancellation", new DOMException("cancelled", "AbortError")],
])(
"reports human %s without starting an executor turn",
async (_label, error) => {
const deps = dependencies();
deps.runConsultation.mockRejectedValue(error);
testSession = await createTestSession({
extensionFactories: [createAngelExtension(deps)],
});
const runner = testSession.session.extensionRunner;
const command = runner.getCommand("angel");
const context = {
...runner.createCommandContext(),
mode: "print" as const,
hasUI: false,
};
const assistantCount = testSession.session.messages.filter(
(message: { role: string }) => message.role === "assistant",
).length;
const write = vi
.spyOn(process.stdout, "write")
.mockImplementation(() => true);
await command?.handler("Investigate this", context);
expect(write).toHaveBeenCalled();
const failure = testSession.session.messages.find(
(message: { role: string; customType?: string }) =>
message.role === "custom" && message.customType === "angel-status",
) as { content?: string } | undefined;
expect(failure?.content).toContain(
error.name === "AbortError" ? "cancelled" : "provider unavailable",
);
expect(
testSession.session.messages.filter(
(message: { role: string }) => message.role === "assistant",
),
).toHaveLength(assistantCount);
},
);
});