import { createRequire } from "node:module"; import { pathToFileURL } from "node:url"; import { stripVTControlCharacters } from "node:util"; import { type AssistantMessage, fauxAssistantMessage, fauxToolCall, type ToolResultMessage, } from "@earendil-works/pi-ai"; import { type BoundaryContextPreview, convertToLlm, type ExtensionRunner, initTheme, } from "@earendil-works/pi-coding-agent"; import type { KeybindingsManager } from "@earendil-works/pi-tui"; import { afterAll, afterEach, beforeAll, describe, expect, it, vi, } from "vitest"; import { calls, createTestSession, says, type TestSession, when, } from "../../../test/harness"; import type { ConsultationOrigin, ConsultationResult } from "../advisor.ts"; import { type AngelDependencies, createAngelExtension } from "../index.ts"; function result(origin: ConsultationOrigin): ConsultationResult { return { advice: `**${origin} advice**\n\nInspect the evidence first.`, metadata: { origin, model: "openai/gpt-4o", thinkingLevel: "max", runtimeMs: 250, tokens: { input: 10, output: 5, cacheRead: 0, cacheWrite: 0, total: 15, }, cost: 0.01, childSessionId: `child-${origin}`, childSessionFile: `/sessions/child-${origin}.jsonl`, }, }; } function assistantCall( toolName: string, toolCallId: string, arguments_: Record = {}, ) { return fauxAssistantMessage( fauxToolCall(toolName, arguments_, { id: toolCallId }), ); } function toolResult( toolName: string, toolCallId: string, isError = true, text = isError ? "failure" : "success", ) { return { role: "toolResult" as const, toolCallId, toolName, content: [{ type: "text" as const, text }], isError, timestamp: Date.now(), }; } const BOUNDARY_CONTEXT: BoundaryContextPreview = { contextEntries: [], contextMessages: [], llmMessages: [], pendingMessages: [], canContinue: false, }; function emitTurnEnd( runner: ExtensionRunner, event: { turnIndex: number; message: AssistantMessage; toolResults: ToolResultMessage[]; }, ) { return runner.emitBoundary( { type: "turn_end", ...event, messageEntryId: `message-${event.turnIndex}`, toolResultEntryIds: event.toolResults.map( (_, index) => `tool-result-${event.turnIndex}-${index}`, ), outcome: "completed", }, () => BOUNDARY_CONTEXT, ); } function theme() { return { fg: (_style: string, text: string) => text, bg: (_style: string, text: string) => text, bold: (text: string) => text, italic: (text: string) => text, strikethrough: (text: string) => text, } as any; } function rendered( component: { render(width: number): string[] } | undefined, ): string { if (!component) throw new Error("expected rendered component"); return stripVTControlCharacters(component.render(120).join("\n")) .replaceAll(/\s+/gu, " ") .trim(); } function dependencies() { const runConsultation = vi.fn< NonNullable >(async (_ctx, request, options) => { options.onProgress?.({ stage: "investigating", message: "Inspecting evidence", }); return result(request.origin); }); return { runConsultation, loadSettings: () => ({ pairs: [{ executor: "openai/gpt-4o", advisor: "openai/gpt-4o" }], thinkingLevel: "max" as const, }), }; } describe("angel extension runtime", () => { let testSession: TestSession | undefined; let testKeybindings: KeybindingsManager; let restoreKeybindings = () => {}; beforeAll(async () => { initTheme("default", false); const codingAgentEntry = import.meta.resolve( "@earendil-works/pi-coding-agent", ); const piTuiEntry = createRequire(codingAgentEntry).resolve( "@earendil-works/pi-tui", ); const piTui = (await import( pathToFileURL(piTuiEntry).href )) as typeof import("@earendil-works/pi-tui"); const previousKeybindings = piTui.getKeybindings(); testKeybindings = new piTui.KeybindingsManager({ "app.tools.expand": { defaultKeys: "ctrl+o", description: "Toggle tool output", }, }); piTui.setKeybindings(testKeybindings); restoreKeybindings = () => piTui.setKeybindings(previousKeybindings); }); afterAll(() => restoreKeybindings()); afterEach(() => { testSession?.dispose(); testSession = undefined; vi.restoreAllMocks(); }); it("loads with a strong tool contract and no bash gate", async () => { const deps = dependencies(); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; const tool = runner.extensions[0]?.tools.get("angel")?.definition; expect(tool?.description).toContain("independent investigation"); expect(tool?.description).toContain("evidence-backed advice"); expect(tool?.promptGuidelines).toHaveLength(4); expect(runner.hasHandlers("tool_call")).toBe(false); }); it("uses the brain icon and native tool expansion across consultation states", async () => { testSession = await createTestSession({ extensionFactories: [createAngelExtension(dependencies())], }); const tool = testSession.session.extensionRunner.extensions[0]?.tools.get( "angel", )?.definition; expect(tool).toBeDefined(); const question = `Which explanation fits ${"the available evidence ".repeat(6)}?`; const context = "Only the integration environment reproduces the failure."; const collapsedCall = rendered( tool?.renderCall?.({ question, context }, theme(), { expanded: false, } as any), ); const expandedCall = rendered( tool?.renderCall?.({ question, context }, theme(), { expanded: true, } as any), ); expect(collapsedCall).toContain("󰧑 angel"); expect(collapsedCall).toContain("ctrl+o to expand"); testKeybindings.setUserBindings({ "app.tools.expand": "alt+o" }); expect( rendered( tool?.renderCall?.({ question, context }, theme(), { expanded: false, } as any), ), ).toContain("alt+o to expand"); testKeybindings.setUserBindings({}); expect(collapsedCall).not.toContain(context); expect(expandedCall).toContain(question); expect(expandedCall).toContain(context); expect(expandedCall).not.toContain("to expand"); expect(() => tool?.renderCall?.({}, theme(), { expanded: false } as any), ).not.toThrow(); const preview = "Provisional investigation text"; const progress = { content: [{ type: "text", text: "Inspecting" }], details: { kind: "progress", progress: { stage: "investigating", message: "Inspecting", preview }, }, }; const collapsedProgress = rendered( tool?.renderResult?.( progress, { expanded: false, isPartial: true }, theme(), {} as any, ), ); const expandedProgress = rendered( tool?.renderResult?.( progress, { expanded: true, isPartial: true }, theme(), {} as any, ), ); expect(collapsedProgress).toContain("󰧑 Inspecting"); expect(collapsedProgress).not.toContain(preview); expect(expandedProgress).toContain(preview); expect(expandedProgress).not.toContain("👼"); const consultation = result("executor"); const complete = { content: [{ type: "text", text: consultation.advice }], details: { kind: "complete", result: consultation }, }; const collapsedComplete = rendered( tool?.renderResult?.( complete, { expanded: false, isPartial: false }, theme(), {} as any, ), ); const expandedComplete = rendered( tool?.renderResult?.( complete, { expanded: true, isPartial: false }, theme(), {} as any, ), ); expect(collapsedComplete).toContain("󰧑 Advice ready"); expect( `${collapsedCall}\n${collapsedComplete}`.match(/to expand/gu), ).toHaveLength(1); expect(expandedComplete).toContain(consultation.metadata.childSessionFile); expect(expandedComplete).not.toContain(consultation.advice); const failure = { content: [{ type: "text", text: "Angel process failed\nEND-DIAGNOSTIC" }], details: {}, }; const collapsedFailure = rendered( tool?.renderResult?.( failure, { expanded: false, isPartial: false }, theme(), {} as any, ), ); const expandedFailure = rendered( tool?.renderResult?.( failure, { expanded: true, isPartial: false }, theme(), {} as any, ), ); expect(collapsedFailure).toBe("Angel failed"); expect(expandedFailure).toContain("END-DIAGNOSTIC"); }); it("publishes and clears its fallback availability status", async () => { testSession = await createTestSession({ extensionFactories: [createAngelExtension(dependencies())], }); expect(testSession.events.uiCallsFor("setStatus").at(-1)?.args[1]).toMatch( /^󰧑 angel: gpt-4o:/u, ); const runner = testSession.session.extensionRunner; await runner .getCommand("angel") ?.handler("off", runner.createCommandContext()); expect( testSession.events.uiCallsFor("setStatus").at(-1)?.args[1], ).toBeUndefined(); }); it("returns tool advice once and appends one UI-only advice entry", async () => { const deps = dependencies(); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); await testSession.run( when("Need a second opinion", [ calls("angel", { question: "Which approach is supported?" }), says("I will use the evidence."), ]), ); expect(deps.runConsultation).toHaveBeenCalledOnce(); expect(deps.runConsultation.mock.calls[0]?.[1]).toMatchObject({ origin: "executor", question: "Which approach is supported?", }); const toolResult = testSession.events.toolResultsFor("angel")[0]; expect(toolResult?.text).toContain("executor advice"); const adviceEntries = testSession.session.sessionManager .getEntries() .filter( (entry: { type: string; customType?: string }) => entry.type === "custom" && entry.customType === "angel-advice", ); expect(adviceEntries).toHaveLength(1); expect( testSession.session.sessionManager .buildSessionContext() .messages.filter( (message: { role: string; customType?: string }) => message.role === "custom" && message.customType === "angel-advice", ), ).toHaveLength(0); }); it.each(["steer", "followUp"] as const)( "keeps executor advice alive across queued %s input", async (delivery) => { const deps = dependencies(); let release: (() => void) | undefined; let signal: AbortSignal | undefined; let markStarted: (() => void) | undefined; const started = new Promise((resolve) => { markStarted = resolve; }); deps.runConsultation.mockImplementation( (_ctx, request, options) => new Promise((resolve) => { signal = options.signal; release = () => resolve(result(request.origin)); markStarted?.(); }), ); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const running = testSession.run( when("Investigate", [ calls("angel", { question: "What is the cause?" }), says("First task finished."), ...(delivery === "followUp" ? [says("Next task finished.")] : []), ]), ); await started; await testSession.session[delivery]("Next task"); expect(signal?.aborted).toBe(false); release?.(); await running; expect(testSession.events.toolResultsFor("angel")[0]?.text).toContain( "executor advice", ); expect( testSession.session.sessionManager .getEntries() .filter( (entry: { type: string; customType?: string }) => entry.type === "custom" && entry.customType === "angel-advice", ), ).toHaveLength(1); expect( testSession.session.messages.some( (message) => message.role === "user" && JSON.stringify(message).includes("Next task"), ), ).toBe(true); expect( testSession.session.sessionManager .buildSessionContext() .messages.filter( (message: { role: string; customType?: string }) => message.role === "custom" && message.customType === "angel-advice", ), ).toHaveLength(0); }, ); it("starts executor work after ordinary input arrives during authentication", async () => { const deps = dependencies(); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; const context = runner.createCommandContext(); const model = context.model; if (!model) throw new Error("test model unavailable"); const auth = await context.modelRegistry.getApiKeyAndHeaders(model); let releaseAuth: ((value: typeof auth) => void) | undefined; vi.spyOn(context.modelRegistry, "getApiKeyAndHeaders").mockImplementation( () => new Promise((resolve) => { releaseAuth = resolve; }), ); const running = testSession.run( when("Investigate", [ calls("angel", { question: "What is the cause?" }), says("Done."), ]), ); await vi.waitFor(() => expect(releaseAuth).toBeDefined()); await runner.emit({ type: "input", text: "Follow up", source: "interactive", }); releaseAuth?.(auth); await running; expect(deps.runConsultation).toHaveBeenCalledOnce(); expect(testSession.events.toolResultsFor("angel")[0]?.text).toContain( "executor advice", ); }); it("injects advice only after the same opaque operation fails again", async () => { const deps = dependencies(); const contexts: unknown[][] = []; testSession = await createTestSession({ propagateErrors: false, extensionFactories: [ (pi) => { pi.registerTool({ name: "failing_tool", label: "Failing Tool", description: "Fail deterministically", parameters: { type: "object", properties: {}, additionalProperties: false, }, async execute() { throw new Error("deterministic failure"); }, }); pi.on("tool_call", (event) => { if (event.toolName === "failing_tool") return { block: true, reason: "deterministic failure" }; }); pi.on("context", (event) => { contexts.push(event.messages); }); }, createAngelExtension(deps), ], }); await testSession.run( when("Recover from this", [ calls("failing_tool"), calls("failing_tool"), says("Recovered after advice."), ]), ); expect(deps.runConsultation).toHaveBeenCalledOnce(); expect(contexts).toHaveLength(3); expect(JSON.stringify(convertToLlm(contexts[1] as never))).not.toContain( "error advice", ); const recoveryContext = JSON.stringify(convertToLlm(contexts[2] as never)); expect(recoveryContext).toContain("error advice"); expect(recoveryContext).not.toContain("child-error"); }); it("references the completed batch without duplicating result bodies", async () => { const deps = dependencies(); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; await emitTurnEnd(runner, { type: "turn_end", turnIndex: 0, message: assistantCall("bash", "private-first", { command: "check" }), toolResults: [toolResult("bash", "private-first")], }); await emitTurnEnd(runner, { type: "turn_end", turnIndex: 1, message: fauxAssistantMessage([ fauxToolCall("bash", { command: "check" }, { id: "private-repeat" }), fauxToolCall( "opaque_tool", { target: "sibling" }, { id: "private-sibling" }, ), ]), toolResults: [ { ...toolResult("bash", "private-repeat", true, "SECRET RESULT BODY"), details: { privateValue: "SECRET DETAILS" }, }, toolResult("opaque_tool", "private-sibling", false, "SECRET SIBLING"), ], }); const request = deps.runConsultation.mock.calls[0]?.[1]; expect(request?.extraContext).toContain("private-repeat"); expect(request?.extraContext).toContain("private-sibling"); expect(request?.extraContext).not.toContain("SECRET RESULT BODY"); expect(request?.extraContext).not.toContain("SECRET DETAILS"); expect(request?.extraContext).not.toContain("SECRET SIBLING"); }); it("consults for further distinct stalled operations without a session quota", async () => { const deps = dependencies(); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; for (let index = 0; index < 6; index++) { const firstA = `${index}-first-a`; const firstB = `${index}-first-b`; await emitTurnEnd(runner, { type: "turn_end", turnIndex: index * 2, message: fauxAssistantMessage([ fauxToolCall("bash", { command: `check-${index}-a` }, { id: firstA }), fauxToolCall("bash", { command: `check-${index}-b` }, { id: firstB }), ]), toolResults: [ toolResult("bash", firstA, true, `failure ${firstA}`), toolResult("bash", firstB, true, `failure ${firstB}`), ], }); const repeatA = `${index}-repeat-a`; const repeatB = `${index}-repeat-b`; await emitTurnEnd(runner, { type: "turn_end", turnIndex: index * 2 + 1, message: fauxAssistantMessage([ fauxToolCall( "bash", { command: `check-${index}-a` }, { id: repeatA }, ), fauxToolCall( "bash", { command: `check-${index}-b` }, { id: repeatB }, ), ]), toolResults: [ toolResult("bash", repeatA, true, `failure ${repeatA}`), toolResult("bash", repeatB, true, `failure ${repeatB}`), ], }); } expect(deps.runConsultation).toHaveBeenCalledTimes(6); for (const [index, call] of deps.runConsultation.mock.calls.entries()) { expect(call[1]).toMatchObject({ origin: "error" }); expect(call[1].question).toContain("2 operations"); expect(call[1].extraContext).toContain(`${index}-repeat-a`); expect(call[1].extraContext).toContain(`${index}-repeat-b`); expect(call[1].extraContext).not.toContain(`failure ${index}-repeat-a`); expect(call[1].extraContext).not.toContain(`failure ${index}-repeat-b`); } const messages = testSession.session.sessionManager .getEntries() .filter( (entry: { type: string; customType?: string }) => entry.type === "custom_message" && entry.customType === "angel-advice", ); expect(messages).toHaveLength(6); }); it("does not consult for successful results, explicit cancellation, or Angel errors", async () => { const deps = dependencies(); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; await emitTurnEnd(runner, { type: "turn_end", turnIndex: 0, message: fauxAssistantMessage([ fauxToolCall("read", { path: "README.md" }, { id: "ok" }), fauxToolCall("bash", { command: "long task" }, { id: "cancel" }), fauxToolCall("angel", { question: "why?" }, { id: "angel-error" }), ]), toolResults: [ toolResult("read", "ok", false, "ok"), toolResult("bash", "cancel", true, "partial output\n\nCommand aborted"), toolResult("angel", "angel-error"), ], }); await emitTurnEnd(runner, { type: "turn_end", turnIndex: 1, message: assistantCall("bash", "cancel-again", { command: "long task" }), toolResults: [ toolResult( "bash", "cancel-again", true, "partial output\n\nOperation aborted", ), ], }); expect(deps.runConsultation).not.toHaveBeenCalled(); }); it("excludes only known low-signal tools registered as Pi built-ins", async () => { const deps = dependencies(); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; for (const toolName of ["read", "edit", "write", "grep", "find", "ls"]) { const arguments_ = { target: `same-${toolName}` }; await emitTurnEnd(runner, { type: "turn_end", turnIndex: 0, message: assistantCall(toolName, `${toolName}-first`, arguments_), toolResults: [toolResult(toolName, `${toolName}-first`)], }); await emitTurnEnd(runner, { type: "turn_end", turnIndex: 1, message: assistantCall(toolName, `${toolName}-repeat`, arguments_), toolResults: [toolResult(toolName, `${toolName}-repeat`)], }); } expect(deps.runConsultation).not.toHaveBeenCalled(); }); it("does not exclude an extension override that uses a built-in name", async () => { const deps = dependencies(); testSession = await createTestSession({ extensionFactories: [ (pi) => { pi.registerTool({ name: "read", label: "Replacement Read", description: "Override used to verify provenance", parameters: { type: "object", properties: {} }, execute: async () => ({ content: [{ type: "text", text: "unused" }], details: {}, }), }); }, createAngelExtension(deps), ], }); const runner = testSession.session.extensionRunner; await emitTurnEnd(runner, { type: "turn_end", turnIndex: 0, message: assistantCall("read", "override-first", { target: "same" }), toolResults: [toolResult("read", "override-first")], }); await emitTurnEnd(runner, { type: "turn_end", turnIndex: 1, message: assistantCall("read", "override-repeat", { target: "same" }), toolResults: [toolResult("read", "override-repeat")], }); expect(deps.runConsultation).toHaveBeenCalledOnce(); }); it("handles unknown extension tools opaquely with canonical arguments", async () => { const deps = dependencies(); testSession = await createTestSession({ extensionFactories: [ (pi) => { pi.registerTool({ name: "opaque_tool", label: "Opaque Tool", description: "An unrelated extension tool", parameters: { type: "object", properties: {} }, execute: async () => ({ content: [{ type: "text", text: "unused" }], details: {}, }), }); }, createAngelExtension(deps), ], }); const runner = testSession.session.extensionRunner; await emitTurnEnd(runner, { type: "turn_end", turnIndex: 0, message: assistantCall("opaque_tool", "opaque-first", { alpha: 1, nested: { beta: 2, gamma: 3 }, }), toolResults: [toolResult("opaque_tool", "opaque-first")], }); expect(deps.runConsultation).not.toHaveBeenCalled(); await emitTurnEnd(runner, { type: "turn_end", turnIndex: 1, message: assistantCall("opaque_tool", "opaque-repeat", { nested: { gamma: 3, beta: 2 }, alpha: 1, }), toolResults: [toolResult("opaque_tool", "opaque-repeat")], }); expect(deps.runConsultation).toHaveBeenCalledOnce(); expect(deps.runConsultation.mock.calls[0]?.[1].question).toContain( "1 operation that failed again", ); }); it("keys opaque operations from arguments after tool-call mutation", async () => { const deps = dependencies(); const executedValues: string[] = []; testSession = await createTestSession({ propagateErrors: true, extensionFactories: [ (pi) => { pi.registerTool({ name: "mutated_tool", label: "Mutated Tool", description: "Receives normalized arguments", parameters: { type: "object", properties: { value: { type: "string" } }, required: ["value"], }, execute: async (_id, params) => { executedValues.push(params.value); throw new Error("deterministic failure"); }, }); pi.on("tool_call", (event) => { if (event.toolName === "mutated_tool") event.input.value = "normalized"; }); }, createAngelExtension(deps), ], }); await testSession.run( when("Exercise normalized arguments", [ calls("mutated_tool", { value: "first" }), calls("mutated_tool", { value: "second" }), says("Recovered after advice."), ]), ); expect(executedValues).toEqual(["normalized", "normalized"]); const finalized = testSession.session.messages.filter( (message) => message.role === "toolResult" && message.toolName === "mutated_tool", ); expect(finalized).toHaveLength(2); expect(finalized.every((message) => message.isError)).toBe(true); expect(deps.runConsultation).toHaveBeenCalledOnce(); }); it("treats changed arguments and matching success as new incidents", async () => { const deps = dependencies(); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; const emit = async ( turnIndex: number, id: string, command: string, isError: boolean, ) => emitTurnEnd(runner, { type: "turn_end", turnIndex, message: assistantCall("bash", id, { command }), toolResults: [toolResult("bash", id, isError)], }); await emit(0, "one-fails", "check one", true); await emit(1, "two-fails", "check two", true); await emit(2, "one-succeeds", "check one", false); await emit(3, "one-fails-new", "check one", true); expect(deps.runConsultation).not.toHaveBeenCalled(); await emit(4, "one-repeats", "check one", true); expect(deps.runConsultation).toHaveBeenCalledOnce(); await emit(5, "one-still-fails", "check one", true); expect(deps.runConsultation).toHaveBeenCalledOnce(); await emit(6, "one-recovers", "check one", false); await emit(7, "one-new-incident", "check one", true); await emit(8, "one-new-repeat", "check one", true); expect(deps.runConsultation).toHaveBeenCalledTimes(2); }); it("preserves unrelated pending failures after consultation", async () => { const deps = dependencies(); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; const emitFailure = async ( turnIndex: number, id: string, command: string, ) => emitTurnEnd(runner, { type: "turn_end", turnIndex, message: assistantCall("bash", id, { command }), toolResults: [toolResult("bash", id)], }); await emitFailure(0, "a-first", "check a"); await emitFailure(1, "b-first", "check b"); await emitFailure(2, "b-repeat", "check b"); expect(deps.runConsultation).toHaveBeenCalledOnce(); await emitFailure(3, "a-repeat", "check a"); expect(deps.runConsultation).toHaveBeenCalledTimes(2); }); it("lets a matching success win over a same-batch failure", async () => { const deps = dependencies(); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; await emitTurnEnd(runner, { type: "turn_end", turnIndex: 0, message: assistantCall("bash", "first", { command: "check" }), toolResults: [toolResult("bash", "first")], }); await emitTurnEnd(runner, { type: "turn_end", turnIndex: 1, message: fauxAssistantMessage([ fauxToolCall("bash", { command: "check" }, { id: "success" }), fauxToolCall("bash", { command: "check" }, { id: "failure" }), ]), toolResults: [ toolResult("bash", "success", false), toolResult("bash", "failure"), ], }); expect(deps.runConsultation).not.toHaveBeenCalled(); await emitTurnEnd(runner, { type: "turn_end", turnIndex: 2, message: assistantCall("bash", "new-first", { command: "check" }), toolResults: [toolResult("bash", "new-first")], }); expect(deps.runConsultation).not.toHaveBeenCalled(); await emitTurnEnd(runner, { type: "turn_end", turnIndex: 3, message: assistantCall("bash", "new-repeat", { command: "check" }), toolResults: [toolResult("bash", "new-repeat")], }); expect(deps.runConsultation).toHaveBeenCalledOnce(); }); it("clears remembered failures when a new human task starts", async () => { const deps = dependencies(); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; await emitTurnEnd(runner, { type: "turn_end", turnIndex: 0, message: assistantCall("bash", "old-task", { command: "check" }), toolResults: [toolResult("bash", "old-task")], }); await runner.emit({ type: "input", text: "A different task", source: "interactive", }); await emitTurnEnd(runner, { type: "turn_end", turnIndex: 0, message: assistantCall("bash", "new-task", { command: "check" }), toolResults: [toolResult("bash", "new-task")], }); expect(deps.runConsultation).not.toHaveBeenCalled(); }); it.each([ ["steer", false], ["followUp", true], ] as const)( "treats queued %s delivery as a new task", async (delivery, endsBeforeDelivery) => { const deps = dependencies(); let releaseFirst: (() => void) | undefined; let markStarted: (() => void) | undefined; const started = new Promise((resolve) => { markStarted = resolve; }); let execution = 0; testSession = await createTestSession({ propagateErrors: true, extensionFactories: [ (pi) => { pi.registerTool({ name: "queued_failure", label: "Queued Failure", description: "Fails around queued user-message delivery", parameters: { type: "object", properties: { value: { type: "string" } }, required: ["value"], }, execute: async () => { execution += 1; if (execution === 1) { markStarted?.(); await new Promise((resolve) => { releaseFirst = resolve; }); } throw new Error("deterministic failure"); }, }); }, createAngelExtension(deps), ], }); const actions = endsBeforeDelivery ? [ calls("queued_failure", { value: "same" }), says("First task stopped."), calls("queued_failure", { value: "same" }), says("Second task stopped."), ] : [ calls("queued_failure", { value: "same" }), calls("queued_failure", { value: "same" }), says("Second task stopped."), ]; const running = testSession.run(when("Initial task", actions)); await started; await testSession.session[delivery]("Replacement task"); releaseFirst?.(); await running; expect(deps.runConsultation).not.toHaveBeenCalled(); }, ); it("rejects an old turn that ends after new input arrives", async () => { const deps = dependencies(); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; await runner.emit({ type: "turn_start", turnIndex: 0, timestamp: Date.now(), }); await runner.emit({ type: "input", text: "Replace the old task", source: "interactive", }); await emitTurnEnd(runner, { type: "turn_end", turnIndex: 0, message: assistantCall("bash", "stale", { command: "check" }), toolResults: [toolResult("bash", "stale")], }); await emitTurnEnd(runner, { type: "turn_end", turnIndex: 0, message: assistantCall("bash", "current", { command: "check" }), toolResults: [toolResult("bash", "current")], }); expect(deps.runConsultation).not.toHaveBeenCalled(); }); it("skips results whose originating structured call is unavailable", async () => { const deps = dependencies(); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; for (let turnIndex = 0; turnIndex < 2; turnIndex++) { await emitTurnEnd(runner, { type: "turn_end", turnIndex, message: fauxAssistantMessage("no structured call"), toolResults: [toolResult("opaque_tool", `missing-${turnIndex}`)], }); } expect(deps.runConsultation).not.toHaveBeenCalled(); }); it("does not track failures while the advisor is unavailable", async () => { const deps = dependencies(); deps.loadSettings = () => ({ pairs: [] }); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; for (let turnIndex = 0; turnIndex < 2; turnIndex++) { const id = `unavailable-${turnIndex}`; await emitTurnEnd(runner, { type: "turn_end", turnIndex, message: assistantCall("bash", id, { command: "check" }), toolResults: [toolResult("bash", id)], }); } expect(testSession.session.getActiveToolNames()).not.toContain("angel"); expect(deps.runConsultation).not.toHaveBeenCalled(); expect( testSession.session.messages.some( (message) => message.role === "custom" && message.customType === "angel-status", ), ).toBe(false); }); it("does not count duplicate failures from one completed turn", async () => { const deps = dependencies(); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; await emitTurnEnd(runner, { type: "turn_end", turnIndex: 0, message: fauxAssistantMessage([ fauxToolCall("bash", { command: "same" }, { id: "same-a" }), fauxToolCall("bash", { command: "same" }, { id: "same-b" }), ]), toolResults: [toolResult("bash", "same-a"), toolResult("bash", "same-b")], }); expect(deps.runConsultation).not.toHaveBeenCalled(); }); it("lets session-local on and off override the configured initial state", async () => { const deps = dependencies(); deps.loadSettings = () => ({ pairs: [{ executor: "openai/gpt-4o", advisor: "openai/gpt-4o" }], enabled: false, }); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; const command = runner.getCommand("angel"); const context = { ...runner.createCommandContext(), mode: "print" as const, hasUI: false, }; expect(testSession.session.getActiveToolNames()).not.toContain("angel"); await command?.handler("on", context); expect(testSession.session.getActiveToolNames()).toContain("angel"); await command?.handler("off", context); expect(testSession.session.getActiveToolNames()).not.toContain("angel"); }); it("clears remembered failures when session control changes", async () => { const deps = dependencies(); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; const command = runner.getCommand("angel"); const context = { ...runner.createCommandContext(), mode: "print" as const, hasUI: false, }; await emitTurnEnd(runner, { type: "turn_end", turnIndex: 0, message: assistantCall("bash", "before-toggle", { command: "check" }), toolResults: [toolResult("bash", "before-toggle")], }); await command?.handler("off", context); await command?.handler("on", context); await emitTurnEnd(runner, { type: "turn_end", turnIndex: 1, message: assistantCall("bash", "after-toggle", { command: "check" }), toolResults: [toolResult("bash", "after-toggle")], }); expect(deps.runConsultation).not.toHaveBeenCalled(); await emitTurnEnd(runner, { type: "turn_end", turnIndex: 2, message: assistantCall("bash", "after-toggle-repeat", { command: "check", }), toolResults: [toolResult("bash", "after-toggle-repeat")], }); expect(deps.runConsultation).toHaveBeenCalledOnce(); }); it("reports advisor authentication failure without starting child work", async () => { const deps = dependencies(); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; const command = runner.getCommand("angel"); const context = { ...runner.createCommandContext(), mode: "print" as const, hasUI: false, }; vi.spyOn(context.modelRegistry, "getApiKeyAndHeaders").mockResolvedValue({ ok: false, error: "missing advisor credentials", }); const write = vi .spyOn(process.stdout, "write") .mockImplementation(() => true); await command?.handler("Can Angel authenticate?", context); expect(write).toHaveBeenCalledWith( expect.stringContaining("missing advisor credentials"), ); expect(deps.runConsultation).not.toHaveBeenCalled(); const failure = testSession.session.messages.find( (message: { role: string; customType?: string }) => message.role === "custom" && message.customType === "angel-status", ) as { content?: string } | undefined; expect(failure?.content).toContain("missing advisor credentials"); }); it("cancels a non-TUI human consultation without disabling Angel", async () => { const deps = dependencies(); let finish: ((value: ConsultationResult) => void) | undefined; deps.runConsultation.mockImplementation( () => new Promise((resolve) => { finish = resolve; }), ); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; const command = runner.getCommand("angel"); const context = { ...runner.createCommandContext(), mode: "rpc" as const, hasUI: false, }; const pending = command?.handler("Investigate this", context); await vi.waitFor(() => expect(finish).toBeDefined()); await command?.handler("cancel", context); finish?.(result("human")); await pending; expect(testSession.session.getActiveToolNames()).toContain("angel"); expect( testSession.session.messages.some( (message) => message.role === "custom" && message.customType === "angel-advice", ), ).toBe(false); }); it("does not deliver in-flight error advice after Angel is disabled", async () => { const deps = dependencies(); let finish: ((value: ConsultationResult) => void) | undefined; deps.runConsultation.mockImplementation( () => new Promise((resolve) => { finish = resolve; }), ); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; await emitTurnEnd(runner, { type: "turn_end", turnIndex: 0, message: assistantCall("bash", "disable-first", { command: "check" }), toolResults: [toolResult("bash", "disable-first")], }); const pending = emitTurnEnd(runner, { type: "turn_end", turnIndex: 1, message: assistantCall("bash", "disable-repeat", { command: "check" }), toolResults: [toolResult("bash", "disable-repeat")], }); await vi.waitFor(() => expect(finish).toBeDefined()); const command = runner.getCommand("angel"); await command?.handler("off", { ...runner.createCommandContext(), mode: "print", hasUI: false, }); finish?.(result("error")); await pending; expect( testSession.session.sessionManager .getEntries() .filter( (entry: { type: string; customType?: string }) => entry.type === "custom_message" && entry.customType === "angel-advice", ), ).toHaveLength(0); }); it("discards stale automatic advice without aborting it on queued input", async () => { const deps = dependencies(); let finish: ((value: ConsultationResult) => void) | undefined; let signal: AbortSignal | undefined; deps.runConsultation.mockImplementation( (_ctx, _request, options) => new Promise((resolve) => { signal = options.signal; finish = resolve; }), ); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; await emitTurnEnd(runner, { type: "turn_end", turnIndex: 0, message: assistantCall("bash", "first", { command: "check" }), toolResults: [toolResult("bash", "first")], }); const pending = emitTurnEnd(runner, { type: "turn_end", turnIndex: 1, message: assistantCall("bash", "repeat", { command: "check" }), toolResults: [toolResult("bash", "repeat")], }); await vi.waitFor(() => expect(finish).toBeDefined()); await runner.emit({ type: "input", text: "A different task", source: "interactive", }); expect(signal?.aborted).toBe(false); finish?.(result("error")); await pending; expect( testSession.session.messages.filter( (message) => message.role === "custom" && message.customType === "angel-advice", ), ).toHaveLength(0); }); it("reports automatic cancellation without steering failure", async () => { const deps = dependencies(); deps.runConsultation.mockRejectedValue( new DOMException("cancelled", "AbortError"), ); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; await emitTurnEnd(runner, { type: "turn_end", turnIndex: 0, message: assistantCall("bash", "cancel-first", { command: "check" }), toolResults: [toolResult("bash", "cancel-first")], }); await emitTurnEnd(runner, { type: "turn_end", turnIndex: 1, message: assistantCall("bash", "cancel-repeat", { command: "check" }), toolResults: [toolResult("bash", "cancel-repeat")], }); const status = testSession.session.messages.find( (message) => message.role === "custom" && message.customType === "angel-status", ); expect(status?.content).toBe("Angel consultation cancelled."); expect(status?.details).toMatchObject({ kind: "cancelled" }); }); it("does not start child work when shutdown aborts authentication", async () => { const deps = dependencies(); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; const context = runner.createCommandContext(); const model = context.model; if (!model) throw new Error("test model unavailable"); const successfulAuth = await context.modelRegistry.getApiKeyAndHeaders(model); let finishAuth: ((value: typeof successfulAuth) => void) | undefined; vi.spyOn(context.modelRegistry, "getApiKeyAndHeaders").mockImplementation( () => new Promise((resolve) => { finishAuth = resolve; }), ); await emitTurnEnd(runner, { type: "turn_end", turnIndex: 0, message: assistantCall("bash", "failed-first", { command: "check" }), toolResults: [toolResult("bash", "failed-first")], }); const pending = emitTurnEnd(runner, { type: "turn_end", turnIndex: 1, message: assistantCall("bash", "failed-repeat", { command: "check" }), toolResults: [toolResult("bash", "failed-repeat")], }); await vi.waitFor(() => expect(finishAuth).toBeDefined()); await runner.emit({ type: "session_shutdown", reason: "switch" }); await runner.emit({ type: "session_start", reason: "switch" }); finishAuth?.(successfulAuth); await pending; expect(deps.runConsultation).not.toHaveBeenCalled(); }); it("does not deliver a stale consultation after session replacement", async () => { const deps = dependencies(); let finish: ((value: ConsultationResult) => void) | undefined; deps.runConsultation.mockImplementation( () => new Promise((resolve) => { finish = resolve; }), ); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; await emitTurnEnd(runner, { type: "turn_end", turnIndex: 0, message: assistantCall("bash", "failed-first", { command: "check" }), toolResults: [toolResult("bash", "failed-first")], }); const pending = emitTurnEnd(runner, { type: "turn_end", turnIndex: 1, message: assistantCall("bash", "failed-repeat", { command: "check" }), toolResults: [toolResult("bash", "failed-repeat")], }); await vi.waitFor(() => expect(finish).toBeDefined()); await runner.emit({ type: "session_shutdown", reason: "switch" }); await runner.emit({ type: "session_start", reason: "switch" }); finish?.(result("error")); await pending; expect( testSession.session.sessionManager .getEntries() .filter( (entry: { type: string; customType?: string }) => entry.type === "custom_message" && entry.customType === "angel-advice", ), ).toHaveLength(0); }); it("does not deliver advice after parent tree navigation", async () => { const deps = dependencies(); let finish: ((value: ConsultationResult) => void) | undefined; deps.runConsultation.mockImplementation( () => new Promise((resolve) => { finish = resolve; }), ); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; await emitTurnEnd(runner, { type: "turn_end", turnIndex: 0, message: assistantCall("bash", "failed-first", { command: "check" }), toolResults: [toolResult("bash", "failed-first")], }); const pending = emitTurnEnd(runner, { type: "turn_end", turnIndex: 1, message: assistantCall("bash", "failed-repeat", { command: "check" }), toolResults: [toolResult("bash", "failed-repeat")], }); await vi.waitFor(() => expect(finish).toBeDefined()); await runner.emit({ type: "session_before_tree", preparation: { targetId: "target", oldLeafId: null, commonAncestorId: null, entriesToSummarize: [], userWantsSummary: false, }, signal: new AbortController().signal, }); finish?.(result("error")); await pending; expect( testSession.session.sessionManager .getEntries() .filter( (entry: { type: string; customType?: string }) => entry.type === "custom_message" && entry.customType === "angel-advice", ), ).toHaveLength(0); }); it("runs a human question without starting an executor turn", async () => { const deps = dependencies(); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const command = testSession.session.extensionRunner.getCommand("angel"); const context = { ...testSession.session.extensionRunner.createCommandContext(), mode: "print" as const, hasUI: false, }; const assistantCount = testSession.session.messages.filter( (message: { role: string }) => message.role === "assistant", ).length; const write = vi .spyOn(process.stdout, "write") .mockImplementation(() => true); await command?.handler("What evidence decides this?", context); expect(write).toHaveBeenCalledWith(`${result("human").advice}\n`); expect(deps.runConsultation).toHaveBeenCalledOnce(); expect(deps.runConsultation.mock.calls[0]?.[1]).toMatchObject({ origin: "human", question: "What evidence decides this?", }); const consultationOptions = deps.runConsultation.mock.calls[0]?.[2]; expect(consultationOptions).toMatchObject({ loadExtensions: false, additionalExtensionPaths: [], }); expect(consultationOptions).not.toHaveProperty("expectedToolNames"); expect( testSession.session.messages.filter( (message: { role: string }) => message.role === "assistant", ), ).toHaveLength(assistantCount); expect( testSession.session.messages.find( (message: { role: string; customType?: string }) => message.role === "custom" && message.customType === "angel-advice", ), ).toBeDefined(); }); it("passes the configured extension whitelist to the child", async () => { const deps = dependencies(); deps.loadSettings = () => ({ pairs: [{ executor: "openai/gpt-4o", advisor: "openai/gpt-4o" }], subagentExtensions: ["web"], }); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; const command = runner.getCommand("angel"); const context = { ...runner.createCommandContext(), mode: "print" as const, hasUI: false, }; vi.spyOn(process.stdout, "write").mockImplementation(() => true); await command?.handler("Use web evidence", context); const options = deps.runConsultation.mock.calls[0]?.[2]; expect(options).toMatchObject({ loadExtensions: false }); expect(options?.additionalExtensionPaths).toHaveLength(1); expect( options?.additionalExtensionPaths?.[0]?.replaceAll("\\", "/"), ).toMatch(/extensions\/web\/index\.ts$/); }); it.each([ ["provider failure", new Error("provider unavailable")], ["cancellation", new DOMException("cancelled", "AbortError")], ])( "reports human %s without starting an executor turn", async (_label, error) => { const deps = dependencies(); deps.runConsultation.mockRejectedValue(error); testSession = await createTestSession({ extensionFactories: [createAngelExtension(deps)], }); const runner = testSession.session.extensionRunner; const command = runner.getCommand("angel"); const context = { ...runner.createCommandContext(), mode: "print" as const, hasUI: false, }; const assistantCount = testSession.session.messages.filter( (message: { role: string }) => message.role === "assistant", ).length; const write = vi .spyOn(process.stdout, "write") .mockImplementation(() => true); await command?.handler("Investigate this", context); expect(write).toHaveBeenCalled(); const failure = testSession.session.messages.find( (message: { role: string; customType?: string }) => message.role === "custom" && message.customType === "angel-status", ) as { content?: string } | undefined; expect(failure?.content).toContain( error.name === "AbortError" ? "cancelled" : "provider unavailable", ); expect( testSession.session.messages.filter( (message: { role: string }) => message.role === "assistant", ), ).toHaveLength(assistantCount); }, ); });