Luigit
repositories / pi-ext

pi-ext

bugabingas pi extensions

owned by admin

extensions/angel/__tests__/harness.test.ts

Raw
import { createRequire } from "node:module";
import { pathToFileURL } from "node:url";
import { stripVTControlCharacters } from "node:util";
import {
	type AssistantMessage,
	fauxAssistantMessage,
	fauxToolCall,
	type ToolResultMessage,
} from "@earendil-works/pi-ai";
import {
	type BoundaryContextPreview,
	convertToLlm,
	type ExtensionRunner,
	initTheme,
} from "@earendil-works/pi-coding-agent";
import type { KeybindingsManager } from "@earendil-works/pi-tui";
import {
	afterAll,
	afterEach,
	beforeAll,
	describe,
	expect,
	it,
	vi,
} from "vitest";
import {
	calls,
	createTestSession,
	says,
	type TestSession,
	when,
} from "../../../test/harness";
import type { ConsultationOrigin, ConsultationResult } from "../advisor.ts";
import { type AngelDependencies, createAngelExtension } from "../index.ts";

function result(origin: ConsultationOrigin): ConsultationResult {
	return {
		advice: `**${origin} advice**\n\nInspect the evidence first.`,
		metadata: {
			origin,
			model: "openai/gpt-4o",
			thinkingLevel: "max",
			runtimeMs: 250,
			tokens: {
				input: 10,
				output: 5,
				cacheRead: 0,
				cacheWrite: 0,
				total: 15,
			},
			cost: 0.01,
			childSessionId: `child-${origin}`,
			childSessionFile: `/sessions/child-${origin}.jsonl`,
		},
	};
}

function assistantCall(
	toolName: string,
	toolCallId: string,
	arguments_: Record<string, unknown> = {},
) {
	return fauxAssistantMessage(
		fauxToolCall(toolName, arguments_, { id: toolCallId }),
	);
}

function toolResult(
	toolName: string,
	toolCallId: string,
	isError = true,
	text = isError ? "failure" : "success",
) {
	return {
		role: "toolResult" as const,
		toolCallId,
		toolName,
		content: [{ type: "text" as const, text }],
		isError,
		timestamp: Date.now(),
	};
}

const BOUNDARY_CONTEXT: BoundaryContextPreview = {
	contextEntries: [],
	contextMessages: [],
	llmMessages: [],
	pendingMessages: [],
	canContinue: false,
};

function emitTurnEnd(
	runner: ExtensionRunner,
	event: {
		turnIndex: number;
		message: AssistantMessage;
		toolResults: ToolResultMessage[];
	},
) {
	return runner.emitBoundary(
		{
			type: "turn_end",
			...event,
			messageEntryId: `message-${event.turnIndex}`,
			toolResultEntryIds: event.toolResults.map(
				(_, index) => `tool-result-${event.turnIndex}-${index}`,
			),
			outcome: "completed",
		},
		() => BOUNDARY_CONTEXT,
	);
}

function theme() {
	return {
		fg: (_style: string, text: string) => text,
		bg: (_style: string, text: string) => text,
		bold: (text: string) => text,
		italic: (text: string) => text,
		strikethrough: (text: string) => text,
	} as any;
}

function rendered(
	component: { render(width: number): string[] } | undefined,
): string {
	if (!component) throw new Error("expected rendered component");
	return stripVTControlCharacters(component.render(120).join("\n"))
		.replaceAll(/\s+/gu, " ")
		.trim();
}

function dependencies() {
	const runConsultation = vi.fn<
		NonNullable<AngelDependencies["runConsultation"]>
	>(async (_ctx, request, options) => {
		options.onProgress?.({
			stage: "investigating",
			message: "Inspecting evidence",
		});
		return result(request.origin);
	});
	return {
		runConsultation,
		loadSettings: () => ({
			pairs: [{ executor: "openai/gpt-4o", advisor: "openai/gpt-4o" }],
			thinkingLevel: "max" as const,
		}),
	};
}

describe("angel extension runtime", () => {
	let testSession: TestSession | undefined;
	let testKeybindings: KeybindingsManager;
	let restoreKeybindings = () => {};

	beforeAll(async () => {
		initTheme("default", false);
		const codingAgentEntry = import.meta.resolve(
			"@earendil-works/pi-coding-agent",
		);
		const piTuiEntry = createRequire(codingAgentEntry).resolve(
			"@earendil-works/pi-tui",
		);
		const piTui = (await import(
			pathToFileURL(piTuiEntry).href
		)) as typeof import("@earendil-works/pi-tui");
		const previousKeybindings = piTui.getKeybindings();
		testKeybindings = new piTui.KeybindingsManager({
			"app.tools.expand": {
				defaultKeys: "ctrl+o",
				description: "Toggle tool output",
			},
		});
		piTui.setKeybindings(testKeybindings);
		restoreKeybindings = () => piTui.setKeybindings(previousKeybindings);
	});

	afterAll(() => restoreKeybindings());

	afterEach(() => {
		testSession?.dispose();
		testSession = undefined;
		vi.restoreAllMocks();
	});

	it("loads with a strong tool contract and no bash gate", async () => {
		const deps = dependencies();
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});

		const runner = testSession.session.extensionRunner;
		const tool = runner.extensions[0]?.tools.get("angel")?.definition;
		expect(tool?.description).toContain("independent investigation");
		expect(tool?.description).toContain("evidence-backed advice");
		expect(tool?.promptGuidelines).toHaveLength(4);
		expect(runner.hasHandlers("tool_call")).toBe(false);
	});

	it("uses the brain icon and native tool expansion across consultation states", async () => {
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(dependencies())],
		});
		const tool =
			testSession.session.extensionRunner.extensions[0]?.tools.get(
				"angel",
			)?.definition;
		expect(tool).toBeDefined();

		const question = `Which explanation fits ${"the available evidence ".repeat(6)}?`;
		const context = "Only the integration environment reproduces the failure.";
		const collapsedCall = rendered(
			tool?.renderCall?.({ question, context }, theme(), {
				expanded: false,
			} as any),
		);
		const expandedCall = rendered(
			tool?.renderCall?.({ question, context }, theme(), {
				expanded: true,
			} as any),
		);
		expect(collapsedCall).toContain("󰧑 angel");
		expect(collapsedCall).toContain("ctrl+o to expand");
		testKeybindings.setUserBindings({ "app.tools.expand": "alt+o" });
		expect(
			rendered(
				tool?.renderCall?.({ question, context }, theme(), {
					expanded: false,
				} as any),
			),
		).toContain("alt+o to expand");
		testKeybindings.setUserBindings({});
		expect(collapsedCall).not.toContain(context);
		expect(expandedCall).toContain(question);
		expect(expandedCall).toContain(context);
		expect(expandedCall).not.toContain("to expand");
		expect(() =>
			tool?.renderCall?.({}, theme(), { expanded: false } as any),
		).not.toThrow();

		const preview = "Provisional investigation text";
		const progress = {
			content: [{ type: "text", text: "Inspecting" }],
			details: {
				kind: "progress",
				progress: { stage: "investigating", message: "Inspecting", preview },
			},
		};
		const collapsedProgress = rendered(
			tool?.renderResult?.(
				progress,
				{ expanded: false, isPartial: true },
				theme(),
				{} as any,
			),
		);
		const expandedProgress = rendered(
			tool?.renderResult?.(
				progress,
				{ expanded: true, isPartial: true },
				theme(),
				{} as any,
			),
		);
		expect(collapsedProgress).toContain("󰧑 Inspecting");
		expect(collapsedProgress).not.toContain(preview);
		expect(expandedProgress).toContain(preview);
		expect(expandedProgress).not.toContain("👼");

		const consultation = result("executor");
		const complete = {
			content: [{ type: "text", text: consultation.advice }],
			details: { kind: "complete", result: consultation },
		};
		const collapsedComplete = rendered(
			tool?.renderResult?.(
				complete,
				{ expanded: false, isPartial: false },
				theme(),
				{} as any,
			),
		);
		const expandedComplete = rendered(
			tool?.renderResult?.(
				complete,
				{ expanded: true, isPartial: false },
				theme(),
				{} as any,
			),
		);
		expect(collapsedComplete).toContain("󰧑 Advice ready");
		expect(
			`${collapsedCall}\n${collapsedComplete}`.match(/to expand/gu),
		).toHaveLength(1);
		expect(expandedComplete).toContain(consultation.metadata.childSessionFile);
		expect(expandedComplete).not.toContain(consultation.advice);

		const failure = {
			content: [{ type: "text", text: "Angel process failed\nEND-DIAGNOSTIC" }],
			details: {},
		};
		const collapsedFailure = rendered(
			tool?.renderResult?.(
				failure,
				{ expanded: false, isPartial: false },
				theme(),
				{} as any,
			),
		);
		const expandedFailure = rendered(
			tool?.renderResult?.(
				failure,
				{ expanded: true, isPartial: false },
				theme(),
				{} as any,
			),
		);
		expect(collapsedFailure).toBe("Angel failed");
		expect(expandedFailure).toContain("END-DIAGNOSTIC");
	});

	it("publishes and clears its fallback availability status", async () => {
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(dependencies())],
		});
		expect(testSession.events.uiCallsFor("setStatus").at(-1)?.args[1]).toMatch(
			/^󰧑 angel: gpt-4o:/u,
		);

		const runner = testSession.session.extensionRunner;
		await runner
			.getCommand("angel")
			?.handler("off", runner.createCommandContext());
		expect(
			testSession.events.uiCallsFor("setStatus").at(-1)?.args[1],
		).toBeUndefined();
	});

	it("returns tool advice once and appends one UI-only advice entry", async () => {
		const deps = dependencies();
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});

		await testSession.run(
			when("Need a second opinion", [
				calls("angel", { question: "Which approach is supported?" }),
				says("I will use the evidence."),
			]),
		);

		expect(deps.runConsultation).toHaveBeenCalledOnce();
		expect(deps.runConsultation.mock.calls[0]?.[1]).toMatchObject({
			origin: "executor",
			question: "Which approach is supported?",
		});
		const toolResult = testSession.events.toolResultsFor("angel")[0];
		expect(toolResult?.text).toContain("executor advice");
		const adviceEntries = testSession.session.sessionManager
			.getEntries()
			.filter(
				(entry: { type: string; customType?: string }) =>
					entry.type === "custom" && entry.customType === "angel-advice",
			);
		expect(adviceEntries).toHaveLength(1);
		expect(
			testSession.session.sessionManager
				.buildSessionContext()
				.messages.filter(
					(message: { role: string; customType?: string }) =>
						message.role === "custom" && message.customType === "angel-advice",
				),
		).toHaveLength(0);
	});

	it.each(["steer", "followUp"] as const)(
		"keeps executor advice alive across queued %s input",
		async (delivery) => {
			const deps = dependencies();
			let release: (() => void) | undefined;
			let signal: AbortSignal | undefined;
			let markStarted: (() => void) | undefined;
			const started = new Promise<void>((resolve) => {
				markStarted = resolve;
			});
			deps.runConsultation.mockImplementation(
				(_ctx, request, options) =>
					new Promise((resolve) => {
						signal = options.signal;
						release = () => resolve(result(request.origin));
						markStarted?.();
					}),
			);
			testSession = await createTestSession({
				extensionFactories: [createAngelExtension(deps)],
			});
			const running = testSession.run(
				when("Investigate", [
					calls("angel", { question: "What is the cause?" }),
					says("First task finished."),
					...(delivery === "followUp" ? [says("Next task finished.")] : []),
				]),
			);
			await started;
			await testSession.session[delivery]("Next task");
			expect(signal?.aborted).toBe(false);
			release?.();
			await running;
			expect(testSession.events.toolResultsFor("angel")[0]?.text).toContain(
				"executor advice",
			);
			expect(
				testSession.session.sessionManager
					.getEntries()
					.filter(
						(entry: { type: string; customType?: string }) =>
							entry.type === "custom" && entry.customType === "angel-advice",
					),
			).toHaveLength(1);
			expect(
				testSession.session.messages.some(
					(message) =>
						message.role === "user" &&
						JSON.stringify(message).includes("Next task"),
				),
			).toBe(true);
			expect(
				testSession.session.sessionManager
					.buildSessionContext()
					.messages.filter(
						(message: { role: string; customType?: string }) =>
							message.role === "custom" &&
							message.customType === "angel-advice",
					),
			).toHaveLength(0);
		},
	);

	it("starts executor work after ordinary input arrives during authentication", async () => {
		const deps = dependencies();
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;
		const context = runner.createCommandContext();
		const model = context.model;
		if (!model) throw new Error("test model unavailable");
		const auth = await context.modelRegistry.getApiKeyAndHeaders(model);
		let releaseAuth: ((value: typeof auth) => void) | undefined;
		vi.spyOn(context.modelRegistry, "getApiKeyAndHeaders").mockImplementation(
			() =>
				new Promise((resolve) => {
					releaseAuth = resolve;
				}),
		);
		const running = testSession.run(
			when("Investigate", [
				calls("angel", { question: "What is the cause?" }),
				says("Done."),
			]),
		);
		await vi.waitFor(() => expect(releaseAuth).toBeDefined());
		await runner.emit({
			type: "input",
			text: "Follow up",
			source: "interactive",
		});
		releaseAuth?.(auth);
		await running;
		expect(deps.runConsultation).toHaveBeenCalledOnce();
		expect(testSession.events.toolResultsFor("angel")[0]?.text).toContain(
			"executor advice",
		);
	});

	it("injects advice only after the same opaque operation fails again", async () => {
		const deps = dependencies();
		const contexts: unknown[][] = [];
		testSession = await createTestSession({
			propagateErrors: false,
			extensionFactories: [
				(pi) => {
					pi.registerTool({
						name: "failing_tool",
						label: "Failing Tool",
						description: "Fail deterministically",
						parameters: {
							type: "object",
							properties: {},
							additionalProperties: false,
						},
						async execute() {
							throw new Error("deterministic failure");
						},
					});
					pi.on("tool_call", (event) => {
						if (event.toolName === "failing_tool")
							return { block: true, reason: "deterministic failure" };
					});
					pi.on("context", (event) => {
						contexts.push(event.messages);
					});
				},
				createAngelExtension(deps),
			],
		});

		await testSession.run(
			when("Recover from this", [
				calls("failing_tool"),
				calls("failing_tool"),
				says("Recovered after advice."),
			]),
		);

		expect(deps.runConsultation).toHaveBeenCalledOnce();
		expect(contexts).toHaveLength(3);
		expect(JSON.stringify(convertToLlm(contexts[1] as never))).not.toContain(
			"error advice",
		);
		const recoveryContext = JSON.stringify(convertToLlm(contexts[2] as never));
		expect(recoveryContext).toContain("error advice");
		expect(recoveryContext).not.toContain("child-error");
	});

	it("references the completed batch without duplicating result bodies", async () => {
		const deps = dependencies();
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 0,
			message: assistantCall("bash", "private-first", { command: "check" }),
			toolResults: [toolResult("bash", "private-first")],
		});
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 1,
			message: fauxAssistantMessage([
				fauxToolCall("bash", { command: "check" }, { id: "private-repeat" }),
				fauxToolCall(
					"opaque_tool",
					{ target: "sibling" },
					{ id: "private-sibling" },
				),
			]),
			toolResults: [
				{
					...toolResult("bash", "private-repeat", true, "SECRET RESULT BODY"),
					details: { privateValue: "SECRET DETAILS" },
				},
				toolResult("opaque_tool", "private-sibling", false, "SECRET SIBLING"),
			],
		});

		const request = deps.runConsultation.mock.calls[0]?.[1];
		expect(request?.extraContext).toContain("private-repeat");
		expect(request?.extraContext).toContain("private-sibling");
		expect(request?.extraContext).not.toContain("SECRET RESULT BODY");
		expect(request?.extraContext).not.toContain("SECRET DETAILS");
		expect(request?.extraContext).not.toContain("SECRET SIBLING");
	});

	it("consults for further distinct stalled operations without a session quota", async () => {
		const deps = dependencies();
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;

		for (let index = 0; index < 6; index++) {
			const firstA = `${index}-first-a`;
			const firstB = `${index}-first-b`;
			await emitTurnEnd(runner, {
				type: "turn_end",
				turnIndex: index * 2,
				message: fauxAssistantMessage([
					fauxToolCall("bash", { command: `check-${index}-a` }, { id: firstA }),
					fauxToolCall("bash", { command: `check-${index}-b` }, { id: firstB }),
				]),
				toolResults: [
					toolResult("bash", firstA, true, `failure ${firstA}`),
					toolResult("bash", firstB, true, `failure ${firstB}`),
				],
			});

			const repeatA = `${index}-repeat-a`;
			const repeatB = `${index}-repeat-b`;
			await emitTurnEnd(runner, {
				type: "turn_end",
				turnIndex: index * 2 + 1,
				message: fauxAssistantMessage([
					fauxToolCall(
						"bash",
						{ command: `check-${index}-a` },
						{ id: repeatA },
					),
					fauxToolCall(
						"bash",
						{ command: `check-${index}-b` },
						{ id: repeatB },
					),
				]),
				toolResults: [
					toolResult("bash", repeatA, true, `failure ${repeatA}`),
					toolResult("bash", repeatB, true, `failure ${repeatB}`),
				],
			});
		}

		expect(deps.runConsultation).toHaveBeenCalledTimes(6);
		for (const [index, call] of deps.runConsultation.mock.calls.entries()) {
			expect(call[1]).toMatchObject({ origin: "error" });
			expect(call[1].question).toContain("2 operations");
			expect(call[1].extraContext).toContain(`${index}-repeat-a`);
			expect(call[1].extraContext).toContain(`${index}-repeat-b`);
			expect(call[1].extraContext).not.toContain(`failure ${index}-repeat-a`);
			expect(call[1].extraContext).not.toContain(`failure ${index}-repeat-b`);
		}
		const messages = testSession.session.sessionManager
			.getEntries()
			.filter(
				(entry: { type: string; customType?: string }) =>
					entry.type === "custom_message" &&
					entry.customType === "angel-advice",
			);
		expect(messages).toHaveLength(6);
	});

	it("does not consult for successful results, explicit cancellation, or Angel errors", async () => {
		const deps = dependencies();
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 0,
			message: fauxAssistantMessage([
				fauxToolCall("read", { path: "README.md" }, { id: "ok" }),
				fauxToolCall("bash", { command: "long task" }, { id: "cancel" }),
				fauxToolCall("angel", { question: "why?" }, { id: "angel-error" }),
			]),
			toolResults: [
				toolResult("read", "ok", false, "ok"),
				toolResult("bash", "cancel", true, "partial output\n\nCommand aborted"),
				toolResult("angel", "angel-error"),
			],
		});
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 1,
			message: assistantCall("bash", "cancel-again", { command: "long task" }),
			toolResults: [
				toolResult(
					"bash",
					"cancel-again",
					true,
					"partial output\n\nOperation aborted",
				),
			],
		});

		expect(deps.runConsultation).not.toHaveBeenCalled();
	});

	it("excludes only known low-signal tools registered as Pi built-ins", async () => {
		const deps = dependencies();
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;

		for (const toolName of ["read", "edit", "write", "grep", "find", "ls"]) {
			const arguments_ = { target: `same-${toolName}` };
			await emitTurnEnd(runner, {
				type: "turn_end",
				turnIndex: 0,
				message: assistantCall(toolName, `${toolName}-first`, arguments_),
				toolResults: [toolResult(toolName, `${toolName}-first`)],
			});
			await emitTurnEnd(runner, {
				type: "turn_end",
				turnIndex: 1,
				message: assistantCall(toolName, `${toolName}-repeat`, arguments_),
				toolResults: [toolResult(toolName, `${toolName}-repeat`)],
			});
		}

		expect(deps.runConsultation).not.toHaveBeenCalled();
	});

	it("does not exclude an extension override that uses a built-in name", async () => {
		const deps = dependencies();
		testSession = await createTestSession({
			extensionFactories: [
				(pi) => {
					pi.registerTool({
						name: "read",
						label: "Replacement Read",
						description: "Override used to verify provenance",
						parameters: { type: "object", properties: {} },
						execute: async () => ({
							content: [{ type: "text", text: "unused" }],
							details: {},
						}),
					});
				},
				createAngelExtension(deps),
			],
		});
		const runner = testSession.session.extensionRunner;
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 0,
			message: assistantCall("read", "override-first", { target: "same" }),
			toolResults: [toolResult("read", "override-first")],
		});
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 1,
			message: assistantCall("read", "override-repeat", { target: "same" }),
			toolResults: [toolResult("read", "override-repeat")],
		});

		expect(deps.runConsultation).toHaveBeenCalledOnce();
	});

	it("handles unknown extension tools opaquely with canonical arguments", async () => {
		const deps = dependencies();
		testSession = await createTestSession({
			extensionFactories: [
				(pi) => {
					pi.registerTool({
						name: "opaque_tool",
						label: "Opaque Tool",
						description: "An unrelated extension tool",
						parameters: { type: "object", properties: {} },
						execute: async () => ({
							content: [{ type: "text", text: "unused" }],
							details: {},
						}),
					});
				},
				createAngelExtension(deps),
			],
		});
		const runner = testSession.session.extensionRunner;

		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 0,
			message: assistantCall("opaque_tool", "opaque-first", {
				alpha: 1,
				nested: { beta: 2, gamma: 3 },
			}),
			toolResults: [toolResult("opaque_tool", "opaque-first")],
		});
		expect(deps.runConsultation).not.toHaveBeenCalled();

		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 1,
			message: assistantCall("opaque_tool", "opaque-repeat", {
				nested: { gamma: 3, beta: 2 },
				alpha: 1,
			}),
			toolResults: [toolResult("opaque_tool", "opaque-repeat")],
		});

		expect(deps.runConsultation).toHaveBeenCalledOnce();
		expect(deps.runConsultation.mock.calls[0]?.[1].question).toContain(
			"1 operation that failed again",
		);
	});

	it("keys opaque operations from arguments after tool-call mutation", async () => {
		const deps = dependencies();
		const executedValues: string[] = [];
		testSession = await createTestSession({
			propagateErrors: true,
			extensionFactories: [
				(pi) => {
					pi.registerTool({
						name: "mutated_tool",
						label: "Mutated Tool",
						description: "Receives normalized arguments",
						parameters: {
							type: "object",
							properties: { value: { type: "string" } },
							required: ["value"],
						},
						execute: async (_id, params) => {
							executedValues.push(params.value);
							throw new Error("deterministic failure");
						},
					});
					pi.on("tool_call", (event) => {
						if (event.toolName === "mutated_tool")
							event.input.value = "normalized";
					});
				},
				createAngelExtension(deps),
			],
		});

		await testSession.run(
			when("Exercise normalized arguments", [
				calls("mutated_tool", { value: "first" }),
				calls("mutated_tool", { value: "second" }),
				says("Recovered after advice."),
			]),
		);

		expect(executedValues).toEqual(["normalized", "normalized"]);
		const finalized = testSession.session.messages.filter(
			(message) =>
				message.role === "toolResult" && message.toolName === "mutated_tool",
		);
		expect(finalized).toHaveLength(2);
		expect(finalized.every((message) => message.isError)).toBe(true);
		expect(deps.runConsultation).toHaveBeenCalledOnce();
	});

	it("treats changed arguments and matching success as new incidents", async () => {
		const deps = dependencies();
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;
		const emit = async (
			turnIndex: number,
			id: string,
			command: string,
			isError: boolean,
		) =>
			emitTurnEnd(runner, {
				type: "turn_end",
				turnIndex,
				message: assistantCall("bash", id, { command }),
				toolResults: [toolResult("bash", id, isError)],
			});

		await emit(0, "one-fails", "check one", true);
		await emit(1, "two-fails", "check two", true);
		await emit(2, "one-succeeds", "check one", false);
		await emit(3, "one-fails-new", "check one", true);
		expect(deps.runConsultation).not.toHaveBeenCalled();
		await emit(4, "one-repeats", "check one", true);
		expect(deps.runConsultation).toHaveBeenCalledOnce();
		await emit(5, "one-still-fails", "check one", true);
		expect(deps.runConsultation).toHaveBeenCalledOnce();
		await emit(6, "one-recovers", "check one", false);
		await emit(7, "one-new-incident", "check one", true);
		await emit(8, "one-new-repeat", "check one", true);
		expect(deps.runConsultation).toHaveBeenCalledTimes(2);
	});

	it("preserves unrelated pending failures after consultation", async () => {
		const deps = dependencies();
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;
		const emitFailure = async (
			turnIndex: number,
			id: string,
			command: string,
		) =>
			emitTurnEnd(runner, {
				type: "turn_end",
				turnIndex,
				message: assistantCall("bash", id, { command }),
				toolResults: [toolResult("bash", id)],
			});

		await emitFailure(0, "a-first", "check a");
		await emitFailure(1, "b-first", "check b");
		await emitFailure(2, "b-repeat", "check b");
		expect(deps.runConsultation).toHaveBeenCalledOnce();
		await emitFailure(3, "a-repeat", "check a");
		expect(deps.runConsultation).toHaveBeenCalledTimes(2);
	});

	it("lets a matching success win over a same-batch failure", async () => {
		const deps = dependencies();
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 0,
			message: assistantCall("bash", "first", { command: "check" }),
			toolResults: [toolResult("bash", "first")],
		});
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 1,
			message: fauxAssistantMessage([
				fauxToolCall("bash", { command: "check" }, { id: "success" }),
				fauxToolCall("bash", { command: "check" }, { id: "failure" }),
			]),
			toolResults: [
				toolResult("bash", "success", false),
				toolResult("bash", "failure"),
			],
		});
		expect(deps.runConsultation).not.toHaveBeenCalled();
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 2,
			message: assistantCall("bash", "new-first", { command: "check" }),
			toolResults: [toolResult("bash", "new-first")],
		});
		expect(deps.runConsultation).not.toHaveBeenCalled();
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 3,
			message: assistantCall("bash", "new-repeat", { command: "check" }),
			toolResults: [toolResult("bash", "new-repeat")],
		});
		expect(deps.runConsultation).toHaveBeenCalledOnce();
	});

	it("clears remembered failures when a new human task starts", async () => {
		const deps = dependencies();
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 0,
			message: assistantCall("bash", "old-task", { command: "check" }),
			toolResults: [toolResult("bash", "old-task")],
		});
		await runner.emit({
			type: "input",
			text: "A different task",
			source: "interactive",
		});
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 0,
			message: assistantCall("bash", "new-task", { command: "check" }),
			toolResults: [toolResult("bash", "new-task")],
		});

		expect(deps.runConsultation).not.toHaveBeenCalled();
	});

	it.each([
		["steer", false],
		["followUp", true],
	] as const)(
		"treats queued %s delivery as a new task",
		async (delivery, endsBeforeDelivery) => {
			const deps = dependencies();
			let releaseFirst: (() => void) | undefined;
			let markStarted: (() => void) | undefined;
			const started = new Promise<void>((resolve) => {
				markStarted = resolve;
			});
			let execution = 0;
			testSession = await createTestSession({
				propagateErrors: true,
				extensionFactories: [
					(pi) => {
						pi.registerTool({
							name: "queued_failure",
							label: "Queued Failure",
							description: "Fails around queued user-message delivery",
							parameters: {
								type: "object",
								properties: { value: { type: "string" } },
								required: ["value"],
							},
							execute: async () => {
								execution += 1;
								if (execution === 1) {
									markStarted?.();
									await new Promise<void>((resolve) => {
										releaseFirst = resolve;
									});
								}
								throw new Error("deterministic failure");
							},
						});
					},
					createAngelExtension(deps),
				],
			});
			const actions = endsBeforeDelivery
				? [
						calls("queued_failure", { value: "same" }),
						says("First task stopped."),
						calls("queued_failure", { value: "same" }),
						says("Second task stopped."),
					]
				: [
						calls("queued_failure", { value: "same" }),
						calls("queued_failure", { value: "same" }),
						says("Second task stopped."),
					];
			const running = testSession.run(when("Initial task", actions));
			await started;
			await testSession.session[delivery]("Replacement task");
			releaseFirst?.();
			await running;

			expect(deps.runConsultation).not.toHaveBeenCalled();
		},
	);

	it("rejects an old turn that ends after new input arrives", async () => {
		const deps = dependencies();
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;
		await runner.emit({
			type: "turn_start",
			turnIndex: 0,
			timestamp: Date.now(),
		});
		await runner.emit({
			type: "input",
			text: "Replace the old task",
			source: "interactive",
		});
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 0,
			message: assistantCall("bash", "stale", { command: "check" }),
			toolResults: [toolResult("bash", "stale")],
		});
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 0,
			message: assistantCall("bash", "current", { command: "check" }),
			toolResults: [toolResult("bash", "current")],
		});

		expect(deps.runConsultation).not.toHaveBeenCalled();
	});

	it("skips results whose originating structured call is unavailable", async () => {
		const deps = dependencies();
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;
		for (let turnIndex = 0; turnIndex < 2; turnIndex++) {
			await emitTurnEnd(runner, {
				type: "turn_end",
				turnIndex,
				message: fauxAssistantMessage("no structured call"),
				toolResults: [toolResult("opaque_tool", `missing-${turnIndex}`)],
			});
		}

		expect(deps.runConsultation).not.toHaveBeenCalled();
	});

	it("does not track failures while the advisor is unavailable", async () => {
		const deps = dependencies();
		deps.loadSettings = () => ({ pairs: [] });
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;
		for (let turnIndex = 0; turnIndex < 2; turnIndex++) {
			const id = `unavailable-${turnIndex}`;
			await emitTurnEnd(runner, {
				type: "turn_end",
				turnIndex,
				message: assistantCall("bash", id, { command: "check" }),
				toolResults: [toolResult("bash", id)],
			});
		}

		expect(testSession.session.getActiveToolNames()).not.toContain("angel");
		expect(deps.runConsultation).not.toHaveBeenCalled();
		expect(
			testSession.session.messages.some(
				(message) =>
					message.role === "custom" && message.customType === "angel-status",
			),
		).toBe(false);
	});

	it("does not count duplicate failures from one completed turn", async () => {
		const deps = dependencies();
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 0,
			message: fauxAssistantMessage([
				fauxToolCall("bash", { command: "same" }, { id: "same-a" }),
				fauxToolCall("bash", { command: "same" }, { id: "same-b" }),
			]),
			toolResults: [toolResult("bash", "same-a"), toolResult("bash", "same-b")],
		});

		expect(deps.runConsultation).not.toHaveBeenCalled();
	});

	it("lets session-local on and off override the configured initial state", async () => {
		const deps = dependencies();
		deps.loadSettings = () => ({
			pairs: [{ executor: "openai/gpt-4o", advisor: "openai/gpt-4o" }],
			enabled: false,
		});
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;
		const command = runner.getCommand("angel");
		const context = {
			...runner.createCommandContext(),
			mode: "print" as const,
			hasUI: false,
		};

		expect(testSession.session.getActiveToolNames()).not.toContain("angel");
		await command?.handler("on", context);
		expect(testSession.session.getActiveToolNames()).toContain("angel");
		await command?.handler("off", context);
		expect(testSession.session.getActiveToolNames()).not.toContain("angel");
	});

	it("clears remembered failures when session control changes", async () => {
		const deps = dependencies();
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;
		const command = runner.getCommand("angel");
		const context = {
			...runner.createCommandContext(),
			mode: "print" as const,
			hasUI: false,
		};
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 0,
			message: assistantCall("bash", "before-toggle", { command: "check" }),
			toolResults: [toolResult("bash", "before-toggle")],
		});
		await command?.handler("off", context);
		await command?.handler("on", context);
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 1,
			message: assistantCall("bash", "after-toggle", { command: "check" }),
			toolResults: [toolResult("bash", "after-toggle")],
		});
		expect(deps.runConsultation).not.toHaveBeenCalled();
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 2,
			message: assistantCall("bash", "after-toggle-repeat", {
				command: "check",
			}),
			toolResults: [toolResult("bash", "after-toggle-repeat")],
		});

		expect(deps.runConsultation).toHaveBeenCalledOnce();
	});

	it("reports advisor authentication failure without starting child work", async () => {
		const deps = dependencies();
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;
		const command = runner.getCommand("angel");
		const context = {
			...runner.createCommandContext(),
			mode: "print" as const,
			hasUI: false,
		};
		vi.spyOn(context.modelRegistry, "getApiKeyAndHeaders").mockResolvedValue({
			ok: false,
			error: "missing advisor credentials",
		});
		const write = vi
			.spyOn(process.stdout, "write")
			.mockImplementation(() => true);

		await command?.handler("Can Angel authenticate?", context);

		expect(write).toHaveBeenCalledWith(
			expect.stringContaining("missing advisor credentials"),
		);
		expect(deps.runConsultation).not.toHaveBeenCalled();
		const failure = testSession.session.messages.find(
			(message: { role: string; customType?: string }) =>
				message.role === "custom" && message.customType === "angel-status",
		) as { content?: string } | undefined;
		expect(failure?.content).toContain("missing advisor credentials");
	});

	it("cancels a non-TUI human consultation without disabling Angel", async () => {
		const deps = dependencies();
		let finish: ((value: ConsultationResult) => void) | undefined;
		deps.runConsultation.mockImplementation(
			() =>
				new Promise((resolve) => {
					finish = resolve;
				}),
		);
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;
		const command = runner.getCommand("angel");
		const context = {
			...runner.createCommandContext(),
			mode: "rpc" as const,
			hasUI: false,
		};
		const pending = command?.handler("Investigate this", context);
		await vi.waitFor(() => expect(finish).toBeDefined());
		await command?.handler("cancel", context);
		finish?.(result("human"));
		await pending;

		expect(testSession.session.getActiveToolNames()).toContain("angel");
		expect(
			testSession.session.messages.some(
				(message) =>
					message.role === "custom" && message.customType === "angel-advice",
			),
		).toBe(false);
	});

	it("does not deliver in-flight error advice after Angel is disabled", async () => {
		const deps = dependencies();
		let finish: ((value: ConsultationResult) => void) | undefined;
		deps.runConsultation.mockImplementation(
			() =>
				new Promise((resolve) => {
					finish = resolve;
				}),
		);
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 0,
			message: assistantCall("bash", "disable-first", { command: "check" }),
			toolResults: [toolResult("bash", "disable-first")],
		});
		const pending = emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 1,
			message: assistantCall("bash", "disable-repeat", { command: "check" }),
			toolResults: [toolResult("bash", "disable-repeat")],
		});
		await vi.waitFor(() => expect(finish).toBeDefined());
		const command = runner.getCommand("angel");
		await command?.handler("off", {
			...runner.createCommandContext(),
			mode: "print",
			hasUI: false,
		});
		finish?.(result("error"));
		await pending;

		expect(
			testSession.session.sessionManager
				.getEntries()
				.filter(
					(entry: { type: string; customType?: string }) =>
						entry.type === "custom_message" &&
						entry.customType === "angel-advice",
				),
		).toHaveLength(0);
	});

	it("discards stale automatic advice without aborting it on queued input", async () => {
		const deps = dependencies();
		let finish: ((value: ConsultationResult) => void) | undefined;
		let signal: AbortSignal | undefined;
		deps.runConsultation.mockImplementation(
			(_ctx, _request, options) =>
				new Promise((resolve) => {
					signal = options.signal;
					finish = resolve;
				}),
		);
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 0,
			message: assistantCall("bash", "first", { command: "check" }),
			toolResults: [toolResult("bash", "first")],
		});
		const pending = emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 1,
			message: assistantCall("bash", "repeat", { command: "check" }),
			toolResults: [toolResult("bash", "repeat")],
		});
		await vi.waitFor(() => expect(finish).toBeDefined());
		await runner.emit({
			type: "input",
			text: "A different task",
			source: "interactive",
		});
		expect(signal?.aborted).toBe(false);
		finish?.(result("error"));
		await pending;
		expect(
			testSession.session.messages.filter(
				(message) =>
					message.role === "custom" && message.customType === "angel-advice",
			),
		).toHaveLength(0);
	});

	it("reports automatic cancellation without steering failure", async () => {
		const deps = dependencies();
		deps.runConsultation.mockRejectedValue(
			new DOMException("cancelled", "AbortError"),
		);
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 0,
			message: assistantCall("bash", "cancel-first", { command: "check" }),
			toolResults: [toolResult("bash", "cancel-first")],
		});
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 1,
			message: assistantCall("bash", "cancel-repeat", { command: "check" }),
			toolResults: [toolResult("bash", "cancel-repeat")],
		});

		const status = testSession.session.messages.find(
			(message) =>
				message.role === "custom" && message.customType === "angel-status",
		);
		expect(status?.content).toBe("Angel consultation cancelled.");
		expect(status?.details).toMatchObject({ kind: "cancelled" });
	});

	it("does not start child work when shutdown aborts authentication", async () => {
		const deps = dependencies();
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;
		const context = runner.createCommandContext();
		const model = context.model;
		if (!model) throw new Error("test model unavailable");
		const successfulAuth =
			await context.modelRegistry.getApiKeyAndHeaders(model);
		let finishAuth: ((value: typeof successfulAuth) => void) | undefined;
		vi.spyOn(context.modelRegistry, "getApiKeyAndHeaders").mockImplementation(
			() =>
				new Promise((resolve) => {
					finishAuth = resolve;
				}),
		);
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 0,
			message: assistantCall("bash", "failed-first", { command: "check" }),
			toolResults: [toolResult("bash", "failed-first")],
		});
		const pending = emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 1,
			message: assistantCall("bash", "failed-repeat", { command: "check" }),
			toolResults: [toolResult("bash", "failed-repeat")],
		});
		await vi.waitFor(() => expect(finishAuth).toBeDefined());
		await runner.emit({ type: "session_shutdown", reason: "switch" });
		await runner.emit({ type: "session_start", reason: "switch" });
		finishAuth?.(successfulAuth);
		await pending;

		expect(deps.runConsultation).not.toHaveBeenCalled();
	});

	it("does not deliver a stale consultation after session replacement", async () => {
		const deps = dependencies();
		let finish: ((value: ConsultationResult) => void) | undefined;
		deps.runConsultation.mockImplementation(
			() =>
				new Promise((resolve) => {
					finish = resolve;
				}),
		);
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 0,
			message: assistantCall("bash", "failed-first", { command: "check" }),
			toolResults: [toolResult("bash", "failed-first")],
		});
		const pending = emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 1,
			message: assistantCall("bash", "failed-repeat", { command: "check" }),
			toolResults: [toolResult("bash", "failed-repeat")],
		});
		await vi.waitFor(() => expect(finish).toBeDefined());
		await runner.emit({ type: "session_shutdown", reason: "switch" });
		await runner.emit({ type: "session_start", reason: "switch" });
		finish?.(result("error"));
		await pending;

		expect(
			testSession.session.sessionManager
				.getEntries()
				.filter(
					(entry: { type: string; customType?: string }) =>
						entry.type === "custom_message" &&
						entry.customType === "angel-advice",
				),
		).toHaveLength(0);
	});

	it("does not deliver advice after parent tree navigation", async () => {
		const deps = dependencies();
		let finish: ((value: ConsultationResult) => void) | undefined;
		deps.runConsultation.mockImplementation(
			() =>
				new Promise((resolve) => {
					finish = resolve;
				}),
		);
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;
		await emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 0,
			message: assistantCall("bash", "failed-first", { command: "check" }),
			toolResults: [toolResult("bash", "failed-first")],
		});
		const pending = emitTurnEnd(runner, {
			type: "turn_end",
			turnIndex: 1,
			message: assistantCall("bash", "failed-repeat", { command: "check" }),
			toolResults: [toolResult("bash", "failed-repeat")],
		});
		await vi.waitFor(() => expect(finish).toBeDefined());
		await runner.emit({
			type: "session_before_tree",
			preparation: {
				targetId: "target",
				oldLeafId: null,
				commonAncestorId: null,
				entriesToSummarize: [],
				userWantsSummary: false,
			},
			signal: new AbortController().signal,
		});
		finish?.(result("error"));
		await pending;

		expect(
			testSession.session.sessionManager
				.getEntries()
				.filter(
					(entry: { type: string; customType?: string }) =>
						entry.type === "custom_message" &&
						entry.customType === "angel-advice",
				),
		).toHaveLength(0);
	});

	it("runs a human question without starting an executor turn", async () => {
		const deps = dependencies();
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const command = testSession.session.extensionRunner.getCommand("angel");
		const context = {
			...testSession.session.extensionRunner.createCommandContext(),
			mode: "print" as const,
			hasUI: false,
		};
		const assistantCount = testSession.session.messages.filter(
			(message: { role: string }) => message.role === "assistant",
		).length;

		const write = vi
			.spyOn(process.stdout, "write")
			.mockImplementation(() => true);
		await command?.handler("What evidence decides this?", context);

		expect(write).toHaveBeenCalledWith(`${result("human").advice}\n`);
		expect(deps.runConsultation).toHaveBeenCalledOnce();
		expect(deps.runConsultation.mock.calls[0]?.[1]).toMatchObject({
			origin: "human",
			question: "What evidence decides this?",
		});
		const consultationOptions = deps.runConsultation.mock.calls[0]?.[2];
		expect(consultationOptions).toMatchObject({
			loadExtensions: false,
			additionalExtensionPaths: [],
		});
		expect(consultationOptions).not.toHaveProperty("expectedToolNames");
		expect(
			testSession.session.messages.filter(
				(message: { role: string }) => message.role === "assistant",
			),
		).toHaveLength(assistantCount);
		expect(
			testSession.session.messages.find(
				(message: { role: string; customType?: string }) =>
					message.role === "custom" && message.customType === "angel-advice",
			),
		).toBeDefined();
	});

	it("passes the configured extension whitelist to the child", async () => {
		const deps = dependencies();
		deps.loadSettings = () => ({
			pairs: [{ executor: "openai/gpt-4o", advisor: "openai/gpt-4o" }],
			subagentExtensions: ["web"],
		});
		testSession = await createTestSession({
			extensionFactories: [createAngelExtension(deps)],
		});
		const runner = testSession.session.extensionRunner;
		const command = runner.getCommand("angel");
		const context = {
			...runner.createCommandContext(),
			mode: "print" as const,
			hasUI: false,
		};
		vi.spyOn(process.stdout, "write").mockImplementation(() => true);

		await command?.handler("Use web evidence", context);

		const options = deps.runConsultation.mock.calls[0]?.[2];
		expect(options).toMatchObject({ loadExtensions: false });
		expect(options?.additionalExtensionPaths).toHaveLength(1);
		expect(
			options?.additionalExtensionPaths?.[0]?.replaceAll("\\", "/"),
		).toMatch(/extensions\/web\/index\.ts$/);
	});

	it.each([
		["provider failure", new Error("provider unavailable")],
		["cancellation", new DOMException("cancelled", "AbortError")],
	])(
		"reports human %s without starting an executor turn",
		async (_label, error) => {
			const deps = dependencies();
			deps.runConsultation.mockRejectedValue(error);
			testSession = await createTestSession({
				extensionFactories: [createAngelExtension(deps)],
			});
			const runner = testSession.session.extensionRunner;
			const command = runner.getCommand("angel");
			const context = {
				...runner.createCommandContext(),
				mode: "print" as const,
				hasUI: false,
			};
			const assistantCount = testSession.session.messages.filter(
				(message: { role: string }) => message.role === "assistant",
			).length;

			const write = vi
				.spyOn(process.stdout, "write")
				.mockImplementation(() => true);
			await command?.handler("Investigate this", context);

			expect(write).toHaveBeenCalled();
			const failure = testSession.session.messages.find(
				(message: { role: string; customType?: string }) =>
					message.role === "custom" && message.customType === "angel-status",
			) as { content?: string } | undefined;
			expect(failure?.content).toContain(
				error.name === "AbortError" ? "cancelled" : "provider unavailable",
			);
			expect(
				testSession.session.messages.filter(
					(message: { role: string }) => message.role === "assistant",
				),
			).toHaveLength(assistantCount);
		},
	);
});