Luigit
repositories / pi-ext

pi-ext

bugabingas pi extensions

owned by admin

extensions/angel/__tests__/advisor.test.ts

Raw
import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs";
import { tmpdir } from "node:os";
import path from "node:path";
import {
	fauxAssistantMessage,
	fauxProvider,
	fauxText,
	fauxToolCall,
	getCurrentSystemPrompt,
	getCurrentTools,
	InMemoryCredentialStore,
} from "@earendil-works/pi-ai";
import {
	ModelRegistry,
	ModelRuntime,
	SessionManager,
} from "@earendil-works/pi-coding-agent";
import { afterEach, describe, expect, it, vi } from "vitest";
import { withProcessEnv } from "../../../test/harness.ts";
import { runConsultation } from "../advisor.ts";

const ZERO_COST = {
	input: 0,
	output: 0,
	cacheRead: 0,
	cacheWrite: 0,
	total: 0,
};

function usage(input: number, output: number) {
	return {
		input,
		output,
		cacheRead: 0,
		cacheWrite: 0,
		totalTokens: input + output,
		cost: ZERO_COST,
	};
}

function assistant(
	content: Parameters<typeof fauxAssistantMessage>[0],
	input: number,
	output: number,
	stopReason: "stop" | "toolUse" = "stop",
) {
	return {
		...fauxAssistantMessage(content, { stopReason }),
		usage: usage(input, output),
	};
}

function environment(root: string): NodeJS.ProcessEnv {
	return {
		...process.env,
		HOME: path.join(root, "home"),
		USERPROFILE: path.join(root, "home"),
		XDG_CONFIG_HOME: path.join(root, "xdg-config"),
		XDG_CACHE_HOME: path.join(root, "xdg-cache"),
		XDG_STATE_HOME: path.join(root, "xdg-state"),
		XDG_DATA_HOME: path.join(root, "xdg-data"),
	};
}

async function fixture(responses: ReturnType<typeof assistant>[]) {
	const root = mkdtempSync(path.join(tmpdir(), "angel-advisor-test-"));
	const cwd = path.join(root, "project");
	const sessionDir = path.join(root, "sessions");
	mkdirSync(path.join(cwd, ".pi", "extensions"), { recursive: true });
	const agentDir = path.join(root, "home", ".pi", "agent");
	mkdirSync(path.join(agentDir, "skills", "evidence-skill"), {
		recursive: true,
	});
	writeFileSync(
		path.join(agentDir, "settings.json"),
		JSON.stringify({
			extensions: [path.resolve("extensions/angel/index.ts")],
			angel: {
				pairs: [
					{
						executor: "angel-test/advisor",
						advisor: "angel-test/advisor",
					},
				],
				thinkingLevel: "max",
			},
		}),
	);
	writeFileSync(
		path.join(agentDir, "skills", "evidence-skill", "SKILL.md"),
		"---\nname: evidence-skill\ndescription: Inspect deterministic evidence.\n---\n\n# Evidence skill\n",
	);
	writeFileSync(
		path.join(cwd, "AGENTS.md"),
		"# Project evidence\n\nRead before advising.\n",
	);
	writeFileSync(
		path.join(cwd, ".pi", "extensions", "probe.ts"),
		`export default function (pi) {
  pi.on("context", (event) => ({
    messages: event.messages.filter(
      (message) => message.role !== "custom" || message.customType !== "protected"
    )
  }));
  pi.registerTool({
    name: "probe",
    label: "Probe",
    description: "Return deterministic project evidence",
    parameters: { type: "object", properties: {}, additionalProperties: false },
    async execute() {
      return { content: [{ type: "text", text: "probe evidence" }], details: {} };
    }
  });
}`,
	);

	const externalExtensionPath = path.join(root, "external-probe.ts");
	writeFileSync(
		externalExtensionPath,
		`export default function (pi) {
  pi.registerTool({
    name: "external_probe",
    label: "External Probe",
    description: "A runtime-provided extension tool",
    parameters: { type: "object", properties: {}, additionalProperties: false },
    async execute() {
      return { content: [{ type: "text", text: "external evidence" }], details: {} };
    }
  });
}`,
	);

	const faux = fauxProvider({
		provider: "angel-test",
		models: [{ id: "advisor", reasoning: true, maxTokens: 8_192 }],
	});
	Object.assign(faux.getModel(), { thinkingLevelMap: { max: "high" } });
	faux.setResponses(responses);
	const runtime = await ModelRuntime.create({
		credentials: new InMemoryCredentialStore(),
		modelsPath: null,
		refreshOnCreate: false,
	});
	runtime.registerNativeProvider(faux.provider);
	const registry = new ModelRegistry(runtime);
	const parent = SessionManager.create(cwd, sessionDir);
	parent.appendMessage({
		role: "user",
		content: "Original parent request",
		timestamp: Date.now(),
	});
	parent.appendMessage({
		...assistant("Parent working answer", 99, 11),
		provider: "openai",
		model: "gpt-4o",
	});
	parent.appendCustomMessageEntry(
		"protected",
		"FILTERED PARENT CONTENT",
		false,
	);

	const ctx = {
		cwd,
		model: faux.getModel(),
		modelRegistry: registry,
		sessionManager: parent,
		getSystemPrompt: () => "PARENT SYSTEM PROMPT WITH PROJECT RULES",
		isProjectTrusted: () => true,
	} as never;
	return { root, cwd, parent, faux, ctx, externalExtensionPath };
}

describe("Angel investigative child session", () => {
	const roots: string[] = [];

	afterEach(() => {
		for (const root of roots.splice(0))
			rmSync(root, { recursive: true, force: true });
	});

	it("uses extension and parent-session tools without preloading history", async () => {
		let firstContext:
			| {
					systemPrompt?: string;
					messages: Array<{ role: string; content: unknown }>;
					tools?: Array<{ name: string }>;
			  }
			| undefined;
		const fixtureData = await fixture([
			assistant(
				[
					fauxToolCall(
						"parent_session",
						{ action: "overview" },
						{ id: "parent-overview" },
					),
					fauxToolCall("probe", {}, { id: "probe-call" }),
				],
				3,
				2,
				"toolUse",
			),
			assistant("**Use the probe evidence.**", 7, 4),
		]);
		roots.push(fixtureData.root);
		fixtureData.faux.setResponses([
			(context) => {
				firstContext = {
					systemPrompt: getCurrentSystemPrompt(context.messages),
					messages: context.messages,
					tools: getCurrentTools(context.messages),
				};
				return assistant(
					[
						fauxToolCall(
							"parent_session",
							{ action: "overview" },
							{ id: "parent-overview" },
						),
						fauxToolCall("probe", {}, { id: "probe-call" }),
					],
					3,
					2,
					"toolUse",
				);
			},
			assistant("**Use the probe evidence.**", 7, 4),
		]);

		const parentAuth = vi.spyOn(
			fixtureData.ctx.modelRegistry,
			"getApiKeyAndHeaders",
		);
		const consultation = await withProcessEnv(
			environment(fixtureData.root),
			() =>
				runConsultation(
					fixtureData.ctx,
					{
						origin: "executor",
						question: "Which evidence should decide?",
					},
					{
						advisor: fixtureData.faux.getModel(),
						thinkingLevel: "max",
						loadExtensions: true,
						additionalExtensionPaths: [fixtureData.externalExtensionPath],
					},
				),
		);

		expect(parentAuth).not.toHaveBeenCalled();
		expect(consultation.advice).toBe("**Use the probe evidence.**");
		expect(consultation.metadata.thinkingLevel).toBe("max");
		expect(consultation.metadata.tokens.input).toBeGreaterThan(0);
		expect(consultation.metadata.tokens.output).toBeGreaterThan(0);
		expect(consultation.metadata.cost).toBeUndefined();
		expect(firstContext?.systemPrompt).toContain("investigative advisor");
		expect(firstContext?.systemPrompt).toContain(
			"invoke only tools exposed in the current runtime",
		);
		expect(firstContext?.systemPrompt).toContain("Project evidence");
		expect(firstContext?.systemPrompt).toContain("evidence-skill");
		expect(firstContext?.tools?.map((tool) => tool.name)).toEqual(
			expect.arrayContaining([
				"read",
				"ls",
				"find",
				"grep",
				"parent_session",
				"probe",
				"external_probe",
			]),
		);
		expect(firstContext?.tools?.map((tool) => tool.name)).not.toContain(
			"write",
		);
		expect(firstContext?.tools?.map((tool) => tool.name)).not.toContain("bash");
		expect(firstContext?.tools?.map((tool) => tool.name)).not.toContain("edit");
		expect(firstContext?.tools?.map((tool) => tool.name)).not.toContain(
			"angel",
		);
		const initialContext = JSON.stringify(firstContext);
		const initialMessages = JSON.stringify(firstContext?.messages);
		expect(initialMessages).toContain(fixtureData.parent.getSessionId());
		expect(initialContext).not.toContain("Original parent request");
		expect(initialContext).not.toContain(
			"PARENT SYSTEM PROMPT WITH PROJECT RULES",
		);
		expect(initialMessages).not.toContain(fixtureData.parent.getSessionFile());
		expect(initialMessages).not.toContain("FILTERED PARENT CONTENT");

		const child = SessionManager.open(consultation.metadata.childSessionFile);
		expect(child.getHeader()?.parentSession).toBe(
			fixtureData.parent.getSessionFile(),
		);
		expect(child.getSessionName()).toContain("Angel · executor");
		expect(child.getEntries()).toContainEqual(
			expect.objectContaining({
				type: "custom",
				customType: "angel-child",
			}),
		);
		expect(child.getEntries()).toContainEqual(
			expect.objectContaining({
				type: "message",
				message: expect.objectContaining({
					role: "toolResult",
					toolName: "probe",
				}),
			}),
		);
		expect(
			child
				.getEntries()
				.some(
					(entry) =>
						entry.type === "message" &&
						entry.message.role === "toolResult" &&
						entry.message.toolName === "parent_session" &&
						JSON.stringify(entry.message.content).includes(
							fixtureData.parent.getSessionId(),
						),
				),
		).toBe(true);
		expect(
			child
				.getEntries()
				.some(
					(entry) =>
						entry.type === "message" &&
						entry.message.role === "assistant" &&
						entry.message.model === "gpt-4o",
				),
		).toBe(false);
		const billedInput = child.getEntries().reduce((total, entry) => {
			if (entry.type !== "message" || entry.message.role !== "assistant")
				return total;
			return total + entry.message.usage.input;
		}, 0);
		expect(consultation.metadata.tokens.input).toBe(billedInput);
	});

	it("loads no configured extensions by default", async () => {
		let toolNames: string[] = [];
		const fixtureData = await fixture([assistant("Core evidence used.", 2, 2)]);
		roots.push(fixtureData.root);
		fixtureData.faux.setResponses([
			(context) => {
				toolNames = getCurrentTools(context.messages).map((tool) => tool.name);
				return assistant("Core evidence used.", 2, 2);
			},
		]);

		await withProcessEnv(environment(fixtureData.root), () =>
			runConsultation(
				fixtureData.ctx,
				{ origin: "executor", question: "Use core evidence" },
				{
					advisor: fixtureData.faux.getModel(),
					thinkingLevel: "max",
				},
			),
		);

		expect(toolNames).toEqual(
			expect.arrayContaining(["read", "parent_session"]),
		);
		expect(toolNames).not.toContain("probe");
		expect(toolNames).not.toContain("angel");
	});

	it("uses one pinned parent identity after the live leaf advances", async () => {
		let firstContext:
			| { messages: Array<{ role: string; content: unknown }> }
			| undefined;
		const fixtureData = await fixture([
			assistant("Pinned evidence used.", 2, 2),
		]);
		roots.push(fixtureData.root);
		const pinnedLeaf = fixtureData.parent.getLeafId();
		fixtureData.faux.setResponses([
			(context) => {
				firstContext = { messages: context.messages };
				return assistant("Pinned evidence used.", 2, 2);
			},
		]);

		const pending = withProcessEnv(environment(fixtureData.root), () =>
			runConsultation(
				fixtureData.ctx,
				{ origin: "executor", question: "Use the pinned parent" },
				{
					advisor: fixtureData.faux.getModel(),
					thinkingLevel: "max",
				},
			),
		);
		fixtureData.parent.appendSessionInfo("Renamed while Angel starts");
		const liveLeaf = fixtureData.parent.getLeafId();
		const consultation = await pending;

		expect(liveLeaf).not.toBe(pinnedLeaf);
		const assignmentContext = JSON.stringify(firstContext?.messages);
		expect(assignmentContext).toContain(`Pinned parent leaf: ${pinnedLeaf}`);
		expect(assignmentContext).not.toContain(`Pinned parent leaf: ${liveLeaf}`);
		const child = SessionManager.open(consultation.metadata.childSessionFile);
		expect(child.getEntries()).toContainEqual(
			expect.objectContaining({
				type: "custom",
				customType: "angel-child",
				data: expect.objectContaining({ parentLeafId: pinnedLeaf }),
			}),
		);
	});

	it("lists parent metadata before retrieving an exact entry", async () => {
		let listedMessages: Array<{ role: string; content: unknown }> = [];
		let retrievedMessages: Array<{ role: string; content: unknown }> = [];
		const fixtureData = await fixture([assistant("Evidence retrieved.", 2, 2)]);
		roots.push(fixtureData.root);
		fixtureData.parent.appendMessage({
			role: "custom",
			customType: "protected-role",
			content: "ROLE CUSTOM SECRET",
			display: false,
			timestamp: Date.now(),
		});
		fixtureData.parent.appendMessage({
			...assistant(
				fauxToolCall("opaque_tool", { target: "same" }, { id: "opaque-call" }),
				1,
				1,
				"toolUse",
			),
			provider: "openai",
			model: "gpt-4o",
		});
		fixtureData.parent.appendMessage({
			role: "bashExecution",
			command: "echo EXCLUDED COMMAND",
			output: "EXCLUDED OUTPUT",
			exitCode: 0,
			cancelled: false,
			truncated: false,
			excludeFromContext: true,
			timestamp: Date.now(),
		});
		const resultEntryId = fixtureData.parent.appendMessage({
			role: "toolResult",
			toolCallId: "opaque-call",
			toolName: "opaque_tool",
			content: [{ type: "text", text: "EXACT TOOL EVIDENCE" }],
			isError: true,
			timestamp: Date.now(),
		});
		fixtureData.faux.setResponses([
			assistant(
				[
					fauxToolCall(
						"parent_session",
						{ action: "list", view: "branch", offset: 0, limit: 20 },
						{ id: "list-parent" },
					),
					fauxToolCall(
						"parent_session",
						{ action: "search", view: "branch", query: "EXCLUDED OUTPUT" },
						{ id: "search-excluded" },
					),
					fauxToolCall(
						"parent_session",
						{
							action: "search",
							view: "branch",
							query: "FILTERED PARENT CONTENT",
						},
						{ id: "search-protected" },
					),
					fauxToolCall(
						"parent_session",
						{ action: "search", view: "branch", query: "ROLE CUSTOM SECRET" },
						{ id: "search-role-custom" },
					),
				],
				2,
				2,
				"toolUse",
			),
			(context) => {
				listedMessages = context.messages as typeof listedMessages;
				return assistant(
					fauxToolCall(
						"parent_session",
						{ action: "get", view: "branch", entryId: resultEntryId },
						{ id: "get-parent" },
					),
					2,
					2,
					"toolUse",
				);
			},
			(context) => {
				retrievedMessages = context.messages as typeof retrievedMessages;
				return assistant("Evidence retrieved.", 2, 2);
			},
		]);

		await withProcessEnv(environment(fixtureData.root), () =>
			runConsultation(
				fixtureData.ctx,
				{ origin: "error", question: "Inspect the failed opaque call" },
				{
					advisor: fixtureData.faux.getModel(),
					thinkingLevel: "max",
				},
			),
		);

		const listed = JSON.stringify(listedMessages);
		expect(listed).toContain(resultEntryId);
		expect(listed).not.toContain("EXACT TOOL EVIDENCE");
		expect(listed).not.toContain("EXCLUDED COMMAND");
		for (const toolCallId of [
			"search-excluded",
			"search-protected",
			"search-role-custom",
		]) {
			const searchResult = listedMessages.find(
				(message) =>
					(message as { toolCallId?: string }).toolCallId === toolCallId,
			);
			expect(JSON.stringify(searchResult?.content)).toContain(
				"Matches in pinned branch: 0",
			);
		}
		expect(JSON.stringify(retrievedMessages)).toContain("EXACT TOOL EVIDENCE");
	});

	it("paginates within long parent entries and searches around the match", async () => {
		let searchedMessages: Array<{ role: string; content: unknown }> = [];
		let pagedMessages: Array<{ role: string; content: unknown }> = [];
		const fixtureData = await fixture([
			assistant("Final line retrieved.", 2, 2),
		]);
		roots.push(fixtureData.root);
		const lines = Array.from({ length: 205 }, (_, index) =>
			index === 204 ? "DISTANT FINAL LINE" : `ordinary line ${index}`,
		);
		const entryId = fixtureData.parent.appendMessage({
			role: "toolResult",
			toolCallId: "long-call",
			toolName: "long_tool",
			content: [{ type: "text", text: lines.join("\n") }],
			isError: true,
			timestamp: Date.now(),
		});
		fixtureData.faux.setResponses([
			assistant(
				fauxToolCall(
					"parent_session",
					{ action: "search", view: "branch", query: "DISTANT FINAL LINE" },
					{ id: "search-final-line" },
				),
				2,
				2,
				"toolUse",
			),
			(context) => {
				searchedMessages = context.messages as typeof searchedMessages;
				return assistant(
					fauxToolCall(
						"parent_session",
						{
							action: "get",
							view: "branch",
							entryId,
							offset: 200,
							limit: 10,
						},
						{ id: "get-final-page" },
					),
					2,
					2,
					"toolUse",
				);
			},
			(context) => {
				pagedMessages = context.messages as typeof pagedMessages;
				return assistant("Final line retrieved.", 2, 2);
			},
		]);

		await withProcessEnv(environment(fixtureData.root), () =>
			runConsultation(
				fixtureData.ctx,
				{ origin: "error", question: "Inspect the end of the long result" },
				{
					advisor: fixtureData.faux.getModel(),
					thinkingLevel: "max",
				},
			),
		);

		expect(JSON.stringify(searchedMessages)).toContain("DISTANT FINAL LINE");
		const paged = JSON.stringify(pagedMessages);
		expect(paged).toContain("Entry text lines 200-204 of 205");
		expect(paged).toContain("DISTANT FINAL LINE");
	});

	it("retrieves markers from a maximum-sized single-line result", async () => {
		let searchedMessages: Array<{ role: string; content: unknown }> = [];
		let pagedMessages: Array<{ role: string; content: unknown }> = [];
		const fixtureData = await fixture([
			assistant("Oversized marker read.", 2, 2),
		]);
		roots.push(fixtureData.root);
		const marker = "OVERSIZED MARKER";
		const body = `${"x".repeat(51_200 - marker.length)}${marker}`;
		const entryId = fixtureData.parent.appendMessage({
			role: "toolResult",
			toolCallId: "oversized-call",
			toolName: "oversized_tool",
			content: [{ type: "text", text: body }],
			isError: true,
			timestamp: Date.now(),
		});
		fixtureData.faux.setResponses([
			assistant(
				fauxToolCall(
					"parent_session",
					{ action: "search", view: "branch", query: marker },
					{ id: "search-oversized" },
				),
				2,
				2,
				"toolUse",
			),
			(context) => {
				searchedMessages = context.messages as typeof searchedMessages;
				return assistant(
					fauxToolCall(
						"parent_session",
						{
							action: "get",
							view: "branch",
							entryId,
							characterOffset: 51_000,
							characterLimit: 200,
						},
						{ id: "get-oversized-tail" },
					),
					2,
					2,
					"toolUse",
				);
			},
			(context) => {
				pagedMessages = context.messages as typeof pagedMessages;
				return assistant("Oversized marker read.", 2, 2);
			},
		]);

		await withProcessEnv(environment(fixtureData.root), () =>
			runConsultation(
				fixtureData.ctx,
				{ origin: "error", question: "Inspect the oversized result" },
				{
					advisor: fixtureData.faux.getModel(),
					thinkingLevel: "max",
				},
			),
		);

		expect(JSON.stringify(searchedMessages)).toContain(marker);
		const paged = JSON.stringify(pagedMessages);
		expect(paged).toContain("Entry text characters 51000-51199 of 51200");
		expect(paged).toContain(marker);
	});

	it("retrieves parent user and tool-result images only on demand", async () => {
		let initialMessages: Array<{ role: string; content: unknown }> = [];
		let retrievedMessages: Array<{ role: string; content: unknown }> = [];
		const fixtureData = await fixture([assistant("Images inspected.", 2, 2)]);
		roots.push(fixtureData.root);
		const userImage = "dXNlci1pbWFnZQ==";
		const toolImage = "dG9vbC1pbWFnZQ==";
		const userEntryId = fixtureData.parent.appendMessage({
			role: "user",
			content: [{ type: "image", data: userImage, mimeType: "image/png" }],
			timestamp: Date.now(),
		});
		const toolEntryId = fixtureData.parent.appendMessage({
			role: "toolResult",
			toolCallId: "image-call",
			toolName: "image_tool",
			content: [{ type: "image", data: toolImage, mimeType: "image/webp" }],
			isError: false,
			timestamp: Date.now(),
		});
		fixtureData.faux.setResponses([
			(context) => {
				initialMessages = context.messages as typeof initialMessages;
				return assistant(
					[
						fauxToolCall(
							"parent_session",
							{ action: "get", view: "branch", entryId: userEntryId },
							{ id: "get-user-image" },
						),
						fauxToolCall(
							"parent_session",
							{ action: "get", view: "branch", entryId: toolEntryId },
							{ id: "get-tool-image" },
						),
					],
					2,
					2,
					"toolUse",
				);
			},
			(context) => {
				retrievedMessages = context.messages as typeof retrievedMessages;
				return assistant("Images inspected.", 2, 2);
			},
		]);

		await withProcessEnv(environment(fixtureData.root), () =>
			runConsultation(
				fixtureData.ctx,
				{ origin: "executor", question: "Inspect both parent images" },
				{
					advisor: fixtureData.faux.getModel(),
					thinkingLevel: "max",
				},
			),
		);

		const initial = JSON.stringify(initialMessages);
		expect(initial).not.toContain(userImage);
		expect(initial).not.toContain(toolImage);
		const retrieved = JSON.stringify(retrievedMessages);
		expect(retrieved).toContain(userImage);
		expect(retrieved).toContain(toolImage);
		expect(retrieved).toContain("1 image");
	});

	it("preserves failure and full-output metadata for parent shell entries", async () => {
		let retrievedMessages: Array<{ role: string; content: unknown }> = [];
		const fixtureData = await fixture([
			assistant("Shell failure inspected.", 2, 2),
		]);
		roots.push(fixtureData.root);
		const entryId = fixtureData.parent.appendMessage({
			role: "bashExecution",
			command: "opaque-command",
			output: "partial output",
			exitCode: 7,
			cancelled: false,
			truncated: true,
			fullOutputPath: "/tmp/opaque-full-output.log",
			timestamp: Date.now(),
		});
		fixtureData.faux.setResponses([
			assistant(
				fauxToolCall(
					"parent_session",
					{ action: "get", view: "branch", entryId },
					{ id: "get-shell-failure" },
				),
				2,
				2,
				"toolUse",
			),
			(context) => {
				retrievedMessages = context.messages as typeof retrievedMessages;
				return assistant("Shell failure inspected.", 2, 2);
			},
		]);

		await withProcessEnv(environment(fixtureData.root), () =>
			runConsultation(
				fixtureData.ctx,
				{ origin: "error", question: "Inspect the failed shell command" },
				{
					advisor: fixtureData.faux.getModel(),
					thinkingLevel: "max",
				},
			),
		);

		const retrieved = JSON.stringify(retrievedMessages);
		expect(retrieved).toContain("Command exited with code 7");
		expect(retrieved).toContain(
			"[Output truncated. Full output: /tmp/opaque-full-output.log]",
		);
	});

	it("retrieves distant parent history only on demand", async () => {
		let initialMessages: Array<{ role: string; content: unknown }> = [];
		let retrievedMessages: Array<{ role: string; content: unknown }> = [];
		const fixtureData = await fixture([
			assistant("All requested history is available.", 2, 2),
		]);
		roots.push(fixtureData.root);
		for (let index = 0; index < 205; index++) {
			fixtureData.parent.appendMessage({
				role: "user",
				content: `historical evidence ${index}`,
				timestamp: Date.now(),
			});
		}
		fixtureData.faux.setResponses([
			(context) => {
				initialMessages = context.messages as typeof initialMessages;
				return assistant(
					[
						fauxToolCall(
							"parent_session",
							{
								action: "search",
								view: "branch",
								query: "historical evidence 0",
							},
							{ id: "search-oldest" },
						),
						fauxToolCall(
							"parent_session",
							{
								action: "search",
								view: "branch",
								query: "historical evidence 204",
							},
							{ id: "search-newest" },
						),
					],
					2,
					2,
					"toolUse",
				);
			},
			(context) => {
				retrievedMessages = context.messages as typeof retrievedMessages;
				return assistant("All requested history is available.", 2, 2);
			},
		]);

		await withProcessEnv(environment(fixtureData.root), () =>
			runConsultation(
				fixtureData.ctx,
				{ origin: "executor", question: "Inspect older evidence" },
				{
					advisor: fixtureData.faux.getModel(),
					thinkingLevel: "max",
				},
			),
		);

		expect(JSON.stringify(initialMessages)).not.toContain(
			"historical evidence 0",
		);
		const retrieved = JSON.stringify(retrievedMessages);
		expect(retrieved).toContain("historical evidence 0");
		expect(retrieved).toContain("historical evidence 204");
	});

	it("retrieves active compaction and branch evidence on demand", async () => {
		let retrievedMessages: Array<{ role: string; content: unknown }> = [];
		const fixtureData = await fixture([assistant("Summaries received.", 2, 2)]);
		roots.push(fixtureData.root);
		const keptId = fixtureData.parent.appendMessage({
			role: "user",
			content: "kept evidence",
			timestamp: Date.now(),
		});
		fixtureData.parent.appendCompaction(
			"ACTIVE COMPACTION SUMMARY",
			keptId,
			12_345,
		);
		const branchFrom = fixtureData.parent.appendMessage({
			role: "user",
			content: "evidence after compaction",
			timestamp: Date.now(),
		});
		fixtureData.parent.appendMessage({
			role: "user",
			content: "abandoned branch evidence",
			timestamp: Date.now(),
		});
		fixtureData.parent.branchWithSummary(branchFrom, "ACTIVE BRANCH SUMMARY");
		fixtureData.parent.appendMessage({
			role: "user",
			content: "active branch evidence",
			timestamp: Date.now(),
		});
		fixtureData.faux.setResponses([
			assistant(
				[
					fauxToolCall(
						"parent_session",
						{
							action: "search",
							query: "ACTIVE COMPACTION SUMMARY",
						},
						{ id: "search-compaction" },
					),
					fauxToolCall(
						"parent_session",
						{ action: "search", query: "ACTIVE BRANCH SUMMARY" },
						{ id: "search-branch" },
					),
					fauxToolCall(
						"parent_session",
						{ action: "search", query: "evidence after compaction" },
						{ id: "search-kept" },
					),
				],
				2,
				2,
				"toolUse",
			),
			(context) => {
				retrievedMessages = context.messages as typeof retrievedMessages;
				return assistant("Summaries received.", 2, 2);
			},
		]);

		await withProcessEnv(environment(fixtureData.root), () =>
			runConsultation(
				fixtureData.ctx,
				{ origin: "executor", question: "Use the active summaries" },
				{
					advisor: fixtureData.faux.getModel(),
					thinkingLevel: "max",
				},
			),
		);

		const retrieved = JSON.stringify(retrievedMessages);
		expect(retrieved).toContain("ACTIVE BRANCH SUMMARY");
		expect(retrieved).toContain("ACTIVE COMPACTION SUMMARY");
		expect(retrieved).toContain("evidence after compaction");
		expect(retrieved).not.toContain("abandoned branch evidence");
	});

	it("reports child extension loading failures instead of silently degrading", async () => {
		const fixtureData = await fixture([assistant("unused", 1, 1)]);
		roots.push(fixtureData.root);
		const brokenExtension = path.join(fixtureData.root, "broken-extension.ts");
		writeFileSync(brokenExtension, "this is not valid TypeScript !!!");

		await expect(
			withProcessEnv(environment(fixtureData.root), () =>
				runConsultation(
					fixtureData.ctx,
					{ origin: "executor", question: "Use every capability" },
					{
						advisor: fixtureData.faux.getModel(),
						thinkingLevel: "max",
						additionalExtensionPaths: [brokenExtension],
					},
				),
			),
		).rejects.toThrow("extension initialization failed");
		expect(fixtureData.faux.state.callCount).toBe(0);
	});

	it("accepts a parent tool that the child loads but leaves inactive", async () => {
		let firstTools: string[] | undefined;
		const fixtureData = await fixture([
			assistant("Child state respected.", 1, 1),
		]);
		roots.push(fixtureData.root);
		fixtureData.faux.setResponses([
			(context) => {
				firstTools = getCurrentTools(context.messages).map((tool) => tool.name);
				return assistant("Child state respected.", 1, 1);
			},
		]);
		const childScopedExtension = path.join(
			fixtureData.root,
			"child-scoped-tool.ts",
		);
		writeFileSync(
			childScopedExtension,
			`export default function (pi) {
  pi.registerTool({
    name: "child_scoped",
    label: "Child scoped",
    description: "Active only when this extension enables it",
    parameters: { type: "object", properties: {}, additionalProperties: false },
    async execute() {
      return { content: [{ type: "text", text: "unused" }], details: {} };
    }
  });
  pi.on("session_start", () => {
    pi.setActiveTools(pi.getActiveTools().filter((name) => name !== "child_scoped"));
  });
}`,
		);

		const consultation = await withProcessEnv(
			environment(fixtureData.root),
			() =>
				runConsultation(
					fixtureData.ctx,
					{ origin: "executor", question: "Respect child tool state" },
					{
						advisor: fixtureData.faux.getModel(),
						thinkingLevel: "max",
						additionalExtensionPaths: [childScopedExtension],
					},
				),
		);

		expect(consultation.advice).toBe("Child state respected.");
		expect(firstTools).not.toContain("child_scoped");
	});

	it("propagates provider failures", async () => {
		const fixtureData = await fixture([assistant("unused", 1, 1)]);
		roots.push(fixtureData.root);
		fixtureData.faux.setResponses([
			() => {
				throw new Error("provider failed");
			},
		]);

		await expect(
			withProcessEnv(environment(fixtureData.root), () =>
				runConsultation(
					fixtureData.ctx,
					{ origin: "human", question: "Investigate safely" },
					{
						advisor: fixtureData.faux.getModel(),
						thinkingLevel: "max",
					},
				),
			),
		).rejects.toThrow("provider failed");
	});

	it("does not accept intermediate investigative text as final advice", async () => {
		const fixtureData = await fixture([
			assistant(
				[
					fauxText("This is only an interim hypothesis."),
					fauxToolCall("probe", {}, { id: "probe-call" }),
				],
				3,
				2,
				"toolUse",
			),
			assistant("", 7, 0),
		]);
		roots.push(fixtureData.root);

		await expect(
			withProcessEnv(environment(fixtureData.root), () =>
				runConsultation(
					fixtureData.ctx,
					{ origin: "error", question: "Diagnose the failure" },
					{
						advisor: fixtureData.faux.getModel(),
						thinkingLevel: "max",
					},
				),
			),
		).rejects.toThrow("without final advice");
	});

	it("cancels after delayed prompt preflight without calling the provider", async () => {
		const fixtureData = await fixture([assistant("unused", 1, 1)]);
		roots.push(fixtureData.root);
		const controller = new AbortController();
		const key = Symbol.for("angel-test-delayed-preflight");
		let markStarted: (() => void) | undefined;
		let release: (() => void) | undefined;
		const started = new Promise<void>((resolve) => {
			markStarted = resolve;
		});
		const gate = new Promise<void>((resolve) => {
			release = resolve;
		});
		const globals = globalThis as typeof globalThis &
			Record<symbol, { started: () => void; gate: Promise<void> }>;
		globals[key] = { started: () => markStarted?.(), gate };
		const delayedExtension = path.join(
			fixtureData.root,
			"delayed-preflight.ts",
		);
		writeFileSync(
			delayedExtension,
			`const state = globalThis[Symbol.for("angel-test-delayed-preflight")];
export default function (pi) {
  pi.on("before_agent_start", async () => {
    state.started();
    await state.gate;
  });
}`,
		);

		try {
			const pending = withProcessEnv(environment(fixtureData.root), () =>
				runConsultation(
					fixtureData.ctx,
					{ origin: "human", question: "Cancel during preflight" },
					{
						advisor: fixtureData.faux.getModel(),
						thinkingLevel: "max",
						signal: controller.signal,
						additionalExtensionPaths: [delayedExtension],
					},
				),
			);
			await started;
			controller.abort();
			release?.();
			await expect(pending).rejects.toMatchObject({ name: "AbortError" });
			expect(fixtureData.faux.state.callCount).toBe(0);
		} finally {
			delete globals[key];
		}
	});

	it("rejects a pre-cancelled consultation before creating child work", async () => {
		const fixtureData = await fixture([assistant("unused", 1, 1)]);
		roots.push(fixtureData.root);
		const controller = new AbortController();
		controller.abort();

		await expect(
			withProcessEnv(environment(fixtureData.root), () =>
				runConsultation(
					fixtureData.ctx,
					{ origin: "human", question: "Do not start" },
					{
						advisor: fixtureData.faux.getModel(),
						thinkingLevel: "max",
						signal: controller.signal,
					},
				),
			),
		).rejects.toMatchObject({ name: "AbortError" });
		expect(fixtureData.faux.state.callCount).toBe(0);
	});
});