Luigit
repositories / pi-ext

pi-ext

bugabingas pi extensions

owned by admin

extensions/ultra/__tests__/harness.test.ts

Raw
import {
	mkdirSync,
	mkdtempSync,
	readdirSync,
	readFileSync,
	rmSync,
	writeFileSync,
} from "node:fs";
import { tmpdir } from "node:os";
import { dirname, join } from "node:path";
import { fileURLToPath } from "node:url";
import {
	type Context,
	fauxAssistantMessage,
	fauxProvider,
	fauxToolCall,
} from "@earendil-works/pi-ai";
import {
	type ExtensionCommandContext,
	initTheme,
	SessionManager,
	type Theme,
} from "@earendil-works/pi-coding-agent";
import {
	createTestSession as createHarnessSession,
	type TestSession,
} from "@marcfargas/pi-test-harness";
import { afterEach, describe, expect, it, vi } from "vitest";
import { createTestSession as createSandboxSession } from "../../../test/harness";
import { DEFAULT_MODEL_TIERS, modelTierPrompt } from "../settings.ts";
import { WorkflowSpecSchema } from "../spec.ts";

let t: TestSession | undefined;
let root: string | undefined;

afterEach(() => {
	t?.dispose();
	if (root) rmSync(root, { recursive: true, force: true });
	t = undefined;
	root = undefined;
	vi.unstubAllEnvs();
	vi.restoreAllMocks();
});

function ultraPath(): string {
	return fileURLToPath(new URL("../index.ts", import.meta.url));
}

async function createTestSession(options: {
	extensions: string[];
	cwd?: string;
}): Promise<TestSession> {
	root = options.cwd ?? mkdtempSync(join(tmpdir(), "ultra-harness-"));
	const agentDir = join(root, "agent");
	mkdirSync(agentDir, { recursive: true });
	vi.stubEnv("PI_CODING_AGENT_DIR", agentDir);
	if (!options.cwd) {
		mkdirSync(join(root, ".pi"), { recursive: true });
		writeFileSync(
			join(root, ".pi", "settings.json"),
			JSON.stringify({ ultra: { autoEnable: false } }),
		);
	}
	return createHarnessSession({ ...options, cwd: root });
}

function saveWorkflow(
	session: TestSession,
	name: string,
	dynamicExtension?: string,
): void {
	const directory = join(session.cwd, ".pi", "workflows");
	mkdirSync(directory, { recursive: true });
	writeFileSync(
		join(directory, `${name}.json`),
		`${JSON.stringify({
			name,
			phases: [
				{
					id: "work",
					kind: "single",
					step: {
						summary: "Work",
						prompt: "Work",
						...(dynamicExtension
							? { dynamicExtensions: [dynamicExtension] }
							: {}),
					},
				},
			],
		})}\n`,
	);
}

function driveOverlay(
	ctx: ExtensionCommandContext,
	interact?: (overlay: { handleInput(data: string): void }) => void,
): void {
	Object.assign(ctx.ui, {
		custom: vi.fn(
			async (
				factory: (
					host: unknown,
					theme: unknown,
					keybindings: unknown,
					done: (value: unknown) => void,
				) => { handleInput(data: string): void },
			) =>
				new Promise((resolve) => {
					const overlay = factory(
						{ requestRender: vi.fn() },
						{},
						{ matches: () => false },
						resolve,
					);
					interact?.(overlay);
				}),
		),
	});
}

describe("ultra extension", () => {
	it("loads only the command and renderer in the real Pi runtime", async () => {
		const indexPath = ultraPath();
		t = await createTestSession({ extensions: [indexPath] });
		const [extension] = t.session.extensionRunner.extensions;
		expect(extension.path).toBe(indexPath);
		expect(extension.tools.has("run_workflow")).toBe(false);
		expect(readFileSync(indexPath, "utf8")).toContain(
			'import("./implementation.ts")',
		);
		expect(extension.commands.has("ultra")).toBe(true);
		expect(extension.commands.has("ultra-runs")).toBe(false);
	});

	it("writes disabled command outcome in its debug sandbox", async () => {
		t = await createSandboxSession({
			env: { PI_ULTRA_DEBUG: "1" },
			extensions: [ultraPath()],
		});
		const extension = t.session.extensionRunner.extensions[0];
		const context = t.session.extensionRunner.createCommandContext();
		for (const handler of extension.handlers.get("session_start"))
			await handler({ type: "session_start" }, context);
		mkdirSync(join(t.cwd, ".pi"), { recursive: true });
		writeFileSync(
			join(t.cwd, ".pi", "settings.json"),
			JSON.stringify({ ultra: { enabled: false } }),
		);
		await extension.commands.get("ultra").handler("", context);
		for (const handler of extension.handlers.get("session_shutdown"))
			await handler({ type: "session_shutdown" }, context);
		const directory = join(t.cwd, ".test-state", "pi-ext", "debug", "ultra");
		const records = readFileSync(
			join(directory, readdirSync(directory)[0]),
			"utf8",
		)
			.trim()
			.split("\n")
			.map((line) => JSON.parse(line));
		expect(
			records.filter((record) => record.event === "command.handle.start"),
		).toHaveLength(1);
		expect(
			records.filter((record) => record.event === "command.handle.finish"),
		).toEqual([expect.objectContaining({ outcome: "disabled" })]);
		expect(records.map((record) => record.event)).toEqual(
			expect.arrayContaining(["session.start", "session.shutdown"]),
		);
	});

	it("auto-enables the workflow tool when configured", async () => {
		root = mkdtempSync(join(tmpdir(), "ultra-auto-enable-"));
		mkdirSync(join(root, ".pi"), { recursive: true });
		writeFileSync(
			join(root, ".pi", "settings.json"),
			JSON.stringify({ ultra: { autoEnable: true } }),
		);

		t = await createTestSession({ cwd: root, extensions: [ultraPath()] });
		const [extension] = t.session.extensionRunner.extensions;

		expect(extension.tools.has("run_workflow")).toBe(true);
		expect(t.session.getActiveToolNames()).toContain("run_workflow");
		expect(t.session.systemPrompt).toContain("- run_workflow:");
	});

	it.each([false, true])(
		"registers the authoring skill independently of tool activation (auto-enable: %s)",
		async (autoEnable) => {
			root = mkdtempSync(join(tmpdir(), "ultra-shared-guidance-"));
			mkdirSync(join(root, ".pi"), { recursive: true });
			writeFileSync(
				join(root, ".pi", "settings.json"),
				JSON.stringify({ ultra: { autoEnable } }),
			);
			t = await createTestSession({ cwd: root, extensions: [ultraPath()] });
			const skillPath = fileURLToPath(
				new URL("../prompts/ultra-authoring/SKILL.md", import.meta.url),
			);
			const skill = t.session.resourceLoader
				.getSkills()
				.skills.find((candidate) => candidate.name === "ultra-authoring");
			expect(skill?.filePath).toBe(skillPath);
			expect(t.session.systemPrompt).toContain(skillPath);
			expect(t.session.systemPrompt).not.toContain(
				readFileSync(skillPath, "utf8").trim(),
			);
			expect(t.session.systemPrompt).not.toContain("## Worked example");

			const [extension] = t.session.extensionRunner.extensions;
			if (!autoEnable) {
				await extension.commands
					.get("ultra")
					.handler("", t.session.extensionRunner.createCommandContext());
			}
			const rules: string[] =
				extension.tools.get("run_workflow").definition.promptGuidelines;
			expect(rules).toEqual([
				"Use run_workflow when decomposition into independent or sequential child-agent steps reduces context or verification risk.",
				"Use dynamic extensions when a workflow needs a reusable, task-specific capability; read the ultra-authoring skill for how to build and load them.",
			]);
			for (const rule of rules)
				expect(t.session.systemPrompt.split(rule)).toHaveLength(2);
			t.session.setActiveToolsByName(
				t.session
					.getActiveToolNames()
					.filter((name) => name !== "run_workflow"),
			);
			expect(t.session.systemPrompt).toContain(skillPath);
			for (const rule of rules)
				expect(t.session.systemPrompt).not.toContain(rule);
			await extension.commands
				.get("ultra")
				.handler("", t.session.extensionRunner.createCommandContext());
			for (const rule of rules)
				expect(t.session.systemPrompt.split(rule)).toHaveLength(2);
			expect(t.session.messages).toHaveLength(0);
		},
	);

	it("does not advertise the skill when Ultra is disabled, even after completion or activation attempts", async () => {
		root = mkdtempSync(join(tmpdir(), "ultra-disabled-skill-"));
		mkdirSync(join(root, ".pi"), { recursive: true });
		writeFileSync(
			join(root, ".pi", "settings.json"),
			JSON.stringify({ ultra: { enabled: false, autoEnable: true } }),
		);
		t = await createTestSession({ cwd: root, extensions: [ultraPath()] });
		const command =
			t.session.extensionRunner.extensions[0].commands.get("ultra");
		await command.getArgumentCompletions("");
		await command.handler("", t.session.extensionRunner.createCommandContext());
		expect(t.session.getActiveToolNames()).not.toContain("run_workflow");
		expect(t.session.systemPrompt).not.toContain("ultra-authoring/SKILL.md");
	});

	it("injects the effective tier policy only while active and refreshes settings each turn", async () => {
		t = await createTestSession({ extensions: [ultraPath()] });
		const runner = t.session.extensionRunner;
		const emit = () =>
			runner.emitBeforeAgentStart("prompt", undefined, {
				cwd: t?.cwd ?? process.cwd(),
			});
		const policy = async () => (await emit()).systemPromptOptions.sections;
		expect((await policy()).ultra_model_tier_policy).toBeUndefined();
		expect((await policy()).ultra_fanout_tool_policy).toBeUndefined();
		const [extension] = runner.extensions;
		await extension.commands
			.get("ultra")
			.handler("", runner.createCommandContext());
		expect((await policy()).ultra_model_tier_policy).toBe(
			modelTierPrompt(DEFAULT_MODEL_TIERS),
		);
		expect((await policy()).ultra_fanout_tool_policy).toContain(
			"read, grep, find, ls, bash",
		);
		writeFileSync(
			join(t.cwd, ".pi", "settings.json"),
			JSON.stringify({
				ultra: {
					fanoutToolAllowlist: ["web"],
					modelTiers: {
						large: { instructions: "Architecture only." },
						fast_extract: {
							model: "custom/extractor",
							thinkingLevel: "low",
							instructions: "Select for literal extraction.",
						},
					},
				},
			}),
		);
		const updated = await policy();
		expect(updated.ultra_model_tier_policy).toContain("Architecture only.");
		expect(updated.ultra_model_tier_policy).not.toContain(
			DEFAULT_MODEL_TIERS.large.instructions,
		);
		expect(updated.ultra_model_tier_policy).toContain(
			"- fast_extract: custom/extractor; low; Select for literal extraction.",
		);
		expect(
			updated.ultra_model_tier_policy?.match(/# ultra model tier policy/g),
		).toHaveLength(1);
		expect(updated.ultra_fanout_tool_policy).toContain("bash, web");
		writeFileSync(
			join(t.cwd, ".pi", "settings.json"),
			JSON.stringify({ ultra: { enabled: false } }),
		);
		expect((await policy()).ultra_model_tier_policy).toBeUndefined();
		expect((await policy()).ultra_fanout_tool_policy).toBeUndefined();
	});

	it("advertises which step.flags enable sub-agent extension tools while active", async () => {
		root = mkdtempSync(join(tmpdir(), "ultra-subagent-flags-"));
		mkdirSync(join(root, ".pi"), { recursive: true });
		writeFileSync(
			join(root, ".pi", "settings.json"),
			JSON.stringify({
				ultra: { subagentExtensions: ["nushell", "chrome-cdp", "ask"] },
			}),
		);
		t = await createTestSession({ cwd: root, extensions: [ultraPath()] });
		const runner = t.session.extensionRunner;
		const sections = async () =>
			(
				await runner.emitBeforeAgentStart("prompt", undefined, {
					cwd: root ?? process.cwd(),
				})
			).systemPromptOptions.sections;
		expect((await sections()).ultra_subagent_flags).toBeUndefined();
		await runner.extensions[0].commands
			.get("ultra")
			.handler("", runner.createCommandContext());
		const prompt = (await sections()).ultra_subagent_flags ?? "";
		expect(prompt).toContain("# ultra sub-agent flags");
		expect(prompt).toContain(
			"- nushell: nu, nu_session; --nushell (boolean): Enable Nushell tools at startup instead of bash and powershell",
		);
		expect(prompt).toContain("- chrome-cdp: chrome_cdp; --chrome (boolean)");
		// ask registers tools but no flags, so it is not listed.
		expect(prompt).not.toContain("- ask:");
	});

	it("rejects unknown tiers before either surface starts work", async () => {
		t = await createTestSession({ extensions: [ultraPath()] });
		const [extension] = t.session.extensionRunner.extensions;
		const ctx = t.session.extensionRunner.createCommandContext();
		await extension.commands.get("ultra").handler("", ctx);
		const spec = {
			name: "unknown-tier",
			phases: [
				{
					id: "first",
					kind: "single",
					step: { summary: "First", prompt: "First", model: "medium" },
				},
				{
					id: "second",
					kind: "single",
					step: { summary: "Second", prompt: "Second", model: "missing_tier" },
				},
			],
		};
		const tool = extension.tools.get("run_workflow").definition;
		await expect(
			tool.execute("call", { spec }, undefined, undefined, ctx),
		).rejects.toThrow('unknown model tier "missing_tier"');
		const directory = join(t.cwd, ".pi", "workflows");
		mkdirSync(directory, { recursive: true });
		writeFileSync(join(directory, "unknown-tier.json"), JSON.stringify(spec));
		const custom = vi.fn();
		Object.assign(ctx.ui, { custom });
		await extension.commands.get("ultra").handler("run unknown-tier", ctx);
		expect(custom).not.toHaveBeenCalled();
		expect(t.events.uiCallsFor("notify").at(-1)?.args).toEqual([
			'ultra: unknown model tier "missing_tier". Configure ultra.modelTiers or use provider/model.',
			"error",
		]);
	});

	it.each(["command", "tool"])(
		"inherits the parent's session directory through the %s surface",
		async (surface) => {
			t = await createTestSession({ extensions: [ultraPath()] });
			saveWorkflow(t, "session-directory");
			const [extension] = t.session.extensionRunner.extensions;
			const sessionDir = join(t.cwd, "custom-sessions");
			const parent = SessionManager.create(t.cwd, sessionDir);
			const ctx = {
				...t.session.extensionRunner.createCommandContext(),
				sessionManager: parent,
			};
			const create = vi
				.spyOn(SessionManager, "create")
				.mockImplementation(() => {
					throw new Error("Stop before provider access");
				});

			if (surface === "command") {
				driveOverlay(ctx);
				await extension.commands
					.get("ultra")
					.handler("run session-directory", ctx);
			} else {
				await extension.commands.get("ultra").handler("", ctx);
				await extension.tools
					.get("run_workflow")
					.definition.execute(
						"call",
						{ name: "session-directory" },
						undefined,
						undefined,
						ctx,
					);
			}

			expect(create).toHaveBeenCalledWith(t.cwd, parent.getSessionDir(), {
				parentSession: parent.getSessionFile(),
			});
		},
	);

	it.each([false, true])(
		"bounds orchestrator content while retaining UI details (oversized: %s)",
		async (oversized) => {
			t = await createTestSession({ extensions: [ultraPath()] });
			const [extension] = t.session.extensionRunner.extensions;
			const ctx = t.session.extensionRunner.createCommandContext();
			await extension.commands.get("ultra").handler("", ctx);
			const answer = oversized ? "界".repeat(30_000) : "Final answer";
			const response = await extension.tools
				.get("run_workflow")
				.definition.execute(
					"call",
					{
						spec: {
							name: "context-budget",
							phases: [
								{
									id: "work",
									kind: "single",
									when: "{args.run}",
									step: { summary: "Unused", prompt: "Unused" },
								},
							],
							return: "{args.answer}",
						},
						args: { run: false, answer },
					},
					undefined,
					undefined,
					ctx,
				);
			const body = JSON.parse(response.content[0].text);
			try {
				expect(response.details.workflowResult.result).toBe(answer);
				expect(response.details.workflowResult.phaseResults).toEqual({
					work: [],
				});
				expect(body.phaseResults).toBeUndefined();
				if (oversized) {
					expect(body.truncated).toBe(true);
					expect(body.fullOutputPath).toBe(response.details.fullOutputPath);
					expect({
						...JSON.parse(readFileSync(body.fullOutputPath, "utf8")),
						fullOutputPath: body.fullOutputPath,
					}).toEqual(response.details.workflowResult);
				} else {
					expect(body.fullOutputPath).toBe(response.details.fullOutputPath);
					expect(
						JSON.parse(readFileSync(body.fullOutputPath, "utf8")).result,
					).toBe(answer);
					expect(body.result).toBe(answer);
					expect(body.phases).toEqual([
						{ id: "work", ran: 0, ok: 0, dropped: 0 },
					]);
					expect(body.truncated).toBeUndefined();
				}
			} finally {
				if (response.details.fullOutputPath)
					rmSync(dirname(response.details.fullOutputPath), {
						recursive: true,
						force: true,
					});
			}
		},
	);

	it.each(["command-output", "review", "research"])(
		"keeps %s command context bounded and its evidence retrievable",
		async (name) => {
			t = await createTestSession({ extensions: [ultraPath()] });
			saveWorkflow(t, name);
			const [extension] = t.session.extensionRunner.extensions;
			const ctx = t.session.extensionRunner.createCommandContext();
			const result = {
				workflow: name,
				phases: [{ id: "work", ran: 1, ok: 1, dropped: 0 }],
				phaseResults: {
					work: [{ ok: true, value: "INTERMEDIATE".repeat(10000) }],
				},
				phaseFailures: {},
				result: [{ status: "blocked", blockers: ["Missing evidence"] }],
				report: { summary: "Human report" },
				steered: false,
				dynamicExtensions: [],
				tokenUsage: {
					input: 1,
					output: 2,
					total: 3,
					cacheRead: 0,
					cacheWrite: 0,
					cost: 0,
				},
			};
			Object.assign(ctx.ui, {
				custom: vi.fn().mockResolvedValue({ status: "completed", result }),
			});
			await extension.commands.get("ultra").handler(`run ${name}`, ctx);
			expect(
				t.events
					.uiCallsFor("notify")
					.some(
						(call) =>
							call.args[0] ===
							`ultra: "${name}" complete · 1 phases · $0.0000 · 2 output.`,
					),
			).toBe(true);
			const message = t.session.messages.find(
				(message) =>
					"customType" in message && message.customType === "ultra-result",
			);
			if (!message || typeof message.content !== "string")
				throw new Error("Missing command result");
			const body = JSON.parse(message.content);
			try {
				expect(body.result).toEqual(result.result);
				expect(body.report).toEqual(result.report);
				expect(body.executionStatus).toBe("finished");
				expect(body.phaseResults).toBeUndefined();
				expect(message.content).not.toContain("INTERMEDIATE");
				expect(message.details).toEqual({
					...result,
					fullOutputPath: body.fullOutputPath,
				});
				expect(JSON.parse(readFileSync(body.fullOutputPath, "utf8"))).toEqual(
					result,
				);
			} finally {
				rmSync(dirname(body.fullOutputPath), { recursive: true, force: true });
			}
		},
	);

	it("completes subcommands first and saved workflows below run", async () => {
		t = await createTestSession({ extensions: [ultraPath()] });
		const [extension] = t.session.extensionRunner.extensions;
		const complete = extension.commands.get("ultra").getArgumentCompletions;
		expect(extension.tools.has("run_workflow")).toBe(false);
		const subcommands = await complete("");
		const workflows = await complete("run ");

		expect(extension.tools.has("run_workflow")).toBe(true);
		expect(t.session.getActiveToolNames()).not.toContain("run_workflow");
		expect(subcommands.map((item: { value: string }) => item.value)).toEqual([
			"exec",
			"run",
		]);
		expect(workflows.map((item: { value: string }) => item.value)).toEqual(
			expect.arrayContaining(["run research", "run review"]),
		);
	});

	it("activates the workflow tool and notifies without starting a turn", async () => {
		t = await createTestSession({ extensions: [ultraPath()] });
		const [extension] = t.session.extensionRunner.extensions;
		const beforeMessages = t.session.messages.length;
		const beforeNotifications = t.events.uiCallsFor("notify").length;

		expect(t.session.getActiveToolNames()).not.toContain("run_workflow");
		expect(t.session.systemPrompt).not.toContain("- run_workflow:");
		await extension.commands
			.get("ultra")
			.handler("", t.session.extensionRunner.createCommandContext());

		expect(t.session.getActiveToolNames()).toContain("run_workflow");
		expect(t.session.systemPrompt).toContain(
			"- run_workflow: run saved or inline multi-agent workflows.",
		);
		expect(t.session.systemPrompt).toContain(
			"Use run_workflow when decomposition into independent or sequential child-agent steps reduces context or verification risk.",
		);
		expect(t.session.messages).toHaveLength(beforeMessages);
		const notifications = t.events.uiCallsFor("notify");
		expect(notifications).toHaveLength(beforeNotifications + 1);
		expect(notifications.at(-1)?.args).toEqual([
			"ultra: run_workflow tool enabled.",
			"info",
		]);
	});

	it("sends exec instructions as the visible user message that starts the turn", async () => {
		t = await createTestSession({ extensions: [ultraPath()] });
		const [extension] = t.session.extensionRunner.extensions;
		const sendUserMessage = vi
			.spyOn(t.session, "sendUserMessage")
			.mockResolvedValue();
		const beforeNotifications = t.events.uiCallsFor("notify").length;

		await extension.commands
			.get("ultra")
			.handler(
				"exec inspect this failure once",
				t.session.extensionRunner.createCommandContext(),
			);

		expect(t.session.getActiveToolNames()).toContain("run_workflow");
		expect(sendUserMessage).toHaveBeenCalledOnce();
		const prompt = sendUserMessage.mock.calls[0]?.[0];
		expect(prompt).toBe(
			"# ultra exec\n\n# user request\n\ninspect this failure once",
		);
		const guidance =
			extension.tools.get("run_workflow").definition.promptGuidelines;
		for (const rule of guidance) {
			expect(t.session.systemPrompt.split(rule)).toHaveLength(2);
			expect(prompt).not.toContain(rule);
		}
		expect(t.session.systemPrompt).not.toContain("# Authoring ultra workflows");
		expect(t.session.systemPrompt).not.toContain("## Worked example");
		expect(prompt).not.toContain("The workflow tool is now available");
		const injected = await t.session.extensionRunner.emitBeforeAgentStart(
			prompt as string,
			undefined,
			{ cwd: t.cwd },
		);
		expect(injected.systemPromptOptions.sections.ultra_model_tier_policy).toBe(
			modelTierPrompt(DEFAULT_MODEL_TIERS),
		);
		expect(injected.systemPromptOptions.forceSystemPrompt).toBeUndefined();
		expect(prompt).not.toContain(DEFAULT_MODEL_TIERS.large.instructions);
		expect(t.events.uiCallsFor("notify")).toHaveLength(beforeNotifications);
	});

	it("renders workflow progress as an overlay without replacing session content", async () => {
		t = await createTestSession({ extensions: [ultraPath()] });
		const [extension] = t.session.extensionRunner.extensions;
		const ctx = t.session.extensionRunner.createCommandContext();
		const custom = vi.fn().mockResolvedValue(null);
		Object.assign(ctx.ui, { custom });

		await extension.commands
			.get("ultra")
			.handler("run review inspect scrolling", ctx);

		expect(custom).toHaveBeenCalledOnce();
		expect(custom.mock.calls[0][1]).toEqual({
			overlay: true,
			overlayOptions: {
				anchor: "bottom-left",
				width: "100%",
				maxHeight: "100%",
			},
		});
	});

	it("surfaces workflow errors instead of misreporting them as user aborts", async () => {
		t = await createTestSession({ extensions: [ultraPath()] });
		saveWorkflow(t, "runtime-error", "missing-live-capability");
		const [extension] = t.session.extensionRunner.extensions;
		const ctx = t.session.extensionRunner.createCommandContext();
		driveOverlay(ctx);

		await extension.commands.get("ultra").handler("run runtime-error", ctx);

		expect(t.events.uiCallsFor("notify").at(-1)?.args).toEqual([
			'ultra: dynamic extension "missing-live-capability" has no canonical revision. Run a builder with all four dynamic-extension tools.',
			"error",
		]);
	});

	it("prepares dynamic extension runs on both command and tool surfaces", async () => {
		const message =
			'ultra: dynamic extension "missing-live-capability" has no canonical revision. Run a builder with all four dynamic-extension tools.';
		t = await createTestSession({ extensions: [ultraPath()] });
		saveWorkflow(t, "runtime-error-command", "missing-live-capability");
		saveWorkflow(t, "runtime-error-tool", "missing-live-capability");
		const [extension] = t.session.extensionRunner.extensions;
		const ctx = t.session.extensionRunner.createCommandContext();
		driveOverlay(ctx);
		await extension.commands
			.get("ultra")
			.handler("run runtime-error-command", ctx);
		expect(t.events.uiCallsFor("notify").at(-1)?.args).toEqual([
			message,
			"error",
		]);

		await extension.commands.get("ultra").handler("", ctx);
		await expect(
			extension.tools
				.get("run_workflow")
				.definition.execute(
					"call",
					{ name: "runtime-error-tool" },
					undefined,
					undefined,
					ctx,
				),
		).rejects.toThrow(message);
	});

	it("reports confirmed whole-run cancellation as aborted while retaining its aggregate", async () => {
		t = await createTestSession({ extensions: [ultraPath()] });
		saveWorkflow(t, "cancelled-run");
		const [extension] = t.session.extensionRunner.extensions;
		const ctx = t.session.extensionRunner.createCommandContext();
		driveOverlay(ctx, (overlay) => {
			overlay.handleInput("\u001b");
			overlay.handleInput("\u001b");
		});

		await extension.commands.get("ultra").handler("run cancelled-run", ctx);

		expect(t.events.uiCallsFor("notify").at(-1)?.args).toEqual([
			'ultra: "cancelled-run" aborted · 1 phases · $0.0000 · 0 output.',
			"warning",
		]);
		expect(
			t.events
				.uiCallsFor("notify")
				.some((call) => String(call.args[0]).includes("complete")),
		).toBe(false);
		const message = t.session.messages.find(
			(message) =>
				"customType" in message && message.customType === "ultra-result",
		);
		const body = JSON.parse(String(message?.content));
		try {
			expect(body.executionStatus).toBe("aborted");
			expect(
				JSON.parse(readFileSync(body.fullOutputPath, "utf8")).aborted,
			).toBe(true);
		} finally {
			rmSync(dirname(body.fullOutputPath), { recursive: true, force: true });
		}
	});

	it("refuses an unknown static sub-agent extension", async () => {
		t = await createTestSession({ extensions: [ultraPath()] });
		mkdirSync(join(t.cwd, ".pi"), { recursive: true });
		writeFileSync(
			join(t.cwd, ".pi", "settings.json"),
			JSON.stringify({
				ultra: { subagentExtensions: ["missing-extension"] },
			}),
		);
		const [extension] = t.session.extensionRunner.extensions;

		await extension.commands
			.get("ultra")
			.handler("run review", t.session.extensionRunner.createCommandContext());

		expect(t.events.uiCallsFor("notify").at(-1)?.args).toEqual([
			"ultra: unknown sub-agent extension: missing-extension",
			"error",
		]);
	});

	it("starts research without knowing which extension supplies research tools", async () => {
		t = await createTestSession({ extensions: [ultraPath()] });
		const [extension] = t.session.extensionRunner.extensions;
		const ctx = t.session.extensionRunner.createCommandContext();
		const custom = vi.fn().mockResolvedValue(null);
		Object.assign(ctx.ui, { custom });

		await extension.commands.get("ultra").handler("run research question", ctx);

		expect(custom).toHaveBeenCalledOnce();
		expect(
			t.events
				.uiCallsFor("notify")
				.some((call) => String(call.args[0]).includes("unavailable tools")),
		).toBe(false);
	});

	it("discovers a workflow saved after the extension loaded without /reload", async () => {
		t = await createTestSession({ extensions: [ultraPath()] });
		const [extension] = t.session.extensionRunner.extensions;
		const complete = () =>
			extension.commands.get("ultra").getArgumentCompletions("run ");
		await complete();

		const previousCwd = process.cwd();
		try {
			process.chdir(t.cwd);
			const directory = join(t.cwd, ".pi", "workflows");
			mkdirSync(directory, { recursive: true });
			writeFileSync(
				join(directory, "fresh.json"),
				`${JSON.stringify({
					name: "fresh",
					phases: [
						{ id: "p", kind: "single", step: { summary: "hi", prompt: "hi" } },
					],
				})}\n`,
			);

			const items = await complete();
			expect(items.map((item: { value: string }) => item.value)).toContain(
				"run fresh",
			);
		} finally {
			process.chdir(previousCwd);
		}
	});

	it.each([
		{
			params: {},
			expected: ['provide either a workflow "name" or an inline "spec"'],
		},
		{
			params: { name: "does-not-exist" },
			expected: ['unknown workflow "does-not-exist"', "Available:", "review"],
		},
		{ params: { args: "not-an-object" }, expected: ["args", "object"] },
		{
			params: {
				spec: {
					name: "invalid",
					phases: [{ id: "work", kind: "single", step: { prompt: "Work" } }],
				},
			},
			expected: ["summary"],
		},
	])(
		"delivers actionable argument errors to the agent and the tool UI: $params",
		async ({ params, expected }) => {
			t = await createTestSession({ extensions: [ultraPath()] });
			const [extension] = t.session.extensionRunner.extensions;
			const ctx = t.session.extensionRunner.createCommandContext();
			await extension.commands.get("ultra").handler("", ctx);
			// Use the real tool boundary: harness run() catches execute errors and
			// returns them as successes when propagateErrors is false.
			const faux = fauxProvider({ provider: "ultra-argument-errors" });
			ctx.modelRegistry.registerProvider(faux.provider);
			await t.session.setModel(faux.getModel());
			const followUp = vi.fn((context: Context) => {
				const result = context.messages.findLast(
					(message) => message.role === "toolResult",
				);
				expect(result?.isError).toBe(true);
				const error = result?.content
					.filter((part) => part.type === "text")
					.map((part) => part.text)
					.join("\n");
				for (const text of expected) expect(error).toContain(text);
				return fauxAssistantMessage("Error received.");
			});
			faux.setResponses([
				fauxAssistantMessage(fauxToolCall("run_workflow", params), {
					stopReason: "toolUse",
				}),
				followUp,
			]);
			await t.session.prompt("Exercise invalid workflow arguments");
			expect(followUp).toHaveBeenCalledOnce();
			const [result] = t.events.toolResultsFor("run_workflow");
			expect(result.isError).toBe(true);
			expect(result.mocked).toBe(false);
			for (const text of expected) expect(result.text).toContain(text);
			const message = t.session.messages.find(
				(message) => message.role === "toolResult",
			);
			expect(message).toMatchObject({
				role: "toolResult",
				isError: true,
				content: result.content,
			});
			expect(t.session.messages.at(-1)).toMatchObject({ role: "assistant" });

			initTheme();
			const theme = {
				fg: (_color: string, text: string) => text,
				bold: (text: string) => text,
			} as Theme;
			const tool = extension.tools.get("run_workflow").definition;
			const collapsed = tool
				.renderResult(message, { expanded: false, isPartial: false }, theme, {
					isError: true,
				})
				.render(120)
				.join("\n");
			expect(collapsed).toContain("ultra · failed:");
			expect(collapsed).not.toContain("0/0 done");
			const expanded = tool
				.renderResult(message, { expanded: true, isPartial: false }, theme, {
					isError: true,
				})
				.render(120)
				.join("\n");
			for (const text of expected) expect(expanded).toContain(text);
		},
	);

	it("exposes the workflow schema while rejecting malformed specs before execution", async () => {
		t = await createTestSession({ extensions: [ultraPath()] });
		const [extension] = t.session.extensionRunner.extensions;
		await extension.commands.get("ultra").getArgumentCompletions("");
		const tool = extension.tools.get("run_workflow").definition;
		const specSchema = tool.parameters.properties.spec;
		expect(tool.renderCall).toBeTypeOf("function");
		expect(specSchema.type).toBe("object");
		expect(specSchema.properties).toHaveProperty("name");
		expect(specSchema.properties).toHaveProperty("phases");
		expect(specSchema.properties).toHaveProperty("report");
		expect(specSchema.additionalProperties).not.toBe(true);

		expect(tool.description).toContain(
			"run saved or inline multi-agent workflows.",
		);
		expect(tool.promptSnippet).toBe(
			"run saved or inline multi-agent workflows.",
		);
		expect(tool.promptGuidelines).toEqual([
			"Use run_workflow when decomposition into independent or sequential child-agent steps reduces context or verification risk.",
			"Use dynamic extensions when a workflow needs a reusable, task-specific capability; read the ultra-authoring skill for how to build and load them.",
		]);
		expect(tool.parameters.properties.background).toBeUndefined();
		expect(JSON.parse(JSON.stringify(specSchema))).toEqual(
			JSON.parse(JSON.stringify(WorkflowSpecSchema)),
		);
		const step = { summary: "Work", prompt: "Work" };
		for (const spec of [
			{},
			{ name: "invalid", phases: [] },
			{ name: "invalid", phases: [{ id: "work", kind: "fanout", step }] },
			{
				name: "invalid",
				phases: [{ id: "work", kind: "single", step: { prompt: "Work" } }],
			},
			{
				name: "invalid",
				phases: [
					{
						id: "work",
						kind: "single",
						step: { ...step, thinkingLevel: "invalid" },
					},
				],
			},
			{
				name: "invalid",
				phases: [
					{ id: "work", kind: "single", step: { ...step, schema: "missing" } },
				],
			},
			{ name: "invalid", phases: [{ id: "args", kind: "single", step }] },
		]) {
			await expect(
				tool.execute(
					"invalid",
					{ spec },
					undefined,
					undefined,
					t.session.extensionRunner.createContext(),
				),
			).rejects.toThrow("Invalid workflow spec");
		}
	});
});