import { mkdirSync, mkdtempSync, readdirSync, readFileSync, rmSync, writeFileSync, } from "node:fs"; import { tmpdir } from "node:os"; import { dirname, join } from "node:path"; import { fileURLToPath } from "node:url"; import { type Context, fauxAssistantMessage, fauxProvider, fauxToolCall, } from "@earendil-works/pi-ai"; import { type ExtensionCommandContext, initTheme, SessionManager, type Theme, } from "@earendil-works/pi-coding-agent"; import { createTestSession as createHarnessSession, type TestSession, } from "@marcfargas/pi-test-harness"; import { afterEach, describe, expect, it, vi } from "vitest"; import { createTestSession as createSandboxSession } from "../../../test/harness"; import { DEFAULT_MODEL_TIERS, modelTierPrompt } from "../settings.ts"; import { WorkflowSpecSchema } from "../spec.ts"; let t: TestSession | undefined; let root: string | undefined; afterEach(() => { t?.dispose(); if (root) rmSync(root, { recursive: true, force: true }); t = undefined; root = undefined; vi.unstubAllEnvs(); vi.restoreAllMocks(); }); function ultraPath(): string { return fileURLToPath(new URL("../index.ts", import.meta.url)); } async function createTestSession(options: { extensions: string[]; cwd?: string; }): Promise { root = options.cwd ?? mkdtempSync(join(tmpdir(), "ultra-harness-")); const agentDir = join(root, "agent"); mkdirSync(agentDir, { recursive: true }); vi.stubEnv("PI_CODING_AGENT_DIR", agentDir); if (!options.cwd) { mkdirSync(join(root, ".pi"), { recursive: true }); writeFileSync( join(root, ".pi", "settings.json"), JSON.stringify({ ultra: { autoEnable: false } }), ); } return createHarnessSession({ ...options, cwd: root }); } function saveWorkflow( session: TestSession, name: string, dynamicExtension?: string, ): void { const directory = join(session.cwd, ".pi", "workflows"); mkdirSync(directory, { recursive: true }); writeFileSync( join(directory, `${name}.json`), `${JSON.stringify({ name, phases: [ { id: "work", kind: "single", step: { summary: "Work", prompt: "Work", ...(dynamicExtension ? { dynamicExtensions: [dynamicExtension] } : {}), }, }, ], })}\n`, ); } function driveOverlay( ctx: ExtensionCommandContext, interact?: (overlay: { handleInput(data: string): void }) => void, ): void { Object.assign(ctx.ui, { custom: vi.fn( async ( factory: ( host: unknown, theme: unknown, keybindings: unknown, done: (value: unknown) => void, ) => { handleInput(data: string): void }, ) => new Promise((resolve) => { const overlay = factory( { requestRender: vi.fn() }, {}, { matches: () => false }, resolve, ); interact?.(overlay); }), ), }); } describe("ultra extension", () => { it("loads only the command and renderer in the real Pi runtime", async () => { const indexPath = ultraPath(); t = await createTestSession({ extensions: [indexPath] }); const [extension] = t.session.extensionRunner.extensions; expect(extension.path).toBe(indexPath); expect(extension.tools.has("run_workflow")).toBe(false); expect(readFileSync(indexPath, "utf8")).toContain( 'import("./implementation.ts")', ); expect(extension.commands.has("ultra")).toBe(true); expect(extension.commands.has("ultra-runs")).toBe(false); }); it("writes disabled command outcome in its debug sandbox", async () => { t = await createSandboxSession({ env: { PI_ULTRA_DEBUG: "1" }, extensions: [ultraPath()], }); const extension = t.session.extensionRunner.extensions[0]; const context = t.session.extensionRunner.createCommandContext(); for (const handler of extension.handlers.get("session_start")) await handler({ type: "session_start" }, context); mkdirSync(join(t.cwd, ".pi"), { recursive: true }); writeFileSync( join(t.cwd, ".pi", "settings.json"), JSON.stringify({ ultra: { enabled: false } }), ); await extension.commands.get("ultra").handler("", context); for (const handler of extension.handlers.get("session_shutdown")) await handler({ type: "session_shutdown" }, context); const directory = join(t.cwd, ".test-state", "pi-ext", "debug", "ultra"); const records = readFileSync( join(directory, readdirSync(directory)[0]), "utf8", ) .trim() .split("\n") .map((line) => JSON.parse(line)); expect( records.filter((record) => record.event === "command.handle.start"), ).toHaveLength(1); expect( records.filter((record) => record.event === "command.handle.finish"), ).toEqual([expect.objectContaining({ outcome: "disabled" })]); expect(records.map((record) => record.event)).toEqual( expect.arrayContaining(["session.start", "session.shutdown"]), ); }); it("auto-enables the workflow tool when configured", async () => { root = mkdtempSync(join(tmpdir(), "ultra-auto-enable-")); mkdirSync(join(root, ".pi"), { recursive: true }); writeFileSync( join(root, ".pi", "settings.json"), JSON.stringify({ ultra: { autoEnable: true } }), ); t = await createTestSession({ cwd: root, extensions: [ultraPath()] }); const [extension] = t.session.extensionRunner.extensions; expect(extension.tools.has("run_workflow")).toBe(true); expect(t.session.getActiveToolNames()).toContain("run_workflow"); expect(t.session.systemPrompt).toContain("- run_workflow:"); }); it.each([false, true])( "registers the authoring skill independently of tool activation (auto-enable: %s)", async (autoEnable) => { root = mkdtempSync(join(tmpdir(), "ultra-shared-guidance-")); mkdirSync(join(root, ".pi"), { recursive: true }); writeFileSync( join(root, ".pi", "settings.json"), JSON.stringify({ ultra: { autoEnable } }), ); t = await createTestSession({ cwd: root, extensions: [ultraPath()] }); const skillPath = fileURLToPath( new URL("../prompts/ultra-authoring/SKILL.md", import.meta.url), ); const skill = t.session.resourceLoader .getSkills() .skills.find((candidate) => candidate.name === "ultra-authoring"); expect(skill?.filePath).toBe(skillPath); expect(t.session.systemPrompt).toContain(skillPath); expect(t.session.systemPrompt).not.toContain( readFileSync(skillPath, "utf8").trim(), ); expect(t.session.systemPrompt).not.toContain("## Worked example"); const [extension] = t.session.extensionRunner.extensions; if (!autoEnable) { await extension.commands .get("ultra") .handler("", t.session.extensionRunner.createCommandContext()); } const rules: string[] = extension.tools.get("run_workflow").definition.promptGuidelines; expect(rules).toEqual([ "Use run_workflow when decomposition into independent or sequential child-agent steps reduces context or verification risk.", "Use dynamic extensions when a workflow needs a reusable, task-specific capability; read the ultra-authoring skill for how to build and load them.", ]); for (const rule of rules) expect(t.session.systemPrompt.split(rule)).toHaveLength(2); t.session.setActiveToolsByName( t.session .getActiveToolNames() .filter((name) => name !== "run_workflow"), ); expect(t.session.systemPrompt).toContain(skillPath); for (const rule of rules) expect(t.session.systemPrompt).not.toContain(rule); await extension.commands .get("ultra") .handler("", t.session.extensionRunner.createCommandContext()); for (const rule of rules) expect(t.session.systemPrompt.split(rule)).toHaveLength(2); expect(t.session.messages).toHaveLength(0); }, ); it("does not advertise the skill when Ultra is disabled, even after completion or activation attempts", async () => { root = mkdtempSync(join(tmpdir(), "ultra-disabled-skill-")); mkdirSync(join(root, ".pi"), { recursive: true }); writeFileSync( join(root, ".pi", "settings.json"), JSON.stringify({ ultra: { enabled: false, autoEnable: true } }), ); t = await createTestSession({ cwd: root, extensions: [ultraPath()] }); const command = t.session.extensionRunner.extensions[0].commands.get("ultra"); await command.getArgumentCompletions(""); await command.handler("", t.session.extensionRunner.createCommandContext()); expect(t.session.getActiveToolNames()).not.toContain("run_workflow"); expect(t.session.systemPrompt).not.toContain("ultra-authoring/SKILL.md"); }); it("injects the effective tier policy only while active and refreshes settings each turn", async () => { t = await createTestSession({ extensions: [ultraPath()] }); const runner = t.session.extensionRunner; const emit = () => runner.emitBeforeAgentStart("prompt", undefined, { cwd: t?.cwd ?? process.cwd(), }); const policy = async () => (await emit()).systemPromptOptions.sections; expect((await policy()).ultra_model_tier_policy).toBeUndefined(); expect((await policy()).ultra_fanout_tool_policy).toBeUndefined(); const [extension] = runner.extensions; await extension.commands .get("ultra") .handler("", runner.createCommandContext()); expect((await policy()).ultra_model_tier_policy).toBe( modelTierPrompt(DEFAULT_MODEL_TIERS), ); expect((await policy()).ultra_fanout_tool_policy).toContain( "read, grep, find, ls, bash", ); writeFileSync( join(t.cwd, ".pi", "settings.json"), JSON.stringify({ ultra: { fanoutToolAllowlist: ["web"], modelTiers: { large: { instructions: "Architecture only." }, fast_extract: { model: "custom/extractor", thinkingLevel: "low", instructions: "Select for literal extraction.", }, }, }, }), ); const updated = await policy(); expect(updated.ultra_model_tier_policy).toContain("Architecture only."); expect(updated.ultra_model_tier_policy).not.toContain( DEFAULT_MODEL_TIERS.large.instructions, ); expect(updated.ultra_model_tier_policy).toContain( "- fast_extract: custom/extractor; low; Select for literal extraction.", ); expect( updated.ultra_model_tier_policy?.match(/# ultra model tier policy/g), ).toHaveLength(1); expect(updated.ultra_fanout_tool_policy).toContain("bash, web"); writeFileSync( join(t.cwd, ".pi", "settings.json"), JSON.stringify({ ultra: { enabled: false } }), ); expect((await policy()).ultra_model_tier_policy).toBeUndefined(); expect((await policy()).ultra_fanout_tool_policy).toBeUndefined(); }); it("advertises which step.flags enable sub-agent extension tools while active", async () => { root = mkdtempSync(join(tmpdir(), "ultra-subagent-flags-")); mkdirSync(join(root, ".pi"), { recursive: true }); writeFileSync( join(root, ".pi", "settings.json"), JSON.stringify({ ultra: { subagentExtensions: ["nushell", "chrome-cdp", "ask"] }, }), ); t = await createTestSession({ cwd: root, extensions: [ultraPath()] }); const runner = t.session.extensionRunner; const sections = async () => ( await runner.emitBeforeAgentStart("prompt", undefined, { cwd: root ?? process.cwd(), }) ).systemPromptOptions.sections; expect((await sections()).ultra_subagent_flags).toBeUndefined(); await runner.extensions[0].commands .get("ultra") .handler("", runner.createCommandContext()); const prompt = (await sections()).ultra_subagent_flags ?? ""; expect(prompt).toContain("# ultra sub-agent flags"); expect(prompt).toContain( "- nushell: nu, nu_session; --nushell (boolean): Enable Nushell tools at startup instead of bash and powershell", ); expect(prompt).toContain("- chrome-cdp: chrome_cdp; --chrome (boolean)"); // ask registers tools but no flags, so it is not listed. expect(prompt).not.toContain("- ask:"); }); it("rejects unknown tiers before either surface starts work", async () => { t = await createTestSession({ extensions: [ultraPath()] }); const [extension] = t.session.extensionRunner.extensions; const ctx = t.session.extensionRunner.createCommandContext(); await extension.commands.get("ultra").handler("", ctx); const spec = { name: "unknown-tier", phases: [ { id: "first", kind: "single", step: { summary: "First", prompt: "First", model: "medium" }, }, { id: "second", kind: "single", step: { summary: "Second", prompt: "Second", model: "missing_tier" }, }, ], }; const tool = extension.tools.get("run_workflow").definition; await expect( tool.execute("call", { spec }, undefined, undefined, ctx), ).rejects.toThrow('unknown model tier "missing_tier"'); const directory = join(t.cwd, ".pi", "workflows"); mkdirSync(directory, { recursive: true }); writeFileSync(join(directory, "unknown-tier.json"), JSON.stringify(spec)); const custom = vi.fn(); Object.assign(ctx.ui, { custom }); await extension.commands.get("ultra").handler("run unknown-tier", ctx); expect(custom).not.toHaveBeenCalled(); expect(t.events.uiCallsFor("notify").at(-1)?.args).toEqual([ 'ultra: unknown model tier "missing_tier". Configure ultra.modelTiers or use provider/model.', "error", ]); }); it.each(["command", "tool"])( "inherits the parent's session directory through the %s surface", async (surface) => { t = await createTestSession({ extensions: [ultraPath()] }); saveWorkflow(t, "session-directory"); const [extension] = t.session.extensionRunner.extensions; const sessionDir = join(t.cwd, "custom-sessions"); const parent = SessionManager.create(t.cwd, sessionDir); const ctx = { ...t.session.extensionRunner.createCommandContext(), sessionManager: parent, }; const create = vi .spyOn(SessionManager, "create") .mockImplementation(() => { throw new Error("Stop before provider access"); }); if (surface === "command") { driveOverlay(ctx); await extension.commands .get("ultra") .handler("run session-directory", ctx); } else { await extension.commands.get("ultra").handler("", ctx); await extension.tools .get("run_workflow") .definition.execute( "call", { name: "session-directory" }, undefined, undefined, ctx, ); } expect(create).toHaveBeenCalledWith(t.cwd, parent.getSessionDir(), { parentSession: parent.getSessionFile(), }); }, ); it.each([false, true])( "bounds orchestrator content while retaining UI details (oversized: %s)", async (oversized) => { t = await createTestSession({ extensions: [ultraPath()] }); const [extension] = t.session.extensionRunner.extensions; const ctx = t.session.extensionRunner.createCommandContext(); await extension.commands.get("ultra").handler("", ctx); const answer = oversized ? "界".repeat(30_000) : "Final answer"; const response = await extension.tools .get("run_workflow") .definition.execute( "call", { spec: { name: "context-budget", phases: [ { id: "work", kind: "single", when: "{args.run}", step: { summary: "Unused", prompt: "Unused" }, }, ], return: "{args.answer}", }, args: { run: false, answer }, }, undefined, undefined, ctx, ); const body = JSON.parse(response.content[0].text); try { expect(response.details.workflowResult.result).toBe(answer); expect(response.details.workflowResult.phaseResults).toEqual({ work: [], }); expect(body.phaseResults).toBeUndefined(); if (oversized) { expect(body.truncated).toBe(true); expect(body.fullOutputPath).toBe(response.details.fullOutputPath); expect({ ...JSON.parse(readFileSync(body.fullOutputPath, "utf8")), fullOutputPath: body.fullOutputPath, }).toEqual(response.details.workflowResult); } else { expect(body.fullOutputPath).toBe(response.details.fullOutputPath); expect( JSON.parse(readFileSync(body.fullOutputPath, "utf8")).result, ).toBe(answer); expect(body.result).toBe(answer); expect(body.phases).toEqual([ { id: "work", ran: 0, ok: 0, dropped: 0 }, ]); expect(body.truncated).toBeUndefined(); } } finally { if (response.details.fullOutputPath) rmSync(dirname(response.details.fullOutputPath), { recursive: true, force: true, }); } }, ); it.each(["command-output", "review", "research"])( "keeps %s command context bounded and its evidence retrievable", async (name) => { t = await createTestSession({ extensions: [ultraPath()] }); saveWorkflow(t, name); const [extension] = t.session.extensionRunner.extensions; const ctx = t.session.extensionRunner.createCommandContext(); const result = { workflow: name, phases: [{ id: "work", ran: 1, ok: 1, dropped: 0 }], phaseResults: { work: [{ ok: true, value: "INTERMEDIATE".repeat(10000) }], }, phaseFailures: {}, result: [{ status: "blocked", blockers: ["Missing evidence"] }], report: { summary: "Human report" }, steered: false, dynamicExtensions: [], tokenUsage: { input: 1, output: 2, total: 3, cacheRead: 0, cacheWrite: 0, cost: 0, }, }; Object.assign(ctx.ui, { custom: vi.fn().mockResolvedValue({ status: "completed", result }), }); await extension.commands.get("ultra").handler(`run ${name}`, ctx); expect( t.events .uiCallsFor("notify") .some( (call) => call.args[0] === `ultra: "${name}" complete · 1 phases · $0.0000 · 2 output.`, ), ).toBe(true); const message = t.session.messages.find( (message) => "customType" in message && message.customType === "ultra-result", ); if (!message || typeof message.content !== "string") throw new Error("Missing command result"); const body = JSON.parse(message.content); try { expect(body.result).toEqual(result.result); expect(body.report).toEqual(result.report); expect(body.executionStatus).toBe("finished"); expect(body.phaseResults).toBeUndefined(); expect(message.content).not.toContain("INTERMEDIATE"); expect(message.details).toEqual({ ...result, fullOutputPath: body.fullOutputPath, }); expect(JSON.parse(readFileSync(body.fullOutputPath, "utf8"))).toEqual( result, ); } finally { rmSync(dirname(body.fullOutputPath), { recursive: true, force: true }); } }, ); it("completes subcommands first and saved workflows below run", async () => { t = await createTestSession({ extensions: [ultraPath()] }); const [extension] = t.session.extensionRunner.extensions; const complete = extension.commands.get("ultra").getArgumentCompletions; expect(extension.tools.has("run_workflow")).toBe(false); const subcommands = await complete(""); const workflows = await complete("run "); expect(extension.tools.has("run_workflow")).toBe(true); expect(t.session.getActiveToolNames()).not.toContain("run_workflow"); expect(subcommands.map((item: { value: string }) => item.value)).toEqual([ "exec", "run", ]); expect(workflows.map((item: { value: string }) => item.value)).toEqual( expect.arrayContaining(["run research", "run review"]), ); }); it("activates the workflow tool and notifies without starting a turn", async () => { t = await createTestSession({ extensions: [ultraPath()] }); const [extension] = t.session.extensionRunner.extensions; const beforeMessages = t.session.messages.length; const beforeNotifications = t.events.uiCallsFor("notify").length; expect(t.session.getActiveToolNames()).not.toContain("run_workflow"); expect(t.session.systemPrompt).not.toContain("- run_workflow:"); await extension.commands .get("ultra") .handler("", t.session.extensionRunner.createCommandContext()); expect(t.session.getActiveToolNames()).toContain("run_workflow"); expect(t.session.systemPrompt).toContain( "- run_workflow: run saved or inline multi-agent workflows.", ); expect(t.session.systemPrompt).toContain( "Use run_workflow when decomposition into independent or sequential child-agent steps reduces context or verification risk.", ); expect(t.session.messages).toHaveLength(beforeMessages); const notifications = t.events.uiCallsFor("notify"); expect(notifications).toHaveLength(beforeNotifications + 1); expect(notifications.at(-1)?.args).toEqual([ "ultra: run_workflow tool enabled.", "info", ]); }); it("sends exec instructions as the visible user message that starts the turn", async () => { t = await createTestSession({ extensions: [ultraPath()] }); const [extension] = t.session.extensionRunner.extensions; const sendUserMessage = vi .spyOn(t.session, "sendUserMessage") .mockResolvedValue(); const beforeNotifications = t.events.uiCallsFor("notify").length; await extension.commands .get("ultra") .handler( "exec inspect this failure once", t.session.extensionRunner.createCommandContext(), ); expect(t.session.getActiveToolNames()).toContain("run_workflow"); expect(sendUserMessage).toHaveBeenCalledOnce(); const prompt = sendUserMessage.mock.calls[0]?.[0]; expect(prompt).toBe( "# ultra exec\n\n# user request\n\ninspect this failure once", ); const guidance = extension.tools.get("run_workflow").definition.promptGuidelines; for (const rule of guidance) { expect(t.session.systemPrompt.split(rule)).toHaveLength(2); expect(prompt).not.toContain(rule); } expect(t.session.systemPrompt).not.toContain("# Authoring ultra workflows"); expect(t.session.systemPrompt).not.toContain("## Worked example"); expect(prompt).not.toContain("The workflow tool is now available"); const injected = await t.session.extensionRunner.emitBeforeAgentStart( prompt as string, undefined, { cwd: t.cwd }, ); expect(injected.systemPromptOptions.sections.ultra_model_tier_policy).toBe( modelTierPrompt(DEFAULT_MODEL_TIERS), ); expect(injected.systemPromptOptions.forceSystemPrompt).toBeUndefined(); expect(prompt).not.toContain(DEFAULT_MODEL_TIERS.large.instructions); expect(t.events.uiCallsFor("notify")).toHaveLength(beforeNotifications); }); it("renders workflow progress as an overlay without replacing session content", async () => { t = await createTestSession({ extensions: [ultraPath()] }); const [extension] = t.session.extensionRunner.extensions; const ctx = t.session.extensionRunner.createCommandContext(); const custom = vi.fn().mockResolvedValue(null); Object.assign(ctx.ui, { custom }); await extension.commands .get("ultra") .handler("run review inspect scrolling", ctx); expect(custom).toHaveBeenCalledOnce(); expect(custom.mock.calls[0][1]).toEqual({ overlay: true, overlayOptions: { anchor: "bottom-left", width: "100%", maxHeight: "100%", }, }); }); it("surfaces workflow errors instead of misreporting them as user aborts", async () => { t = await createTestSession({ extensions: [ultraPath()] }); saveWorkflow(t, "runtime-error", "missing-live-capability"); const [extension] = t.session.extensionRunner.extensions; const ctx = t.session.extensionRunner.createCommandContext(); driveOverlay(ctx); await extension.commands.get("ultra").handler("run runtime-error", ctx); expect(t.events.uiCallsFor("notify").at(-1)?.args).toEqual([ 'ultra: dynamic extension "missing-live-capability" has no canonical revision. Run a builder with all four dynamic-extension tools.', "error", ]); }); it("prepares dynamic extension runs on both command and tool surfaces", async () => { const message = 'ultra: dynamic extension "missing-live-capability" has no canonical revision. Run a builder with all four dynamic-extension tools.'; t = await createTestSession({ extensions: [ultraPath()] }); saveWorkflow(t, "runtime-error-command", "missing-live-capability"); saveWorkflow(t, "runtime-error-tool", "missing-live-capability"); const [extension] = t.session.extensionRunner.extensions; const ctx = t.session.extensionRunner.createCommandContext(); driveOverlay(ctx); await extension.commands .get("ultra") .handler("run runtime-error-command", ctx); expect(t.events.uiCallsFor("notify").at(-1)?.args).toEqual([ message, "error", ]); await extension.commands.get("ultra").handler("", ctx); await expect( extension.tools .get("run_workflow") .definition.execute( "call", { name: "runtime-error-tool" }, undefined, undefined, ctx, ), ).rejects.toThrow(message); }); it("reports confirmed whole-run cancellation as aborted while retaining its aggregate", async () => { t = await createTestSession({ extensions: [ultraPath()] }); saveWorkflow(t, "cancelled-run"); const [extension] = t.session.extensionRunner.extensions; const ctx = t.session.extensionRunner.createCommandContext(); driveOverlay(ctx, (overlay) => { overlay.handleInput("\u001b"); overlay.handleInput("\u001b"); }); await extension.commands.get("ultra").handler("run cancelled-run", ctx); expect(t.events.uiCallsFor("notify").at(-1)?.args).toEqual([ 'ultra: "cancelled-run" aborted · 1 phases · $0.0000 · 0 output.', "warning", ]); expect( t.events .uiCallsFor("notify") .some((call) => String(call.args[0]).includes("complete")), ).toBe(false); const message = t.session.messages.find( (message) => "customType" in message && message.customType === "ultra-result", ); const body = JSON.parse(String(message?.content)); try { expect(body.executionStatus).toBe("aborted"); expect( JSON.parse(readFileSync(body.fullOutputPath, "utf8")).aborted, ).toBe(true); } finally { rmSync(dirname(body.fullOutputPath), { recursive: true, force: true }); } }); it("refuses an unknown static sub-agent extension", async () => { t = await createTestSession({ extensions: [ultraPath()] }); mkdirSync(join(t.cwd, ".pi"), { recursive: true }); writeFileSync( join(t.cwd, ".pi", "settings.json"), JSON.stringify({ ultra: { subagentExtensions: ["missing-extension"] }, }), ); const [extension] = t.session.extensionRunner.extensions; await extension.commands .get("ultra") .handler("run review", t.session.extensionRunner.createCommandContext()); expect(t.events.uiCallsFor("notify").at(-1)?.args).toEqual([ "ultra: unknown sub-agent extension: missing-extension", "error", ]); }); it("starts research without knowing which extension supplies research tools", async () => { t = await createTestSession({ extensions: [ultraPath()] }); const [extension] = t.session.extensionRunner.extensions; const ctx = t.session.extensionRunner.createCommandContext(); const custom = vi.fn().mockResolvedValue(null); Object.assign(ctx.ui, { custom }); await extension.commands.get("ultra").handler("run research question", ctx); expect(custom).toHaveBeenCalledOnce(); expect( t.events .uiCallsFor("notify") .some((call) => String(call.args[0]).includes("unavailable tools")), ).toBe(false); }); it("discovers a workflow saved after the extension loaded without /reload", async () => { t = await createTestSession({ extensions: [ultraPath()] }); const [extension] = t.session.extensionRunner.extensions; const complete = () => extension.commands.get("ultra").getArgumentCompletions("run "); await complete(); const previousCwd = process.cwd(); try { process.chdir(t.cwd); const directory = join(t.cwd, ".pi", "workflows"); mkdirSync(directory, { recursive: true }); writeFileSync( join(directory, "fresh.json"), `${JSON.stringify({ name: "fresh", phases: [ { id: "p", kind: "single", step: { summary: "hi", prompt: "hi" } }, ], })}\n`, ); const items = await complete(); expect(items.map((item: { value: string }) => item.value)).toContain( "run fresh", ); } finally { process.chdir(previousCwd); } }); it.each([ { params: {}, expected: ['provide either a workflow "name" or an inline "spec"'], }, { params: { name: "does-not-exist" }, expected: ['unknown workflow "does-not-exist"', "Available:", "review"], }, { params: { args: "not-an-object" }, expected: ["args", "object"] }, { params: { spec: { name: "invalid", phases: [{ id: "work", kind: "single", step: { prompt: "Work" } }], }, }, expected: ["summary"], }, ])( "delivers actionable argument errors to the agent and the tool UI: $params", async ({ params, expected }) => { t = await createTestSession({ extensions: [ultraPath()] }); const [extension] = t.session.extensionRunner.extensions; const ctx = t.session.extensionRunner.createCommandContext(); await extension.commands.get("ultra").handler("", ctx); // Use the real tool boundary: harness run() catches execute errors and // returns them as successes when propagateErrors is false. const faux = fauxProvider({ provider: "ultra-argument-errors" }); ctx.modelRegistry.registerProvider(faux.provider); await t.session.setModel(faux.getModel()); const followUp = vi.fn((context: Context) => { const result = context.messages.findLast( (message) => message.role === "toolResult", ); expect(result?.isError).toBe(true); const error = result?.content .filter((part) => part.type === "text") .map((part) => part.text) .join("\n"); for (const text of expected) expect(error).toContain(text); return fauxAssistantMessage("Error received."); }); faux.setResponses([ fauxAssistantMessage(fauxToolCall("run_workflow", params), { stopReason: "toolUse", }), followUp, ]); await t.session.prompt("Exercise invalid workflow arguments"); expect(followUp).toHaveBeenCalledOnce(); const [result] = t.events.toolResultsFor("run_workflow"); expect(result.isError).toBe(true); expect(result.mocked).toBe(false); for (const text of expected) expect(result.text).toContain(text); const message = t.session.messages.find( (message) => message.role === "toolResult", ); expect(message).toMatchObject({ role: "toolResult", isError: true, content: result.content, }); expect(t.session.messages.at(-1)).toMatchObject({ role: "assistant" }); initTheme(); const theme = { fg: (_color: string, text: string) => text, bold: (text: string) => text, } as Theme; const tool = extension.tools.get("run_workflow").definition; const collapsed = tool .renderResult(message, { expanded: false, isPartial: false }, theme, { isError: true, }) .render(120) .join("\n"); expect(collapsed).toContain("ultra · failed:"); expect(collapsed).not.toContain("0/0 done"); const expanded = tool .renderResult(message, { expanded: true, isPartial: false }, theme, { isError: true, }) .render(120) .join("\n"); for (const text of expected) expect(expanded).toContain(text); }, ); it("exposes the workflow schema while rejecting malformed specs before execution", async () => { t = await createTestSession({ extensions: [ultraPath()] }); const [extension] = t.session.extensionRunner.extensions; await extension.commands.get("ultra").getArgumentCompletions(""); const tool = extension.tools.get("run_workflow").definition; const specSchema = tool.parameters.properties.spec; expect(tool.renderCall).toBeTypeOf("function"); expect(specSchema.type).toBe("object"); expect(specSchema.properties).toHaveProperty("name"); expect(specSchema.properties).toHaveProperty("phases"); expect(specSchema.properties).toHaveProperty("report"); expect(specSchema.additionalProperties).not.toBe(true); expect(tool.description).toContain( "run saved or inline multi-agent workflows.", ); expect(tool.promptSnippet).toBe( "run saved or inline multi-agent workflows.", ); expect(tool.promptGuidelines).toEqual([ "Use run_workflow when decomposition into independent or sequential child-agent steps reduces context or verification risk.", "Use dynamic extensions when a workflow needs a reusable, task-specific capability; read the ultra-authoring skill for how to build and load them.", ]); expect(tool.parameters.properties.background).toBeUndefined(); expect(JSON.parse(JSON.stringify(specSchema))).toEqual( JSON.parse(JSON.stringify(WorkflowSpecSchema)), ); const step = { summary: "Work", prompt: "Work" }; for (const spec of [ {}, { name: "invalid", phases: [] }, { name: "invalid", phases: [{ id: "work", kind: "fanout", step }] }, { name: "invalid", phases: [{ id: "work", kind: "single", step: { prompt: "Work" } }], }, { name: "invalid", phases: [ { id: "work", kind: "single", step: { ...step, thinkingLevel: "invalid" }, }, ], }, { name: "invalid", phases: [ { id: "work", kind: "single", step: { ...step, schema: "missing" } }, ], }, { name: "invalid", phases: [{ id: "args", kind: "single", step }] }, ]) { await expect( tool.execute( "invalid", { spec }, undefined, undefined, t.session.extensionRunner.createContext(), ), ).rejects.toThrow("Invalid workflow spec"); } }); });