repositories / pi-ext
pi-ext
bugabingas pi extensions
owned by admin
extensions/ultra/__tests__/harness.test.ts
Rawimport {
mkdirSync,
mkdtempSync,
readdirSync,
readFileSync,
rmSync,
writeFileSync,
} from "node:fs";
import { tmpdir } from "node:os";
import { dirname, join } from "node:path";
import { fileURLToPath } from "node:url";
import {
type Context,
fauxAssistantMessage,
fauxProvider,
fauxToolCall,
} from "@earendil-works/pi-ai";
import {
type ExtensionCommandContext,
initTheme,
SessionManager,
type Theme,
} from "@earendil-works/pi-coding-agent";
import {
createTestSession as createHarnessSession,
type TestSession,
} from "@marcfargas/pi-test-harness";
import { afterEach, describe, expect, it, vi } from "vitest";
import { createTestSession as createSandboxSession } from "../../../test/harness";
import { DEFAULT_MODEL_TIERS, modelTierPrompt } from "../settings.ts";
import { WorkflowSpecSchema } from "../spec.ts";
let t: TestSession | undefined;
let root: string | undefined;
afterEach(() => {
t?.dispose();
if (root) rmSync(root, { recursive: true, force: true });
t = undefined;
root = undefined;
vi.unstubAllEnvs();
vi.restoreAllMocks();
});
function ultraPath(): string {
return fileURLToPath(new URL("../index.ts", import.meta.url));
}
async function createTestSession(options: {
extensions: string[];
cwd?: string;
}): Promise<TestSession> {
root = options.cwd ?? mkdtempSync(join(tmpdir(), "ultra-harness-"));
const agentDir = join(root, "agent");
mkdirSync(agentDir, { recursive: true });
vi.stubEnv("PI_CODING_AGENT_DIR", agentDir);
if (!options.cwd) {
mkdirSync(join(root, ".pi"), { recursive: true });
writeFileSync(
join(root, ".pi", "settings.json"),
JSON.stringify({ ultra: { autoEnable: false } }),
);
}
return createHarnessSession({ ...options, cwd: root });
}
function saveWorkflow(
session: TestSession,
name: string,
dynamicExtension?: string,
): void {
const directory = join(session.cwd, ".pi", "workflows");
mkdirSync(directory, { recursive: true });
writeFileSync(
join(directory, `${name}.json`),
`${JSON.stringify({
name,
phases: [
{
id: "work",
kind: "single",
step: {
summary: "Work",
prompt: "Work",
...(dynamicExtension
? { dynamicExtensions: [dynamicExtension] }
: {}),
},
},
],
})}\n`,
);
}
function driveOverlay(
ctx: ExtensionCommandContext,
interact?: (overlay: { handleInput(data: string): void }) => void,
): void {
Object.assign(ctx.ui, {
custom: vi.fn(
async (
factory: (
host: unknown,
theme: unknown,
keybindings: unknown,
done: (value: unknown) => void,
) => { handleInput(data: string): void },
) =>
new Promise((resolve) => {
const overlay = factory(
{ requestRender: vi.fn() },
{},
{ matches: () => false },
resolve,
);
interact?.(overlay);
}),
),
});
}
describe("ultra extension", () => {
it("loads only the command and renderer in the real Pi runtime", async () => {
const indexPath = ultraPath();
t = await createTestSession({ extensions: [indexPath] });
const [extension] = t.session.extensionRunner.extensions;
expect(extension.path).toBe(indexPath);
expect(extension.tools.has("run_workflow")).toBe(false);
expect(readFileSync(indexPath, "utf8")).toContain(
'import("./implementation.ts")',
);
expect(extension.commands.has("ultra")).toBe(true);
expect(extension.commands.has("ultra-runs")).toBe(false);
});
it("writes disabled command outcome in its debug sandbox", async () => {
t = await createSandboxSession({
env: { PI_ULTRA_DEBUG: "1" },
extensions: [ultraPath()],
});
const extension = t.session.extensionRunner.extensions[0];
const context = t.session.extensionRunner.createCommandContext();
for (const handler of extension.handlers.get("session_start"))
await handler({ type: "session_start" }, context);
mkdirSync(join(t.cwd, ".pi"), { recursive: true });
writeFileSync(
join(t.cwd, ".pi", "settings.json"),
JSON.stringify({ ultra: { enabled: false } }),
);
await extension.commands.get("ultra").handler("", context);
for (const handler of extension.handlers.get("session_shutdown"))
await handler({ type: "session_shutdown" }, context);
const directory = join(t.cwd, ".test-state", "pi-ext", "debug", "ultra");
const records = readFileSync(
join(directory, readdirSync(directory)[0]),
"utf8",
)
.trim()
.split("\n")
.map((line) => JSON.parse(line));
expect(
records.filter((record) => record.event === "command.handle.start"),
).toHaveLength(1);
expect(
records.filter((record) => record.event === "command.handle.finish"),
).toEqual([expect.objectContaining({ outcome: "disabled" })]);
expect(records.map((record) => record.event)).toEqual(
expect.arrayContaining(["session.start", "session.shutdown"]),
);
});
it("auto-enables the workflow tool when configured", async () => {
root = mkdtempSync(join(tmpdir(), "ultra-auto-enable-"));
mkdirSync(join(root, ".pi"), { recursive: true });
writeFileSync(
join(root, ".pi", "settings.json"),
JSON.stringify({ ultra: { autoEnable: true } }),
);
t = await createTestSession({ cwd: root, extensions: [ultraPath()] });
const [extension] = t.session.extensionRunner.extensions;
expect(extension.tools.has("run_workflow")).toBe(true);
expect(t.session.getActiveToolNames()).toContain("run_workflow");
expect(t.session.systemPrompt).toContain("- run_workflow:");
});
it.each([false, true])(
"registers the authoring skill independently of tool activation (auto-enable: %s)",
async (autoEnable) => {
root = mkdtempSync(join(tmpdir(), "ultra-shared-guidance-"));
mkdirSync(join(root, ".pi"), { recursive: true });
writeFileSync(
join(root, ".pi", "settings.json"),
JSON.stringify({ ultra: { autoEnable } }),
);
t = await createTestSession({ cwd: root, extensions: [ultraPath()] });
const skillPath = fileURLToPath(
new URL("../prompts/ultra-authoring/SKILL.md", import.meta.url),
);
const skill = t.session.resourceLoader
.getSkills()
.skills.find((candidate) => candidate.name === "ultra-authoring");
expect(skill?.filePath).toBe(skillPath);
expect(t.session.systemPrompt).toContain(skillPath);
expect(t.session.systemPrompt).not.toContain(
readFileSync(skillPath, "utf8").trim(),
);
expect(t.session.systemPrompt).not.toContain("## Worked example");
const [extension] = t.session.extensionRunner.extensions;
if (!autoEnable) {
await extension.commands
.get("ultra")
.handler("", t.session.extensionRunner.createCommandContext());
}
const rules: string[] =
extension.tools.get("run_workflow").definition.promptGuidelines;
expect(rules).toEqual([
"Use run_workflow when decomposition into independent or sequential child-agent steps reduces context or verification risk.",
"Use dynamic extensions when a workflow needs a reusable, task-specific capability; read the ultra-authoring skill for how to build and load them.",
]);
for (const rule of rules)
expect(t.session.systemPrompt.split(rule)).toHaveLength(2);
t.session.setActiveToolsByName(
t.session
.getActiveToolNames()
.filter((name) => name !== "run_workflow"),
);
expect(t.session.systemPrompt).toContain(skillPath);
for (const rule of rules)
expect(t.session.systemPrompt).not.toContain(rule);
await extension.commands
.get("ultra")
.handler("", t.session.extensionRunner.createCommandContext());
for (const rule of rules)
expect(t.session.systemPrompt.split(rule)).toHaveLength(2);
expect(t.session.messages).toHaveLength(0);
},
);
it("does not advertise the skill when Ultra is disabled, even after completion or activation attempts", async () => {
root = mkdtempSync(join(tmpdir(), "ultra-disabled-skill-"));
mkdirSync(join(root, ".pi"), { recursive: true });
writeFileSync(
join(root, ".pi", "settings.json"),
JSON.stringify({ ultra: { enabled: false, autoEnable: true } }),
);
t = await createTestSession({ cwd: root, extensions: [ultraPath()] });
const command =
t.session.extensionRunner.extensions[0].commands.get("ultra");
await command.getArgumentCompletions("");
await command.handler("", t.session.extensionRunner.createCommandContext());
expect(t.session.getActiveToolNames()).not.toContain("run_workflow");
expect(t.session.systemPrompt).not.toContain("ultra-authoring/SKILL.md");
});
it("injects the effective tier policy only while active and refreshes settings each turn", async () => {
t = await createTestSession({ extensions: [ultraPath()] });
const runner = t.session.extensionRunner;
const emit = () =>
runner.emitBeforeAgentStart("prompt", undefined, {
cwd: t?.cwd ?? process.cwd(),
});
const policy = async () => (await emit()).systemPromptOptions.sections;
expect((await policy()).ultra_model_tier_policy).toBeUndefined();
expect((await policy()).ultra_fanout_tool_policy).toBeUndefined();
const [extension] = runner.extensions;
await extension.commands
.get("ultra")
.handler("", runner.createCommandContext());
expect((await policy()).ultra_model_tier_policy).toBe(
modelTierPrompt(DEFAULT_MODEL_TIERS),
);
expect((await policy()).ultra_fanout_tool_policy).toContain(
"read, grep, find, ls, bash",
);
writeFileSync(
join(t.cwd, ".pi", "settings.json"),
JSON.stringify({
ultra: {
fanoutToolAllowlist: ["web"],
modelTiers: {
large: { instructions: "Architecture only." },
fast_extract: {
model: "custom/extractor",
thinkingLevel: "low",
instructions: "Select for literal extraction.",
},
},
},
}),
);
const updated = await policy();
expect(updated.ultra_model_tier_policy).toContain("Architecture only.");
expect(updated.ultra_model_tier_policy).not.toContain(
DEFAULT_MODEL_TIERS.large.instructions,
);
expect(updated.ultra_model_tier_policy).toContain(
"- fast_extract: custom/extractor; low; Select for literal extraction.",
);
expect(
updated.ultra_model_tier_policy?.match(/# ultra model tier policy/g),
).toHaveLength(1);
expect(updated.ultra_fanout_tool_policy).toContain("bash, web");
writeFileSync(
join(t.cwd, ".pi", "settings.json"),
JSON.stringify({ ultra: { enabled: false } }),
);
expect((await policy()).ultra_model_tier_policy).toBeUndefined();
expect((await policy()).ultra_fanout_tool_policy).toBeUndefined();
});
it("advertises which step.flags enable sub-agent extension tools while active", async () => {
root = mkdtempSync(join(tmpdir(), "ultra-subagent-flags-"));
mkdirSync(join(root, ".pi"), { recursive: true });
writeFileSync(
join(root, ".pi", "settings.json"),
JSON.stringify({
ultra: { subagentExtensions: ["nushell", "chrome-cdp", "ask"] },
}),
);
t = await createTestSession({ cwd: root, extensions: [ultraPath()] });
const runner = t.session.extensionRunner;
const sections = async () =>
(
await runner.emitBeforeAgentStart("prompt", undefined, {
cwd: root ?? process.cwd(),
})
).systemPromptOptions.sections;
expect((await sections()).ultra_subagent_flags).toBeUndefined();
await runner.extensions[0].commands
.get("ultra")
.handler("", runner.createCommandContext());
const prompt = (await sections()).ultra_subagent_flags ?? "";
expect(prompt).toContain("# ultra sub-agent flags");
expect(prompt).toContain(
"- nushell: nu, nu_session; --nushell (boolean): Enable Nushell tools at startup instead of bash and powershell",
);
expect(prompt).toContain("- chrome-cdp: chrome_cdp; --chrome (boolean)");
// ask registers tools but no flags, so it is not listed.
expect(prompt).not.toContain("- ask:");
});
it("rejects unknown tiers before either surface starts work", async () => {
t = await createTestSession({ extensions: [ultraPath()] });
const [extension] = t.session.extensionRunner.extensions;
const ctx = t.session.extensionRunner.createCommandContext();
await extension.commands.get("ultra").handler("", ctx);
const spec = {
name: "unknown-tier",
phases: [
{
id: "first",
kind: "single",
step: { summary: "First", prompt: "First", model: "medium" },
},
{
id: "second",
kind: "single",
step: { summary: "Second", prompt: "Second", model: "missing_tier" },
},
],
};
const tool = extension.tools.get("run_workflow").definition;
await expect(
tool.execute("call", { spec }, undefined, undefined, ctx),
).rejects.toThrow('unknown model tier "missing_tier"');
const directory = join(t.cwd, ".pi", "workflows");
mkdirSync(directory, { recursive: true });
writeFileSync(join(directory, "unknown-tier.json"), JSON.stringify(spec));
const custom = vi.fn();
Object.assign(ctx.ui, { custom });
await extension.commands.get("ultra").handler("run unknown-tier", ctx);
expect(custom).not.toHaveBeenCalled();
expect(t.events.uiCallsFor("notify").at(-1)?.args).toEqual([
'ultra: unknown model tier "missing_tier". Configure ultra.modelTiers or use provider/model.',
"error",
]);
});
it.each(["command", "tool"])(
"inherits the parent's session directory through the %s surface",
async (surface) => {
t = await createTestSession({ extensions: [ultraPath()] });
saveWorkflow(t, "session-directory");
const [extension] = t.session.extensionRunner.extensions;
const sessionDir = join(t.cwd, "custom-sessions");
const parent = SessionManager.create(t.cwd, sessionDir);
const ctx = {
...t.session.extensionRunner.createCommandContext(),
sessionManager: parent,
};
const create = vi
.spyOn(SessionManager, "create")
.mockImplementation(() => {
throw new Error("Stop before provider access");
});
if (surface === "command") {
driveOverlay(ctx);
await extension.commands
.get("ultra")
.handler("run session-directory", ctx);
} else {
await extension.commands.get("ultra").handler("", ctx);
await extension.tools
.get("run_workflow")
.definition.execute(
"call",
{ name: "session-directory" },
undefined,
undefined,
ctx,
);
}
expect(create).toHaveBeenCalledWith(t.cwd, parent.getSessionDir(), {
parentSession: parent.getSessionFile(),
});
},
);
it.each([false, true])(
"bounds orchestrator content while retaining UI details (oversized: %s)",
async (oversized) => {
t = await createTestSession({ extensions: [ultraPath()] });
const [extension] = t.session.extensionRunner.extensions;
const ctx = t.session.extensionRunner.createCommandContext();
await extension.commands.get("ultra").handler("", ctx);
const answer = oversized ? "界".repeat(30_000) : "Final answer";
const response = await extension.tools
.get("run_workflow")
.definition.execute(
"call",
{
spec: {
name: "context-budget",
phases: [
{
id: "work",
kind: "single",
when: "{args.run}",
step: { summary: "Unused", prompt: "Unused" },
},
],
return: "{args.answer}",
},
args: { run: false, answer },
},
undefined,
undefined,
ctx,
);
const body = JSON.parse(response.content[0].text);
try {
expect(response.details.workflowResult.result).toBe(answer);
expect(response.details.workflowResult.phaseResults).toEqual({
work: [],
});
expect(body.phaseResults).toBeUndefined();
if (oversized) {
expect(body.truncated).toBe(true);
expect(body.fullOutputPath).toBe(response.details.fullOutputPath);
expect({
...JSON.parse(readFileSync(body.fullOutputPath, "utf8")),
fullOutputPath: body.fullOutputPath,
}).toEqual(response.details.workflowResult);
} else {
expect(body.fullOutputPath).toBe(response.details.fullOutputPath);
expect(
JSON.parse(readFileSync(body.fullOutputPath, "utf8")).result,
).toBe(answer);
expect(body.result).toBe(answer);
expect(body.phases).toEqual([
{ id: "work", ran: 0, ok: 0, dropped: 0 },
]);
expect(body.truncated).toBeUndefined();
}
} finally {
if (response.details.fullOutputPath)
rmSync(dirname(response.details.fullOutputPath), {
recursive: true,
force: true,
});
}
},
);
it.each(["command-output", "review", "research"])(
"keeps %s command context bounded and its evidence retrievable",
async (name) => {
t = await createTestSession({ extensions: [ultraPath()] });
saveWorkflow(t, name);
const [extension] = t.session.extensionRunner.extensions;
const ctx = t.session.extensionRunner.createCommandContext();
const result = {
workflow: name,
phases: [{ id: "work", ran: 1, ok: 1, dropped: 0 }],
phaseResults: {
work: [{ ok: true, value: "INTERMEDIATE".repeat(10000) }],
},
phaseFailures: {},
result: [{ status: "blocked", blockers: ["Missing evidence"] }],
report: { summary: "Human report" },
steered: false,
dynamicExtensions: [],
tokenUsage: {
input: 1,
output: 2,
total: 3,
cacheRead: 0,
cacheWrite: 0,
cost: 0,
},
};
Object.assign(ctx.ui, {
custom: vi.fn().mockResolvedValue({ status: "completed", result }),
});
await extension.commands.get("ultra").handler(`run ${name}`, ctx);
expect(
t.events
.uiCallsFor("notify")
.some(
(call) =>
call.args[0] ===
`ultra: "${name}" complete · 1 phases · $0.0000 · 2 output.`,
),
).toBe(true);
const message = t.session.messages.find(
(message) =>
"customType" in message && message.customType === "ultra-result",
);
if (!message || typeof message.content !== "string")
throw new Error("Missing command result");
const body = JSON.parse(message.content);
try {
expect(body.result).toEqual(result.result);
expect(body.report).toEqual(result.report);
expect(body.executionStatus).toBe("finished");
expect(body.phaseResults).toBeUndefined();
expect(message.content).not.toContain("INTERMEDIATE");
expect(message.details).toEqual({
...result,
fullOutputPath: body.fullOutputPath,
});
expect(JSON.parse(readFileSync(body.fullOutputPath, "utf8"))).toEqual(
result,
);
} finally {
rmSync(dirname(body.fullOutputPath), { recursive: true, force: true });
}
},
);
it("completes subcommands first and saved workflows below run", async () => {
t = await createTestSession({ extensions: [ultraPath()] });
const [extension] = t.session.extensionRunner.extensions;
const complete = extension.commands.get("ultra").getArgumentCompletions;
expect(extension.tools.has("run_workflow")).toBe(false);
const subcommands = await complete("");
const workflows = await complete("run ");
expect(extension.tools.has("run_workflow")).toBe(true);
expect(t.session.getActiveToolNames()).not.toContain("run_workflow");
expect(subcommands.map((item: { value: string }) => item.value)).toEqual([
"exec",
"run",
]);
expect(workflows.map((item: { value: string }) => item.value)).toEqual(
expect.arrayContaining(["run research", "run review"]),
);
});
it("activates the workflow tool and notifies without starting a turn", async () => {
t = await createTestSession({ extensions: [ultraPath()] });
const [extension] = t.session.extensionRunner.extensions;
const beforeMessages = t.session.messages.length;
const beforeNotifications = t.events.uiCallsFor("notify").length;
expect(t.session.getActiveToolNames()).not.toContain("run_workflow");
expect(t.session.systemPrompt).not.toContain("- run_workflow:");
await extension.commands
.get("ultra")
.handler("", t.session.extensionRunner.createCommandContext());
expect(t.session.getActiveToolNames()).toContain("run_workflow");
expect(t.session.systemPrompt).toContain(
"- run_workflow: run saved or inline multi-agent workflows.",
);
expect(t.session.systemPrompt).toContain(
"Use run_workflow when decomposition into independent or sequential child-agent steps reduces context or verification risk.",
);
expect(t.session.messages).toHaveLength(beforeMessages);
const notifications = t.events.uiCallsFor("notify");
expect(notifications).toHaveLength(beforeNotifications + 1);
expect(notifications.at(-1)?.args).toEqual([
"ultra: run_workflow tool enabled.",
"info",
]);
});
it("sends exec instructions as the visible user message that starts the turn", async () => {
t = await createTestSession({ extensions: [ultraPath()] });
const [extension] = t.session.extensionRunner.extensions;
const sendUserMessage = vi
.spyOn(t.session, "sendUserMessage")
.mockResolvedValue();
const beforeNotifications = t.events.uiCallsFor("notify").length;
await extension.commands
.get("ultra")
.handler(
"exec inspect this failure once",
t.session.extensionRunner.createCommandContext(),
);
expect(t.session.getActiveToolNames()).toContain("run_workflow");
expect(sendUserMessage).toHaveBeenCalledOnce();
const prompt = sendUserMessage.mock.calls[0]?.[0];
expect(prompt).toBe(
"# ultra exec\n\n# user request\n\ninspect this failure once",
);
const guidance =
extension.tools.get("run_workflow").definition.promptGuidelines;
for (const rule of guidance) {
expect(t.session.systemPrompt.split(rule)).toHaveLength(2);
expect(prompt).not.toContain(rule);
}
expect(t.session.systemPrompt).not.toContain("# Authoring ultra workflows");
expect(t.session.systemPrompt).not.toContain("## Worked example");
expect(prompt).not.toContain("The workflow tool is now available");
const injected = await t.session.extensionRunner.emitBeforeAgentStart(
prompt as string,
undefined,
{ cwd: t.cwd },
);
expect(injected.systemPromptOptions.sections.ultra_model_tier_policy).toBe(
modelTierPrompt(DEFAULT_MODEL_TIERS),
);
expect(injected.systemPromptOptions.forceSystemPrompt).toBeUndefined();
expect(prompt).not.toContain(DEFAULT_MODEL_TIERS.large.instructions);
expect(t.events.uiCallsFor("notify")).toHaveLength(beforeNotifications);
});
it("renders workflow progress as an overlay without replacing session content", async () => {
t = await createTestSession({ extensions: [ultraPath()] });
const [extension] = t.session.extensionRunner.extensions;
const ctx = t.session.extensionRunner.createCommandContext();
const custom = vi.fn().mockResolvedValue(null);
Object.assign(ctx.ui, { custom });
await extension.commands
.get("ultra")
.handler("run review inspect scrolling", ctx);
expect(custom).toHaveBeenCalledOnce();
expect(custom.mock.calls[0][1]).toEqual({
overlay: true,
overlayOptions: {
anchor: "bottom-left",
width: "100%",
maxHeight: "100%",
},
});
});
it("surfaces workflow errors instead of misreporting them as user aborts", async () => {
t = await createTestSession({ extensions: [ultraPath()] });
saveWorkflow(t, "runtime-error", "missing-live-capability");
const [extension] = t.session.extensionRunner.extensions;
const ctx = t.session.extensionRunner.createCommandContext();
driveOverlay(ctx);
await extension.commands.get("ultra").handler("run runtime-error", ctx);
expect(t.events.uiCallsFor("notify").at(-1)?.args).toEqual([
'ultra: dynamic extension "missing-live-capability" has no canonical revision. Run a builder with all four dynamic-extension tools.',
"error",
]);
});
it("prepares dynamic extension runs on both command and tool surfaces", async () => {
const message =
'ultra: dynamic extension "missing-live-capability" has no canonical revision. Run a builder with all four dynamic-extension tools.';
t = await createTestSession({ extensions: [ultraPath()] });
saveWorkflow(t, "runtime-error-command", "missing-live-capability");
saveWorkflow(t, "runtime-error-tool", "missing-live-capability");
const [extension] = t.session.extensionRunner.extensions;
const ctx = t.session.extensionRunner.createCommandContext();
driveOverlay(ctx);
await extension.commands
.get("ultra")
.handler("run runtime-error-command", ctx);
expect(t.events.uiCallsFor("notify").at(-1)?.args).toEqual([
message,
"error",
]);
await extension.commands.get("ultra").handler("", ctx);
await expect(
extension.tools
.get("run_workflow")
.definition.execute(
"call",
{ name: "runtime-error-tool" },
undefined,
undefined,
ctx,
),
).rejects.toThrow(message);
});
it("reports confirmed whole-run cancellation as aborted while retaining its aggregate", async () => {
t = await createTestSession({ extensions: [ultraPath()] });
saveWorkflow(t, "cancelled-run");
const [extension] = t.session.extensionRunner.extensions;
const ctx = t.session.extensionRunner.createCommandContext();
driveOverlay(ctx, (overlay) => {
overlay.handleInput("\u001b");
overlay.handleInput("\u001b");
});
await extension.commands.get("ultra").handler("run cancelled-run", ctx);
expect(t.events.uiCallsFor("notify").at(-1)?.args).toEqual([
'ultra: "cancelled-run" aborted · 1 phases · $0.0000 · 0 output.',
"warning",
]);
expect(
t.events
.uiCallsFor("notify")
.some((call) => String(call.args[0]).includes("complete")),
).toBe(false);
const message = t.session.messages.find(
(message) =>
"customType" in message && message.customType === "ultra-result",
);
const body = JSON.parse(String(message?.content));
try {
expect(body.executionStatus).toBe("aborted");
expect(
JSON.parse(readFileSync(body.fullOutputPath, "utf8")).aborted,
).toBe(true);
} finally {
rmSync(dirname(body.fullOutputPath), { recursive: true, force: true });
}
});
it("refuses an unknown static sub-agent extension", async () => {
t = await createTestSession({ extensions: [ultraPath()] });
mkdirSync(join(t.cwd, ".pi"), { recursive: true });
writeFileSync(
join(t.cwd, ".pi", "settings.json"),
JSON.stringify({
ultra: { subagentExtensions: ["missing-extension"] },
}),
);
const [extension] = t.session.extensionRunner.extensions;
await extension.commands
.get("ultra")
.handler("run review", t.session.extensionRunner.createCommandContext());
expect(t.events.uiCallsFor("notify").at(-1)?.args).toEqual([
"ultra: unknown sub-agent extension: missing-extension",
"error",
]);
});
it("starts research without knowing which extension supplies research tools", async () => {
t = await createTestSession({ extensions: [ultraPath()] });
const [extension] = t.session.extensionRunner.extensions;
const ctx = t.session.extensionRunner.createCommandContext();
const custom = vi.fn().mockResolvedValue(null);
Object.assign(ctx.ui, { custom });
await extension.commands.get("ultra").handler("run research question", ctx);
expect(custom).toHaveBeenCalledOnce();
expect(
t.events
.uiCallsFor("notify")
.some((call) => String(call.args[0]).includes("unavailable tools")),
).toBe(false);
});
it("discovers a workflow saved after the extension loaded without /reload", async () => {
t = await createTestSession({ extensions: [ultraPath()] });
const [extension] = t.session.extensionRunner.extensions;
const complete = () =>
extension.commands.get("ultra").getArgumentCompletions("run ");
await complete();
const previousCwd = process.cwd();
try {
process.chdir(t.cwd);
const directory = join(t.cwd, ".pi", "workflows");
mkdirSync(directory, { recursive: true });
writeFileSync(
join(directory, "fresh.json"),
`${JSON.stringify({
name: "fresh",
phases: [
{ id: "p", kind: "single", step: { summary: "hi", prompt: "hi" } },
],
})}\n`,
);
const items = await complete();
expect(items.map((item: { value: string }) => item.value)).toContain(
"run fresh",
);
} finally {
process.chdir(previousCwd);
}
});
it.each([
{
params: {},
expected: ['provide either a workflow "name" or an inline "spec"'],
},
{
params: { name: "does-not-exist" },
expected: ['unknown workflow "does-not-exist"', "Available:", "review"],
},
{ params: { args: "not-an-object" }, expected: ["args", "object"] },
{
params: {
spec: {
name: "invalid",
phases: [{ id: "work", kind: "single", step: { prompt: "Work" } }],
},
},
expected: ["summary"],
},
])(
"delivers actionable argument errors to the agent and the tool UI: $params",
async ({ params, expected }) => {
t = await createTestSession({ extensions: [ultraPath()] });
const [extension] = t.session.extensionRunner.extensions;
const ctx = t.session.extensionRunner.createCommandContext();
await extension.commands.get("ultra").handler("", ctx);
// Use the real tool boundary: harness run() catches execute errors and
// returns them as successes when propagateErrors is false.
const faux = fauxProvider({ provider: "ultra-argument-errors" });
ctx.modelRegistry.registerProvider(faux.provider);
await t.session.setModel(faux.getModel());
const followUp = vi.fn((context: Context) => {
const result = context.messages.findLast(
(message) => message.role === "toolResult",
);
expect(result?.isError).toBe(true);
const error = result?.content
.filter((part) => part.type === "text")
.map((part) => part.text)
.join("\n");
for (const text of expected) expect(error).toContain(text);
return fauxAssistantMessage("Error received.");
});
faux.setResponses([
fauxAssistantMessage(fauxToolCall("run_workflow", params), {
stopReason: "toolUse",
}),
followUp,
]);
await t.session.prompt("Exercise invalid workflow arguments");
expect(followUp).toHaveBeenCalledOnce();
const [result] = t.events.toolResultsFor("run_workflow");
expect(result.isError).toBe(true);
expect(result.mocked).toBe(false);
for (const text of expected) expect(result.text).toContain(text);
const message = t.session.messages.find(
(message) => message.role === "toolResult",
);
expect(message).toMatchObject({
role: "toolResult",
isError: true,
content: result.content,
});
expect(t.session.messages.at(-1)).toMatchObject({ role: "assistant" });
initTheme();
const theme = {
fg: (_color: string, text: string) => text,
bold: (text: string) => text,
} as Theme;
const tool = extension.tools.get("run_workflow").definition;
const collapsed = tool
.renderResult(message, { expanded: false, isPartial: false }, theme, {
isError: true,
})
.render(120)
.join("\n");
expect(collapsed).toContain("ultra · failed:");
expect(collapsed).not.toContain("0/0 done");
const expanded = tool
.renderResult(message, { expanded: true, isPartial: false }, theme, {
isError: true,
})
.render(120)
.join("\n");
for (const text of expected) expect(expanded).toContain(text);
},
);
it("exposes the workflow schema while rejecting malformed specs before execution", async () => {
t = await createTestSession({ extensions: [ultraPath()] });
const [extension] = t.session.extensionRunner.extensions;
await extension.commands.get("ultra").getArgumentCompletions("");
const tool = extension.tools.get("run_workflow").definition;
const specSchema = tool.parameters.properties.spec;
expect(tool.renderCall).toBeTypeOf("function");
expect(specSchema.type).toBe("object");
expect(specSchema.properties).toHaveProperty("name");
expect(specSchema.properties).toHaveProperty("phases");
expect(specSchema.properties).toHaveProperty("report");
expect(specSchema.additionalProperties).not.toBe(true);
expect(tool.description).toContain(
"run saved or inline multi-agent workflows.",
);
expect(tool.promptSnippet).toBe(
"run saved or inline multi-agent workflows.",
);
expect(tool.promptGuidelines).toEqual([
"Use run_workflow when decomposition into independent or sequential child-agent steps reduces context or verification risk.",
"Use dynamic extensions when a workflow needs a reusable, task-specific capability; read the ultra-authoring skill for how to build and load them.",
]);
expect(tool.parameters.properties.background).toBeUndefined();
expect(JSON.parse(JSON.stringify(specSchema))).toEqual(
JSON.parse(JSON.stringify(WorkflowSpecSchema)),
);
const step = { summary: "Work", prompt: "Work" };
for (const spec of [
{},
{ name: "invalid", phases: [] },
{ name: "invalid", phases: [{ id: "work", kind: "fanout", step }] },
{
name: "invalid",
phases: [{ id: "work", kind: "single", step: { prompt: "Work" } }],
},
{
name: "invalid",
phases: [
{
id: "work",
kind: "single",
step: { ...step, thinkingLevel: "invalid" },
},
],
},
{
name: "invalid",
phases: [
{ id: "work", kind: "single", step: { ...step, schema: "missing" } },
],
},
{ name: "invalid", phases: [{ id: "args", kind: "single", step }] },
]) {
await expect(
tool.execute(
"invalid",
{ spec },
undefined,
undefined,
t.session.extensionRunner.createContext(),
),
).rejects.toThrow("Invalid workflow spec");
}
});
});