import { readFile, rm, stat } from "node:fs/promises"; import { dirname } from "node:path"; import { stripVTControlCharacters } from "node:util"; import { DEFAULT_MAX_BYTES, DEFAULT_MAX_LINES, getMarkdownTheme, initTheme, } from "@earendil-works/pi-coding-agent"; import { Markdown } from "@earendil-works/pi-tui"; import { afterEach, beforeAll, describe, expect, it, vi } from "vitest"; import type { WorkflowResult } from "../engine.ts"; import type { AgentProgressEvent, UltraProgressEvent } from "../progress.ts"; import type { DiscoveredWorkflow, WorkflowSpec } from "../spec.ts"; import { completeUltraCommand, completeWorkflowNames, formatWorkflowToolResult as formatOutput, formatUltraResultMarkdown, makeToolProgressSink, mapPositionalArgs, parseCommandLine, parseUltraCommand, resolveWorkflowInput, } from "../wire.ts"; const artifacts: string[] = []; afterEach(async () => { await Promise.all( artifacts .splice(0) .map((path) => rm(dirname(path), { recursive: true, force: true })), ); }); async function formatWorkflowToolResult(result: WorkflowResult) { const output = await formatOutput(result); artifacts.push(output.fullOutputPath); return output; } const validSpec = (name: string): WorkflowSpec => ({ name, phases: [{ id: "p", kind: "single", step: { summary: "hi", prompt: "hi" } }], }); const specWithParams = (name: string, params: string[]): WorkflowSpec => ({ name, args: { params }, phases: [ { id: "p", kind: "single", step: { summary: "hi {args.base}", prompt: "hi {args.base}" }, }, ], }); const discovered = ( name: string, source: DiscoveredWorkflow["source"], ): DiscoveredWorkflow => ({ name, source, path: `/${source}/${name}.json`, spec: validSpec(name), }); // --------------------------------------------------------------------------- // makeToolProgressSink — folds engine events, pushes through onUpdate // --------------------------------------------------------------------------- describe("formatWorkflowToolResult", () => { const result: WorkflowResult = { workflow: "context-budget", phases: [{ id: "review", ran: 2, ok: 1, dropped: 1 }], phaseResults: { review: [{ ok: true, value: "INTERMEDIATE".repeat(DEFAULT_MAX_BYTES) }], }, phaseFailures: { review: [ { agentId: "review#1", index: 1, code: "transport", message: "Provider unavailable", retryable: true, attempts: 1, item: "BULKY INPUT", transcriptTail: "PRIVATE TRANSCRIPT", }, ], }, result: [{ answer: "FINAL ANSWER" }], report: { summary: "HUMAN REPORT" }, steered: false, tokenUsage: { input: 1, output: 2, total: 3, cacheRead: 0, cacheWrite: 0, cost: 0.01, }, dynamicExtensions: [], }; it("returns final output once, retaining diagnostics and metadata without intermediate payloads", async () => { const output = await formatWorkflowToolResult(result); const body = JSON.parse(output.text); expect(body.result).toEqual(result.result); expect(body.report).toEqual(result.report); expect(output.text.match(/FINAL ANSWER/g)).toHaveLength(1); expect(output.text.match(/HUMAN REPORT/g)).toHaveLength(1); expect(body.phases).toEqual(result.phases); expect(body.tokenUsage).toEqual(result.tokenUsage); expect(body.steered).toBe(false); expect(body.dynamicExtensions).toEqual([]); expect(body.phaseResults).toBeUndefined(); expect(body.phaseFailures.review).toEqual([ { agentId: "review#1", index: 1, code: "transport", message: "Provider unavailable", retryable: true, attempts: 1, }, ]); for (const excluded of [ "INTERMEDIATE", "BULKY INPUT", "PRIVATE TRANSCRIPT", ]) expect(output.text).not.toContain(excluded); expect(body.fullOutputPath).toBe(output.fullOutputPath); expect(body.executionStatus).toBe("incomplete"); expect(JSON.parse(await readFile(output.fullOutputPath, "utf8"))).toEqual( result, ); expect( JSON.parse(await readFile(output.fullOutputPath, "utf8")).report, ).toEqual(result.report); expect(result.phaseFailures.review[0].transcriptTail).toBe( "PRIVATE TRANSCRIPT", ); }); it("preserves semantic blockers without calling a finished execution successful", async () => { const full = { ...result, phases: [], phaseFailures: {}, result: [{ status: "blocked", blockers: ["No reproduction"] }], }; const output = await formatWorkflowToolResult(full); expect(JSON.parse(output.text)).toMatchObject({ executionStatus: "finished", result: full.result, fullOutputPath: output.fullOutputPath, }); expect(JSON.parse(await readFile(output.fullOutputPath, "utf8"))).toEqual( full, ); const cancelled = await formatWorkflowToolResult({ ...full, aborted: true, }); expect(JSON.parse(cancelled.text).executionStatus).toBe("aborted"); }); it.each([ ["UTF-8 bytes", "界".repeat(DEFAULT_MAX_BYTES / 2)], ["lines", Array.from({ length: DEFAULT_MAX_LINES }, () => "line")], ])( "bounds %s and preserves the full aggregate in a private file", async (_limit, value) => { const full = { ...result, result: value }; const output = await formatWorkflowToolResult(full); try { expect(Buffer.byteLength(output.text)).toBeLessThanOrEqual( DEFAULT_MAX_BYTES, ); expect(output.text.split("\n").length).toBeLessThanOrEqual( DEFAULT_MAX_LINES, ); expect(JSON.parse(output.text)).toMatchObject({ truncated: true, fullOutputPath: output.fullOutputPath, }); expect(output.fullOutputPath).toBeTypeOf("string"); const path = output.fullOutputPath; if (!path) throw new Error("Missing full output file"); expect(JSON.parse(await readFile(path, "utf8"))).toEqual(full); if (process.platform !== "win32") expect((await stat(path)).mode & 0o777).toBe(0o600); } finally { if (output.fullOutputPath) await rm(dirname(output.fullOutputPath), { recursive: true, force: true, }); } }, ); }); describe("formatUltraResultMarkdown", () => { beforeAll(() => initTheme()); function renderedLines(markdown: string): string[] { return new Markdown(markdown, 0, 0, getMarkdownTheme()) .render(200) .map(stripVTControlCharacters); } it("renders text reports as native Markdown documents", () => { const report = "# HUMAN REPORT\n\n- **Finding:** native rendering"; expect( formatUltraResultMarkdown({ workflow: "custom", phases: [{ id: "work", ran: 1, ok: 1, dropped: 0 }], phaseResults: {}, phaseFailures: {}, result: [{ secret: "AGENT ONLY" }], report, steered: false, tokenUsage: { input: 10, output: 20, total: 30, cacheRead: 0, cacheWrite: 0, cost: 0.4, }, }), ).toBe(report); }); it("renders only the configured structured report in collapsed and expanded views", () => { const markdown = formatUltraResultMarkdown({ workflow: "custom", phases: [{ id: "work", ran: 1, ok: 1, dropped: 0 }], phaseResults: {}, phaseFailures: {}, result: [{ secret: "AGENT ONLY" }], report: { summary: "HUMAN REPORT", findings: ["one"] }, steered: false, tokenUsage: { input: 10, output: 20, total: 30, cacheRead: 0, cacheWrite: 0, cost: 0.4, }, }); expect(markdown).toContain("HUMAN REPORT"); expect(markdown).not.toContain("AGENT ONLY"); expect(markdown).not.toContain("Usage:"); for (const expanded of [false, true]) { const output = formatUltraResultMarkdown( { workflow: "custom", phases: [{ id: "work", ran: 1, ok: 1, dropped: 0 }], phaseResults: {}, phaseFailures: {}, result: [{ secret: "AGENT ONLY" }], report: { summary: "HUMAN REPORT", findings: ["one"] }, steered: false, tokenUsage: { input: 10, output: 20, total: 30, cacheRead: 0, cacheWrite: 0, cost: 0.4, }, }, expanded, ); expect(output).toContain("HUMAN REPORT"); expect(output).not.toContain("AGENT ONLY"); expect(output).not.toContain("Usage:"); expect(output).not.toContain("Workflow details"); } }); it("renders compact drops and expanded typed diagnostics", () => { const failure = { code: "transport" as const, message: "provider error", retryable: true, attempts: 3, model: "openai/gpt", thinkingLevel: "high" as const, transcriptTail: "last bad response", }; const result = { workflow: "review", phases: [ { id: "review", ran: 3, ok: 2, dropped: 1 }, { id: "verify", ran: 2, ok: 2, dropped: 0 }, ], phaseResults: { review: [{ ok: false as const, value: null, failure }], verify: [ { ok: true as const, value: { title: "possible issue", status: "refuted", file: "a.ts", why: "not reproducible", }, }, ], }, phaseFailures: { review: [{ agentId: "review#0", index: 0, item: "scope", ...failure }], verify: [], }, result: [], steered: false, }; const collapsed = formatUltraResultMarkdown(result); expect(collapsed).toContain("# ultra review"); expect(collapsed).toContain("No verified findings."); expect(collapsed).toContain("1 candidate finding refuted."); expect(collapsed).toContain("Review incomplete: 1 step dropped."); expect(collapsed).toContain( "Inspect dropped steps and rerun review before relying on the result.", ); expect(collapsed).toContain("review: 2/3 ok, 1 dropped"); const lines = renderedLines(collapsed); const failureLine = lines.findIndex((line) => line.includes("provider error"), ); expect(failureLine).toBeGreaterThanOrEqual(0); expect(lines[failureLine]).not.toContain("review#0"); expect(lines[failureLine + 1]).toContain("review#0 [transport]"); expect(collapsed).not.toContain("last bad response"); expect(collapsed).toContain("Refuted findings"); const expanded = formatUltraResultMarkdown(result, true); expect(expanded).toContain("retryable: yes"); expect(expanded).toContain("attempts: 3"); expect(expanded).toContain("model: openai/gpt"); expect(expanded).toContain("**Item**\n - scope"); expect(expanded).toContain("last bad response"); }); it.each(["unresolved", undefined])( "does not label uncertain verdicts as refuted (status: %s)", (status) => { const value = { id: "security-1", title: "Needs evidence", file: "a.ts", why: "Cannot inspect target", ...(status ? { status } : { real: false }), }; const markdown = formatUltraResultMarkdown({ workflow: "review", phases: [], phaseFailures: {}, phaseResults: { verify: [{ ok: true, value }] }, result: [], steered: false, } as unknown as WorkflowResult); expect(markdown).toContain("## Unresolved findings"); expect(markdown).toContain("Cannot inspect target"); expect(markdown).not.toContain("## Refuted findings"); expect(markdown).not.toContain("No code changes recommended"); }, ); it("shows report blockers, unresolved findings, and the evidence path", () => { const markdown = formatUltraResultMarkdown({ workflow: "review", phases: [], phaseResults: {}, phaseFailures: {}, steered: false, fullOutputPath: "/evidence/result.json", result: [ { status: "blocked", analysis: "Review incomplete", findings: [], blockers: ["Missing target"], unresolved: [ { id: "correctness-1", title: "Unverified claim", file: "a.ts", why: "Missing evidence", }, ], recommendations: ["Provide target"], }, ], } as unknown as WorkflowResult); expect(markdown).toContain("Status: blocked"); expect(markdown).toContain("## Blockers\n- Missing target"); expect(markdown).toContain("## Unresolved findings"); expect(markdown).toContain("Unverified claim"); expect(markdown).toContain("/evidence/result.json"); }); it("bounds collapsed failure messages", () => { const message = "x".repeat(500); const failure = { code: "transport" as const, message, retryable: true, attempts: 1, }; const markdown = formatUltraResultMarkdown({ workflow: "bounded", phases: [{ id: "p", ran: 1, ok: 0, dropped: 1 }], phaseResults: { p: [{ ok: false, value: null, failure }] }, phaseFailures: { p: [{ agentId: "p#0", index: 0, ...failure }], }, result: [], steered: false, tokenUsage: { input: 0, output: 0, total: 0, cacheRead: 0, cacheWrite: 0, cost: 0, }, }); expect(markdown).toContain(`${"x".repeat(159)}…`); expect(markdown).not.toContain(message); }); it("renders final result titles before raw JSON", () => { const markdown = formatUltraResultMarkdown({ workflow: "review", phases: [{ id: "verify", ran: 1, ok: 1, dropped: 0 }], phaseResults: {}, phaseFailures: {}, result: [{ title: "real bug", file: "src/a.ts", why: "confirmed" }], steered: false, }); expect(markdown).toContain("Verified findings"); expect(markdown).toContain("real bug"); expect(markdown).toContain("src/a.ts"); expect(markdown).not.toContain("```json"); const lines = renderedLines(markdown); const finding = lines.findIndex((line) => line.includes("real bug")); expect(finding).toBeGreaterThanOrEqual(0); expect(lines[finding]).not.toContain("src/a.ts"); expect(lines[finding + 1]).toContain("File: src/a.ts"); expect(lines[finding + 2]).toContain("confirmed"); }); it("renders synthesized review analysis and concrete recommendations", () => { const markdown = formatUltraResultMarkdown({ workflow: "review", phases: [{ id: "report", ran: 1, ok: 1, dropped: 0 }], phaseResults: {}, phaseFailures: {}, result: [ { analysis: "The implementation is cohesive, but one boundary check is missing.", findings: [ { title: "Missing boundary check", file: "src/parser.ts", why: "Malformed input reaches the decoder.", severity: "high", }, ], recommendations: [ "Reject malformed lengths before decoding.", "Add the malformed-input regression test, then rerun review.", ], }, ], steered: false, }); expect(markdown).toContain("## Analysis"); expect(markdown).toContain( "The implementation is cohesive, but one boundary check is missing.", ); expect(markdown).toContain("## Verified findings"); expect(markdown).toContain("Missing boundary check"); expect(markdown).toContain("## Recommendations"); expect(markdown).toContain("Reject malformed lengths before decoding."); expect(markdown).not.toContain("Continue with the final results above."); expect(markdown).not.toContain("```json"); }); it("renders a cited research report instead of dumping JSON", () => { const result = { workflow: "research", phases: [{ id: "report", ran: 1, ok: 1, dropped: 0 }], phaseResults: {}, phaseFailures: {}, result: [ { answer_status: "partially answered", executive_summary: "The evidence supports the main claim with one caveat.", analysis: "Primary and independent sources converge.", findings: [ { claim: "The main claim is supported", evidence: "Two independent primary sources agree.", confidence: "high", citations: ["https://example.com/primary"], }, ], contradictions: ["One older source disagrees."], gaps: ["No current regional data."], recommendations: ["Collect regional data before deciding."], sources: [ { title: "Primary source", url: "https://example.com/primary", source_type: "primary", published_at: "2026-08-01", }, ], }, ], steered: false, }; const markdown = formatUltraResultMarkdown(result); expect(markdown).toContain("## Answer"); expect(markdown).toContain("Status: partially answered"); expect(markdown).toContain("## Findings"); expect(markdown).toContain("Sources: https://example.com/primary"); expect(markdown).toContain("## Contradictions"); expect(markdown).toContain("## Research gaps"); expect(markdown).toContain("## Sources"); expect(markdown).not.toContain("```json"); const lines = renderedLines(markdown); const answer = lines.findIndex((line) => line.includes("The evidence supports the main claim"), ); expect(answer).toBeGreaterThanOrEqual(0); expect(lines[answer]).not.toContain("Status:"); expect(lines[answer + 1]).toContain("Status: partially answered"); const finding = lines.findIndex((line) => line.includes("The main claim is supported"), ); expect(finding).toBeGreaterThanOrEqual(0); expect(lines[finding]).not.toContain("high"); expect(lines[finding + 1]).toContain("Confidence: high"); const source = lines.findIndex((line) => line.includes("Primary source")); expect(source).toBeGreaterThanOrEqual(0); expect(lines[source]).not.toContain("2026-08-01"); expect(lines[source + 1]).toContain("primary · 2026-08-01"); expect(lines[source + 1]).toContain("https://example.com/primary"); }); it.each([false, true])( "renders arbitrary result fields and failure inputs (expanded=%s)", (expanded) => { const markdown = formatUltraResultMarkdown( { workflow: "arbitrary", phases: [{ id: "p", ran: 1, ok: 0, dropped: 1 }], phaseResults: {}, phaseFailures: { p: [ { agentId: "p#0", index: 0, code: "missing-output", message: "No output", retryable: false, attempts: 1, item: { task: "integration", files: ["a.ts"] }, transcriptTail: '{"status":"unfinished"}', }, ], }, result: [ { verdict: { accepted: false }, evidence: ["Observed behavior"] }, { title: "Titled result", extra: { retained: true } }, "plain result", 42, ], steered: false, }, expanded, ); const lines = renderedLines(markdown).join("\n"); expect(lines).toContain("Verdict:"); expect(lines).toContain("Accepted: false"); expect(lines).toContain("Observed behavior"); expect(lines).toContain("Extra:"); expect(lines).toContain("Retained: true"); expect(lines).toContain("plain result"); expect(lines).not.toContain('"plain result"'); expect(lines).not.toContain('"verdict":'); expect(lines).not.toContain('"task":'); expect(markdown).not.toContain("```json"); if (expanded) { expect(lines).toContain("Task: integration"); expect(lines).toContain("Status: unfinished"); } }, ); it.each([false, true])( "renders cost/output first and expands input/cache (expanded=%s)", (expanded) => { const markdown = formatUltraResultMarkdown( { workflow: "review", phases: [], phaseResults: {}, phaseFailures: {}, result: [], steered: false, tokenUsage: { input: 100, output: 40, total: 140, cacheRead: 20, cacheWrite: 5, cost: 0.1234, }, }, expanded, ); expect(markdown).toContain("Usage: $0.1234 · 40 output"); expect(markdown).not.toContain("140 tokens"); expect(markdown.indexOf("Usage:")).toBeLessThan( markdown.indexOf("No phases."), ); const lines = renderedLines(markdown); const usage = lines.findIndex((line) => line.includes("Usage:")); expect(lines[usage]).toContain("$0.1234 · 40 output"); if (expanded) expect(lines[usage + 1]).toContain( "100 uncached input · 20 cache read · 5 cache write", ); else { expect(markdown).not.toContain("uncached input"); expect(markdown).not.toContain("cache read"); } }, ); it("renders one concise final dynamic-extension block", () => { const markdown = formatUltraResultMarkdown({ workflow: "consumer", phases: [], phaseResults: {}, phaseFailures: {}, result: [], steered: false, tokenUsage: { input: 0, output: 0, total: 0, cacheRead: 0, cacheWrite: 0, cost: 0, }, dynamicExtensions: [ { name: "probe-capability", description: "Handles deterministic probe values.", selector: "probe-capability@revision", sourcePath: "/snapshot/probe-capability/revision", debugLogPaths: ["/logs/child/probe-capability.jsonl"], }, ], }); expect(markdown).toContain("## Dynamic extensions"); expect(markdown).toContain("probe-capability"); expect(markdown).toContain("/snapshot/probe-capability/revision"); expect(markdown).not.toContain("/logs/child/probe-capability.jsonl"); expect(markdown).not.toContain("probe-capability@revision"); const lines = renderedLines(markdown); const extension = lines.findIndex((line) => line.includes("Handles deterministic probe values."), ); expect(extension).toBeGreaterThanOrEqual(0); expect(lines[extension]).not.toContain("Source:"); expect(lines[extension + 1]).toContain( "Source: /snapshot/probe-capability/revision", ); }); it("renders complete workflow details instead of expanded raw JSON", () => { const result = { workflow: "review", phases: [], phaseResults: {}, phaseFailures: {}, result: [], steered: false, }; expect(formatUltraResultMarkdown(result)).not.toContain("Workflow details"); const expanded = formatUltraResultMarkdown(result, true); expect(expanded).toContain("## Workflow details"); expect(expanded).toContain("**Phase Results:**"); expect(expanded).toContain("**Steered:** false"); expect(expanded).not.toContain("```json"); expect(expanded).not.toContain("Raw JSON"); }); }); // --------------------------------------------------------------------------- // Human summaries stay in the renderer, not the model-facing command content. // --------------------------------------------------------------------------- describe("human workflow summaries", () => { it("formats a human summary independently of the model-facing envelope", () => { const content = formatUltraResultMarkdown({ workflow: "review", phases: [ { id: "review", ran: 3, ok: 3, dropped: 0 }, { id: "verify", ran: 1, ok: 1, dropped: 0 }, ], phaseResults: { review: [ { ok: true, value: { findings: [{ title: "candidate", file: "a.ts" }] }, }, ], verify: [{ ok: true, value: { title: "real bug", real: true } }], }, phaseFailures: { review: [], verify: [] }, result: [{ title: "real bug", real: true }], steered: false, }); expect(content).toContain("# ultra review"); expect(content).toContain("1 verified finding."); expect(content).toContain("Verified findings"); expect(content).toContain("real bug"); expect(content).toContain("Address verified findings, then rerun review."); expect(content).not.toContain("```json"); expect(content).not.toContain("phaseResults"); }); it("makes an empty result explicit instead of looking like no run happened", () => { const content = formatUltraResultMarkdown({ workflow: "review", phases: [{ id: "review", ran: 3, ok: 3, dropped: 0 }], phaseResults: { review: [{ ok: true, value: { findings: [] } }], verify: [ { ok: true, value: { title: "speculative issue", file: "a.ts", status: "refuted", why: "not reproducible", }, }, ], }, phaseFailures: { review: [] }, result: [], steered: false, }); expect(content).toContain("No verified findings."); expect(content).toContain("1 candidate finding refuted."); expect(content).toContain("No code changes recommended from this review."); expect(content).toContain("Continue normal project validation."); expect(content).not.toContain("Result is empty"); expect(content).not.toContain('"result": []'); }); it("renders structured details for non-review workflows", () => { const content = formatUltraResultMarkdown({ workflow: "feature", phases: [], phaseResults: {}, phaseFailures: {}, result: { plan: "ship it" }, steered: false, }); expect(content).toContain("# ultra feature"); expect(content).toContain("**Plan:** ship it"); }); }); // --------------------------------------------------------------------------- // makeToolProgressSink — folds engine events, pushes through onUpdate // --------------------------------------------------------------------------- describe("makeToolProgressSink", () => { const ev = (over: Partial): AgentProgressEvent => ({ agentId: "a#0", phase: "p", kind: "start", ...over, }); it("publishes all planned phases and queued agents on the first update", () => { const onUpdate = vi.fn(); const { sink } = makeToolProgressSink(onUpdate); const plan: UltraProgressEvent = { kind: "plan", startedAt: 1000, phases: [ { phase: "one", status: "resolved", agentIds: ["one#0"] }, { phase: "verify", status: "agents-pending", agentIds: [] }, ], }; sink(plan); const first = onUpdate.mock.calls[0][0]; expect( first.details.phases.map((phase: { phase: string }) => phase.phase), ).toEqual(["one", "verify"]); expect(first.details.rows).toEqual([ expect.objectContaining({ agentId: "one#0", status: "queued" }), ]); expect(first.content[0].text).toContain("1 queued"); expect(first.content[0].text).toContain("1 later"); }); it("calls onUpdate with { content, details: reducerState } on every event, accumulating", () => { const onUpdate = vi.fn(); const { sink, getState } = makeToolProgressSink(onUpdate); sink(ev({ agentId: "a#0", kind: "start" })); sink(ev({ agentId: "a#1", kind: "start" })); sink(ev({ agentId: "a#0", kind: "end", status: "done" })); expect(onUpdate).toHaveBeenCalledTimes(3); const last = onUpdate.mock.calls[2][0] as { content: unknown[]; details: { total: number; done: number }; }; expect(Array.isArray(last.content)).toBe(true); expect(last.details.total).toBe(2); expect(last.details.done).toBe(1); expect(getState().total).toBe(2); }); it("emits fresh details each call (renderResult tracks live state, not a frozen snapshot)", () => { const seen: number[] = []; const onUpdate = vi.fn((r: { details: { total: number } }) => seen.push(r.details.total), ); const { sink } = makeToolProgressSink(onUpdate); sink(ev({ agentId: "a#0" })); sink(ev({ agentId: "a#1" })); expect(seen).toEqual([1, 2]); // details changed between calls }); it("attaches the final aggregate only after progress completes", () => { const onUpdate = vi.fn(); const progress = makeToolProgressSink(onUpdate); progress.sink(ev({ kind: "start" })); expect(progress.getState().workflowResult).toBeUndefined(); const result = { workflow: "done", phases: [{ id: "p", ran: 1, ok: 1, dropped: 0 }], phaseResults: {}, phaseFailures: {}, result: [{ summary: "finished" }], steered: false, tokenUsage: { input: 1, output: 1, total: 2, cacheRead: 0, cacheWrite: 0, cost: 0, }, }; expect(progress.finish(result).workflowResult).toEqual(result); expect(() => structuredClone(progress.getState())).not.toThrow(); }); it("strips non-cloneable `controls` so Pi can structured-clone details + result (regression)", () => { // The runner attaches live `controls` functions to every agent event. // Pi structured-clones the tool `details` (per-tick onUpdate + final // result); a function in the state // throws "The object can not be cloned." and fails the whole run. const noopControls = { steer: () => {}, abort: () => {} }; const updates: { details: unknown }[] = []; const onUpdate = vi.fn((u: { details: unknown }) => updates.push(u)); const { sink, getState } = makeToolProgressSink(onUpdate); sink( ev({ agentId: "a#0", kind: "prompt", summary: "Review security", prompt: "SECRET EXACT PROMPT", controls: noopControls, }), ); sink(ev({ agentId: "a#0", kind: "start", controls: noopControls })); sink( ev({ agentId: "a#0", kind: "end", status: "done", controls: noopControls, }), ); // Both the live onUpdate payload and the final state must clone cleanly. expect(() => structuredClone(updates[0].details)).not.toThrow(); expect(() => structuredClone(getState())).not.toThrow(); // And the run is still tracked correctly. expect(getState().rows[0]).toMatchObject({ controls: undefined, prompt: undefined, summary: "Review security", }); expect(JSON.stringify(getState())).not.toContain("SECRET EXACT PROMPT"); expect(getState().done).toBe(1); }); }); // --------------------------------------------------------------------------- // completeWorkflowNames — getArgumentCompletions logic // --------------------------------------------------------------------------- describe("completeWorkflowNames", () => { const list = [ discovered("ultra-review", "bundled"), discovered("audit", "project"), discovered("ui-sweep", "global"), ]; it("filters by prefix and returns {value,label} items", () => { expect(completeWorkflowNames(list, "u")).toEqual([ { value: "ultra-review", label: "ultra-review", description: "bundled" }, { value: "ui-sweep", label: "ui-sweep", description: "global" }, ]); }); it("is case-insensitive and returns all names for an empty prefix", () => { expect(completeWorkflowNames(list, "")?.map((i) => i.value)).toEqual([ "ultra-review", "audit", "ui-sweep", ]); expect(completeWorkflowNames(list, "AUD")?.map((i) => i.value)).toEqual([ "audit", ]); }); it("returns null when nothing matches", () => { expect(completeWorkflowNames(list, "zzz")).toBeNull(); }); }); describe("completeUltraCommand", () => { const list = [ discovered("review", "bundled"), discovered("repair", "project"), ]; it("completes only subcommands at the top level", () => { expect(completeUltraCommand(list, "")?.map((item) => item.value)).toEqual([ "exec", "run", ]); expect(completeUltraCommand(list, "r")?.map((item) => item.value)).toEqual([ "run", ]); expect(completeUltraCommand(list, "h")).toBeNull(); }); it("completes workflow names only below run", () => { expect( completeUltraCommand(list, "run re")?.map((item) => item.value), ).toEqual(["run review", "run repair"]); expect(completeUltraCommand(list, "exec re")).toBeNull(); }); }); // --------------------------------------------------------------------------- // resolveWorkflowInput — name (persisted) vs inline spec (ephemeral) // --------------------------------------------------------------------------- describe("resolveWorkflowInput", () => { const list = [ discovered("ultra-review", "bundled"), discovered("audit", "project"), ]; it("parses an inline spec (ephemeral mode)", () => { const out = resolveWorkflowInput( { spec: validSpec("inline"), args: { x: 1 } }, list, ); expect(out.spec.name).toBe("inline"); expect(out.args).toEqual({ x: 1 }); }); it("looks a named workflow up among discovered (persisted mode)", () => { const out = resolveWorkflowInput({ name: "audit" }, list); expect(out.spec.name).toBe("audit"); expect(out.args).toBeUndefined(); }); it("maps a bare positional remainder to the spec's declared args.params", () => { const wf: DiscoveredWorkflow = { name: "rev", source: "bundled", path: "/bundled/rev.json", spec: specWithParams("rev", ["base"]), }; const out = resolveWorkflowInput({ name: "rev" }, [wf], "main"); expect(out.args).toEqual({ base: "main" }); }); it("ignores a positional remainder when the spec declares no params", () => { const out = resolveWorkflowInput({ name: "audit" }, list, "main"); expect(out.args).toBeUndefined(); }); it("prefers an explicit JSON args object over the positional remainder", () => { const wf: DiscoveredWorkflow = { name: "rev", source: "bundled", path: "/bundled/rev.json", spec: specWithParams("rev", ["base"]), }; const out = resolveWorkflowInput( { name: "rev", args: { base: "dev" } }, [wf], "main", ); expect(out.args).toEqual({ base: "dev" }); }); it("applies the same configured fanout policy to inline and saved specs", () => { const raw = { name: "research", phases: [ { id: "shop", kind: "fanout", over: ["one"], step: { summary: "Search", prompt: "Do not mutate.", tools: ["web"] }, }, ], }; const saved: DiscoveredWorkflow = { name: raw.name, source: "project", path: "/project/research.json", spec: raw as unknown as WorkflowSpec, }; for (const input of [{ spec: raw }, { name: raw.name }]) { expect(() => resolveWorkflowInput(input, [saved])).toThrow( /outside the fanout allowlist/, ); expect(() => resolveWorkflowInput(input, [saved], undefined, ["read", "web"]), ).not.toThrow(); } }); it("throws with the available names for an unknown name", () => { expect(() => resolveWorkflowInput({ name: "missing" }, list)).toThrow( /ultra-review|audit/, ); }); it("throws when neither name nor spec is provided", () => { expect(() => resolveWorkflowInput({}, list)).toThrow( /name.*spec|spec.*name/i, ); }); it("propagates a parse error for a malformed inline spec", () => { expect(() => resolveWorkflowInput({ spec: { phases: [] } }, list), ).toThrow(); }); it("validates a named workflow's spec, surfacing parseWorkflow's error on a malformed saved file", () => { // Discovery stores the raw JSON.parse output cast to WorkflowSpec — a saved // file can be JSON-valid with a `name` yet structurally malformed (no // `phases`). Resolving by name must run parseWorkflow so `/ultra run ` // fails fast with the TypeBox path error, not a cryptic mid-run crash. const malformed: DiscoveredWorkflow = { name: "broken", source: "project", path: "/project/broken.json", spec: { name: "broken" } as unknown as WorkflowSpec, }; expect(() => resolveWorkflowInput({ name: "broken" }, [malformed])).toThrow( /Invalid workflow spec/, ); }); }); // --------------------------------------------------------------------------- // command parsing // --------------------------------------------------------------------------- describe("parseUltraCommand", () => { it("routes every feature through an explicit subcommand", () => { expect(parseUltraCommand(" ")).toEqual({ subcommand: "activate" }); expect(parseUltraCommand("exec fix the race")).toEqual({ subcommand: "exec", instructions: "fix the race", }); expect(parseUltraCommand("run review main")).toEqual({ subcommand: "run", invocation: "review main", }); }); it("rejects missing arguments and unknown commands", () => { for (const input of [ "exec", "run", "review main", "runs", "history", "history extra", ]) { expect(parseUltraCommand(input)).toEqual({ subcommand: "invalid" }); } }); }); describe("parseCommandLine", () => { it("returns {} for an empty line", () => { expect(parseCommandLine(" ")).toEqual({}); }); it("extracts a bare name", () => { expect(parseCommandLine("ultra-review")).toEqual({ name: "ultra-review" }); }); it("parses a trailing JSON object as args (retaining the raw rest)", () => { expect(parseCommandLine('ultra-review {"base":"main","n":2}')).toEqual({ name: "ultra-review", args: { base: "main", n: 2 }, rest: '{"base":"main","n":2}', }); }); it("keeps non-JSON trailing text as `rest` (for positional mapping)", () => { expect(parseCommandLine("ultra-review main")).toEqual({ name: "ultra-review", rest: "main", }); expect(parseCommandLine("ultra-review [1,2]")).toEqual({ name: "ultra-review", rest: "[1,2]", }); }); }); // --------------------------------------------------------------------------- // mapPositionalArgs — bare `/ultra run ` → declared args.params // --------------------------------------------------------------------------- describe("mapPositionalArgs", () => { it("fills declared params in order from whitespace tokens", () => { expect(mapPositionalArgs("main", ["base"])).toEqual({ base: "main" }); expect(mapPositionalArgs("a b", ["x", "y"])).toEqual({ x: "a", y: "b" }); }); it("soaks remaining tokens into the last declared param (free-text tail)", () => { // single free-text param keeps the whole remainder expect(mapPositionalArgs("add a dark-mode toggle", ["task"])).toEqual({ task: "add a dark-mode toggle", }); // only the LAST param soaks; earlier params take one token each expect(mapPositionalArgs("main fix the bug", ["base", "task"])).toEqual({ base: "main", task: "fix the bug", }); }); it("returns undefined for an empty remainder or no declared params", () => { expect(mapPositionalArgs(" ", ["base"])).toBeUndefined(); expect(mapPositionalArgs("main", [])).toBeUndefined(); }); });