Luigit
repositories / pi-ext

pi-ext

bugabingas pi extensions

owned by admin

extensions/ultra/__tests__/wire.test.ts

Raw
import { readFile, rm, stat } from "node:fs/promises";
import { dirname } from "node:path";
import { stripVTControlCharacters } from "node:util";
import {
	DEFAULT_MAX_BYTES,
	DEFAULT_MAX_LINES,
	getMarkdownTheme,
	initTheme,
} from "@earendil-works/pi-coding-agent";
import { Markdown } from "@earendil-works/pi-tui";
import { afterEach, beforeAll, describe, expect, it, vi } from "vitest";
import type { WorkflowResult } from "../engine.ts";
import type { AgentProgressEvent, UltraProgressEvent } from "../progress.ts";
import type { DiscoveredWorkflow, WorkflowSpec } from "../spec.ts";
import {
	completeUltraCommand,
	completeWorkflowNames,
	formatWorkflowToolResult as formatOutput,
	formatUltraResultMarkdown,
	makeToolProgressSink,
	mapPositionalArgs,
	parseCommandLine,
	parseUltraCommand,
	resolveWorkflowInput,
} from "../wire.ts";

const artifacts: string[] = [];
afterEach(async () => {
	await Promise.all(
		artifacts
			.splice(0)
			.map((path) => rm(dirname(path), { recursive: true, force: true })),
	);
});
async function formatWorkflowToolResult(result: WorkflowResult) {
	const output = await formatOutput(result);
	artifacts.push(output.fullOutputPath);
	return output;
}

const validSpec = (name: string): WorkflowSpec => ({
	name,
	phases: [{ id: "p", kind: "single", step: { summary: "hi", prompt: "hi" } }],
});

const specWithParams = (name: string, params: string[]): WorkflowSpec => ({
	name,
	args: { params },
	phases: [
		{
			id: "p",
			kind: "single",
			step: { summary: "hi {args.base}", prompt: "hi {args.base}" },
		},
	],
});

const discovered = (
	name: string,
	source: DiscoveredWorkflow["source"],
): DiscoveredWorkflow => ({
	name,
	source,
	path: `/${source}/${name}.json`,
	spec: validSpec(name),
});

// ---------------------------------------------------------------------------
// makeToolProgressSink — folds engine events, pushes through onUpdate
// ---------------------------------------------------------------------------

describe("formatWorkflowToolResult", () => {
	const result: WorkflowResult = {
		workflow: "context-budget",
		phases: [{ id: "review", ran: 2, ok: 1, dropped: 1 }],
		phaseResults: {
			review: [{ ok: true, value: "INTERMEDIATE".repeat(DEFAULT_MAX_BYTES) }],
		},
		phaseFailures: {
			review: [
				{
					agentId: "review#1",
					index: 1,
					code: "transport",
					message: "Provider unavailable",
					retryable: true,
					attempts: 1,
					item: "BULKY INPUT",
					transcriptTail: "PRIVATE TRANSCRIPT",
				},
			],
		},
		result: [{ answer: "FINAL ANSWER" }],
		report: { summary: "HUMAN REPORT" },
		steered: false,
		tokenUsage: {
			input: 1,
			output: 2,
			total: 3,
			cacheRead: 0,
			cacheWrite: 0,
			cost: 0.01,
		},
		dynamicExtensions: [],
	};

	it("returns final output once, retaining diagnostics and metadata without intermediate payloads", async () => {
		const output = await formatWorkflowToolResult(result);
		const body = JSON.parse(output.text);
		expect(body.result).toEqual(result.result);
		expect(body.report).toEqual(result.report);
		expect(output.text.match(/FINAL ANSWER/g)).toHaveLength(1);
		expect(output.text.match(/HUMAN REPORT/g)).toHaveLength(1);
		expect(body.phases).toEqual(result.phases);
		expect(body.tokenUsage).toEqual(result.tokenUsage);
		expect(body.steered).toBe(false);
		expect(body.dynamicExtensions).toEqual([]);
		expect(body.phaseResults).toBeUndefined();
		expect(body.phaseFailures.review).toEqual([
			{
				agentId: "review#1",
				index: 1,
				code: "transport",
				message: "Provider unavailable",
				retryable: true,
				attempts: 1,
			},
		]);
		for (const excluded of [
			"INTERMEDIATE",
			"BULKY INPUT",
			"PRIVATE TRANSCRIPT",
		])
			expect(output.text).not.toContain(excluded);
		expect(body.fullOutputPath).toBe(output.fullOutputPath);
		expect(body.executionStatus).toBe("incomplete");
		expect(JSON.parse(await readFile(output.fullOutputPath, "utf8"))).toEqual(
			result,
		);
		expect(
			JSON.parse(await readFile(output.fullOutputPath, "utf8")).report,
		).toEqual(result.report);
		expect(result.phaseFailures.review[0].transcriptTail).toBe(
			"PRIVATE TRANSCRIPT",
		);
	});

	it("preserves semantic blockers without calling a finished execution successful", async () => {
		const full = {
			...result,
			phases: [],
			phaseFailures: {},
			result: [{ status: "blocked", blockers: ["No reproduction"] }],
		};
		const output = await formatWorkflowToolResult(full);
		expect(JSON.parse(output.text)).toMatchObject({
			executionStatus: "finished",
			result: full.result,
			fullOutputPath: output.fullOutputPath,
		});
		expect(JSON.parse(await readFile(output.fullOutputPath, "utf8"))).toEqual(
			full,
		);
		const cancelled = await formatWorkflowToolResult({
			...full,
			aborted: true,
		});
		expect(JSON.parse(cancelled.text).executionStatus).toBe("aborted");
	});

	it.each([
		["UTF-8 bytes", "界".repeat(DEFAULT_MAX_BYTES / 2)],
		["lines", Array.from({ length: DEFAULT_MAX_LINES }, () => "line")],
	])(
		"bounds %s and preserves the full aggregate in a private file",
		async (_limit, value) => {
			const full = { ...result, result: value };
			const output = await formatWorkflowToolResult(full);
			try {
				expect(Buffer.byteLength(output.text)).toBeLessThanOrEqual(
					DEFAULT_MAX_BYTES,
				);
				expect(output.text.split("\n").length).toBeLessThanOrEqual(
					DEFAULT_MAX_LINES,
				);
				expect(JSON.parse(output.text)).toMatchObject({
					truncated: true,
					fullOutputPath: output.fullOutputPath,
				});
				expect(output.fullOutputPath).toBeTypeOf("string");
				const path = output.fullOutputPath;
				if (!path) throw new Error("Missing full output file");
				expect(JSON.parse(await readFile(path, "utf8"))).toEqual(full);
				if (process.platform !== "win32")
					expect((await stat(path)).mode & 0o777).toBe(0o600);
			} finally {
				if (output.fullOutputPath)
					await rm(dirname(output.fullOutputPath), {
						recursive: true,
						force: true,
					});
			}
		},
	);
});

describe("formatUltraResultMarkdown", () => {
	beforeAll(() => initTheme());

	function renderedLines(markdown: string): string[] {
		return new Markdown(markdown, 0, 0, getMarkdownTheme())
			.render(200)
			.map(stripVTControlCharacters);
	}

	it("renders text reports as native Markdown documents", () => {
		const report = "# HUMAN REPORT\n\n- **Finding:** native rendering";
		expect(
			formatUltraResultMarkdown({
				workflow: "custom",
				phases: [{ id: "work", ran: 1, ok: 1, dropped: 0 }],
				phaseResults: {},
				phaseFailures: {},
				result: [{ secret: "AGENT ONLY" }],
				report,
				steered: false,
				tokenUsage: {
					input: 10,
					output: 20,
					total: 30,
					cacheRead: 0,
					cacheWrite: 0,
					cost: 0.4,
				},
			}),
		).toBe(report);
	});

	it("renders only the configured structured report in collapsed and expanded views", () => {
		const markdown = formatUltraResultMarkdown({
			workflow: "custom",
			phases: [{ id: "work", ran: 1, ok: 1, dropped: 0 }],
			phaseResults: {},
			phaseFailures: {},
			result: [{ secret: "AGENT ONLY" }],
			report: { summary: "HUMAN REPORT", findings: ["one"] },
			steered: false,
			tokenUsage: {
				input: 10,
				output: 20,
				total: 30,
				cacheRead: 0,
				cacheWrite: 0,
				cost: 0.4,
			},
		});
		expect(markdown).toContain("HUMAN REPORT");
		expect(markdown).not.toContain("AGENT ONLY");
		expect(markdown).not.toContain("Usage:");
		for (const expanded of [false, true]) {
			const output = formatUltraResultMarkdown(
				{
					workflow: "custom",
					phases: [{ id: "work", ran: 1, ok: 1, dropped: 0 }],
					phaseResults: {},
					phaseFailures: {},
					result: [{ secret: "AGENT ONLY" }],
					report: { summary: "HUMAN REPORT", findings: ["one"] },
					steered: false,
					tokenUsage: {
						input: 10,
						output: 20,
						total: 30,
						cacheRead: 0,
						cacheWrite: 0,
						cost: 0.4,
					},
				},
				expanded,
			);
			expect(output).toContain("HUMAN REPORT");
			expect(output).not.toContain("AGENT ONLY");
			expect(output).not.toContain("Usage:");
			expect(output).not.toContain("Workflow details");
		}
	});

	it("renders compact drops and expanded typed diagnostics", () => {
		const failure = {
			code: "transport" as const,
			message: "provider error",
			retryable: true,
			attempts: 3,
			model: "openai/gpt",
			thinkingLevel: "high" as const,
			transcriptTail: "last bad response",
		};
		const result = {
			workflow: "review",
			phases: [
				{ id: "review", ran: 3, ok: 2, dropped: 1 },
				{ id: "verify", ran: 2, ok: 2, dropped: 0 },
			],
			phaseResults: {
				review: [{ ok: false as const, value: null, failure }],
				verify: [
					{
						ok: true as const,
						value: {
							title: "possible issue",
							status: "refuted",
							file: "a.ts",
							why: "not reproducible",
						},
					},
				],
			},
			phaseFailures: {
				review: [{ agentId: "review#0", index: 0, item: "scope", ...failure }],
				verify: [],
			},
			result: [],
			steered: false,
		};

		const collapsed = formatUltraResultMarkdown(result);
		expect(collapsed).toContain("# ultra review");
		expect(collapsed).toContain("No verified findings.");
		expect(collapsed).toContain("1 candidate finding refuted.");
		expect(collapsed).toContain("Review incomplete: 1 step dropped.");
		expect(collapsed).toContain(
			"Inspect dropped steps and rerun review before relying on the result.",
		);
		expect(collapsed).toContain("review: 2/3 ok, 1 dropped");
		const lines = renderedLines(collapsed);
		const failureLine = lines.findIndex((line) =>
			line.includes("provider error"),
		);
		expect(failureLine).toBeGreaterThanOrEqual(0);
		expect(lines[failureLine]).not.toContain("review#0");
		expect(lines[failureLine + 1]).toContain("review#0 [transport]");
		expect(collapsed).not.toContain("last bad response");
		expect(collapsed).toContain("Refuted findings");

		const expanded = formatUltraResultMarkdown(result, true);
		expect(expanded).toContain("retryable: yes");
		expect(expanded).toContain("attempts: 3");
		expect(expanded).toContain("model: openai/gpt");
		expect(expanded).toContain("**Item**\n  - scope");
		expect(expanded).toContain("last bad response");
	});

	it.each(["unresolved", undefined])(
		"does not label uncertain verdicts as refuted (status: %s)",
		(status) => {
			const value = {
				id: "security-1",
				title: "Needs evidence",
				file: "a.ts",
				why: "Cannot inspect target",
				...(status ? { status } : { real: false }),
			};
			const markdown = formatUltraResultMarkdown({
				workflow: "review",
				phases: [],
				phaseFailures: {},
				phaseResults: { verify: [{ ok: true, value }] },
				result: [],
				steered: false,
			} as unknown as WorkflowResult);
			expect(markdown).toContain("## Unresolved findings");
			expect(markdown).toContain("Cannot inspect target");
			expect(markdown).not.toContain("## Refuted findings");
			expect(markdown).not.toContain("No code changes recommended");
		},
	);

	it("shows report blockers, unresolved findings, and the evidence path", () => {
		const markdown = formatUltraResultMarkdown({
			workflow: "review",
			phases: [],
			phaseResults: {},
			phaseFailures: {},
			steered: false,
			fullOutputPath: "/evidence/result.json",
			result: [
				{
					status: "blocked",
					analysis: "Review incomplete",
					findings: [],
					blockers: ["Missing target"],
					unresolved: [
						{
							id: "correctness-1",
							title: "Unverified claim",
							file: "a.ts",
							why: "Missing evidence",
						},
					],
					recommendations: ["Provide target"],
				},
			],
		} as unknown as WorkflowResult);
		expect(markdown).toContain("Status: blocked");
		expect(markdown).toContain("## Blockers\n- Missing target");
		expect(markdown).toContain("## Unresolved findings");
		expect(markdown).toContain("Unverified claim");
		expect(markdown).toContain("/evidence/result.json");
	});

	it("bounds collapsed failure messages", () => {
		const message = "x".repeat(500);
		const failure = {
			code: "transport" as const,
			message,
			retryable: true,
			attempts: 1,
		};
		const markdown = formatUltraResultMarkdown({
			workflow: "bounded",
			phases: [{ id: "p", ran: 1, ok: 0, dropped: 1 }],
			phaseResults: { p: [{ ok: false, value: null, failure }] },
			phaseFailures: {
				p: [{ agentId: "p#0", index: 0, ...failure }],
			},
			result: [],
			steered: false,
			tokenUsage: {
				input: 0,
				output: 0,
				total: 0,
				cacheRead: 0,
				cacheWrite: 0,
				cost: 0,
			},
		});

		expect(markdown).toContain(`${"x".repeat(159)}…`);
		expect(markdown).not.toContain(message);
	});

	it("renders final result titles before raw JSON", () => {
		const markdown = formatUltraResultMarkdown({
			workflow: "review",
			phases: [{ id: "verify", ran: 1, ok: 1, dropped: 0 }],
			phaseResults: {},
			phaseFailures: {},
			result: [{ title: "real bug", file: "src/a.ts", why: "confirmed" }],
			steered: false,
		});

		expect(markdown).toContain("Verified findings");
		expect(markdown).toContain("real bug");
		expect(markdown).toContain("src/a.ts");
		expect(markdown).not.toContain("```json");
		const lines = renderedLines(markdown);
		const finding = lines.findIndex((line) => line.includes("real bug"));
		expect(finding).toBeGreaterThanOrEqual(0);
		expect(lines[finding]).not.toContain("src/a.ts");
		expect(lines[finding + 1]).toContain("File: src/a.ts");
		expect(lines[finding + 2]).toContain("confirmed");
	});

	it("renders synthesized review analysis and concrete recommendations", () => {
		const markdown = formatUltraResultMarkdown({
			workflow: "review",
			phases: [{ id: "report", ran: 1, ok: 1, dropped: 0 }],
			phaseResults: {},
			phaseFailures: {},
			result: [
				{
					analysis:
						"The implementation is cohesive, but one boundary check is missing.",
					findings: [
						{
							title: "Missing boundary check",
							file: "src/parser.ts",
							why: "Malformed input reaches the decoder.",
							severity: "high",
						},
					],
					recommendations: [
						"Reject malformed lengths before decoding.",
						"Add the malformed-input regression test, then rerun review.",
					],
				},
			],
			steered: false,
		});

		expect(markdown).toContain("## Analysis");
		expect(markdown).toContain(
			"The implementation is cohesive, but one boundary check is missing.",
		);
		expect(markdown).toContain("## Verified findings");
		expect(markdown).toContain("Missing boundary check");
		expect(markdown).toContain("## Recommendations");
		expect(markdown).toContain("Reject malformed lengths before decoding.");
		expect(markdown).not.toContain("Continue with the final results above.");
		expect(markdown).not.toContain("```json");
	});

	it("renders a cited research report instead of dumping JSON", () => {
		const result = {
			workflow: "research",
			phases: [{ id: "report", ran: 1, ok: 1, dropped: 0 }],
			phaseResults: {},
			phaseFailures: {},
			result: [
				{
					answer_status: "partially answered",
					executive_summary:
						"The evidence supports the main claim with one caveat.",
					analysis: "Primary and independent sources converge.",
					findings: [
						{
							claim: "The main claim is supported",
							evidence: "Two independent primary sources agree.",
							confidence: "high",
							citations: ["https://example.com/primary"],
						},
					],
					contradictions: ["One older source disagrees."],
					gaps: ["No current regional data."],
					recommendations: ["Collect regional data before deciding."],
					sources: [
						{
							title: "Primary source",
							url: "https://example.com/primary",
							source_type: "primary",
							published_at: "2026-08-01",
						},
					],
				},
			],
			steered: false,
		};

		const markdown = formatUltraResultMarkdown(result);
		expect(markdown).toContain("## Answer");
		expect(markdown).toContain("Status: partially answered");
		expect(markdown).toContain("## Findings");
		expect(markdown).toContain("Sources: https://example.com/primary");
		expect(markdown).toContain("## Contradictions");
		expect(markdown).toContain("## Research gaps");
		expect(markdown).toContain("## Sources");
		expect(markdown).not.toContain("```json");
		const lines = renderedLines(markdown);
		const answer = lines.findIndex((line) =>
			line.includes("The evidence supports the main claim"),
		);
		expect(answer).toBeGreaterThanOrEqual(0);
		expect(lines[answer]).not.toContain("Status:");
		expect(lines[answer + 1]).toContain("Status: partially answered");
		const finding = lines.findIndex((line) =>
			line.includes("The main claim is supported"),
		);
		expect(finding).toBeGreaterThanOrEqual(0);
		expect(lines[finding]).not.toContain("high");
		expect(lines[finding + 1]).toContain("Confidence: high");
		const source = lines.findIndex((line) => line.includes("Primary source"));
		expect(source).toBeGreaterThanOrEqual(0);
		expect(lines[source]).not.toContain("2026-08-01");
		expect(lines[source + 1]).toContain("primary · 2026-08-01");
		expect(lines[source + 1]).toContain("https://example.com/primary");
	});

	it.each([false, true])(
		"renders arbitrary result fields and failure inputs (expanded=%s)",
		(expanded) => {
			const markdown = formatUltraResultMarkdown(
				{
					workflow: "arbitrary",
					phases: [{ id: "p", ran: 1, ok: 0, dropped: 1 }],
					phaseResults: {},
					phaseFailures: {
						p: [
							{
								agentId: "p#0",
								index: 0,
								code: "missing-output",
								message: "No output",
								retryable: false,
								attempts: 1,
								item: { task: "integration", files: ["a.ts"] },
								transcriptTail: '{"status":"unfinished"}',
							},
						],
					},
					result: [
						{ verdict: { accepted: false }, evidence: ["Observed behavior"] },
						{ title: "Titled result", extra: { retained: true } },
						"plain result",
						42,
					],
					steered: false,
				},
				expanded,
			);
			const lines = renderedLines(markdown).join("\n");
			expect(lines).toContain("Verdict:");
			expect(lines).toContain("Accepted: false");
			expect(lines).toContain("Observed behavior");
			expect(lines).toContain("Extra:");
			expect(lines).toContain("Retained: true");
			expect(lines).toContain("plain result");
			expect(lines).not.toContain('"plain result"');
			expect(lines).not.toContain('"verdict":');
			expect(lines).not.toContain('"task":');
			expect(markdown).not.toContain("```json");
			if (expanded) {
				expect(lines).toContain("Task: integration");
				expect(lines).toContain("Status: unfinished");
			}
		},
	);

	it.each([false, true])(
		"renders cost/output first and expands input/cache (expanded=%s)",
		(expanded) => {
			const markdown = formatUltraResultMarkdown(
				{
					workflow: "review",
					phases: [],
					phaseResults: {},
					phaseFailures: {},
					result: [],
					steered: false,
					tokenUsage: {
						input: 100,
						output: 40,
						total: 140,
						cacheRead: 20,
						cacheWrite: 5,
						cost: 0.1234,
					},
				},
				expanded,
			);

			expect(markdown).toContain("Usage: $0.1234 · 40 output");
			expect(markdown).not.toContain("140 tokens");
			expect(markdown.indexOf("Usage:")).toBeLessThan(
				markdown.indexOf("No phases."),
			);
			const lines = renderedLines(markdown);
			const usage = lines.findIndex((line) => line.includes("Usage:"));
			expect(lines[usage]).toContain("$0.1234 · 40 output");
			if (expanded)
				expect(lines[usage + 1]).toContain(
					"100 uncached input · 20 cache read · 5 cache write",
				);
			else {
				expect(markdown).not.toContain("uncached input");
				expect(markdown).not.toContain("cache read");
			}
		},
	);

	it("renders one concise final dynamic-extension block", () => {
		const markdown = formatUltraResultMarkdown({
			workflow: "consumer",
			phases: [],
			phaseResults: {},
			phaseFailures: {},
			result: [],
			steered: false,
			tokenUsage: {
				input: 0,
				output: 0,
				total: 0,
				cacheRead: 0,
				cacheWrite: 0,
				cost: 0,
			},
			dynamicExtensions: [
				{
					name: "probe-capability",
					description: "Handles deterministic probe values.",
					selector: "probe-capability@revision",
					sourcePath: "/snapshot/probe-capability/revision",
					debugLogPaths: ["/logs/child/probe-capability.jsonl"],
				},
			],
		});

		expect(markdown).toContain("## Dynamic extensions");
		expect(markdown).toContain("probe-capability");
		expect(markdown).toContain("/snapshot/probe-capability/revision");
		expect(markdown).not.toContain("/logs/child/probe-capability.jsonl");
		expect(markdown).not.toContain("probe-capability@revision");
		const lines = renderedLines(markdown);
		const extension = lines.findIndex((line) =>
			line.includes("Handles deterministic probe values."),
		);
		expect(extension).toBeGreaterThanOrEqual(0);
		expect(lines[extension]).not.toContain("Source:");
		expect(lines[extension + 1]).toContain(
			"Source: /snapshot/probe-capability/revision",
		);
	});

	it("renders complete workflow details instead of expanded raw JSON", () => {
		const result = {
			workflow: "review",
			phases: [],
			phaseResults: {},
			phaseFailures: {},
			result: [],
			steered: false,
		};

		expect(formatUltraResultMarkdown(result)).not.toContain("Workflow details");
		const expanded = formatUltraResultMarkdown(result, true);
		expect(expanded).toContain("## Workflow details");
		expect(expanded).toContain("**Phase Results:**");
		expect(expanded).toContain("**Steered:** false");
		expect(expanded).not.toContain("```json");
		expect(expanded).not.toContain("Raw JSON");
	});
});

// ---------------------------------------------------------------------------
// Human summaries stay in the renderer, not the model-facing command content.
// ---------------------------------------------------------------------------

describe("human workflow summaries", () => {
	it("formats a human summary independently of the model-facing envelope", () => {
		const content = formatUltraResultMarkdown({
			workflow: "review",
			phases: [
				{ id: "review", ran: 3, ok: 3, dropped: 0 },
				{ id: "verify", ran: 1, ok: 1, dropped: 0 },
			],
			phaseResults: {
				review: [
					{
						ok: true,
						value: { findings: [{ title: "candidate", file: "a.ts" }] },
					},
				],
				verify: [{ ok: true, value: { title: "real bug", real: true } }],
			},
			phaseFailures: { review: [], verify: [] },
			result: [{ title: "real bug", real: true }],
			steered: false,
		});

		expect(content).toContain("# ultra review");
		expect(content).toContain("1 verified finding.");
		expect(content).toContain("Verified findings");
		expect(content).toContain("real bug");
		expect(content).toContain("Address verified findings, then rerun review.");
		expect(content).not.toContain("```json");
		expect(content).not.toContain("phaseResults");
	});

	it("makes an empty result explicit instead of looking like no run happened", () => {
		const content = formatUltraResultMarkdown({
			workflow: "review",
			phases: [{ id: "review", ran: 3, ok: 3, dropped: 0 }],
			phaseResults: {
				review: [{ ok: true, value: { findings: [] } }],
				verify: [
					{
						ok: true,
						value: {
							title: "speculative issue",
							file: "a.ts",
							status: "refuted",
							why: "not reproducible",
						},
					},
				],
			},
			phaseFailures: { review: [] },
			result: [],
			steered: false,
		});

		expect(content).toContain("No verified findings.");
		expect(content).toContain("1 candidate finding refuted.");
		expect(content).toContain("No code changes recommended from this review.");
		expect(content).toContain("Continue normal project validation.");
		expect(content).not.toContain("Result is empty");
		expect(content).not.toContain('"result": []');
	});

	it("renders structured details for non-review workflows", () => {
		const content = formatUltraResultMarkdown({
			workflow: "feature",
			phases: [],
			phaseResults: {},
			phaseFailures: {},
			result: { plan: "ship it" },
			steered: false,
		});

		expect(content).toContain("# ultra feature");
		expect(content).toContain("**Plan:** ship it");
	});
});

// ---------------------------------------------------------------------------
// makeToolProgressSink — folds engine events, pushes through onUpdate
// ---------------------------------------------------------------------------

describe("makeToolProgressSink", () => {
	const ev = (over: Partial<AgentProgressEvent>): AgentProgressEvent => ({
		agentId: "a#0",
		phase: "p",
		kind: "start",
		...over,
	});

	it("publishes all planned phases and queued agents on the first update", () => {
		const onUpdate = vi.fn();
		const { sink } = makeToolProgressSink(onUpdate);
		const plan: UltraProgressEvent = {
			kind: "plan",
			startedAt: 1000,
			phases: [
				{ phase: "one", status: "resolved", agentIds: ["one#0"] },
				{ phase: "verify", status: "agents-pending", agentIds: [] },
			],
		};

		sink(plan);

		const first = onUpdate.mock.calls[0][0];
		expect(
			first.details.phases.map((phase: { phase: string }) => phase.phase),
		).toEqual(["one", "verify"]);
		expect(first.details.rows).toEqual([
			expect.objectContaining({ agentId: "one#0", status: "queued" }),
		]);
		expect(first.content[0].text).toContain("1 queued");
		expect(first.content[0].text).toContain("1 later");
	});

	it("calls onUpdate with { content, details: reducerState } on every event, accumulating", () => {
		const onUpdate = vi.fn();
		const { sink, getState } = makeToolProgressSink(onUpdate);

		sink(ev({ agentId: "a#0", kind: "start" }));
		sink(ev({ agentId: "a#1", kind: "start" }));
		sink(ev({ agentId: "a#0", kind: "end", status: "done" }));

		expect(onUpdate).toHaveBeenCalledTimes(3);
		const last = onUpdate.mock.calls[2][0] as {
			content: unknown[];
			details: { total: number; done: number };
		};
		expect(Array.isArray(last.content)).toBe(true);
		expect(last.details.total).toBe(2);
		expect(last.details.done).toBe(1);
		expect(getState().total).toBe(2);
	});

	it("emits fresh details each call (renderResult tracks live state, not a frozen snapshot)", () => {
		const seen: number[] = [];
		const onUpdate = vi.fn((r: { details: { total: number } }) =>
			seen.push(r.details.total),
		);
		const { sink } = makeToolProgressSink(onUpdate);

		sink(ev({ agentId: "a#0" }));
		sink(ev({ agentId: "a#1" }));

		expect(seen).toEqual([1, 2]); // details changed between calls
	});

	it("attaches the final aggregate only after progress completes", () => {
		const onUpdate = vi.fn();
		const progress = makeToolProgressSink(onUpdate);
		progress.sink(ev({ kind: "start" }));
		expect(progress.getState().workflowResult).toBeUndefined();
		const result = {
			workflow: "done",
			phases: [{ id: "p", ran: 1, ok: 1, dropped: 0 }],
			phaseResults: {},
			phaseFailures: {},
			result: [{ summary: "finished" }],
			steered: false,
			tokenUsage: {
				input: 1,
				output: 1,
				total: 2,
				cacheRead: 0,
				cacheWrite: 0,
				cost: 0,
			},
		};
		expect(progress.finish(result).workflowResult).toEqual(result);
		expect(() => structuredClone(progress.getState())).not.toThrow();
	});

	it("strips non-cloneable `controls` so Pi can structured-clone details + result (regression)", () => {
		// The runner attaches live `controls` functions to every agent event.
		// Pi structured-clones the tool `details` (per-tick onUpdate + final
		// result); a function in the state
		// throws "The object can not be cloned." and fails the whole run.
		const noopControls = { steer: () => {}, abort: () => {} };
		const updates: { details: unknown }[] = [];
		const onUpdate = vi.fn((u: { details: unknown }) => updates.push(u));
		const { sink, getState } = makeToolProgressSink(onUpdate);

		sink(
			ev({
				agentId: "a#0",
				kind: "prompt",
				summary: "Review security",
				prompt: "SECRET EXACT PROMPT",
				controls: noopControls,
			}),
		);
		sink(ev({ agentId: "a#0", kind: "start", controls: noopControls }));
		sink(
			ev({
				agentId: "a#0",
				kind: "end",
				status: "done",
				controls: noopControls,
			}),
		);

		// Both the live onUpdate payload and the final state must clone cleanly.
		expect(() => structuredClone(updates[0].details)).not.toThrow();
		expect(() => structuredClone(getState())).not.toThrow();
		// And the run is still tracked correctly.
		expect(getState().rows[0]).toMatchObject({
			controls: undefined,
			prompt: undefined,
			summary: "Review security",
		});
		expect(JSON.stringify(getState())).not.toContain("SECRET EXACT PROMPT");
		expect(getState().done).toBe(1);
	});
});

// ---------------------------------------------------------------------------
// completeWorkflowNames — getArgumentCompletions logic
// ---------------------------------------------------------------------------

describe("completeWorkflowNames", () => {
	const list = [
		discovered("ultra-review", "bundled"),
		discovered("audit", "project"),
		discovered("ui-sweep", "global"),
	];

	it("filters by prefix and returns {value,label} items", () => {
		expect(completeWorkflowNames(list, "u")).toEqual([
			{ value: "ultra-review", label: "ultra-review", description: "bundled" },
			{ value: "ui-sweep", label: "ui-sweep", description: "global" },
		]);
	});

	it("is case-insensitive and returns all names for an empty prefix", () => {
		expect(completeWorkflowNames(list, "")?.map((i) => i.value)).toEqual([
			"ultra-review",
			"audit",
			"ui-sweep",
		]);
		expect(completeWorkflowNames(list, "AUD")?.map((i) => i.value)).toEqual([
			"audit",
		]);
	});

	it("returns null when nothing matches", () => {
		expect(completeWorkflowNames(list, "zzz")).toBeNull();
	});
});

describe("completeUltraCommand", () => {
	const list = [
		discovered("review", "bundled"),
		discovered("repair", "project"),
	];

	it("completes only subcommands at the top level", () => {
		expect(completeUltraCommand(list, "")?.map((item) => item.value)).toEqual([
			"exec",
			"run",
		]);
		expect(completeUltraCommand(list, "r")?.map((item) => item.value)).toEqual([
			"run",
		]);
		expect(completeUltraCommand(list, "h")).toBeNull();
	});

	it("completes workflow names only below run", () => {
		expect(
			completeUltraCommand(list, "run re")?.map((item) => item.value),
		).toEqual(["run review", "run repair"]);
		expect(completeUltraCommand(list, "exec re")).toBeNull();
	});
});

// ---------------------------------------------------------------------------
// resolveWorkflowInput — name (persisted) vs inline spec (ephemeral)
// ---------------------------------------------------------------------------

describe("resolveWorkflowInput", () => {
	const list = [
		discovered("ultra-review", "bundled"),
		discovered("audit", "project"),
	];

	it("parses an inline spec (ephemeral mode)", () => {
		const out = resolveWorkflowInput(
			{ spec: validSpec("inline"), args: { x: 1 } },
			list,
		);
		expect(out.spec.name).toBe("inline");
		expect(out.args).toEqual({ x: 1 });
	});

	it("looks a named workflow up among discovered (persisted mode)", () => {
		const out = resolveWorkflowInput({ name: "audit" }, list);
		expect(out.spec.name).toBe("audit");
		expect(out.args).toBeUndefined();
	});

	it("maps a bare positional remainder to the spec's declared args.params", () => {
		const wf: DiscoveredWorkflow = {
			name: "rev",
			source: "bundled",
			path: "/bundled/rev.json",
			spec: specWithParams("rev", ["base"]),
		};
		const out = resolveWorkflowInput({ name: "rev" }, [wf], "main");
		expect(out.args).toEqual({ base: "main" });
	});

	it("ignores a positional remainder when the spec declares no params", () => {
		const out = resolveWorkflowInput({ name: "audit" }, list, "main");
		expect(out.args).toBeUndefined();
	});

	it("prefers an explicit JSON args object over the positional remainder", () => {
		const wf: DiscoveredWorkflow = {
			name: "rev",
			source: "bundled",
			path: "/bundled/rev.json",
			spec: specWithParams("rev", ["base"]),
		};
		const out = resolveWorkflowInput(
			{ name: "rev", args: { base: "dev" } },
			[wf],
			"main",
		);
		expect(out.args).toEqual({ base: "dev" });
	});

	it("applies the same configured fanout policy to inline and saved specs", () => {
		const raw = {
			name: "research",
			phases: [
				{
					id: "shop",
					kind: "fanout",
					over: ["one"],
					step: { summary: "Search", prompt: "Do not mutate.", tools: ["web"] },
				},
			],
		};
		const saved: DiscoveredWorkflow = {
			name: raw.name,
			source: "project",
			path: "/project/research.json",
			spec: raw as unknown as WorkflowSpec,
		};
		for (const input of [{ spec: raw }, { name: raw.name }]) {
			expect(() => resolveWorkflowInput(input, [saved])).toThrow(
				/outside the fanout allowlist/,
			);
			expect(() =>
				resolveWorkflowInput(input, [saved], undefined, ["read", "web"]),
			).not.toThrow();
		}
	});

	it("throws with the available names for an unknown name", () => {
		expect(() => resolveWorkflowInput({ name: "missing" }, list)).toThrow(
			/ultra-review|audit/,
		);
	});

	it("throws when neither name nor spec is provided", () => {
		expect(() => resolveWorkflowInput({}, list)).toThrow(
			/name.*spec|spec.*name/i,
		);
	});

	it("propagates a parse error for a malformed inline spec", () => {
		expect(() =>
			resolveWorkflowInput({ spec: { phases: [] } }, list),
		).toThrow();
	});

	it("validates a named workflow's spec, surfacing parseWorkflow's error on a malformed saved file", () => {
		// Discovery stores the raw JSON.parse output cast to WorkflowSpec — a saved
		// file can be JSON-valid with a `name` yet structurally malformed (no
		// `phases`). Resolving by name must run parseWorkflow so `/ultra run <name>`
		// fails fast with the TypeBox path error, not a cryptic mid-run crash.
		const malformed: DiscoveredWorkflow = {
			name: "broken",
			source: "project",
			path: "/project/broken.json",
			spec: { name: "broken" } as unknown as WorkflowSpec,
		};
		expect(() => resolveWorkflowInput({ name: "broken" }, [malformed])).toThrow(
			/Invalid workflow spec/,
		);
	});
});

// ---------------------------------------------------------------------------
// command parsing
// ---------------------------------------------------------------------------

describe("parseUltraCommand", () => {
	it("routes every feature through an explicit subcommand", () => {
		expect(parseUltraCommand(" ")).toEqual({ subcommand: "activate" });
		expect(parseUltraCommand("exec fix the race")).toEqual({
			subcommand: "exec",
			instructions: "fix the race",
		});
		expect(parseUltraCommand("run review main")).toEqual({
			subcommand: "run",
			invocation: "review main",
		});
	});

	it("rejects missing arguments and unknown commands", () => {
		for (const input of [
			"exec",
			"run",
			"review main",
			"runs",
			"history",
			"history extra",
		]) {
			expect(parseUltraCommand(input)).toEqual({ subcommand: "invalid" });
		}
	});
});

describe("parseCommandLine", () => {
	it("returns {} for an empty line", () => {
		expect(parseCommandLine("   ")).toEqual({});
	});

	it("extracts a bare name", () => {
		expect(parseCommandLine("ultra-review")).toEqual({ name: "ultra-review" });
	});

	it("parses a trailing JSON object as args (retaining the raw rest)", () => {
		expect(parseCommandLine('ultra-review {"base":"main","n":2}')).toEqual({
			name: "ultra-review",
			args: { base: "main", n: 2 },
			rest: '{"base":"main","n":2}',
		});
	});

	it("keeps non-JSON trailing text as `rest` (for positional mapping)", () => {
		expect(parseCommandLine("ultra-review main")).toEqual({
			name: "ultra-review",
			rest: "main",
		});
		expect(parseCommandLine("ultra-review [1,2]")).toEqual({
			name: "ultra-review",
			rest: "[1,2]",
		});
	});
});

// ---------------------------------------------------------------------------
// mapPositionalArgs — bare `/ultra run <name> <tok...>` → declared args.params
// ---------------------------------------------------------------------------

describe("mapPositionalArgs", () => {
	it("fills declared params in order from whitespace tokens", () => {
		expect(mapPositionalArgs("main", ["base"])).toEqual({ base: "main" });
		expect(mapPositionalArgs("a   b", ["x", "y"])).toEqual({ x: "a", y: "b" });
	});

	it("soaks remaining tokens into the last declared param (free-text tail)", () => {
		// single free-text param keeps the whole remainder
		expect(mapPositionalArgs("add a dark-mode toggle", ["task"])).toEqual({
			task: "add a dark-mode toggle",
		});
		// only the LAST param soaks; earlier params take one token each
		expect(mapPositionalArgs("main fix the bug", ["base", "task"])).toEqual({
			base: "main",
			task: "fix the bug",
		});
	});

	it("returns undefined for an empty remainder or no declared params", () => {
		expect(mapPositionalArgs("   ", ["base"])).toBeUndefined();
		expect(mapPositionalArgs("main", [])).toBeUndefined();
	});
});