Luigit
repositories / pi-ext

pi-ext

bugabingas pi extensions

owned by admin

extensions/ultra/__tests__/research.test.ts

Raw
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import { fileURLToPath } from "node:url";
import { describe, expect, it } from "vitest";
import { runWorkflow } from "../engine.ts";
import { resolveSelector } from "../interp.ts";
import { compileSchema, discoverWorkflows, parseWorkflow } from "../spec.ts";

const here = path.dirname(fileURLToPath(import.meta.url));
const RESEARCH_PATH = path.join(here, "..", "workflows", "research.json");
function loadRaw(): string {
	return fs.readFileSync(RESEARCH_PATH, "utf8");
}

describe("research shipped workflow", () => {
	it("parses the shipped five-phase research pipeline", () => {
		const spec = parseWorkflow(loadRaw());
		expect(spec.name).toBe("research");
		expect(spec.args?.params).toEqual(["question"]);
		expect(spec.phases.map((phase) => phase.id)).toEqual([
			"plan",
			"gather",
			"audit",
			"verify",
			"report",
		]);
		expect(spec.return).toBe("{report.results}");
		expect(spec.report).toBe("{report.results}");
	});

	it("compiles every named output schema", () => {
		const spec = parseWorkflow(loadRaw());
		for (const schema of Object.values(spec.schemas ?? {}))
			expect(compileSchema(schema)).toBeDefined();
	});

	it("fans out breadth, audits centrally, then fans out targeted verification", () => {
		const spec = parseWorkflow(loadRaw());
		const phase = (id: string) =>
			spec.phases.find((candidate) => candidate.id === id);

		expect(phase("gather")?.over).toBe("{plan.results[].angles[]}");
		expect(phase("audit")?.step.prompt).toContain("{gather.results}");
		expect(phase("verify")?.when).toBe(
			"{audit.results[].verification_tasks[]}",
		);
		expect(phase("verify")?.over).toBe(
			"{audit.results[].verification_tasks[]}",
		);
		expect(phase("report")?.step.prompt).toContain("{verify.results}");

		const plan = spec.schemas?.ResearchPlan as {
			properties?: { angles?: { maxItems?: number } };
		};
		const audit = spec.schemas?.Audit as {
			properties?: { verification_tasks?: { maxItems?: number } };
		};
		expect(plan.properties?.angles?.maxItems).toBe(4);
		expect(audit.properties?.verification_tasks?.maxItems).toBe(3);
		expect(phase("gather")?.step.prompt).toContain(
			"Keep each source retrieval targeted",
		);
		expect(phase("verify")?.step.prompt).toContain(
			"at most 3 substantive sources",
		);
	});

	it("enables repository instruction reads without pinning extension-managed research tools", () => {
		const spec = parseWorkflow(loadRaw());
		for (const phase of spec.phases) {
			expect(phase.step.tools).toEqual(["read", "grep", "find", "ls"]);
			if (phase.kind === "fanout")
				expect(phase.step.prompt).toMatch(/do not modify/i);
		}
	});

	it("selects bounded scouting and reasoning-intensive tiers without pinning thinking", () => {
		const spec = parseWorkflow(loadRaw());
		expect(
			Object.fromEntries(
				spec.phases.map((phase) => [phase.id, phase.step.model]),
			),
		).toEqual({
			plan: "large",
			gather: "small",
			audit: "large",
			verify: "large",
			report: "large",
		});
		for (const phase of spec.phases)
			expect(phase.step.thinkingLevel).toBeUndefined();
	});

	it("requires traceable claims and a decision-ready final report", () => {
		const spec = parseWorkflow(loadRaw());
		const report = spec.schemas?.ResearchReport as {
			required?: string[];
			properties?: Record<string, unknown>;
		};
		for (const field of [
			"answer_status",
			"executive_summary",
			"analysis",
			"blockers",
			"findings",
			"contradictions",
			"gaps",
			"recommendations",
			"sources",
		])
			expect(report.required).toContain(field);
		expect(report.properties).toHaveProperty("findings");

		const value = {
			answer_status: "fully answered",
			executive_summary: "Answer",
			analysis: "Analysis",
			blockers: [],
			findings: [],
			contradictions: [],
			gaps: [],
			recommendations: [],
			sources: [],
		};
		for (const selector of [spec.return, spec.report])
			expect(
				resolveSelector(selector ?? "", { results: { report: [value] } }),
			).toEqual([value]);
	});

	it("preserves question plus item packets, instruction obligations and explicit gaps", () => {
		const spec = parseWorkflow(loadRaw());
		for (const id of ["gather", "verify"]) {
			const phase = spec.phases.find((candidate) => candidate.id === id);
			expect(phase?.step.prompt).toContain("{args.question}");
			expect(phase?.step.prompt).toContain("{item}");
		}
		for (const phase of spec.phases) {
			expect(phase.step.prompt).toContain("AGENTS.md");
			expect(phase.step.prompt).toContain("instructions");
			expect(phase.step.prompt).toContain("unsupported assumptions");
			expect(phase.step.prompt).toContain("overflow");
		}
		const report = spec.phases.find((phase) => phase.id === "report");
		expect(report?.when).toBeUndefined();
		for (const id of ["plan", "gather", "audit", "verify"]) {
			expect(report?.step.prompt).toContain(`{${id}.results}`);
			expect(report?.step.prompt).toContain(`{${id}.failures}`);
		}
		const validate = compileSchema(spec.schemas?.Verification ?? {});
		const verdict = {
			id: "v1",
			claim: "Claim",
			verdict: "unresolved",
			explanation: "Source unavailable",
			sources: [],
		};
		expect(validate.Check(verdict)).toBe(true);
		expect(validate.Check({ ...verdict, verdict: "probably" })).toBe(false);
		expect(validate.Check({ ...verdict, explanation: "x".repeat(4001) })).toBe(
			false,
		);
	});

	it.each(["plan", "gather", "audit", "verify"])(
		"passes %s failures into an explicit simulated unresolved report",
		async (failedAt) => {
			const spec = parseWorkflow(loadRaw());
			const question = "Which storage choice meets the constraints?";
			const reason = `${failedAt}: source access denied`;
			const instructions = [
				"Use primary evidence; AGENTS.md: read-only inspection",
			];
			const angle = {
				id: "a1",
				question: "What is supported?",
				source_strategy: "Primary docs",
				opposition_strategy: "Find limitations",
				instructions,
			};
			const task = {
				id: "v1",
				claim: "Claim to check",
				why: "Single source: https://example.com/spec",
				preferred_sources: "Primary docs",
				opposition_query: "Contrary evidence",
				instructions,
			};
			const report = {
				answer_status: "unresolved",
				executive_summary: "Insufficient evidence",
				analysis: reason,
				blockers: [reason],
				findings: [],
				contradictions: [],
				gaps: [reason],
				recommendations: ["Restore source access and verify"],
				sources: [],
			};
			const result = await runWorkflow(
				spec,
				{ question },
				{
					concurrency: 3,
					stepRunner: async ({ phase, step, item }) => {
						if (phase === failedAt)
							return {
								ok: false,
								value: null,
								failure: {
									code: "transport",
									message: reason,
									retryable: true,
									attempts: 1,
								},
							};
						let value: unknown;
						if (phase === "plan")
							value = {
								question,
								answer_shape: "Comparison",
								blockers: [],
								gaps: [],
								angles: [angle, { ...angle, id: "a2" }, { ...angle, id: "a3" }],
							};
						else if (phase === "gather") {
							expect(step.prompt).toContain(question);
							expect(step.prompt).toContain(instructions[0]);
							const packet = item as typeof angle;
							value = {
								angle_id: packet.id,
								angle: packet.question,
								findings: [],
								contradictions: [],
								gaps: ["No authoritative evidence"],
								leads: [],
							};
						} else if (phase === "audit") {
							if (failedAt === "plan" || failedAt === "gather")
								expect(step.prompt).toContain(reason);
							value = {
								assessment: "Evidence incomplete",
								verification_tasks: failedAt === "plan" ? [] : [task],
								unresolved_gaps: [reason],
							};
						} else if (phase === "verify") {
							expect(step.prompt).toContain(question);
							expect(step.prompt).toContain(instructions[0]);
							expect(step.prompt).toContain(task.why);
							value = {
								id: task.id,
								claim: task.claim,
								verdict: "unresolved",
								explanation: "Source inaccessible",
								sources: [],
							};
						} else {
							expect(phase).toBe("report");
							expect(step.prompt).toContain(reason);
							expect(step.prompt).toContain('"code":"transport"');
							value = report;
						}
						expect(
							compileSchema(spec.schemas?.[String(step.schema)] ?? {}).Check(
								value,
							),
						).toBe(true);
						return { ok: true, value };
					},
				},
			);
			expect(result.result).toEqual([report]);
			if (failedAt === "plan") expect(result.phaseResults.gather).toEqual([]);
			if (failedAt === "plan" || failedAt === "audit")
				expect(result.phaseResults.verify).toEqual([]);
		},
	);

	it("is discoverable as a bundled workflow", () => {
		const cwd = fs.mkdtempSync(path.join(os.tmpdir(), "ultra-cwd-"));
		const home = fs.mkdtempSync(path.join(os.tmpdir(), "ultra-home-"));
		try {
			const research = discoverWorkflows(cwd, { home }).find(
				(workflow) => workflow.name === "research",
			);
			expect(research?.source).toBe("bundled");
		} finally {
			fs.rmSync(cwd, { recursive: true, force: true });
			fs.rmSync(home, { recursive: true, force: true });
		}
	});
});