repositories / pi-ext
pi-ext
bugabingas pi extensions
owned by admin
extensions/decay/__tests__/integration.test.ts
Rawimport {
fauxAssistantMessage,
fauxProvider,
fauxToolCall,
getCurrentTools,
type TranscriptContext,
} from "@earendil-works/pi-ai";
import { SettingsManager } from "@earendil-works/pi-coding-agent";
import { afterEach, describe, expect, it, vi } from "vitest";
import { createTestSession, type TestSession } from "../../../test/harness";
import decayExtension from "../index";
function preparation() {
return {
firstKeptEntryId: "kept-entry",
messagesToSummarize: [
{
role: "user",
content: [{ type: "text", text: "Keep the exact path src/main.ts" }],
timestamp: 1,
},
],
turnPrefixMessages: [],
isSplitTurn: false,
tokensBefore: 42_000,
previousSummary: undefined,
fileOps: {
read: new Set(["README.md"]),
written: new Set<string>(),
edited: new Set(["src/main.ts"]),
},
settings: {
enabled: true,
reserveTokens: 16_384,
keepRecentTokens: 20_000,
},
};
}
function compactionEvent(
signal = new AbortController().signal,
overrides: Record<string, unknown> = {},
) {
return {
type: "session_before_compact" as const,
preparation: preparation(),
branchEntries: [
{
type: "message" as const,
id: "kept-entry",
parentId: null,
timestamp: new Date(1).toISOString(),
message: {
role: "user" as const,
content: "k".repeat(96_000),
timestamp: 1,
},
},
],
customInstructions: "focus on exact paths",
reason: "manual" as const,
willRetry: false,
signal,
...overrides,
};
}
function classifierMessage() {
return fauxAssistantMessage(
fauxToolCall("record_chunk", {
chunk: 1,
of: 1,
atoms: [
{
key: "path:src/main.ts",
text: "Continue work in src/main.ts",
kind: "exact",
status: "active",
},
],
}),
{ stopReason: "toolUse" },
);
}
describe("Decay real extension hook", () => {
let testSession: TestSession | undefined;
afterEach(() => {
testSession?.dispose();
testSession = undefined;
vi.restoreAllMocks();
});
async function setup() {
testSession = await createTestSession({
extensionFactories: [decayExtension],
});
const faux = fauxProvider({
provider: "openai",
models: [
{ id: "gpt-4o", maxTokens: 16_384 },
{ id: "classifier", maxTokens: 16_384 },
{ id: "summarizer", maxTokens: 8_192 },
],
});
testSession.session.extensionRunner
.createContext()
.modelRegistry.registerProvider(faux.provider);
return { testSession, faux };
}
it("classifies, merges, summarizes, persists details, and preserves Pi's tail", async () => {
const { testSession, faux } = await setup();
expect(testSession.session.extensionRunner.createContext().mode).not.toBe(
"tui",
);
const streamSimple = vi.spyOn(faux.provider, "streamSimple");
const stream = vi.spyOn(faux.provider, "stream");
const contexts: TranscriptContext[] = [];
faux.setResponses([
(context) => {
contexts.push(context);
return classifierMessage();
},
(context) => {
contexts.push(context);
return fauxAssistantMessage("## Goal\n\nContinue implementation.");
},
]);
const result = await testSession.session.extensionRunner.emit(
compactionEvent(),
);
expect(result.compaction).toEqual(
expect.objectContaining({
firstKeptEntryId: "kept-entry",
tokensBefore: 42_000,
estimatedTokensAfter: expect.any(Number),
summary: expect.stringContaining("Continue implementation"),
}),
);
expect(result.compaction.summary).toContain(
"<read-files>\nREADME.md\n</read-files>",
);
expect(result.compaction.summary).toContain(
"<modified-files>\nsrc/main.ts\n</modified-files>",
);
expect(result.compaction.details.decay).toEqual(
expect.objectContaining({
version: 1,
chunkCount: 1,
keepRecentTokens: 20_000,
estimatedKeptTokens: 24_000,
atoms: [expect.objectContaining({ key: "path:src/main.ts" })],
}),
);
expect(result.compaction.usage.totalTokens).toBeGreaterThan(0);
expect(getCurrentTools(contexts[0].messages)).toHaveLength(1);
expect(JSON.stringify(contexts[0])).toContain("focus on exact paths");
expect(JSON.stringify(contexts[1])).toContain("allocation");
expect(streamSimple).toHaveBeenCalledTimes(2);
expect(stream).not.toHaveBeenCalled();
await testSession.session.extensionRunner.emit({
type: "session_compact",
compactionEntry: {
...result.compaction,
type: "compaction",
id: "compact",
parentId: "parent",
timestamp: new Date().toISOString(),
},
fromExtension: true,
reason: "manual",
willRetry: false,
});
expect(testSession.events.uiCallsFor("notify").at(-1)?.args.at(0)).toMatch(
/Decay · gpt-4o · 1\/1 classified · ~24,000 kept \(target 20,000\)/,
);
});
it("preserves real prepared messages through the Decay hook and context rebuild", async () => {
const { testSession, faux } = await setup();
const manager = testSession.session.sessionManager;
const old = { role: "user", content: "old ".repeat(12_000), timestamp: 1 };
const kept = [
{ role: "user", content: "keep-a ".repeat(7_000), timestamp: 2 },
{ role: "user", content: "keep-b ".repeat(7_000), timestamp: 3 },
];
manager.appendMessage(old);
for (const message of kept) manager.appendMessage(message);
let classifierContext: TranscriptContext | undefined;
faux.setResponses([
(context) => {
classifierContext = context;
return classifierMessage();
},
fauxAssistantMessage("New summary."),
]);
await testSession.session.compact();
const messages = manager.buildSessionContext().messages;
expect(messages.slice(1)).toEqual(kept);
expect(JSON.stringify(classifierContext)).not.toContain("keep-a");
expect(JSON.stringify(classifierContext)).not.toContain("keep-b");
const checkpoint = manager.getBranch().at(-1);
expect(checkpoint.details.decay.estimatedKeptTokens).toBe(24_500);
expect(checkpoint.details.decay.keepRecentTokens).toBe(20_000);
});
it("uses a separately configured summarizer model", async () => {
const { testSession, faux } = await setup();
vi.spyOn(SettingsManager, "create").mockReturnValue(
SettingsManager.inMemory({
decay: {
model: "openai/classifier",
summarizerModel: "openai/summarizer",
},
}),
);
const streamSimple = vi.spyOn(faux.provider, "streamSimple");
faux.setResponses([
classifierMessage(),
fauxAssistantMessage("## Goal\n\nSeparate summarizer."),
]);
const result = await testSession.session.extensionRunner.emit(
compactionEvent(),
);
expect(result.compaction.summary).toContain("Separate summarizer");
expect(streamSimple.mock.calls.map(([model]) => model.id)).toEqual([
"classifier",
"summarizer",
]);
expect(streamSimple.mock.calls[1]?.[2]?.maxTokens).toBe(8_192);
});
it("disables reasoning only for Luna classification", async () => {
testSession = await createTestSession({
extensionFactories: [decayExtension],
});
const faux = fauxProvider({
provider: "openai-codex",
models: [{ id: "gpt-6-luna", reasoning: true, maxTokens: 16_384 }],
});
const registry =
testSession.session.extensionRunner.createContext().modelRegistry;
registry.registerProvider(faux.provider);
await registry.refresh({ allowNetwork: false });
vi.spyOn(SettingsManager, "create").mockReturnValue(
SettingsManager.inMemory({
decay: { model: "openai-codex/gpt-6-luna" },
}),
);
const streamSimple = vi.spyOn(faux.provider, "streamSimple");
expect(
registry
.getAvailable()
.some(
(model) =>
model.provider === "openai-codex" && model.id === "gpt-6-luna",
),
).toBe(true);
faux.setResponses([
classifierMessage(),
fauxAssistantMessage("## Goal\n\nContinue implementation."),
]);
const result = await testSession.session.extensionRunner.emit(
compactionEvent(),
);
expect(streamSimple.mock.calls.map(([model]) => model.id)).toEqual([
"gpt-6-luna",
"gpt-6-luna",
]);
expect(result.compaction.details.decay.model).toBe(
"openai-codex/gpt-6-luna",
);
expect(
streamSimple.mock.calls.map(([, , options]) => options?.reasoning),
).toEqual([undefined, "low"]);
});
it("falls back to current for a classifier error", async () => {
const { testSession, faux } = await setup();
vi.spyOn(SettingsManager, "create").mockReturnValue(
SettingsManager.inMemory({ decay: { model: "openai/classifier" } }),
);
const streamSimple = vi.spyOn(faux.provider, "streamSimple");
faux.setResponses([
fauxAssistantMessage([], {
stopReason: "error",
errorMessage: "429 quota exceeded",
}),
classifierMessage(),
fauxAssistantMessage("## Goal\n\nCurrent model recovered."),
]);
const result = await testSession.session.extensionRunner.emit(
compactionEvent(),
);
expect(result.compaction.summary).toContain("Current model recovered");
expect(streamSimple.mock.calls.map(([model]) => model.id)).toEqual([
"classifier",
"gpt-4o",
"gpt-4o",
]);
expect(result.compaction.details.decay.model).toBe("openai/gpt-4o");
});
it("falls back to current for a summarizer quota error", async () => {
const { testSession, faux } = await setup();
vi.spyOn(SettingsManager, "create").mockReturnValue(
SettingsManager.inMemory({
decay: {
model: "openai/classifier",
summarizerModel: "openai/summarizer",
},
}),
);
const streamSimple = vi.spyOn(faux.provider, "streamSimple");
faux.setResponses([
classifierMessage(),
fauxAssistantMessage([], {
stopReason: "error",
errorMessage: "quota exceeded",
}),
fauxAssistantMessage("## Goal\n\nCurrent summarizer recovered."),
]);
const result = await testSession.session.extensionRunner.emit(
compactionEvent(),
);
expect(result.compaction.summary).toContain("Current summarizer recovered");
expect(streamSimple.mock.calls.map(([model]) => model.id)).toEqual([
"classifier",
"summarizer",
"gpt-4o",
]);
});
it("runs a passive fixed-height widget without taking editor input", async () => {
const { testSession, faux } = await setup();
faux.setResponses([
classifierMessage(),
fauxAssistantMessage("## Goal\n\nWidget path."),
]);
const runner = testSession.session.extensionRunner;
const baseContext = runner.createContext();
const terminalWrite = vi.fn();
const custom = vi.fn();
const setEditorText = vi.fn();
let widgetOptions: unknown;
let widgetCleared = false;
let component:
| {
render(width: number): string[];
dispose?(): void;
state: {
candidates: unknown[];
selectedKeys: string[];
summaryTokens: number;
};
}
| undefined;
runner.createContext = () => ({
...baseContext,
mode: "tui",
hasUI: true,
ui: {
...baseContext.ui,
custom,
setEditorText,
getEditorText: () => "steering draft",
setWidget: (
_key: string,
content: Function | undefined,
options: unknown,
) => {
if (!content) {
widgetCleared = true;
return;
}
widgetOptions = options;
component = content(
{
terminal: { rows: 30, write: terminalWrite },
requestRender: vi.fn(),
},
{
fg: (_color: string, text: string) => text,
bg: (_color: string, text: string) => text,
bold: (text: string) => text,
inverse: (text: string) => text,
},
);
},
},
});
const result = await runner.emit(compactionEvent());
const rendered = component?.render(80) ?? [];
expect(result.compaction.summary).toContain("Widget path");
expect(widgetOptions).toEqual({ placement: "aboveEditor" });
expect(widgetCleared).toBe(true);
expect(rendered).toHaveLength(5);
expect(component?.state.candidates).toHaveLength(1);
expect(component?.state.selectedKeys).toHaveLength(1);
expect(component?.state.summaryTokens).toBeGreaterThan(0);
expect(rendered.join("\n")).toContain("DECAY");
expect(rendered.join("\n")).toContain("↵ queues a message");
expect(custom).not.toHaveBeenCalled();
expect(setEditorText).not.toHaveBeenCalled();
expect(terminalWrite).not.toHaveBeenCalled();
});
it("retries only the missing shard while retaining earlier classifications", async () => {
const { testSession, faux } = await setup();
const calls: number[] = [];
const messages = [
{ role: "user", content: "first" },
{ role: "assistant", content: [{ type: "text", text: "done" }] },
{ role: "user", content: "second" },
];
faux.setResponses([
fauxAssistantMessage(
fauxToolCall("record_chunk", {
chunk: 1,
of: 2,
atoms: [
{ key: "first", text: "First", kind: "state", status: "active" },
],
}),
{ stopReason: "toolUse" },
),
(context) => {
const user = context.messages.findLast(
(message) => message.role === "user",
);
if (!user || typeof user.content === "string")
throw new Error("missing input");
const text = user.content.find((item) => item.type === "text");
if (text?.type !== "text") throw new Error("missing input");
const payload = JSON.parse(text.text);
calls.push(payload.chunks[0].chunk);
expect(payload.priorCatalog).toEqual([
expect.objectContaining({ key: "first" }),
]);
return fauxAssistantMessage("omitted");
},
fauxAssistantMessage(
fauxToolCall("record_chunk", {
chunk: 2,
of: 2,
atoms: [
{ key: "second", text: "Second", kind: "state", status: "active" },
],
}),
{ stopReason: "toolUse" },
),
fauxAssistantMessage("## Goal\n\nBoth classified."),
]);
const prep = preparation();
prep.messagesToSummarize = messages as never;
const result = await testSession.session.extensionRunner.emit(
compactionEvent(new AbortController().signal, { preparation: prep }),
);
expect(calls).toEqual([2]);
expect(
result.compaction.details.decay.atoms.map(
(atom: { key: string }) => atom.key,
),
).toEqual(["first", "second"]);
});
it("retries an incomplete response instead of committing its partial record", async () => {
const { testSession, faux } = await setup();
const calls = vi.spyOn(faux.provider, "streamSimple");
faux.setResponses([
fauxAssistantMessage(
fauxToolCall("record_chunk", {
chunk: 1,
of: 1,
atoms: [
{
key: "partial",
text: "Incomplete",
kind: "state",
status: "active",
},
],
}),
{ stopReason: "length" },
),
classifierMessage(),
fauxAssistantMessage("## Goal\n\nComplete."),
]);
const result = await testSession.session.extensionRunner.emit(
compactionEvent(),
);
expect(calls).toHaveBeenCalledTimes(3);
expect(
result.compaction.details.decay.atoms.map(
(atom: { key: string }) => atom.key,
),
).toEqual(["path:src/main.ts"]);
});
it("rejects records for a different shard", async () => {
const { testSession, faux } = await setup();
faux.setResponses([
fauxAssistantMessage(
[
fauxToolCall("record_chunk", { chunk: 1, of: 2, atoms: [] }),
fauxToolCall("record_chunk", { chunk: 2, of: 2, atoms: [] }),
],
{ stopReason: "toolUse" },
),
]);
const prep = preparation();
prep.messagesToSummarize = [
{ role: "user", content: "first" },
{ role: "assistant", content: [{ type: "text", text: "done" }] },
{ role: "user", content: "second" },
] as never;
expect(
await testSession.session.extensionRunner.emit(
compactionEvent(new AbortController().signal, { preparation: prep }),
),
).toBeUndefined();
const diagnostic = testSession.session.sessionManager
.getBranch()
.find(
(entry) =>
entry.type === "custom" && entry.customType === "decay-diagnostic",
);
expect(diagnostic).toMatchObject({
data: {
stage: "classifier",
reason: "Decay classifier returned a record for another chunk",
},
});
});
it("returns nothing so Pi defaults when classification coverage is incomplete", async () => {
const { testSession, faux } = await setup();
faux.setResponses([fauxAssistantMessage("no tool records")]);
expect(
await testSession.session.extensionRunner.emit(compactionEvent()),
).toBeUndefined();
expect(
testSession.events.uiCallsFor("notify").at(-1)?.args.at(0),
).toContain("using default compaction");
});
it("persists an agent-invisible diagnostic before provider fallback", async () => {
const { testSession, faux } = await setup();
faux.setResponses([
fauxAssistantMessage([], {
stopReason: "error",
errorMessage: "controlled provider failure",
}),
]);
const messagesBefore = JSON.stringify(testSession.session.messages);
expect(
await testSession.session.extensionRunner.emit(compactionEvent()),
).toBeUndefined();
expect(
testSession.events.uiCallsFor("notify").at(-1)?.args.at(0),
).toContain("controlled provider failure");
const diagnostic = testSession.session.sessionManager
.getBranch()
.find(
(entry) =>
entry.type === "custom" && entry.customType === "decay-diagnostic",
);
expect(diagnostic).toMatchObject({
type: "custom",
customType: "decay-diagnostic",
data: {
version: 1,
outcome: "fallback",
stage: "classifier",
model: expect.any(String),
reason: "controlled provider failure",
timestamp: expect.any(Number),
},
});
expect(JSON.stringify(testSession.session.messages)).toBe(messagesBefore);
expect(JSON.stringify(diagnostic)).not.toContain("stack");
});
it("salvages a deterministic summary when the summarizer fails", async () => {
const { testSession, faux } = await setup();
faux.setResponses([
classifierMessage(),
fauxAssistantMessage(fauxToolCall("unexpected", {}), {
stopReason: "toolUse",
}),
]);
const result = await testSession.session.extensionRunner.emit(
compactionEvent(),
);
const compaction = (
result as
| { compaction: { summary: string; details: unknown } }
| undefined
)?.compaction;
expect(compaction?.summary).toContain("Decay summarizer unavailable");
expect(compaction?.summary).toContain("src/main.ts");
expect(compaction?.details).toMatchObject({
decay: { degraded: "summarizer" },
});
const diagnostic = testSession.session.sessionManager
.getBranch()
.find(
(entry) =>
entry.type === "custom" && entry.customType === "decay-diagnostic",
);
expect(diagnostic).toMatchObject({
data: { outcome: "salvage", stage: "summarizer" },
});
});
it("gap-fills chunks the classifier omitted after a transient failure", async () => {
const { testSession, faux } = await setup();
faux.setResponses([
fauxAssistantMessage([], {
stopReason: "error",
errorMessage: "429 rate limit exceeded",
}),
classifierMessage(),
fauxAssistantMessage("## Goal\n\nRecovered."),
]);
const result = await testSession.session.extensionRunner.emit(
compactionEvent(),
);
expect(
(result as { compaction: { summary: string } } | undefined)?.compaction
.summary,
).toContain("Recovered.");
});
it("reuses prior keys for paraphrased repeated compactions", async () => {
const { testSession, faux } = await setup();
faux.setResponses([
classifierMessage(),
fauxAssistantMessage("## Goal\n\nFirst."),
]);
const first = await testSession.session.extensionRunner.emit(
compactionEvent(),
);
let secondClassifierContext: TranscriptContext | undefined;
faux.setResponses([
(context) => {
secondClassifierContext = context;
return classifierMessage();
},
fauxAssistantMessage("## Goal\n\nSecond."),
]);
const second = await testSession.session.extensionRunner.emit(
compactionEvent(new AbortController().signal, {
preparation: {
...preparation(),
messagesToSummarize: [
{
role: "user",
content: [
{
type: "text",
text: "Resume implementation using the same src/main.ts location",
},
],
timestamp: 2,
},
],
previousSummary: first.compaction.summary,
fileOps: {
read: new Set(["second.md"]),
written: new Set<string>(),
edited: new Set<string>(),
},
},
branchEntries: [
{
type: "compaction",
id: "first",
parentId: null,
timestamp: new Date().toISOString(),
...first.compaction,
},
],
}),
);
expect(second.compaction.details.decay.atoms[0]).toEqual(
expect.objectContaining({ seenCount: 2, firstSeen: 1, lastSeen: 2 }),
);
expect(secondClassifierContext).toBeDefined();
const classifierUserMessage = secondClassifierContext?.messages.findLast(
(message) => message.role === "user",
);
if (
!classifierUserMessage ||
typeof classifierUserMessage.content === "string"
)
throw new Error("missing classifier payload");
const classifierText = classifierUserMessage.content.find(
(item) => item.type === "text",
);
if (classifierText?.type !== "text")
throw new Error("missing classifier payload");
const classifierPayload = JSON.parse(classifierText.text);
expect(classifierPayload.priorCatalog).toEqual([
expect.objectContaining({
key: "path:src/main.ts",
text: "Continue work in src/main.ts",
}),
]);
expect(classifierPayload.priorCatalog[0]).not.toHaveProperty("seenCount");
expect(classifierPayload.chunks[0].text).toContain(
"Resume implementation using the same src/main.ts location",
);
expect(second.compaction.summary).toContain("README.md");
expect(second.compaction.summary).toContain("second.md");
expect(second.compaction.summary).toContain("src/main.ts");
});
it.each([
["threshold", false],
["overflow", true],
] as const)(
"preserves %s compaction semantics while replacing only the summary",
async (reason, willRetry) => {
const { testSession, faux } = await setup();
faux.setResponses([
classifierMessage(),
fauxAssistantMessage("## Goal\n\nContinue."),
]);
const result = await testSession.session.extensionRunner.emit(
compactionEvent(new AbortController().signal, { reason, willRetry }),
);
expect(result.compaction.firstKeptEntryId).toBe("kept-entry");
expect(result.compaction.tokensBefore).toBe(42_000);
},
);
it("classifies discarded split-turn prefix separately while preserving the prepared suffix boundary", async () => {
const { testSession, faux } = await setup();
const contexts: TranscriptContext[] = [];
faux.setResponses([
(context) => {
contexts.push(context);
return fauxAssistantMessage(
fauxToolCall("record_chunk", {
chunk: 1,
of: 2,
atoms: [
{
key: "history",
text: "Prior history",
kind: "state",
status: "active",
},
],
}),
{ stopReason: "toolUse" },
);
},
(context) => {
contexts.push(context);
return fauxAssistantMessage(
fauxToolCall("record_chunk", {
chunk: 2,
of: 2,
atoms: [
{
key: "split",
text: "Discarded turn prefix",
kind: "state",
status: "active",
},
],
}),
{ stopReason: "toolUse" },
);
},
fauxAssistantMessage("## Goal\n\nContinue split turn."),
]);
const splitPreparation = preparation();
splitPreparation.isSplitTurn = true;
splitPreparation.turnPrefixMessages = [
{
role: "assistant",
content: [{ type: "text", text: "discarded assistant prefix" }],
api: "faux",
provider: "faux",
model: "faux",
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
total: 0,
},
},
stopReason: "stop",
timestamp: 2,
},
] as never;
const result = await testSession.session.extensionRunner.emit(
compactionEvent(new AbortController().signal, {
preparation: splitPreparation,
}),
);
expect(result.compaction.firstKeptEntryId).toBe("kept-entry");
expect(result.compaction.details.decay.chunkCount).toBe(2);
const payloads = contexts.map((context) => {
const user = context.messages.findLast(
(message) => message.role === "user",
);
if (!user || typeof user.content === "string")
throw new Error("missing classifier input");
const text = user.content.find((item) => item.type === "text");
if (text?.type !== "text") throw new Error("missing classifier input");
return JSON.parse(text.text);
});
expect(
payloads.map((payload) =>
payload.chunks.map((item: { chunk: number }) => item.chunk),
),
).toEqual([[1], [2]]);
expect(payloads[1].priorCatalog).toEqual([
expect.objectContaining({ key: "history" }),
]);
expect(
result.compaction.details.decay.atoms.map(
(atom: { key: string }) => atom.key,
),
).toEqual(["history", "split"]);
});
it("returns cancellation instead of starting default compaction after abort", async () => {
const { testSession, faux } = await setup();
faux.setResponses([classifierMessage()]);
const controller = new AbortController();
controller.abort();
expect(
await testSession.session.extensionRunner.emit(
compactionEvent(controller.signal),
),
).toEqual({ cancel: true });
expect(
testSession.session.sessionManager
.getBranch()
.some(
(entry) =>
entry.type === "custom" && entry.customType === "decay-diagnostic",
),
).toBe(false);
});
});