repositories / pi-ext
pi-ext
bugabingas pi extensions
owned by admin
extensions/gratis/__tests__/gratis.test.ts
Rawimport { mkdtempSync, rmSync, writeFileSync } from "node:fs";
import { tmpdir } from "node:os";
import path from "node:path";
import {
type Api,
type AssistantMessage,
isContextOverflow,
type Model,
} from "@earendil-works/pi-ai";
import { afterEach, describe, expect, it, vi } from "vitest";
import { createTestSession, type TestSession } from "../../../test/harness";
import { createGratis } from "../index.ts";
import {
anthropicText,
contextOverflow,
googleText,
liveModel,
openAIText,
type Recorded,
type Reply,
type Responder,
rateLimited,
startUpstream,
} from "./fake-upstream.ts";
type Registry = {
refresh(options: {
allowNetwork?: boolean;
providers?: string[];
}): Promise<unknown>;
getAll(): Model<Api>[];
getAvailable(): Model<Api>[];
find(provider: string, id: string): Model<Api> | undefined;
streamSimple(
model: Model<Api>,
context: unknown,
): { result(): Promise<AssistantMessage> };
getApiKeyForProvider(provider: string): Promise<string | undefined>;
};
const KILO_MODELS = "/kilo/api/gateway/models";
const KILO_CHAT = "/kilo/api/gateway/chat/completions";
const OPENROUTER_MODELS = "/openrouter/api/v1/models";
const OPENROUTER_CHAT = "/openrouter/api/v1/chat/completions";
const OPENROUTER_MESSAGES = "/openrouter/api/v1/messages";
const GOOGLE = "/google/v1beta/models/";
const openrouterCatalog = {
data: [
liveModel("qwen/qwen3.8-27b:free"),
liveModel("anthropic/claude-trial:free"),
liveModel("openrouter/free", {
context_length: 200_000,
top_provider: { context_length: null, max_completion_tokens: null },
}),
liveModel("qwen/qwen3.8-27b", {
pricing: { prompt: "0.0000002", completion: "0.0000008" },
}),
],
};
const kiloCatalog = {
data: [
liveModel("minimax/minimax-m3:free"),
liveModel("vendor/paid-model", {
pricing: { prompt: "0.000001", completion: "0.000002" },
}),
liveModel("vendor/fake-free:free", {
pricing: { prompt: "0.1", completion: "0" },
}),
liveModel("vendor/no-tools:free", { supported_parameters: ["max_tokens"] }),
liveModel("openrouter/free", {
context_length: 200_000,
top_provider: { context_length: null, max_completion_tokens: null },
}),
liveModel("kilo-auto/free", {
context_length: 256_000,
top_provider: { context_length: 256_000, max_completion_tokens: 32_768 },
}),
],
};
let session: TestSession | undefined;
let upstream: Awaited<ReturnType<typeof startUpstream>> | undefined;
let agentDir: string | undefined;
afterEach(async () => {
session?.dispose();
session = undefined;
await upstream?.close();
upstream = undefined;
if (agentDir) rmSync(agentDir, { recursive: true, force: true });
agentDir = undefined;
});
interface Setup {
respond: Responder;
env?: Record<string, string>;
auth?: Record<string, unknown>;
firstResponseDeadlineMs?: number;
routingStatePath?: string;
}
async function setup({
respond,
env = {},
auth,
firstResponseDeadlineMs,
routingStatePath,
}: Setup) {
upstream = await startUpstream(respond);
agentDir = mkdtempSync(path.join(tmpdir(), "gratis-agent-"));
if (auth)
writeFileSync(path.join(agentDir, "auth.json"), JSON.stringify(auth));
session = await createTestSession({
extensionFactories: [
createGratis({
resolveUrl: upstream.resolveUrl,
...(firstResponseDeadlineMs === undefined
? {}
: { firstResponseDeadlineMs }),
...(routingStatePath === undefined ? {} : { routingStatePath }),
}),
],
env: { PI_CODING_AGENT_DIR: agentDir, ...env },
});
const context = session.session.extensionRunner.createContext();
const registry = context.modelRegistry as Registry;
await registry.refresh({ allowNetwork: true, providers: ["grts"] });
return { registry, context, upstream };
}
const grtsIds = (registry: Registry) =>
registry
.getAll()
.filter((model) => model.provider === "grts")
.map((model) => model.id);
const ask = (registry: Registry, id: string) => {
const model = registry.find("grts", id);
if (!model) throw new Error(`missing grts/${id}`);
return registry
.streamSimple(model, {
messages: [{ role: "user", content: "hi", timestamp: Date.now() }],
})
.result();
};
const kiloOnly: Responder = (request) => {
if (request.path === KILO_MODELS) return { status: 200, json: kiloCatalog };
if (request.path === KILO_CHAT) return openAIText("kilo says hi");
return undefined;
};
const chats = (requests: Recorded[], prefix: string) =>
requests.filter(
(request) => request.method === "POST" && request.path.startsWith(prefix),
);
describe("gratis keyless kilo", () => {
it("registers only free kilo models plus auto when no keys are set", async () => {
const { registry } = await setup({ respond: kiloOnly });
expect(grtsIds(registry)).toEqual([
"auto",
"kilo/kilo-auto/free",
"kilo/openrouter/free",
"kilo/minimax/minimax-m3:free",
]);
expect(
registry.getAvailable().filter((model) => model.provider === "grts"),
).toHaveLength(4);
});
it("labels every model free and data-caveated", async () => {
const { registry } = await setup({ respond: kiloOnly });
for (const model of registry
.getAll()
.filter((entry) => entry.provider === "grts")) {
expect(model.cost).toEqual({
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
});
expect(model.name).toContain("prompts may be logged");
}
});
it("advertises published virtual-model parameters and falls back only when absent", async () => {
const { registry } = await setup({ respond: kiloOnly });
expect(registry.find("grts", "kilo/kilo-auto/free")).toMatchObject({
contextWindow: 256_000,
maxTokens: 32_768,
});
expect(registry.find("grts", "kilo/openrouter/free")).toMatchObject({
contextWindow: 200_000,
maxTokens: 8_192,
});
});
it("stays invisible when the kilo catalog is unreachable and no key is set", async () => {
const { registry } = await setup({
respond: () => ({ status: 503, json: { error: "down" } }),
});
expect(grtsIds(registry)).toEqual([]);
});
it("yields no kilo models and no error for an empty free pool", async () => {
const { registry } = await setup({
respond: (request) =>
request.path === KILO_MODELS
? { status: 200, json: { data: [liveModel("vendor/paid")] } }
: undefined,
});
expect(grtsIds(registry)).toEqual([]);
});
it("streams a pinned kilo model anonymously under the grts identity", async () => {
const { registry, upstream } = await setup({ respond: kiloOnly });
const message = await ask(registry, "kilo/kilo-auto/free");
expect(message.stopReason).toBe("stop");
expect(message.content).toEqual([{ type: "text", text: "kilo says hi" }]);
expect(message.provider).toBe("grts");
expect(message.model).toBe("kilo/kilo-auto/free");
const [chat] = chats(upstream.requests, KILO_CHAT);
expect(chat?.body?.model).toBe("kilo-auto/free");
expect(chat?.headers.authorization).toBeUndefined();
});
it("never forces reasoning off on router models whose upstream may require it", async () => {
const { registry, upstream } = await setup({
respond: (request) =>
request.path === KILO_MODELS
? {
status: 200,
json: {
data: [
liveModel("openrouter/free", {
supported_parameters: ["tools", "reasoning"],
}),
],
},
}
: kiloOnly(request),
});
await ask(registry, "kilo/openrouter/free");
const body = chats(upstream.requests, KILO_CHAT)[0]?.body ?? {};
expect(body).not.toHaveProperty("reasoning");
});
it("sends an optional kilo key when one is set", async () => {
const { registry, upstream } = await setup({
respond: kiloOnly,
env: { GRATIS_KILO_TOKEN: "kg-value" },
});
await ask(registry, "kilo/kilo-auto/free");
expect(chats(upstream.requests, KILO_CHAT)[0]?.headers.authorization).toBe(
"Bearer kg-value",
);
});
it("reads the native kilo key when no gratis kilo key is set", async () => {
const { registry, upstream } = await setup({
respond: kiloOnly,
env: { KILO_API_KEY: "kn-value" },
});
await ask(registry, "kilo/kilo-auto/free");
expect(chats(upstream.requests, KILO_CHAT)[0]?.headers.authorization).toBe(
"Bearer kn-value",
);
});
});
const everyBackend: Responder = (request) => {
if (request.path === KILO_MODELS) return { status: 200, json: kiloCatalog };
if (request.path === OPENROUTER_MODELS)
return { status: 200, json: openrouterCatalog };
if (request.path === KILO_CHAT) return openAIText("kilo says hi");
if (request.path === OPENROUTER_CHAT) return openAIText("openrouter says hi");
if (request.path.startsWith(OPENROUTER_MESSAGES))
return anthropicText("anthropic shape says hi");
if (request.path.startsWith(GOOGLE)) return googleText("gemini says hi");
return undefined;
};
const backendsOf = (registry: Registry) => [
...new Set(
grtsIds(registry)
.filter((id) => id !== "auto")
.map((id) => id.slice(0, id.indexOf("/"))),
),
];
describe("gratis keyed backends", () => {
it("shows only the keyed backend next to kilo for one native key", async () => {
const { registry } = await setup({
respond: everyBackend,
env: { GROQ_API_KEY: "gq-value" },
});
expect(backendsOf(registry)).toEqual(["groq", "kilo"]);
expect(grtsIds(registry)).toContain("groq/openai/gpt-oss-120b");
});
it("registers every keyed backend with its curated catalog", async () => {
const { registry } = await setup({
respond: everyBackend,
env: {
GEMINI_API_KEY: "gm-value",
OPENROUTER_API_KEY: "on-value",
GROQ_API_KEY: "gq-value",
MISTRAL_API_KEY: "ms-value",
},
});
expect(backendsOf(registry)).toEqual([
"google",
"groq",
"mistral",
"openrouter",
"kilo",
]);
expect(registry.find("grts", "mistral/devstral-latest")?.api).toBe(
"mistral-conversations",
);
expect(registry.find("grts", "google/gemini-3.8-flash")?.api).toBe(
"google-generative-ai",
);
});
it("lets GRATIS_OPENROUTER_TOKEN override the native OpenRouter key", async () => {
const { registry, upstream } = await setup({
respond: everyBackend,
env: {
GRATIS_OPENROUTER_TOKEN: "og-value",
OPENROUTER_API_KEY: "on-value",
},
});
const message = await ask(registry, "openrouter/qwen/qwen3.8-27b:free");
expect(message.content).toEqual([
{ type: "text", text: "openrouter says hi" },
]);
expect(
chats(upstream.requests, OPENROUTER_CHAT)[0]?.headers.authorization,
).toBe("Bearer og-value");
});
it("keeps only free OpenRouter models, virtual router first", async () => {
const { registry } = await setup({
respond: everyBackend,
env: { OPENROUTER_API_KEY: "on-value" },
});
expect(
grtsIds(registry).filter((id) => id.startsWith("openrouter/")),
).toEqual([
"openrouter/openrouter/free",
"openrouter/qwen/qwen3.8-27b:free",
"openrouter/anthropic/claude-trial:free",
]);
});
it("yields no OpenRouter models and no error for zero free variants", async () => {
const { registry } = await setup({
respond: (request) =>
request.path === OPENROUTER_MODELS
? { status: 200, json: { data: [liveModel("vendor/paid")] } }
: everyBackend(request),
env: { OPENROUTER_API_KEY: "on-value" },
});
expect(backendsOf(registry)).toEqual(["kilo"]);
});
it("falls back to the static OpenRouter router when discovery fails", async () => {
const { registry } = await setup({
respond: (request) =>
request.path === OPENROUTER_MODELS
? { status: 500, json: { error: "down" } }
: everyBackend(request),
env: { OPENROUTER_API_KEY: "on-value" },
});
expect(
grtsIds(registry).filter((id) => id.startsWith("openrouter/")),
).toEqual(["openrouter/openrouter/free"]);
});
it("registers a backend from a key in Pi's credential store", async () => {
const { registry, upstream } = await setup({
respond: everyBackend,
auth: {
openrouter: { type: "api_key", key: "os-value" },
},
});
// Pi emits session_start while creating the session; lookup runs in the background.
await expect
.poll(() => backendsOf(registry))
.toEqual(["openrouter", "kilo"]);
const message = await ask(registry, "openrouter/openrouter/free");
expect(message.stopReason).toBe("stop");
expect(
chats(upstream.requests, OPENROUTER_CHAT)[0]?.headers.authorization,
).toBe("Bearer os-value");
});
it("streams google-shaped entries through the Google API", async () => {
const { registry, upstream } = await setup({
respond: everyBackend,
env: { GEMINI_API_KEY: "gm-value" },
});
const message = await ask(registry, "google/gemini-3.8-flash");
expect(message.stopReason).toBe("stop");
expect(message.content).toEqual([{ type: "text", text: "gemini says hi" }]);
const [request] = chats(upstream.requests, GOOGLE);
expect(request?.path).toContain("gemini-3.8-flash:streamGenerateContent");
expect(request?.headers["x-goog-api-key"]).toBe("gm-value");
});
it("never asks newer Gemini Flash models for the unsupported MINIMAL thinking level", async () => {
const { registry, upstream } = await setup({
respond: everyBackend,
env: { GEMINI_API_KEY: "gm-value" },
});
await ask(registry, "google/gemini-3.8-flash");
const [request] = chats(upstream.requests, GOOGLE);
const config = JSON.stringify(request?.body?.generationConfig ?? {});
expect(config).not.toMatch(/MINIMAL/i);
});
it("streams anthropic-shaped entries through the Anthropic API", async () => {
const { registry, upstream } = await setup({
respond: everyBackend,
env: { OPENROUTER_API_KEY: "on-value" },
});
const id = "openrouter/anthropic/claude-trial:free";
expect(registry.find("grts", id)?.api).toBe("anthropic-messages");
const message = await ask(registry, id);
expect(message.stopReason).toBe("stop");
expect(message.content).toEqual([
{ type: "text", text: "anthropic shape says hi" },
]);
expect(chats(upstream.requests, OPENROUTER_MESSAGES)[0]?.body?.model).toBe(
"anthropic/claude-trial:free",
);
});
});
const GROQ_CHAT = "/groq/openai/v1/chat/completions";
const GROQ_CANDIDATES = [
"groq:openai/gpt-oss-120b",
"groq:llama-3.3-70b-versatile",
"groq:openai/gpt-oss-20b",
"groq:llama-3.1-8b-instant",
];
/** Kilo and Groq fakes whose chat replies depend on the upstream model. */
function cascade(
kilo: (model: string) => Reply,
groq: (model: string) => Reply = () => openAIText("groq says hi"),
): Responder {
return (request) => {
if (request.path === KILO_MODELS) return { status: 200, json: kiloCatalog };
const model = String(request.body?.model);
if (request.path === KILO_CHAT) return kilo(model);
if (request.path === GROQ_CHAT) return groq(model);
return undefined;
};
}
const chatModels = (requests: Recorded[]) =>
requests
.filter((request) => request.method === "POST")
.map((request) => `${request.path.split("/")[1]}:${request.body?.model}`);
describe("gratis auto router", () => {
it("advertises the tightest candidate context and output limits", async () => {
const { registry } = await setup({
respond: cascade(() => openAIText("ok")),
env: { GROQ_API_KEY: "gq-value" },
});
expect(registry.find("grts", "auto")).toMatchObject({
contextWindow: 131_072,
maxTokens: 8_192,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
});
});
it("tries the virtual kilo routers first", async () => {
const { registry, upstream } = await setup({
respond: cascade(() => openAIText("kilo says hi")),
});
const message = await ask(registry, "auto");
expect(message.content).toEqual([{ type: "text", text: "kilo says hi" }]);
expect(message.model).toBe("auto");
expect(chatModels(upstream.requests)).toEqual(["kilo:kilo-auto/free"]);
});
it("completes on the next kilo candidate after upstream saturation, then skips the saturated one", async () => {
const { registry, upstream } = await setup({
respond: cascade((model) =>
model === "kilo-auto/free" ? rateLimited : openAIText("router says hi"),
),
});
const first = await ask(registry, "auto");
const second = await ask(registry, "auto");
expect(first.content).toEqual([{ type: "text", text: "router says hi" }]);
expect(second.stopReason).toBe("stop");
// kilo-auto/free cools on its own: the second request costs no round trip to it.
expect(chatModels(upstream.requests)).toEqual([
"kilo:kilo-auto/free",
"kilo:openrouter/free",
"kilo:openrouter/free",
]);
});
it("moves to the next backend when groq is exhausted and then skips the cooling backend", async () => {
const { registry, upstream } = await setup({
respond: cascade(
() => openAIText("kilo says hi"),
() => rateLimited,
),
env: { GROQ_API_KEY: "gq-value" },
});
const first = await ask(registry, "auto");
const second = await ask(registry, "auto");
expect(first.stopReason).toBe("stop");
expect(first.content).toEqual([{ type: "text", text: "kilo says hi" }]);
expect(second.content).toEqual([{ type: "text", text: "kilo says hi" }]);
expect(chatModels(upstream.requests)).toEqual([
...GROQ_CANDIDATES,
"kilo:kilo-auto/free",
"kilo:kilo-auto/free",
]);
});
it("advances on context overflow without cooling while a next backend remains", async () => {
const { registry, upstream } = await setup({
respond: cascade(
() => openAIText("kilo says hi"),
() => contextOverflow,
),
env: { GROQ_API_KEY: "gq-value" },
});
const first = await ask(registry, "auto");
await ask(registry, "auto");
expect(first.stopReason).toBe("stop");
expect(first.content).toEqual([{ type: "text", text: "kilo says hi" }]);
// Groq was not cooled: the second request tries it again.
expect(
chatModels(upstream.requests).filter((hop) => hop.startsWith("groq:")),
).toHaveLength(8);
});
it("treats already-normalized overflow text as overflow", async () => {
const { registry, upstream } = await setup({
respond: cascade(
() => openAIText("kilo says hi"),
() => ({
status: 400,
json: { error: { message: "context_length_exceeded: upstream" } },
}),
),
env: { GROQ_API_KEY: "gq-value" },
});
const first = await ask(registry, "auto");
await ask(registry, "auto");
expect(first.content).toEqual([{ type: "text", text: "kilo says hi" }]);
expect(
chatModels(upstream.requests).filter((hop) => hop.startsWith("groq:")),
).toHaveLength(8);
});
it("reports overflow Pi can compact when every backend overflows", async () => {
const { registry } = await setup({
respond: cascade(
() => contextOverflow,
() => contextOverflow,
),
env: { GROQ_API_KEY: "gq-value" },
});
const message = await ask(registry, "auto");
expect(message.stopReason).toBe("error");
expect(message.errorMessage).toMatch(/^context_length_exceeded: /);
});
it("fails only when every backend is cooling and lists them", async () => {
const { registry } = await setup({
respond: cascade(
() => rateLimited,
() => ({
status: 429,
json: { error: { message: "You exceeded your current quota" } },
}),
),
env: { GROQ_API_KEY: "gq-value" },
});
const first = await ask(registry, "auto");
const second = await ask(registry, "auto");
expect(first.stopReason).toBe("error");
expect(first.errorMessage).toContain("kilo exhausted (rate)");
expect(first.errorMessage).toContain("groq exhausted (quota)");
expect(second.errorMessage).toContain("kilo cooling");
expect(second.errorMessage).toContain("groq cooling");
});
it("makes a cooled backend available again after its rate window", async () => {
vi.useFakeTimers({ toFake: ["Date"] });
try {
let kiloDown = false;
const { registry, upstream } = await setup({
respond: cascade(
() => (kiloDown ? rateLimited : openAIText("kilo says hi")),
() => rateLimited,
),
env: { GROQ_API_KEY: "gq-value" },
});
await ask(registry, "auto");
const groqHops = () =>
chatModels(upstream.requests).filter((hop) => hop.startsWith("groq:"))
.length;
expect(groqHops()).toBe(4);
// Groq is now demoted behind Kilo, but no longer cooling: it is tried when Kilo fails.
kiloDown = true;
vi.setSystemTime(Date.now() + 60_000 + 1);
await ask(registry, "auto");
expect(groqHops()).toBe(8);
} finally {
vi.useRealTimers();
}
});
it("replays same-model history with its native reasoning state", async () => {
const { registry, upstream } = await setup({ respond: kiloOnly });
const id = "kilo/kilo-auto/free";
const model = registry.find("grts", id);
if (!model) throw new Error("missing model");
const previous = await ask(registry, id);
previous.content = [
{
type: "thinking",
thinking: "plan",
thinkingSignature: JSON.stringify([
{ type: "reasoning.text", text: "plan", index: 0 },
]),
},
{ type: "text", text: "kilo says hi" },
];
await registry
.streamSimple(model, {
messages: [
{ role: "user", content: "hi", timestamp: 1 },
previous,
{ role: "user", content: "again", timestamp: 2 },
],
})
.result();
const replay = chats(upstream.requests, KILO_CHAT).at(-1)?.body
?.messages as Record<string, unknown>[];
expect(replay[1]?.reasoning_details).toEqual([
{ type: "reasoning.text", text: "plan", index: 0 },
]);
});
it("moves on when a backend reports upstream high demand", async () => {
const { registry } = await setup({
respond: cascade(() => ({
status: 503,
json: {
error: {
code: 503,
message: "This model is currently experiencing high demand.",
status: "UNAVAILABLE",
},
},
})),
env: { GROQ_API_KEY: "gq-value" },
});
const message = await ask(registry, "auto");
expect(message.content).toEqual([{ type: "text", text: "groq says hi" }]);
});
it("surfaces other upstream errors instead of cascading", async () => {
const { registry, upstream } = await setup({
respond: cascade(
() => openAIText("kilo says hi"),
() => ({
status: 401,
json: { error: { message: "invalid credentials" } },
}),
),
env: { GROQ_API_KEY: "gq-value" },
});
const message = await ask(registry, "auto");
expect(message.stopReason).toBe("error");
expect(message.errorMessage).toContain("invalid credentials");
expect(chatModels(upstream.requests)).toEqual(["groq:openai/gpt-oss-120b"]);
});
it("moves past a hop that streams nothing before its deadline", async () => {
const { registry, upstream } = await setup({
respond: cascade(
() => openAIText("kilo says hi"),
() => ({ hang: true }),
),
env: { GROQ_API_KEY: "gq-value" },
firstResponseDeadlineMs: 150,
});
const started = Date.now();
const first = await ask(registry, "auto");
const second = await ask(registry, "auto");
expect(first.content).toEqual([{ type: "text", text: "kilo says hi" }]);
expect(second.content).toEqual([{ type: "text", text: "kilo says hi" }]);
// One silent hop marks Groq congested: no waiting on its siblings, then Groq cools.
expect(chatModels(upstream.requests)).toEqual([
"groq:openai/gpt-oss-120b",
"kilo:kilo-auto/free",
"kilo:kilo-auto/free",
]);
expect(Date.now() - started).toBeLessThan(2_000);
});
it("skips a listed model the account cannot use instead of failing", async () => {
const notFound: Reply = {
status: 404,
json: {
status: 404,
title: "Not Found",
detail: "Function 'abc': Not found for account 'xyz'",
},
};
const { registry, upstream } = await setup({
respond: cascade(
() => openAIText("kilo says hi"),
(model) =>
model === "openai/gpt-oss-120b"
? notFound
: openAIText("groq says hi"),
),
env: { GROQ_API_KEY: "gq-value" },
});
const first = await ask(registry, "auto");
const second = await ask(registry, "auto");
expect(first.content).toEqual([{ type: "text", text: "groq says hi" }]);
expect(second.content).toEqual([{ type: "text", text: "groq says hi" }]);
expect(chatModels(upstream.requests)).toEqual([
"groq:openai/gpt-oss-120b",
"groq:llama-3.3-70b-versatile",
"groq:llama-3.3-70b-versatile",
]);
});
it("still surfaces a pinned model's not-found error", async () => {
const { registry } = await setup({
respond: cascade(
() => openAIText("kilo says hi"),
() => ({
status: 404,
json: { error: { message: "model not found" } },
}),
),
env: { GROQ_API_KEY: "gq-value" },
});
const message = await ask(registry, "groq/openai/gpt-oss-120b");
expect(message.stopReason).toBe("error");
expect(message.errorMessage).toContain("model not found");
});
it("completes in the last-resort pass when every backend is cooling", async () => {
let recovered = false;
const { registry, upstream } = await setup({
respond: cascade(
() => rateLimited,
() => (recovered ? openAIText("groq is back") : rateLimited),
),
env: { GROQ_API_KEY: "gq-value" },
});
const failed = await ask(registry, "auto");
expect(failed.stopReason).toBe("error");
const before = upstream.requests.length;
recovered = true;
const message = await ask(registry, "auto");
expect(message.content).toEqual([{ type: "text", text: "groq is back" }]);
expect(chatModels(upstream.requests.slice(before))).toEqual([
"groq:openai/gpt-oss-120b",
]);
});
it("returns the upstream error for a pinned model that 429s", async () => {
const { registry, upstream } = await setup({
respond: cascade(() => rateLimited),
env: { GROQ_API_KEY: "gq-value" },
});
const message = await ask(registry, "kilo/minimax/minimax-m3:free");
expect(message.stopReason).toBe("error");
expect(message.errorMessage).toContain("Rate limit exceeded");
expect(chatModels(upstream.requests)).toEqual([
"kilo:minimax/minimax-m3:free",
]);
});
it("normalizes unrecognized overflow errors so Pi compacts", async () => {
const { registry } = await setup({
respond: cascade(() => ({
status: 400,
json: {
error: {
message:
"Requested 300000 tokens but this free model's context window is 262144.",
},
},
})),
});
const message = await ask(registry, "kilo/minimax/minimax-m3:free");
expect(isContextOverflow(message)).toBe(false);
const runner = session?.session.extensionRunner;
const [handler] = runner.extensions[0].handlers.get("message_end");
const result = await handler(
{ type: "message_end", message },
runner.createContext(),
);
expect(result?.message?.errorMessage).toMatch(/^context_length_exceeded: /);
expect(isContextOverflow(result.message)).toBe(true);
});
});
const NVIDIA_MODELS = "/nvidia/v1/models";
const NVIDIA_CHAT = "/nvidia/v1/chat/completions";
const ZAI_CHAT = "/zai/api/paas/v4/chat/completions";
const HETZNER_MODELS = "/hetzner/api/v1/models";
const HETZNER_CHAT = "/hetzner/api/v1/chat/completions";
const nvidiaLive = {
data: [
{ id: "z-ai/glm-5.3", object: "model" },
{ id: "moonshotai/kimi-k2.6", object: "model" },
{ id: "nvidia/nemotron-3-ultra-550b-a55b", object: "model" },
{ id: "vendor/unknown-model", object: "model" },
],
};
const hetznerLive = {
data: [
{ id: "Qwen/Qwen3.6-35B-A3B-FP8", max_model_len: 131_072 },
{ id: "Vendor/New-Model", max_model_len: 65_536 },
],
};
const addedBackends: Responder = (request) => {
if (request.path === KILO_MODELS) return { status: 200, json: kiloCatalog };
if (request.path === NVIDIA_MODELS) return { status: 200, json: nvidiaLive };
if (request.path === HETZNER_MODELS)
return request.headers.authorization === "Bearer hz-value"
? { status: 200, json: hetznerLive }
: { status: 401, json: { error: "unauthorized" } };
if (request.path === NVIDIA_CHAT) return openAIText("nvidia says hi");
if (request.path === ZAI_CHAT) return openAIText("zai says hi");
if (request.path === HETZNER_CHAT) return openAIText("hetzner says hi");
return undefined;
};
const idsOf = (registry: Registry, backend: string) =>
grtsIds(registry).filter((id) => id.startsWith(`${backend}/`));
describe("gratis added backends", () => {
it("keeps only curated free NVIDIA models that are live, in curated rank", async () => {
const { registry } = await setup({
respond: addedBackends,
env: { NVIDIA_API_KEY: "nv-value" },
});
expect(backendsOf(registry)).toEqual(["nvidia", "kilo"]);
expect(idsOf(registry, "nvidia")).toEqual([
"nvidia/z-ai/glm-5.3",
"nvidia/moonshotai/kimi-k2.6",
]);
});
it("falls back to the curated NVIDIA list when the live list is unreachable", async () => {
const { registry } = await setup({
respond: (request) =>
request.path === NVIDIA_MODELS
? { status: 503, json: { error: "down" } }
: addedBackends(request),
env: { NVIDIA_API_KEY: "nv-value" },
});
expect(idsOf(registry, "nvidia")).toEqual([
"nvidia/moonshotai/kimi-k3",
"nvidia/z-ai/glm-5.3",
"nvidia/moonshotai/kimi-k2.6",
"nvidia/z-ai/glm-5.3-flash",
"nvidia/nvidia/nemotron-3.5-lightning-30b-a3b",
"nvidia/openai/gpt-oss-20b",
]);
});
it("streams a pinned NVIDIA model with the NVIDIA key", async () => {
const { registry, upstream } = await setup({
respond: addedBackends,
env: { GRATIS_NVIDIA_TOKEN: "nvg-value", NVIDIA_API_KEY: "nv-value" },
});
const message = await ask(registry, "nvidia/z-ai/glm-5.3");
expect(message.content).toEqual([{ type: "text", text: "nvidia says hi" }]);
const [chat] = chats(upstream.requests, NVIDIA_CHAT);
expect(chat?.body?.model).toBe("z-ai/glm-5.3");
expect(chat?.headers.authorization).toBe("Bearer nvg-value");
});
it("registers exactly the free Z.ai Flash models on the general endpoint", async () => {
const { registry, upstream } = await setup({
respond: addedBackends,
env: { ZAI_API_KEY: "zai-value" },
});
expect(idsOf(registry, "zai")).toEqual([
"zai/glm-4.7-flash",
"zai/glm-4.5-flash",
"zai/glm-4.6v-flash",
]);
const message = await ask(registry, "zai/glm-4.7-flash");
expect(message.content).toEqual([{ type: "text", text: "zai says hi" }]);
const posts = upstream.requests.filter((r) => r.method === "POST");
expect(posts.map((r) => r.path)).toEqual([ZAI_CHAT]);
expect(posts.some((r) => r.path.includes("/coding/"))).toBe(false);
expect(posts[0]?.headers.authorization).toBe("Bearer zai-value");
});
it("follows Hetzner's keyed live list, keeping curated metadata for known ids", async () => {
const { registry, upstream } = await setup({
respond: addedBackends,
env: { HETZNER_INFERENCE_API_KEY: "hz-value" },
});
expect(idsOf(registry, "hetzner")).toEqual([
"hetzner/Qwen/Qwen3.6-35B-A3B-FP8",
"hetzner/Vendor/New-Model",
]);
expect(
registry.find("grts", "hetzner/Qwen/Qwen3.6-35B-A3B-FP8"),
).toMatchObject({
contextWindow: 131_072,
maxTokens: 65_536,
input: ["text", "image"],
});
expect(registry.find("grts", "hetzner/Vendor/New-Model")).toMatchObject({
contextWindow: 65_536,
maxTokens: 8_192,
input: ["text"],
});
const discovery = upstream.requests.find((r) => r.path === HETZNER_MODELS);
expect(discovery?.headers.authorization).toBe("Bearer hz-value");
const message = await ask(registry, "hetzner/Qwen/Qwen3.6-35B-A3B-FP8");
expect(message.content).toEqual([
{ type: "text", text: "hetzner says hi" },
]);
});
it("falls back to the static Hetzner list when the live list is unreachable", async () => {
const { registry } = await setup({
respond: (request) =>
request.path === HETZNER_MODELS
? { status: 500, json: { error: "down" } }
: addedBackends(request),
env: { GRATIS_HETZNER_TOKEN: "hz-value" },
});
expect(idsOf(registry, "hetzner")).toEqual([
"hetzner/Qwen/Qwen3.6-35B-A3B-FP8",
"hetzner/Qwen3.8-27B",
]);
});
it("never calls Hetzner or lists it without a key", async () => {
const { registry, upstream } = await setup({ respond: addedBackends });
expect(backendsOf(registry)).toEqual(["kilo"]);
expect(upstream.requests.some((r) => r.path.startsWith("/hetzner/"))).toBe(
false,
);
});
it("orders every backend by auto priority when all are keyed", async () => {
const { registry } = await setup({
respond: (request) =>
request.path === OPENROUTER_MODELS
? { status: 200, json: openrouterCatalog }
: addedBackends(request),
env: {
GEMINI_API_KEY: "gm-value",
NVIDIA_API_KEY: "nv-value",
OPENROUTER_API_KEY: "on-value",
GROQ_API_KEY: "gq-value",
MISTRAL_API_KEY: "ms-value",
ZAI_API_KEY: "zai-value",
HETZNER_INFERENCE_API_KEY: "hz-value",
},
});
expect(backendsOf(registry)).toEqual([
"google",
"zai",
"groq",
"mistral",
"nvidia",
"openrouter",
"hetzner",
"kilo",
]);
});
it("cools a Z.ai backend reporting insufficient balance and moves on", async () => {
const { registry, upstream } = await setup({
respond: (request) => {
if (request.path === KILO_MODELS)
return { status: 200, json: { data: [] } };
if (request.path === ZAI_CHAT)
return {
status: 429,
json: { error: { code: "1113", message: "Insufficient balance" } },
};
return addedBackends(request);
},
env: {
ZAI_API_KEY: "zai-value",
HETZNER_INFERENCE_API_KEY: "hz-value",
},
});
const first = await ask(registry, "auto");
const second = await ask(registry, "auto");
expect(first.content).toEqual([{ type: "text", text: "hetzner says hi" }]);
expect(second.content).toEqual([{ type: "text", text: "hetzner says hi" }]);
expect(chats(upstream.requests, ZAI_CHAT)).toHaveLength(2);
});
});
const statusOf = async () => {
const runner = session?.session.extensionRunner;
const command = runner.extensions[0].commands.get("gratis");
await command.handler("", runner.createContext());
return String(session?.events.uiCallsFor("notify").at(-1)?.args[0]);
};
describe("gratis served model", () => {
it("names the model a virtual router resolved to, under the requested model", async () => {
const { registry } = await setup({
respond: (request) =>
request.path === KILO_CHAT
? openAIText("router says hi", "dots-studio/dots-3-note-preview:free")
: kiloOnly(request),
});
const pinned = await ask(registry, "kilo/kilo-auto/free");
const auto = await ask(registry, "auto");
expect(pinned).toMatchObject({
model: "kilo/kilo-auto/free",
responseModel: "kilo/dots-studio/dots-3-note-preview:free",
});
expect(auto).toMatchObject({
model: "auto",
responseModel: "kilo/dots-studio/dots-3-note-preview:free",
});
});
it("leaves responseModel unset when a pinned model served itself", async () => {
const { registry } = await setup({
respond: (request) =>
request.path === GROQ_CHAT
? openAIText("groq says hi", "openai/gpt-oss-120b")
: kiloOnly(request),
env: { GROQ_API_KEY: "gq-value" },
});
const message = await ask(registry, "groq/openai/gpt-oss-120b");
expect(message.model).toBe("groq/openai/gpt-oss-120b");
expect(message.responseModel).toBeUndefined();
});
it("falls back to the hop model when the transport reports none", async () => {
const { registry } = await setup({
respond: (request) =>
request.path === KILO_MODELS
? { status: 200, json: { data: [] } }
: everyBackend(request),
env: { GEMINI_API_KEY: "gm-value" },
});
const message = await ask(registry, "auto");
expect(message.responseModel).toBe("google/gemini-3.8-flash");
});
});
describe("gratis status command", () => {
it("reports backends, missing keys, and that nothing ran yet", async () => {
await setup({ respond: kiloOnly, env: { GROQ_API_KEY: "gq-value" } });
const text = await statusOf();
expect(text).toContain("Backends: groq 6, kilo 3");
expect(text).toContain(
"No key: google, zai, mistral, nvidia, openrouter, hetzner",
);
expect(text).toContain("Cooling: none");
expect(text).toContain("Last: no grts request yet");
});
it("reports the last auto route, what it skipped, and cooling backends", async () => {
const { registry } = await setup({
respond: cascade(
() => openAIText("kilo says hi", "dots-studio/dots-3:free"),
() => rateLimited,
),
env: { GROQ_API_KEY: "gq-value" },
});
await ask(registry, "auto");
const text = await statusOf();
expect(text).toMatch(
/Last: auto → kilo\/dots-studio\/dots-3:free · ok · 5 hops · \d+s ago/,
);
expect(text).toContain("Skipped: groq exhausted (rate)");
expect(text).toMatch(/Cooling: groq \(1m left\)/);
});
it("reports a failed pinned route as an error without a served model", async () => {
const { registry } = await setup({
respond: cascade(() => rateLimited),
});
await ask(registry, "kilo/minimax/minimax-m3:free");
const text = await statusOf();
expect(text).toMatch(
/Last: kilo\/minimax\/minimax-m3:free → kilo\/minimax\/minimax-m3:free · error · 1 hop/,
);
});
});
describe("gratis performance-aware routing", () => {
it("cuts a hop sooner when its model usually starts fast", async () => {
let groqCalls = 0;
const { registry } = await setup({
respond: cascade(
() => openAIText("kilo says hi"),
() => (++groqCalls <= 3 ? openAIText("groq says hi") : { hang: true }),
),
env: { GROQ_API_KEY: "gq-value" },
firstResponseDeadlineMs: 4_000,
});
for (let i = 0; i < 3; i++) await ask(registry, "auto");
const started = Date.now();
const message = await ask(registry, "auto");
expect(message.content).toEqual([{ type: "text", text: "kilo says hi" }]);
const waited = Date.now() - started;
expect(waited).toBeGreaterThanOrEqual(2_900);
expect(waited).toBeLessThan(3_800);
});
it("tries a repeatedly failing model after its healthy siblings", async () => {
vi.useFakeTimers({ toFake: ["Date"] });
try {
const { registry, upstream } = await setup({
respond: cascade(
() => openAIText("kilo says hi"),
(model) =>
model === "openai/gpt-oss-120b"
? rateLimited
: openAIText("groq says hi"),
),
env: { GROQ_API_KEY: "gq-value" },
});
for (let i = 0; i < 3; i++) {
await ask(registry, "auto");
vi.setSystemTime(Date.now() + 61_000);
}
const before = upstream.requests.length;
const message = await ask(registry, "auto");
expect(message.content).toEqual([{ type: "text", text: "groq says hi" }]);
expect(chatModels(upstream.requests.slice(before))).toEqual([
"groq:llama-3.3-70b-versatile",
]);
} finally {
vi.useRealTimers();
}
});
it("shows measured models in the status command", async () => {
const { registry } = await setup({ respond: kiloOnly });
await ask(registry, "auto");
const text = await statusOf();
expect(text).toMatch(
/Models:\n {2}kilo\/kilo-auto\/free · first token \d+\.\ds · 0% fail · 1 hop/,
);
});
});
const gratisCommand = async (args: string) => {
const runner = session?.session.extensionRunner;
const command = runner.extensions[0].commands.get("gratis");
await command.handler(args, runner.createContext());
const call = session?.events.uiCallsFor("notify").at(-1);
return { text: String(call?.args[0]), level: call?.args[1] };
};
const completions = (prefix: string) =>
session?.session.extensionRunner.extensions[0].commands
.get("gratis")
.getArgumentCompletions(prefix);
describe("gratis manual cooling", () => {
const kiloRouter: Responder = (request) =>
request.path === KILO_CHAT
? openAIText("kilo says hi", "dots-studio/dots-3:free")
: kiloOnly(request);
it("asks for a model when nothing has answered yet", async () => {
await setup({ respond: kiloRouter });
const reply = await gratisCommand("cool");
expect(reply.level).toBe("warning");
expect(reply.text).toContain("name one: /gratis cool <model>");
});
it("cools the candidate that answered last, naming what it answered as", async () => {
const { registry, upstream } = await setup({ respond: kiloRouter });
await ask(registry, "auto");
const reply = await gratisCommand("cool");
const before = upstream.requests.length;
await ask(registry, "auto");
expect(reply).toEqual({
level: "info",
text: "grts/auto skips kilo/kilo-auto/free for 60 min (it answered as kilo/dots-studio/dots-3:free; gratis can only skip kilo/kilo-auto/free itself)",
});
expect(chatModels(upstream.requests.slice(before))).toEqual([
"kilo:openrouter/free",
]);
expect((await gratisCommand("")).text).toContain(
"Manually cooled: kilo/kilo-auto/free (60m left)",
);
});
it("keeps a manually cooled model out of the last-resort pass", async () => {
const { registry, upstream } = await setup({
respond: cascade((model) =>
model === "kilo-auto/free" ? openAIText("unwanted") : rateLimited,
),
});
await gratisCommand("cool kilo/kilo-auto/free 30");
const message = await ask(registry, "auto");
await ask(registry, "auto");
expect(message.stopReason).toBe("error");
expect(chatModels(upstream.requests)).not.toContain("kilo:kilo-auto/free");
});
it("ends manual cooldowns on uncool", async () => {
const { registry, upstream } = await setup({ respond: kiloRouter });
await gratisCommand("cool kilo/kilo-auto/free");
const one = await gratisCommand("uncool kilo/kilo-auto/free");
await ask(registry, "auto");
await gratisCommand("cool kilo/openrouter/free");
await gratisCommand("cool kilo/minimax/minimax-m3:free");
const all = await gratisCommand("uncool");
const none = await gratisCommand("uncool");
expect(one.text).toBe("kilo/kilo-auto/free is back in grts/auto");
expect(chatModels(upstream.requests)).toEqual(["kilo:kilo-auto/free"]);
expect(all.text).toBe("2 manually cooled models are back in grts/auto");
expect(none).toEqual({
level: "warning",
text: "No model is manually cooled",
});
});
it("still answers when the cooled model is pinned", async () => {
const { registry } = await setup({ respond: kiloRouter });
await gratisCommand("cool kilo/kilo-auto/free");
const message = await ask(registry, "kilo/kilo-auto/free");
expect(message.content).toEqual([{ type: "text", text: "kilo says hi" }]);
});
it("rejects unknown models, auto, and bad durations", async () => {
await setup({ respond: kiloRouter });
expect(await gratisCommand("cool kilo/nope")).toEqual({
level: "warning",
text: "kilo/nope is not a current grts model",
});
expect((await gratisCommand("cool auto")).text).toBe(
"auto is not a current grts model",
);
for (const minutes of ["0", "1.5", "abc", "1441"])
expect(
(await gratisCommand(`cool kilo/kilo-auto/free ${minutes}`)).text,
).toBe("Minutes must be a whole number from 1 to 1440");
expect((await gratisCommand("warm")).text).toBe(
"Usage: /gratis [cool [model] [minutes] | uncool [model]]",
);
});
it("autocompletes subcommands, models, and manually cooled models", async () => {
await setup({ respond: kiloRouter });
await gratisCommand("cool kilo/openrouter/free");
expect(await completions("c")).toEqual([{ value: "cool", label: "cool" }]);
expect(await completions("cool kilo/k")).toEqual([
{ value: "cool kilo/kilo-auto/free", label: "kilo/kilo-auto/free" },
]);
expect(await completions("uncool ")).toEqual([
{ value: "uncool kilo/openrouter/free", label: "kilo/openrouter/free" },
]);
expect(await completions("cool kilo/kilo-auto/free 3")).toBeNull();
});
});
const ZAI_PRICING_PATH = "/zaidocs/guides/overview/pricing.md";
const MODELS_DEV_PATH = "/modelsdev/api.json";
const freeRow = (name: string) => `| ${name} | Free | Free | Free | Free |`;
const pricingPage = (...rows: string[]) =>
[
"### Text Models",
"| Model | Input | Cached Input | Cached Input Storage | Output |",
"| :- | :- | :- | :- | :- |",
"| GLM-4.7 | $0.6 | $0.11 | Limited-time Free | $2.2 |",
...rows,
].join("\n");
const modelsDev = {
zai: {
api: "https://api.z.ai/api/paas/v4",
models: {
"glm-9-flash": {
name: "GLM 9 Flash",
cost: { input: 0, output: 0 },
limit: { context: 300_000, output: 64_000 },
modalities: { input: ["text"] },
},
"glm-9-pro": {
name: "GLM 9 Pro",
cost: { input: 2, output: 8 },
limit: { context: 300_000, output: 64_000 },
},
},
},
};
describe("gratis fresh catalogs", () => {
it("lists Google models from Pi's catalog with Pi's thinking levels", async () => {
const { registry, context } = await setup({
respond: kiloOnly,
env: { GEMINI_API_KEY: "gm-value" },
});
const pi = context.modelRegistry.find("google", "gemini-3.7-flash");
const ids = idsOf(registry, "google");
expect(ids[0]).toBe("google/gemini-3.8-flash");
expect(ids).toContain("google/gemini-3.6-flash");
expect(ids.some((id) => /latest|preview/.test(id))).toBe(false);
expect(
registry.find("grts", "google/gemini-3.7-flash")?.thinkingLevelMap,
).toEqual(pi?.thinkingLevelMap);
});
it("shows a model added to Pi's catalog without a gratis update", async () => {
const { registry, context } = await setup({
respond: kiloOnly,
env: { GROQ_API_KEY: "gq-value" },
});
expect(idsOf(registry, "groq")).not.toContain("groq/brand-new-model");
context.modelRegistry.registerProvider("groq", {
baseUrl: "https://api.groq.com/openai/v1",
apiKey: "gq-value",
api: "openai-completions",
models: [
{
id: "brand-new-model",
name: "Brand New",
reasoning: false,
input: ["text"],
cost: { input: 0.1, output: 0.1, cacheRead: 0, cacheWrite: 0 },
contextWindow: 131_072,
maxTokens: 16_384,
},
],
});
expect(idsOf(registry, "groq")).toEqual(["groq/brand-new-model"]);
expect(registry.find("grts", "groq/brand-new-model")?.cost.input).toBe(0);
});
it("follows Z.ai's pricing page, confirming unknown free ids on models.dev", async () => {
const { registry, upstream } = await setup({
respond: (request) => {
if (request.path === ZAI_PRICING_PATH)
return {
status: 200,
text: pricingPage(
freeRow("GLM-4.7-Flash"),
freeRow("GLM-9-Flash"),
freeRow("GLM-9-Pro"),
),
};
if (request.path === MODELS_DEV_PATH)
return { status: 200, json: modelsDev };
return kiloOnly(request);
},
env: { ZAI_API_KEY: "zai-value" },
});
expect(idsOf(registry, "zai")).toEqual([
"zai/glm-4.7-flash",
"zai/glm-9-flash",
]);
expect(registry.find("grts", "zai/glm-9-flash")).toMatchObject({
baseUrl: "https://api.z.ai/api/paas/v4",
contextWindow: 300_000,
});
expect(
upstream.requests.filter((request) => request.path === MODELS_DEV_PATH),
).toHaveLength(1);
});
it("lists no Z.ai models when the page lists none as Free", async () => {
const { registry } = await setup({
respond: (request) =>
request.path === ZAI_PRICING_PATH
? { status: 200, text: pricingPage() }
: kiloOnly(request),
env: { ZAI_API_KEY: "zai-value" },
});
expect(idsOf(registry, "zai")).toEqual([]);
});
it("keeps the last confirmed Z.ai list when the page becomes unreachable", async () => {
let pageUp = true;
const { registry } = await setup({
respond: (request) =>
request.path === ZAI_PRICING_PATH
? pageUp
? { status: 200, text: pricingPage(freeRow("GLM-4.5-Flash")) }
: { status: 503, json: { error: "down" } }
: kiloOnly(request),
env: { ZAI_API_KEY: "zai-value" },
});
expect(idsOf(registry, "zai")).toEqual(["zai/glm-4.5-flash"]);
pageUp = false;
await registry.refresh({ allowNetwork: true, providers: ["grts"] });
expect(idsOf(registry, "zai")).toEqual(["zai/glm-4.5-flash"]);
});
});
describe("gratis learning across time", () => {
it("keeps a twice-rejected model behind its siblings after the cooldowns end", async () => {
vi.useFakeTimers({ toFake: ["Date"] });
try {
const { registry, upstream } = await setup({ respond: kiloOnly });
await gratisCommand("cool kilo/kilo-auto/free");
await gratisCommand("cool kilo/kilo-auto/free");
vi.setSystemTime(Date.now() + 61 * 60_000);
await ask(registry, "auto");
expect(chatModels(upstream.requests)).toEqual(["kilo:openrouter/free"]);
} finally {
vi.useRealTimers();
}
});
it("still applies a demotion learned before a restart", async () => {
const stateDir = mkdtempSync(path.join(tmpdir(), "gratis-routing-"));
const routingStatePath = path.join(stateDir, "routing.json");
const respond = cascade(
() => openAIText("kilo says hi"),
(model) =>
model === "openai/gpt-oss-120b"
? rateLimited
: openAIText("groq says hi"),
);
vi.useFakeTimers({ toFake: ["Date"] });
try {
const first = await setup({
respond,
env: { GROQ_API_KEY: "gq-value" },
routingStatePath,
});
for (let i = 0; i < 3; i++) {
await ask(first.registry, "auto");
vi.setSystemTime(Date.now() + 61_000);
}
await session?.session.extensionRunner.emit({
type: "session_shutdown",
reason: "quit",
});
session?.dispose();
session = undefined;
await upstream?.close();
upstream = undefined;
const second = await setup({
respond,
env: { GROQ_API_KEY: "gq-value" },
routingStatePath,
});
// Wait for the background load, then send exactly one request: a second
// request would skip the model through its fresh 429 cooldown anyway.
await expect
.poll(statusOf)
.toMatch(/groq\/openai\/gpt-oss-120b · .*demoted/);
const before = second.upstream.requests.length;
await ask(second.registry, "auto");
expect(chatModels(second.upstream.requests.slice(before))[0]).toBe(
"groq:llama-3.3-70b-versatile",
);
} finally {
vi.useRealTimers();
rmSync(stateDir, { recursive: true, force: true });
}
});
});