import { mkdtempSync, rmSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import path from "node:path"; import { type Api, type AssistantMessage, isContextOverflow, type Model, } from "@earendil-works/pi-ai"; import { afterEach, describe, expect, it, vi } from "vitest"; import { createTestSession, type TestSession } from "../../../test/harness"; import { createGratis } from "../index.ts"; import { anthropicText, contextOverflow, googleText, liveModel, openAIText, type Recorded, type Reply, type Responder, rateLimited, startUpstream, } from "./fake-upstream.ts"; type Registry = { refresh(options: { allowNetwork?: boolean; providers?: string[]; }): Promise; getAll(): Model[]; getAvailable(): Model[]; find(provider: string, id: string): Model | undefined; streamSimple( model: Model, context: unknown, ): { result(): Promise }; getApiKeyForProvider(provider: string): Promise; }; const KILO_MODELS = "/kilo/api/gateway/models"; const KILO_CHAT = "/kilo/api/gateway/chat/completions"; const OPENROUTER_MODELS = "/openrouter/api/v1/models"; const OPENROUTER_CHAT = "/openrouter/api/v1/chat/completions"; const OPENROUTER_MESSAGES = "/openrouter/api/v1/messages"; const GOOGLE = "/google/v1beta/models/"; const openrouterCatalog = { data: [ liveModel("qwen/qwen3.8-27b:free"), liveModel("anthropic/claude-trial:free"), liveModel("openrouter/free", { context_length: 200_000, top_provider: { context_length: null, max_completion_tokens: null }, }), liveModel("qwen/qwen3.8-27b", { pricing: { prompt: "0.0000002", completion: "0.0000008" }, }), ], }; const kiloCatalog = { data: [ liveModel("minimax/minimax-m3:free"), liveModel("vendor/paid-model", { pricing: { prompt: "0.000001", completion: "0.000002" }, }), liveModel("vendor/fake-free:free", { pricing: { prompt: "0.1", completion: "0" }, }), liveModel("vendor/no-tools:free", { supported_parameters: ["max_tokens"] }), liveModel("openrouter/free", { context_length: 200_000, top_provider: { context_length: null, max_completion_tokens: null }, }), liveModel("kilo-auto/free", { context_length: 256_000, top_provider: { context_length: 256_000, max_completion_tokens: 32_768 }, }), ], }; let session: TestSession | undefined; let upstream: Awaited> | undefined; let agentDir: string | undefined; afterEach(async () => { session?.dispose(); session = undefined; await upstream?.close(); upstream = undefined; if (agentDir) rmSync(agentDir, { recursive: true, force: true }); agentDir = undefined; }); interface Setup { respond: Responder; env?: Record; auth?: Record; firstResponseDeadlineMs?: number; routingStatePath?: string; } async function setup({ respond, env = {}, auth, firstResponseDeadlineMs, routingStatePath, }: Setup) { upstream = await startUpstream(respond); agentDir = mkdtempSync(path.join(tmpdir(), "gratis-agent-")); if (auth) writeFileSync(path.join(agentDir, "auth.json"), JSON.stringify(auth)); session = await createTestSession({ extensionFactories: [ createGratis({ resolveUrl: upstream.resolveUrl, ...(firstResponseDeadlineMs === undefined ? {} : { firstResponseDeadlineMs }), ...(routingStatePath === undefined ? {} : { routingStatePath }), }), ], env: { PI_CODING_AGENT_DIR: agentDir, ...env }, }); const context = session.session.extensionRunner.createContext(); const registry = context.modelRegistry as Registry; await registry.refresh({ allowNetwork: true, providers: ["grts"] }); return { registry, context, upstream }; } const grtsIds = (registry: Registry) => registry .getAll() .filter((model) => model.provider === "grts") .map((model) => model.id); const ask = (registry: Registry, id: string) => { const model = registry.find("grts", id); if (!model) throw new Error(`missing grts/${id}`); return registry .streamSimple(model, { messages: [{ role: "user", content: "hi", timestamp: Date.now() }], }) .result(); }; const kiloOnly: Responder = (request) => { if (request.path === KILO_MODELS) return { status: 200, json: kiloCatalog }; if (request.path === KILO_CHAT) return openAIText("kilo says hi"); return undefined; }; const chats = (requests: Recorded[], prefix: string) => requests.filter( (request) => request.method === "POST" && request.path.startsWith(prefix), ); describe("gratis keyless kilo", () => { it("registers only free kilo models plus auto when no keys are set", async () => { const { registry } = await setup({ respond: kiloOnly }); expect(grtsIds(registry)).toEqual([ "auto", "kilo/kilo-auto/free", "kilo/openrouter/free", "kilo/minimax/minimax-m3:free", ]); expect( registry.getAvailable().filter((model) => model.provider === "grts"), ).toHaveLength(4); }); it("labels every model free and data-caveated", async () => { const { registry } = await setup({ respond: kiloOnly }); for (const model of registry .getAll() .filter((entry) => entry.provider === "grts")) { expect(model.cost).toEqual({ input: 0, output: 0, cacheRead: 0, cacheWrite: 0, }); expect(model.name).toContain("prompts may be logged"); } }); it("advertises published virtual-model parameters and falls back only when absent", async () => { const { registry } = await setup({ respond: kiloOnly }); expect(registry.find("grts", "kilo/kilo-auto/free")).toMatchObject({ contextWindow: 256_000, maxTokens: 32_768, }); expect(registry.find("grts", "kilo/openrouter/free")).toMatchObject({ contextWindow: 200_000, maxTokens: 8_192, }); }); it("stays invisible when the kilo catalog is unreachable and no key is set", async () => { const { registry } = await setup({ respond: () => ({ status: 503, json: { error: "down" } }), }); expect(grtsIds(registry)).toEqual([]); }); it("yields no kilo models and no error for an empty free pool", async () => { const { registry } = await setup({ respond: (request) => request.path === KILO_MODELS ? { status: 200, json: { data: [liveModel("vendor/paid")] } } : undefined, }); expect(grtsIds(registry)).toEqual([]); }); it("streams a pinned kilo model anonymously under the grts identity", async () => { const { registry, upstream } = await setup({ respond: kiloOnly }); const message = await ask(registry, "kilo/kilo-auto/free"); expect(message.stopReason).toBe("stop"); expect(message.content).toEqual([{ type: "text", text: "kilo says hi" }]); expect(message.provider).toBe("grts"); expect(message.model).toBe("kilo/kilo-auto/free"); const [chat] = chats(upstream.requests, KILO_CHAT); expect(chat?.body?.model).toBe("kilo-auto/free"); expect(chat?.headers.authorization).toBeUndefined(); }); it("never forces reasoning off on router models whose upstream may require it", async () => { const { registry, upstream } = await setup({ respond: (request) => request.path === KILO_MODELS ? { status: 200, json: { data: [ liveModel("openrouter/free", { supported_parameters: ["tools", "reasoning"], }), ], }, } : kiloOnly(request), }); await ask(registry, "kilo/openrouter/free"); const body = chats(upstream.requests, KILO_CHAT)[0]?.body ?? {}; expect(body).not.toHaveProperty("reasoning"); }); it("sends an optional kilo key when one is set", async () => { const { registry, upstream } = await setup({ respond: kiloOnly, env: { GRATIS_KILO_TOKEN: "kg-value" }, }); await ask(registry, "kilo/kilo-auto/free"); expect(chats(upstream.requests, KILO_CHAT)[0]?.headers.authorization).toBe( "Bearer kg-value", ); }); it("reads the native kilo key when no gratis kilo key is set", async () => { const { registry, upstream } = await setup({ respond: kiloOnly, env: { KILO_API_KEY: "kn-value" }, }); await ask(registry, "kilo/kilo-auto/free"); expect(chats(upstream.requests, KILO_CHAT)[0]?.headers.authorization).toBe( "Bearer kn-value", ); }); }); const everyBackend: Responder = (request) => { if (request.path === KILO_MODELS) return { status: 200, json: kiloCatalog }; if (request.path === OPENROUTER_MODELS) return { status: 200, json: openrouterCatalog }; if (request.path === KILO_CHAT) return openAIText("kilo says hi"); if (request.path === OPENROUTER_CHAT) return openAIText("openrouter says hi"); if (request.path.startsWith(OPENROUTER_MESSAGES)) return anthropicText("anthropic shape says hi"); if (request.path.startsWith(GOOGLE)) return googleText("gemini says hi"); return undefined; }; const backendsOf = (registry: Registry) => [ ...new Set( grtsIds(registry) .filter((id) => id !== "auto") .map((id) => id.slice(0, id.indexOf("/"))), ), ]; describe("gratis keyed backends", () => { it("shows only the keyed backend next to kilo for one native key", async () => { const { registry } = await setup({ respond: everyBackend, env: { GROQ_API_KEY: "gq-value" }, }); expect(backendsOf(registry)).toEqual(["groq", "kilo"]); expect(grtsIds(registry)).toContain("groq/openai/gpt-oss-120b"); }); it("registers every keyed backend with its curated catalog", async () => { const { registry } = await setup({ respond: everyBackend, env: { GEMINI_API_KEY: "gm-value", OPENROUTER_API_KEY: "on-value", GROQ_API_KEY: "gq-value", MISTRAL_API_KEY: "ms-value", }, }); expect(backendsOf(registry)).toEqual([ "google", "groq", "mistral", "openrouter", "kilo", ]); expect(registry.find("grts", "mistral/devstral-latest")?.api).toBe( "mistral-conversations", ); expect(registry.find("grts", "google/gemini-3.8-flash")?.api).toBe( "google-generative-ai", ); }); it("lets GRATIS_OPENROUTER_TOKEN override the native OpenRouter key", async () => { const { registry, upstream } = await setup({ respond: everyBackend, env: { GRATIS_OPENROUTER_TOKEN: "og-value", OPENROUTER_API_KEY: "on-value", }, }); const message = await ask(registry, "openrouter/qwen/qwen3.8-27b:free"); expect(message.content).toEqual([ { type: "text", text: "openrouter says hi" }, ]); expect( chats(upstream.requests, OPENROUTER_CHAT)[0]?.headers.authorization, ).toBe("Bearer og-value"); }); it("keeps only free OpenRouter models, virtual router first", async () => { const { registry } = await setup({ respond: everyBackend, env: { OPENROUTER_API_KEY: "on-value" }, }); expect( grtsIds(registry).filter((id) => id.startsWith("openrouter/")), ).toEqual([ "openrouter/openrouter/free", "openrouter/qwen/qwen3.8-27b:free", "openrouter/anthropic/claude-trial:free", ]); }); it("yields no OpenRouter models and no error for zero free variants", async () => { const { registry } = await setup({ respond: (request) => request.path === OPENROUTER_MODELS ? { status: 200, json: { data: [liveModel("vendor/paid")] } } : everyBackend(request), env: { OPENROUTER_API_KEY: "on-value" }, }); expect(backendsOf(registry)).toEqual(["kilo"]); }); it("falls back to the static OpenRouter router when discovery fails", async () => { const { registry } = await setup({ respond: (request) => request.path === OPENROUTER_MODELS ? { status: 500, json: { error: "down" } } : everyBackend(request), env: { OPENROUTER_API_KEY: "on-value" }, }); expect( grtsIds(registry).filter((id) => id.startsWith("openrouter/")), ).toEqual(["openrouter/openrouter/free"]); }); it("registers a backend from a key in Pi's credential store", async () => { const { registry, upstream } = await setup({ respond: everyBackend, auth: { openrouter: { type: "api_key", key: "os-value" }, }, }); // Pi emits session_start while creating the session; lookup runs in the background. await expect .poll(() => backendsOf(registry)) .toEqual(["openrouter", "kilo"]); const message = await ask(registry, "openrouter/openrouter/free"); expect(message.stopReason).toBe("stop"); expect( chats(upstream.requests, OPENROUTER_CHAT)[0]?.headers.authorization, ).toBe("Bearer os-value"); }); it("streams google-shaped entries through the Google API", async () => { const { registry, upstream } = await setup({ respond: everyBackend, env: { GEMINI_API_KEY: "gm-value" }, }); const message = await ask(registry, "google/gemini-3.8-flash"); expect(message.stopReason).toBe("stop"); expect(message.content).toEqual([{ type: "text", text: "gemini says hi" }]); const [request] = chats(upstream.requests, GOOGLE); expect(request?.path).toContain("gemini-3.8-flash:streamGenerateContent"); expect(request?.headers["x-goog-api-key"]).toBe("gm-value"); }); it("never asks newer Gemini Flash models for the unsupported MINIMAL thinking level", async () => { const { registry, upstream } = await setup({ respond: everyBackend, env: { GEMINI_API_KEY: "gm-value" }, }); await ask(registry, "google/gemini-3.8-flash"); const [request] = chats(upstream.requests, GOOGLE); const config = JSON.stringify(request?.body?.generationConfig ?? {}); expect(config).not.toMatch(/MINIMAL/i); }); it("streams anthropic-shaped entries through the Anthropic API", async () => { const { registry, upstream } = await setup({ respond: everyBackend, env: { OPENROUTER_API_KEY: "on-value" }, }); const id = "openrouter/anthropic/claude-trial:free"; expect(registry.find("grts", id)?.api).toBe("anthropic-messages"); const message = await ask(registry, id); expect(message.stopReason).toBe("stop"); expect(message.content).toEqual([ { type: "text", text: "anthropic shape says hi" }, ]); expect(chats(upstream.requests, OPENROUTER_MESSAGES)[0]?.body?.model).toBe( "anthropic/claude-trial:free", ); }); }); const GROQ_CHAT = "/groq/openai/v1/chat/completions"; const GROQ_CANDIDATES = [ "groq:openai/gpt-oss-120b", "groq:llama-3.3-70b-versatile", "groq:openai/gpt-oss-20b", "groq:llama-3.1-8b-instant", ]; /** Kilo and Groq fakes whose chat replies depend on the upstream model. */ function cascade( kilo: (model: string) => Reply, groq: (model: string) => Reply = () => openAIText("groq says hi"), ): Responder { return (request) => { if (request.path === KILO_MODELS) return { status: 200, json: kiloCatalog }; const model = String(request.body?.model); if (request.path === KILO_CHAT) return kilo(model); if (request.path === GROQ_CHAT) return groq(model); return undefined; }; } const chatModels = (requests: Recorded[]) => requests .filter((request) => request.method === "POST") .map((request) => `${request.path.split("/")[1]}:${request.body?.model}`); describe("gratis auto router", () => { it("advertises the tightest candidate context and output limits", async () => { const { registry } = await setup({ respond: cascade(() => openAIText("ok")), env: { GROQ_API_KEY: "gq-value" }, }); expect(registry.find("grts", "auto")).toMatchObject({ contextWindow: 131_072, maxTokens: 8_192, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, }); }); it("tries the virtual kilo routers first", async () => { const { registry, upstream } = await setup({ respond: cascade(() => openAIText("kilo says hi")), }); const message = await ask(registry, "auto"); expect(message.content).toEqual([{ type: "text", text: "kilo says hi" }]); expect(message.model).toBe("auto"); expect(chatModels(upstream.requests)).toEqual(["kilo:kilo-auto/free"]); }); it("completes on the next kilo candidate after upstream saturation, then skips the saturated one", async () => { const { registry, upstream } = await setup({ respond: cascade((model) => model === "kilo-auto/free" ? rateLimited : openAIText("router says hi"), ), }); const first = await ask(registry, "auto"); const second = await ask(registry, "auto"); expect(first.content).toEqual([{ type: "text", text: "router says hi" }]); expect(second.stopReason).toBe("stop"); // kilo-auto/free cools on its own: the second request costs no round trip to it. expect(chatModels(upstream.requests)).toEqual([ "kilo:kilo-auto/free", "kilo:openrouter/free", "kilo:openrouter/free", ]); }); it("moves to the next backend when groq is exhausted and then skips the cooling backend", async () => { const { registry, upstream } = await setup({ respond: cascade( () => openAIText("kilo says hi"), () => rateLimited, ), env: { GROQ_API_KEY: "gq-value" }, }); const first = await ask(registry, "auto"); const second = await ask(registry, "auto"); expect(first.stopReason).toBe("stop"); expect(first.content).toEqual([{ type: "text", text: "kilo says hi" }]); expect(second.content).toEqual([{ type: "text", text: "kilo says hi" }]); expect(chatModels(upstream.requests)).toEqual([ ...GROQ_CANDIDATES, "kilo:kilo-auto/free", "kilo:kilo-auto/free", ]); }); it("advances on context overflow without cooling while a next backend remains", async () => { const { registry, upstream } = await setup({ respond: cascade( () => openAIText("kilo says hi"), () => contextOverflow, ), env: { GROQ_API_KEY: "gq-value" }, }); const first = await ask(registry, "auto"); await ask(registry, "auto"); expect(first.stopReason).toBe("stop"); expect(first.content).toEqual([{ type: "text", text: "kilo says hi" }]); // Groq was not cooled: the second request tries it again. expect( chatModels(upstream.requests).filter((hop) => hop.startsWith("groq:")), ).toHaveLength(8); }); it("treats already-normalized overflow text as overflow", async () => { const { registry, upstream } = await setup({ respond: cascade( () => openAIText("kilo says hi"), () => ({ status: 400, json: { error: { message: "context_length_exceeded: upstream" } }, }), ), env: { GROQ_API_KEY: "gq-value" }, }); const first = await ask(registry, "auto"); await ask(registry, "auto"); expect(first.content).toEqual([{ type: "text", text: "kilo says hi" }]); expect( chatModels(upstream.requests).filter((hop) => hop.startsWith("groq:")), ).toHaveLength(8); }); it("reports overflow Pi can compact when every backend overflows", async () => { const { registry } = await setup({ respond: cascade( () => contextOverflow, () => contextOverflow, ), env: { GROQ_API_KEY: "gq-value" }, }); const message = await ask(registry, "auto"); expect(message.stopReason).toBe("error"); expect(message.errorMessage).toMatch(/^context_length_exceeded: /); }); it("fails only when every backend is cooling and lists them", async () => { const { registry } = await setup({ respond: cascade( () => rateLimited, () => ({ status: 429, json: { error: { message: "You exceeded your current quota" } }, }), ), env: { GROQ_API_KEY: "gq-value" }, }); const first = await ask(registry, "auto"); const second = await ask(registry, "auto"); expect(first.stopReason).toBe("error"); expect(first.errorMessage).toContain("kilo exhausted (rate)"); expect(first.errorMessage).toContain("groq exhausted (quota)"); expect(second.errorMessage).toContain("kilo cooling"); expect(second.errorMessage).toContain("groq cooling"); }); it("makes a cooled backend available again after its rate window", async () => { vi.useFakeTimers({ toFake: ["Date"] }); try { let kiloDown = false; const { registry, upstream } = await setup({ respond: cascade( () => (kiloDown ? rateLimited : openAIText("kilo says hi")), () => rateLimited, ), env: { GROQ_API_KEY: "gq-value" }, }); await ask(registry, "auto"); const groqHops = () => chatModels(upstream.requests).filter((hop) => hop.startsWith("groq:")) .length; expect(groqHops()).toBe(4); // Groq is now demoted behind Kilo, but no longer cooling: it is tried when Kilo fails. kiloDown = true; vi.setSystemTime(Date.now() + 60_000 + 1); await ask(registry, "auto"); expect(groqHops()).toBe(8); } finally { vi.useRealTimers(); } }); it("replays same-model history with its native reasoning state", async () => { const { registry, upstream } = await setup({ respond: kiloOnly }); const id = "kilo/kilo-auto/free"; const model = registry.find("grts", id); if (!model) throw new Error("missing model"); const previous = await ask(registry, id); previous.content = [ { type: "thinking", thinking: "plan", thinkingSignature: JSON.stringify([ { type: "reasoning.text", text: "plan", index: 0 }, ]), }, { type: "text", text: "kilo says hi" }, ]; await registry .streamSimple(model, { messages: [ { role: "user", content: "hi", timestamp: 1 }, previous, { role: "user", content: "again", timestamp: 2 }, ], }) .result(); const replay = chats(upstream.requests, KILO_CHAT).at(-1)?.body ?.messages as Record[]; expect(replay[1]?.reasoning_details).toEqual([ { type: "reasoning.text", text: "plan", index: 0 }, ]); }); it("moves on when a backend reports upstream high demand", async () => { const { registry } = await setup({ respond: cascade(() => ({ status: 503, json: { error: { code: 503, message: "This model is currently experiencing high demand.", status: "UNAVAILABLE", }, }, })), env: { GROQ_API_KEY: "gq-value" }, }); const message = await ask(registry, "auto"); expect(message.content).toEqual([{ type: "text", text: "groq says hi" }]); }); it("surfaces other upstream errors instead of cascading", async () => { const { registry, upstream } = await setup({ respond: cascade( () => openAIText("kilo says hi"), () => ({ status: 401, json: { error: { message: "invalid credentials" } }, }), ), env: { GROQ_API_KEY: "gq-value" }, }); const message = await ask(registry, "auto"); expect(message.stopReason).toBe("error"); expect(message.errorMessage).toContain("invalid credentials"); expect(chatModels(upstream.requests)).toEqual(["groq:openai/gpt-oss-120b"]); }); it("moves past a hop that streams nothing before its deadline", async () => { const { registry, upstream } = await setup({ respond: cascade( () => openAIText("kilo says hi"), () => ({ hang: true }), ), env: { GROQ_API_KEY: "gq-value" }, firstResponseDeadlineMs: 150, }); const started = Date.now(); const first = await ask(registry, "auto"); const second = await ask(registry, "auto"); expect(first.content).toEqual([{ type: "text", text: "kilo says hi" }]); expect(second.content).toEqual([{ type: "text", text: "kilo says hi" }]); // One silent hop marks Groq congested: no waiting on its siblings, then Groq cools. expect(chatModels(upstream.requests)).toEqual([ "groq:openai/gpt-oss-120b", "kilo:kilo-auto/free", "kilo:kilo-auto/free", ]); expect(Date.now() - started).toBeLessThan(2_000); }); it("skips a listed model the account cannot use instead of failing", async () => { const notFound: Reply = { status: 404, json: { status: 404, title: "Not Found", detail: "Function 'abc': Not found for account 'xyz'", }, }; const { registry, upstream } = await setup({ respond: cascade( () => openAIText("kilo says hi"), (model) => model === "openai/gpt-oss-120b" ? notFound : openAIText("groq says hi"), ), env: { GROQ_API_KEY: "gq-value" }, }); const first = await ask(registry, "auto"); const second = await ask(registry, "auto"); expect(first.content).toEqual([{ type: "text", text: "groq says hi" }]); expect(second.content).toEqual([{ type: "text", text: "groq says hi" }]); expect(chatModels(upstream.requests)).toEqual([ "groq:openai/gpt-oss-120b", "groq:llama-3.3-70b-versatile", "groq:llama-3.3-70b-versatile", ]); }); it("still surfaces a pinned model's not-found error", async () => { const { registry } = await setup({ respond: cascade( () => openAIText("kilo says hi"), () => ({ status: 404, json: { error: { message: "model not found" } }, }), ), env: { GROQ_API_KEY: "gq-value" }, }); const message = await ask(registry, "groq/openai/gpt-oss-120b"); expect(message.stopReason).toBe("error"); expect(message.errorMessage).toContain("model not found"); }); it("completes in the last-resort pass when every backend is cooling", async () => { let recovered = false; const { registry, upstream } = await setup({ respond: cascade( () => rateLimited, () => (recovered ? openAIText("groq is back") : rateLimited), ), env: { GROQ_API_KEY: "gq-value" }, }); const failed = await ask(registry, "auto"); expect(failed.stopReason).toBe("error"); const before = upstream.requests.length; recovered = true; const message = await ask(registry, "auto"); expect(message.content).toEqual([{ type: "text", text: "groq is back" }]); expect(chatModels(upstream.requests.slice(before))).toEqual([ "groq:openai/gpt-oss-120b", ]); }); it("returns the upstream error for a pinned model that 429s", async () => { const { registry, upstream } = await setup({ respond: cascade(() => rateLimited), env: { GROQ_API_KEY: "gq-value" }, }); const message = await ask(registry, "kilo/minimax/minimax-m3:free"); expect(message.stopReason).toBe("error"); expect(message.errorMessage).toContain("Rate limit exceeded"); expect(chatModels(upstream.requests)).toEqual([ "kilo:minimax/minimax-m3:free", ]); }); it("normalizes unrecognized overflow errors so Pi compacts", async () => { const { registry } = await setup({ respond: cascade(() => ({ status: 400, json: { error: { message: "Requested 300000 tokens but this free model's context window is 262144.", }, }, })), }); const message = await ask(registry, "kilo/minimax/minimax-m3:free"); expect(isContextOverflow(message)).toBe(false); const runner = session?.session.extensionRunner; const [handler] = runner.extensions[0].handlers.get("message_end"); const result = await handler( { type: "message_end", message }, runner.createContext(), ); expect(result?.message?.errorMessage).toMatch(/^context_length_exceeded: /); expect(isContextOverflow(result.message)).toBe(true); }); }); const NVIDIA_MODELS = "/nvidia/v1/models"; const NVIDIA_CHAT = "/nvidia/v1/chat/completions"; const ZAI_CHAT = "/zai/api/paas/v4/chat/completions"; const HETZNER_MODELS = "/hetzner/api/v1/models"; const HETZNER_CHAT = "/hetzner/api/v1/chat/completions"; const nvidiaLive = { data: [ { id: "z-ai/glm-5.3", object: "model" }, { id: "moonshotai/kimi-k2.6", object: "model" }, { id: "nvidia/nemotron-3-ultra-550b-a55b", object: "model" }, { id: "vendor/unknown-model", object: "model" }, ], }; const hetznerLive = { data: [ { id: "Qwen/Qwen3.6-35B-A3B-FP8", max_model_len: 131_072 }, { id: "Vendor/New-Model", max_model_len: 65_536 }, ], }; const addedBackends: Responder = (request) => { if (request.path === KILO_MODELS) return { status: 200, json: kiloCatalog }; if (request.path === NVIDIA_MODELS) return { status: 200, json: nvidiaLive }; if (request.path === HETZNER_MODELS) return request.headers.authorization === "Bearer hz-value" ? { status: 200, json: hetznerLive } : { status: 401, json: { error: "unauthorized" } }; if (request.path === NVIDIA_CHAT) return openAIText("nvidia says hi"); if (request.path === ZAI_CHAT) return openAIText("zai says hi"); if (request.path === HETZNER_CHAT) return openAIText("hetzner says hi"); return undefined; }; const idsOf = (registry: Registry, backend: string) => grtsIds(registry).filter((id) => id.startsWith(`${backend}/`)); describe("gratis added backends", () => { it("keeps only curated free NVIDIA models that are live, in curated rank", async () => { const { registry } = await setup({ respond: addedBackends, env: { NVIDIA_API_KEY: "nv-value" }, }); expect(backendsOf(registry)).toEqual(["nvidia", "kilo"]); expect(idsOf(registry, "nvidia")).toEqual([ "nvidia/z-ai/glm-5.3", "nvidia/moonshotai/kimi-k2.6", ]); }); it("falls back to the curated NVIDIA list when the live list is unreachable", async () => { const { registry } = await setup({ respond: (request) => request.path === NVIDIA_MODELS ? { status: 503, json: { error: "down" } } : addedBackends(request), env: { NVIDIA_API_KEY: "nv-value" }, }); expect(idsOf(registry, "nvidia")).toEqual([ "nvidia/moonshotai/kimi-k3", "nvidia/z-ai/glm-5.3", "nvidia/moonshotai/kimi-k2.6", "nvidia/z-ai/glm-5.3-flash", "nvidia/nvidia/nemotron-3.5-lightning-30b-a3b", "nvidia/openai/gpt-oss-20b", ]); }); it("streams a pinned NVIDIA model with the NVIDIA key", async () => { const { registry, upstream } = await setup({ respond: addedBackends, env: { GRATIS_NVIDIA_TOKEN: "nvg-value", NVIDIA_API_KEY: "nv-value" }, }); const message = await ask(registry, "nvidia/z-ai/glm-5.3"); expect(message.content).toEqual([{ type: "text", text: "nvidia says hi" }]); const [chat] = chats(upstream.requests, NVIDIA_CHAT); expect(chat?.body?.model).toBe("z-ai/glm-5.3"); expect(chat?.headers.authorization).toBe("Bearer nvg-value"); }); it("registers exactly the free Z.ai Flash models on the general endpoint", async () => { const { registry, upstream } = await setup({ respond: addedBackends, env: { ZAI_API_KEY: "zai-value" }, }); expect(idsOf(registry, "zai")).toEqual([ "zai/glm-4.7-flash", "zai/glm-4.5-flash", "zai/glm-4.6v-flash", ]); const message = await ask(registry, "zai/glm-4.7-flash"); expect(message.content).toEqual([{ type: "text", text: "zai says hi" }]); const posts = upstream.requests.filter((r) => r.method === "POST"); expect(posts.map((r) => r.path)).toEqual([ZAI_CHAT]); expect(posts.some((r) => r.path.includes("/coding/"))).toBe(false); expect(posts[0]?.headers.authorization).toBe("Bearer zai-value"); }); it("follows Hetzner's keyed live list, keeping curated metadata for known ids", async () => { const { registry, upstream } = await setup({ respond: addedBackends, env: { HETZNER_INFERENCE_API_KEY: "hz-value" }, }); expect(idsOf(registry, "hetzner")).toEqual([ "hetzner/Qwen/Qwen3.6-35B-A3B-FP8", "hetzner/Vendor/New-Model", ]); expect( registry.find("grts", "hetzner/Qwen/Qwen3.6-35B-A3B-FP8"), ).toMatchObject({ contextWindow: 131_072, maxTokens: 65_536, input: ["text", "image"], }); expect(registry.find("grts", "hetzner/Vendor/New-Model")).toMatchObject({ contextWindow: 65_536, maxTokens: 8_192, input: ["text"], }); const discovery = upstream.requests.find((r) => r.path === HETZNER_MODELS); expect(discovery?.headers.authorization).toBe("Bearer hz-value"); const message = await ask(registry, "hetzner/Qwen/Qwen3.6-35B-A3B-FP8"); expect(message.content).toEqual([ { type: "text", text: "hetzner says hi" }, ]); }); it("falls back to the static Hetzner list when the live list is unreachable", async () => { const { registry } = await setup({ respond: (request) => request.path === HETZNER_MODELS ? { status: 500, json: { error: "down" } } : addedBackends(request), env: { GRATIS_HETZNER_TOKEN: "hz-value" }, }); expect(idsOf(registry, "hetzner")).toEqual([ "hetzner/Qwen/Qwen3.6-35B-A3B-FP8", "hetzner/Qwen3.8-27B", ]); }); it("never calls Hetzner or lists it without a key", async () => { const { registry, upstream } = await setup({ respond: addedBackends }); expect(backendsOf(registry)).toEqual(["kilo"]); expect(upstream.requests.some((r) => r.path.startsWith("/hetzner/"))).toBe( false, ); }); it("orders every backend by auto priority when all are keyed", async () => { const { registry } = await setup({ respond: (request) => request.path === OPENROUTER_MODELS ? { status: 200, json: openrouterCatalog } : addedBackends(request), env: { GEMINI_API_KEY: "gm-value", NVIDIA_API_KEY: "nv-value", OPENROUTER_API_KEY: "on-value", GROQ_API_KEY: "gq-value", MISTRAL_API_KEY: "ms-value", ZAI_API_KEY: "zai-value", HETZNER_INFERENCE_API_KEY: "hz-value", }, }); expect(backendsOf(registry)).toEqual([ "google", "zai", "groq", "mistral", "nvidia", "openrouter", "hetzner", "kilo", ]); }); it("cools a Z.ai backend reporting insufficient balance and moves on", async () => { const { registry, upstream } = await setup({ respond: (request) => { if (request.path === KILO_MODELS) return { status: 200, json: { data: [] } }; if (request.path === ZAI_CHAT) return { status: 429, json: { error: { code: "1113", message: "Insufficient balance" } }, }; return addedBackends(request); }, env: { ZAI_API_KEY: "zai-value", HETZNER_INFERENCE_API_KEY: "hz-value", }, }); const first = await ask(registry, "auto"); const second = await ask(registry, "auto"); expect(first.content).toEqual([{ type: "text", text: "hetzner says hi" }]); expect(second.content).toEqual([{ type: "text", text: "hetzner says hi" }]); expect(chats(upstream.requests, ZAI_CHAT)).toHaveLength(2); }); }); const statusOf = async () => { const runner = session?.session.extensionRunner; const command = runner.extensions[0].commands.get("gratis"); await command.handler("", runner.createContext()); return String(session?.events.uiCallsFor("notify").at(-1)?.args[0]); }; describe("gratis served model", () => { it("names the model a virtual router resolved to, under the requested model", async () => { const { registry } = await setup({ respond: (request) => request.path === KILO_CHAT ? openAIText("router says hi", "dots-studio/dots-3-note-preview:free") : kiloOnly(request), }); const pinned = await ask(registry, "kilo/kilo-auto/free"); const auto = await ask(registry, "auto"); expect(pinned).toMatchObject({ model: "kilo/kilo-auto/free", responseModel: "kilo/dots-studio/dots-3-note-preview:free", }); expect(auto).toMatchObject({ model: "auto", responseModel: "kilo/dots-studio/dots-3-note-preview:free", }); }); it("leaves responseModel unset when a pinned model served itself", async () => { const { registry } = await setup({ respond: (request) => request.path === GROQ_CHAT ? openAIText("groq says hi", "openai/gpt-oss-120b") : kiloOnly(request), env: { GROQ_API_KEY: "gq-value" }, }); const message = await ask(registry, "groq/openai/gpt-oss-120b"); expect(message.model).toBe("groq/openai/gpt-oss-120b"); expect(message.responseModel).toBeUndefined(); }); it("falls back to the hop model when the transport reports none", async () => { const { registry } = await setup({ respond: (request) => request.path === KILO_MODELS ? { status: 200, json: { data: [] } } : everyBackend(request), env: { GEMINI_API_KEY: "gm-value" }, }); const message = await ask(registry, "auto"); expect(message.responseModel).toBe("google/gemini-3.8-flash"); }); }); describe("gratis status command", () => { it("reports backends, missing keys, and that nothing ran yet", async () => { await setup({ respond: kiloOnly, env: { GROQ_API_KEY: "gq-value" } }); const text = await statusOf(); expect(text).toContain("Backends: groq 6, kilo 3"); expect(text).toContain( "No key: google, zai, mistral, nvidia, openrouter, hetzner", ); expect(text).toContain("Cooling: none"); expect(text).toContain("Last: no grts request yet"); }); it("reports the last auto route, what it skipped, and cooling backends", async () => { const { registry } = await setup({ respond: cascade( () => openAIText("kilo says hi", "dots-studio/dots-3:free"), () => rateLimited, ), env: { GROQ_API_KEY: "gq-value" }, }); await ask(registry, "auto"); const text = await statusOf(); expect(text).toMatch( /Last: auto → kilo\/dots-studio\/dots-3:free · ok · 5 hops · \d+s ago/, ); expect(text).toContain("Skipped: groq exhausted (rate)"); expect(text).toMatch(/Cooling: groq \(1m left\)/); }); it("reports a failed pinned route as an error without a served model", async () => { const { registry } = await setup({ respond: cascade(() => rateLimited), }); await ask(registry, "kilo/minimax/minimax-m3:free"); const text = await statusOf(); expect(text).toMatch( /Last: kilo\/minimax\/minimax-m3:free → kilo\/minimax\/minimax-m3:free · error · 1 hop/, ); }); }); describe("gratis performance-aware routing", () => { it("cuts a hop sooner when its model usually starts fast", async () => { let groqCalls = 0; const { registry } = await setup({ respond: cascade( () => openAIText("kilo says hi"), () => (++groqCalls <= 3 ? openAIText("groq says hi") : { hang: true }), ), env: { GROQ_API_KEY: "gq-value" }, firstResponseDeadlineMs: 4_000, }); for (let i = 0; i < 3; i++) await ask(registry, "auto"); const started = Date.now(); const message = await ask(registry, "auto"); expect(message.content).toEqual([{ type: "text", text: "kilo says hi" }]); const waited = Date.now() - started; expect(waited).toBeGreaterThanOrEqual(2_900); expect(waited).toBeLessThan(3_800); }); it("tries a repeatedly failing model after its healthy siblings", async () => { vi.useFakeTimers({ toFake: ["Date"] }); try { const { registry, upstream } = await setup({ respond: cascade( () => openAIText("kilo says hi"), (model) => model === "openai/gpt-oss-120b" ? rateLimited : openAIText("groq says hi"), ), env: { GROQ_API_KEY: "gq-value" }, }); for (let i = 0; i < 3; i++) { await ask(registry, "auto"); vi.setSystemTime(Date.now() + 61_000); } const before = upstream.requests.length; const message = await ask(registry, "auto"); expect(message.content).toEqual([{ type: "text", text: "groq says hi" }]); expect(chatModels(upstream.requests.slice(before))).toEqual([ "groq:llama-3.3-70b-versatile", ]); } finally { vi.useRealTimers(); } }); it("shows measured models in the status command", async () => { const { registry } = await setup({ respond: kiloOnly }); await ask(registry, "auto"); const text = await statusOf(); expect(text).toMatch( /Models:\n {2}kilo\/kilo-auto\/free · first token \d+\.\ds · 0% fail · 1 hop/, ); }); }); const gratisCommand = async (args: string) => { const runner = session?.session.extensionRunner; const command = runner.extensions[0].commands.get("gratis"); await command.handler(args, runner.createContext()); const call = session?.events.uiCallsFor("notify").at(-1); return { text: String(call?.args[0]), level: call?.args[1] }; }; const completions = (prefix: string) => session?.session.extensionRunner.extensions[0].commands .get("gratis") .getArgumentCompletions(prefix); describe("gratis manual cooling", () => { const kiloRouter: Responder = (request) => request.path === KILO_CHAT ? openAIText("kilo says hi", "dots-studio/dots-3:free") : kiloOnly(request); it("asks for a model when nothing has answered yet", async () => { await setup({ respond: kiloRouter }); const reply = await gratisCommand("cool"); expect(reply.level).toBe("warning"); expect(reply.text).toContain("name one: /gratis cool "); }); it("cools the candidate that answered last, naming what it answered as", async () => { const { registry, upstream } = await setup({ respond: kiloRouter }); await ask(registry, "auto"); const reply = await gratisCommand("cool"); const before = upstream.requests.length; await ask(registry, "auto"); expect(reply).toEqual({ level: "info", text: "grts/auto skips kilo/kilo-auto/free for 60 min (it answered as kilo/dots-studio/dots-3:free; gratis can only skip kilo/kilo-auto/free itself)", }); expect(chatModels(upstream.requests.slice(before))).toEqual([ "kilo:openrouter/free", ]); expect((await gratisCommand("")).text).toContain( "Manually cooled: kilo/kilo-auto/free (60m left)", ); }); it("keeps a manually cooled model out of the last-resort pass", async () => { const { registry, upstream } = await setup({ respond: cascade((model) => model === "kilo-auto/free" ? openAIText("unwanted") : rateLimited, ), }); await gratisCommand("cool kilo/kilo-auto/free 30"); const message = await ask(registry, "auto"); await ask(registry, "auto"); expect(message.stopReason).toBe("error"); expect(chatModels(upstream.requests)).not.toContain("kilo:kilo-auto/free"); }); it("ends manual cooldowns on uncool", async () => { const { registry, upstream } = await setup({ respond: kiloRouter }); await gratisCommand("cool kilo/kilo-auto/free"); const one = await gratisCommand("uncool kilo/kilo-auto/free"); await ask(registry, "auto"); await gratisCommand("cool kilo/openrouter/free"); await gratisCommand("cool kilo/minimax/minimax-m3:free"); const all = await gratisCommand("uncool"); const none = await gratisCommand("uncool"); expect(one.text).toBe("kilo/kilo-auto/free is back in grts/auto"); expect(chatModels(upstream.requests)).toEqual(["kilo:kilo-auto/free"]); expect(all.text).toBe("2 manually cooled models are back in grts/auto"); expect(none).toEqual({ level: "warning", text: "No model is manually cooled", }); }); it("still answers when the cooled model is pinned", async () => { const { registry } = await setup({ respond: kiloRouter }); await gratisCommand("cool kilo/kilo-auto/free"); const message = await ask(registry, "kilo/kilo-auto/free"); expect(message.content).toEqual([{ type: "text", text: "kilo says hi" }]); }); it("rejects unknown models, auto, and bad durations", async () => { await setup({ respond: kiloRouter }); expect(await gratisCommand("cool kilo/nope")).toEqual({ level: "warning", text: "kilo/nope is not a current grts model", }); expect((await gratisCommand("cool auto")).text).toBe( "auto is not a current grts model", ); for (const minutes of ["0", "1.5", "abc", "1441"]) expect( (await gratisCommand(`cool kilo/kilo-auto/free ${minutes}`)).text, ).toBe("Minutes must be a whole number from 1 to 1440"); expect((await gratisCommand("warm")).text).toBe( "Usage: /gratis [cool [model] [minutes] | uncool [model]]", ); }); it("autocompletes subcommands, models, and manually cooled models", async () => { await setup({ respond: kiloRouter }); await gratisCommand("cool kilo/openrouter/free"); expect(await completions("c")).toEqual([{ value: "cool", label: "cool" }]); expect(await completions("cool kilo/k")).toEqual([ { value: "cool kilo/kilo-auto/free", label: "kilo/kilo-auto/free" }, ]); expect(await completions("uncool ")).toEqual([ { value: "uncool kilo/openrouter/free", label: "kilo/openrouter/free" }, ]); expect(await completions("cool kilo/kilo-auto/free 3")).toBeNull(); }); }); const ZAI_PRICING_PATH = "/zaidocs/guides/overview/pricing.md"; const MODELS_DEV_PATH = "/modelsdev/api.json"; const freeRow = (name: string) => `| ${name} | Free | Free | Free | Free |`; const pricingPage = (...rows: string[]) => [ "### Text Models", "| Model | Input | Cached Input | Cached Input Storage | Output |", "| :- | :- | :- | :- | :- |", "| GLM-4.7 | $0.6 | $0.11 | Limited-time Free | $2.2 |", ...rows, ].join("\n"); const modelsDev = { zai: { api: "https://api.z.ai/api/paas/v4", models: { "glm-9-flash": { name: "GLM 9 Flash", cost: { input: 0, output: 0 }, limit: { context: 300_000, output: 64_000 }, modalities: { input: ["text"] }, }, "glm-9-pro": { name: "GLM 9 Pro", cost: { input: 2, output: 8 }, limit: { context: 300_000, output: 64_000 }, }, }, }, }; describe("gratis fresh catalogs", () => { it("lists Google models from Pi's catalog with Pi's thinking levels", async () => { const { registry, context } = await setup({ respond: kiloOnly, env: { GEMINI_API_KEY: "gm-value" }, }); const pi = context.modelRegistry.find("google", "gemini-3.7-flash"); const ids = idsOf(registry, "google"); expect(ids[0]).toBe("google/gemini-3.8-flash"); expect(ids).toContain("google/gemini-3.6-flash"); expect(ids.some((id) => /latest|preview/.test(id))).toBe(false); expect( registry.find("grts", "google/gemini-3.7-flash")?.thinkingLevelMap, ).toEqual(pi?.thinkingLevelMap); }); it("shows a model added to Pi's catalog without a gratis update", async () => { const { registry, context } = await setup({ respond: kiloOnly, env: { GROQ_API_KEY: "gq-value" }, }); expect(idsOf(registry, "groq")).not.toContain("groq/brand-new-model"); context.modelRegistry.registerProvider("groq", { baseUrl: "https://api.groq.com/openai/v1", apiKey: "gq-value", api: "openai-completions", models: [ { id: "brand-new-model", name: "Brand New", reasoning: false, input: ["text"], cost: { input: 0.1, output: 0.1, cacheRead: 0, cacheWrite: 0 }, contextWindow: 131_072, maxTokens: 16_384, }, ], }); expect(idsOf(registry, "groq")).toEqual(["groq/brand-new-model"]); expect(registry.find("grts", "groq/brand-new-model")?.cost.input).toBe(0); }); it("follows Z.ai's pricing page, confirming unknown free ids on models.dev", async () => { const { registry, upstream } = await setup({ respond: (request) => { if (request.path === ZAI_PRICING_PATH) return { status: 200, text: pricingPage( freeRow("GLM-4.7-Flash"), freeRow("GLM-9-Flash"), freeRow("GLM-9-Pro"), ), }; if (request.path === MODELS_DEV_PATH) return { status: 200, json: modelsDev }; return kiloOnly(request); }, env: { ZAI_API_KEY: "zai-value" }, }); expect(idsOf(registry, "zai")).toEqual([ "zai/glm-4.7-flash", "zai/glm-9-flash", ]); expect(registry.find("grts", "zai/glm-9-flash")).toMatchObject({ baseUrl: "https://api.z.ai/api/paas/v4", contextWindow: 300_000, }); expect( upstream.requests.filter((request) => request.path === MODELS_DEV_PATH), ).toHaveLength(1); }); it("lists no Z.ai models when the page lists none as Free", async () => { const { registry } = await setup({ respond: (request) => request.path === ZAI_PRICING_PATH ? { status: 200, text: pricingPage() } : kiloOnly(request), env: { ZAI_API_KEY: "zai-value" }, }); expect(idsOf(registry, "zai")).toEqual([]); }); it("keeps the last confirmed Z.ai list when the page becomes unreachable", async () => { let pageUp = true; const { registry } = await setup({ respond: (request) => request.path === ZAI_PRICING_PATH ? pageUp ? { status: 200, text: pricingPage(freeRow("GLM-4.5-Flash")) } : { status: 503, json: { error: "down" } } : kiloOnly(request), env: { ZAI_API_KEY: "zai-value" }, }); expect(idsOf(registry, "zai")).toEqual(["zai/glm-4.5-flash"]); pageUp = false; await registry.refresh({ allowNetwork: true, providers: ["grts"] }); expect(idsOf(registry, "zai")).toEqual(["zai/glm-4.5-flash"]); }); }); describe("gratis learning across time", () => { it("keeps a twice-rejected model behind its siblings after the cooldowns end", async () => { vi.useFakeTimers({ toFake: ["Date"] }); try { const { registry, upstream } = await setup({ respond: kiloOnly }); await gratisCommand("cool kilo/kilo-auto/free"); await gratisCommand("cool kilo/kilo-auto/free"); vi.setSystemTime(Date.now() + 61 * 60_000); await ask(registry, "auto"); expect(chatModels(upstream.requests)).toEqual(["kilo:openrouter/free"]); } finally { vi.useRealTimers(); } }); it("still applies a demotion learned before a restart", async () => { const stateDir = mkdtempSync(path.join(tmpdir(), "gratis-routing-")); const routingStatePath = path.join(stateDir, "routing.json"); const respond = cascade( () => openAIText("kilo says hi"), (model) => model === "openai/gpt-oss-120b" ? rateLimited : openAIText("groq says hi"), ); vi.useFakeTimers({ toFake: ["Date"] }); try { const first = await setup({ respond, env: { GROQ_API_KEY: "gq-value" }, routingStatePath, }); for (let i = 0; i < 3; i++) { await ask(first.registry, "auto"); vi.setSystemTime(Date.now() + 61_000); } await session?.session.extensionRunner.emit({ type: "session_shutdown", reason: "quit", }); session?.dispose(); session = undefined; await upstream?.close(); upstream = undefined; const second = await setup({ respond, env: { GROQ_API_KEY: "gq-value" }, routingStatePath, }); // Wait for the background load, then send exactly one request: a second // request would skip the model through its fresh 429 cooldown anyway. await expect .poll(statusOf) .toMatch(/groq\/openai\/gpt-oss-120b · .*demoted/); const before = second.upstream.requests.length; await ask(second.registry, "auto"); expect(chatModels(second.upstream.requests.slice(before))[0]).toBe( "groq:llama-3.3-70b-versatile", ); } finally { vi.useRealTimers(); rmSync(stateDir, { recursive: true, force: true }); } }); });