import type { Provider, RefreshModelsContext, SimpleStreamOptions, TranscriptContext, } from "@earendil-works/pi-ai"; import type { ExtensionAPI, ExtensionContext, } from "@earendil-works/pi-coding-agent"; import { AUTO_CANDIDATES_PER_BACKEND, AUTO_ID, AUTO_MIN_CONTEXT_WINDOW, BACKEND_INFO, BACKENDS, type Backend, backendOf, DISCOVERY_NEEDS_KEY, DISCOVERY_URL, fromPiCatalog, type GratisModel, LIVE_BACKENDS, type LiveBackend, liveModelIds, MODELS_DEV_URL, modelsDevZaiEntry, PI_CATALOG_BACKENDS, type PiCatalogBackend, PROVIDER, parseCatalog, parseZaiPricing, staticModels, unknownZaiIds, ZAI_PRICING_URL, type ZaiEntry, zaiFreeModels, } from "./src/catalog.ts"; import { registerGratisCommand } from "./src/command.ts"; import { closeDebug, dbg, span } from "./src/debug.ts"; import { createRouter, isUnrecognizedOverflow, OVERFLOW_PREFIX, } from "./src/router.ts"; import { loadRoutingState, routingStatePath, saveRoutingState, } from "./src/state.ts"; const isLive = (backend: Backend | undefined): backend is LiveBackend => (LIVE_BACKENDS as readonly string[]).includes(backend ?? ""); const DISCOVERY_TIMEOUT_MS = 10_000; export interface GratisOptions { /** Test seam: redirect upstream URLs to local fakes. */ resolveUrl?: (url: string) => string; /** Test seam: shorten `grts/auto`'s per-hop first-response deadline. */ firstResponseDeadlineMs?: number; /** Test seam: where learned routing persists; defaults to the user state dir. */ routingStatePath?: string; } /** Batch writes: save this long after the last routing change. */ const SAVE_DELAY_MS = 2_000; function envKey(backend: Backend): string | undefined { const info = BACKEND_INFO[backend]; // Empty values carry no key material, so they do not shadow later steps. return ( process.env[info.gratisEnv] || process.env[info.nativeEnv] || undefined ); } export function createGratis(options: GratisOptions = {}) { const resolveUrl = options.resolveUrl ?? ((url: string) => url); return function gratis(pi: ExtensionAPI): void { // Steps 1-2 run at load; step 3 (Pi's credential store) needs a session. const envKeys = new Map(); for (const backend of BACKENDS) { const key = envKey(backend); if (key) envKeys.set(backend, key); } let storeKeys = new Map(); let discovered: Partial> = {}; // Captured at session start: Pi's built-in catalogs refresh at runtime. let registry: ExtensionContext["modelRegistry"] | undefined; let nvidiaLive: Set | undefined; const zaiConfirmed = new Map(); let generation = 0; const key = (backend: Backend): string | undefined | null => envKeys.get(backend) ?? storeKeys.get(backend) ?? (BACKEND_INFO[backend].keyless ? undefined : null); /** * Pi's runtime-refreshed built-in catalog for a backend. Reads one provider's * list, never the registry's full list, which would include grts itself. */ const piCatalog = (backend: PiCatalogBackend): GratisModel[] => { const native = BACKEND_INFO[backend].nativeProvider; try { const list = native ? registry?.getProvider(native)?.getModels() : undefined; return list ? fromPiCatalog( backend, list, backend === "nvidia" ? nvidiaServed() : undefined, ) : []; } catch { return []; } }; /** NVIDIA ids known to be served: live list, else last confirmed, else curated. */ const nvidiaServed = (): ReadonlySet => nvidiaLive ?? new Set( (discovered.nvidia ?? staticModels("nvidia")).map((model) => model.id.slice("nvidia/".length), ), ); const backendModels = (backend: Backend): GratisModel[] => { // Keyless Kilo has no static list: unreachable means absent, not broken. if (backend === "kilo") return discovered.kilo ?? []; if (key(backend) === null) return []; if ((PI_CATALOG_BACKENDS as readonly string[]).includes(backend)) { const fresh = piCatalog(backend as PiCatalogBackend); if (fresh.length > 0) return fresh; } if (isLive(backend)) return discovered[backend] ?? staticModels(backend); return staticModels(backend); }; const pickCandidates = (list: GratisModel[]): GratisModel[] => list .filter((model) => model.contextWindow >= AUTO_MIN_CONTEXT_WINDOW) .slice(0, AUTO_CANDIDATES_PER_BACKEND); const candidates = (backend: Backend): GratisModel[] => pickCandidates(backendModels(backend)); const models = (): GratisModel[] => { // Resolve each backend once; pinned models and auto candidates share it. const byBackend = BACKENDS.map(backendModels); const pinned = byBackend.flat(); const pool = byBackend.flatMap(pickCandidates); if (pool.length === 0) return pinned; // Pi keeps attributes per model entry, so auto advertises the tightest candidate. const auto: GratisModel = { id: AUTO_ID, name: "Auto (free tiers, prompts may be logged)", api: "openai-completions", provider: PROVIDER, baseUrl: "https://grts.invalid", reasoning: false, input: pool.every((model) => model.input.includes("image")) ? ["text", "image"] : ["text"], cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: Math.min(...pool.map((model) => model.contextWindow)), maxTokens: Math.min(...pool.map((model) => model.maxTokens)), }; return [auto, ...pinned]; }; const router = createRouter({ key, candidates, resolveUrl, ...(options.firstResponseDeadlineMs === undefined ? {} : { firstResponseDeadlineMs: options.firstResponseDeadlineMs }), }); const fetchOk = async ( url: string, signal: AbortSignal, init?: RequestInit, ) => { const response = await fetch(resolveUrl(url), { ...init, signal: AbortSignal.any([ signal, AbortSignal.timeout(DISCOVERY_TIMEOUT_MS), ]), }); return response.ok ? response : undefined; }; /** * Z.ai: the first-party pricing page decides what is free; an id gratis does * not know also needs models.dev at cost 0, since a wrong claim bills the user. */ async function discoverZai( signal: AbortSignal, ): Promise { try { const page = await fetchOk(ZAI_PRICING_URL, signal); const pricing = page ? parseZaiPricing(await page.text()) : undefined; if (!pricing) return undefined; const unknown = unknownZaiIds(pricing).filter( (id) => !zaiConfirmed.has(id), ); if (unknown.length > 0) { const response = await fetchOk(MODELS_DEV_URL, signal).catch( () => undefined, ); const payload = response ? await response.json() : undefined; for (const id of unknown) { const entry = payload ? modelsDevZaiEntry(payload, id) : undefined; if (entry) zaiConfirmed.set(id, entry); } } return zaiFreeModels(pricing, zaiConfirmed); } catch { return undefined; } } async function discover( backend: LiveBackend, signal: AbortSignal, ): Promise { if (backend === "zai") return discoverZai(signal); const token = key(backend); if (DISCOVERY_NEEDS_KEY[backend] && !token) return undefined; try { const response = await fetchOk(DISCOVERY_URL[backend], signal, { headers: DISCOVERY_NEEDS_KEY[backend] && token ? { Authorization: `Bearer ${token}` } : {}, }); if (!response) return undefined; const payload = await response.json(); if (backend === "nvidia") nvidiaLive = liveModelIds(payload); return parseCatalog(backend, payload); } catch { return undefined; } } async function refreshModels(context: RefreshModelsContext) { const end = span?.("catalog.refresh", { network: context.allowNetwork }); if (context.stored) { const restored: Partial> = {}; for (const model of context.stored.models) { const backend = backendOf(model); if (!isLive(backend)) continue; const list = restored[backend] ?? []; list.push(model as GratisModel); restored[backend] = list; } const published = await context.publish({ update: () => { discovered = restored; }, }); if (!published) return end?.("finish", { outcome: "offline" }); } if (!context.allowNetwork || context.signal.aborted) return end?.("finish", { outcome: "offline" }); // Unkeyed backends are not fetched; keyless Kilo always refreshes. const fetched = await Promise.all( LIVE_BACKENDS.map((backend) => backend === "kilo" || key(backend) !== null ? discover(backend, context.signal) : undefined, ), ); if (context.signal.aborted) return end?.("finish", { outcome: "aborted" }); const next: Partial> = {}; LIVE_BACKENDS.forEach((backend, index) => { // A failed or skipped fetch keeps the last confirmed list rather than guessing. const models = fetched[index] ?? (backend === "kilo" ? undefined : discovered[backend]); if (models) next[backend] = models; }); // Unreachable Kilo means no Kilo models: visible-but-broken is worse than absent. next.kilo ??= []; await context.publish({ persist: { models: LIVE_BACKENDS.flatMap((backend) => next[backend] ?? []), checkedAt: Date.now(), }, update: () => { discovered = next; }, }); end?.("finish", { outcome: "ok", kilo: next.kilo.length, openrouter: next.openrouter?.length ?? 0, }); } const provider: Provider = { id: PROVIDER, name: "Gratis", auth: { apiKey: { name: "Gratis free tiers", // Keyless Kilo keeps grts configured; empty model lists keep it invisible. resolve: async () => ({ auth: { apiKey: PROVIDER }, source: "gratis", }), }, }, getModels: models, refreshModels, // Pi's agent loop uses streamSimple; raw stream options map onto the same route. stream: (model, context: TranscriptContext, streamOptions) => router.streamSimple( model, context, streamOptions as SimpleStreamOptions | undefined, ), streamSimple: (model, context, streamOptions) => router.streamSimple(model, context, streamOptions), }; pi.registerProvider(provider); // Learned routing survives restarts; startup never waits on the file. const statePath = options.routingStatePath ?? routingStatePath(); let dirty = false; let saveTimer: ReturnType | undefined; const flush = async () => { clearTimeout(saveTimer); saveTimer = undefined; if (!dirty) return; dirty = false; await saveRoutingState(statePath, router.learned.export()).catch( () => undefined, ); }; const loaded = loadRoutingState(statePath).then((state) => router.learned.import(state), ); router.learned.onChange(() => { dirty = true; if (saveTimer) return; saveTimer = setTimeout(() => void flush(), SAVE_DELAY_MS); saveTimer.unref?.(); }); registerGratisCommand(pi, { router, models, backendModels, key }); async function resolveStoreKeys( modelRegistry: ExtensionContext["modelRegistry"], ticket: number, ) { const lookups = BACKENDS.flatMap((backend) => { const native = BACKEND_INFO[backend].nativeProvider; if (!native || envKeys.has(backend)) return []; return [ modelRegistry.getApiKeyForProvider(native).then( (stored) => [backend, stored] as const, () => [backend, undefined] as const, ), ]; }); const next = new Map(); for (const [backend, stored] of await Promise.all(lookups)) if (stored) next.set(backend, stored); if (ticket !== generation) return; dbg?.("keys.resolve", { keyed: envKeys.size + next.size }); const changed = next.size !== storeKeys.size || [...next].some(([backend, value]) => storeKeys.get(backend) !== value); if (!changed) return; storeKeys = next; // Refresh republishes the model set and discovers newly keyed live catalogs. await modelRegistry .refresh({ providers: [PROVIDER] }) .catch(() => undefined); } pi.on("session_start", (_event, ctx) => { dbg?.("session.start", { keyed: envKeys.size }); // No refresh here: a new refresh aborts an in-flight one (Pi's startup // network refresh), and Pi re-snapshots after it anyway. registry = ctx.modelRegistry; // Credential lookup may refresh OAuth tokens; never block startup on it. void resolveStoreKeys(registry, ++generation); }); // Pi compacts and retries only on overflow errors it recognizes. pi.on("message_end", (event) => { const message = event.message; if ( message.role !== "assistant" || message.stopReason !== "error" || message.provider !== PROVIDER || !isUnrecognizedOverflow(message) ) return; return { message: { ...message, errorMessage: `${OVERFLOW_PREFIX}: ${message.errorMessage}`, }, }; }); pi.on("session_shutdown", async () => { dbg?.("session.shutdown"); registry = undefined; generation++; await loaded; await flush(); closeDebug(); }); }; } export default createGratis();