import type { Api, Model } from "@earendil-works/pi-ai"; export const PROVIDER = "grts"; export const AUTO_ID = "auto"; /** * Cascade priority for `grts/auto`, answer quality first; also the display order. * NVIDIA sits behind Mistral: its strongest free models queue past the deadline * (measured twice). Keyless Kilo trails: always there, but its anonymous router * lands on weak models. */ export const BACKENDS = [ "google", "zai", "groq", "mistral", "nvidia", "openrouter", "hetzner", "kilo", ] as const; export type Backend = (typeof BACKENDS)[number]; export type GratisModel = Model; type Input = ("text" | "image")[]; interface BackendInfo { label: string; /** Backend-specific key: step 1 of key resolution. */ gratisEnv: string; /** Native env of the backend: step 2. */ nativeEnv: string; /** Pi provider id for credential-store lookup: step 3. */ nativeProvider?: string; /** Keyless backends register without any key. */ keyless: boolean; /** Advisory cooldowns after observed 429 or quota exhaustion. */ rateCooldownMs: number; quotaCooldownMs: number; } const MINUTE = 60_000; const HOUR = 60 * MINUTE; export const BACKEND_INFO: Record = { kilo: { label: "Kilo", gratisEnv: "GRATIS_KILO_TOKEN", nativeEnv: "KILO_API_KEY", keyless: true, rateCooldownMs: 5 * MINUTE, quotaCooldownMs: HOUR, }, google: { label: "Google", gratisEnv: "GRATIS_GOOGLE_TOKEN", nativeEnv: "GEMINI_API_KEY", nativeProvider: "google", keyless: false, rateCooldownMs: MINUTE, quotaCooldownMs: HOUR, }, nvidia: { label: "NVIDIA", gratisEnv: "GRATIS_NVIDIA_TOKEN", nativeEnv: "NVIDIA_API_KEY", nativeProvider: "nvidia", keyless: false, rateCooldownMs: MINUTE, quotaCooldownMs: HOUR, }, openrouter: { label: "OpenRouter", gratisEnv: "GRATIS_OPENROUTER_TOKEN", nativeEnv: "OPENROUTER_API_KEY", nativeProvider: "openrouter", keyless: false, rateCooldownMs: MINUTE, quotaCooldownMs: HOUR, }, groq: { label: "Groq", gratisEnv: "GRATIS_GROQ_TOKEN", nativeEnv: "GROQ_API_KEY", nativeProvider: "groq", keyless: false, rateCooldownMs: MINUTE, quotaCooldownMs: HOUR, }, mistral: { label: "Mistral", gratisEnv: "GRATIS_MISTRAL_TOKEN", nativeEnv: "MISTRAL_API_KEY", nativeProvider: "mistral", keyless: false, rateCooldownMs: MINUTE, quotaCooldownMs: HOUR, }, zai: { label: "Z.ai", gratisEnv: "GRATIS_ZAI_TOKEN", nativeEnv: "ZAI_API_KEY", nativeProvider: "zai", keyless: false, rateCooldownMs: MINUTE, quotaCooldownMs: HOUR, }, hetzner: { label: "Hetzner", gratisEnv: "GRATIS_HETZNER_TOKEN", nativeEnv: "HETZNER_INFERENCE_API_KEY", // Pi has no built-in Hetzner provider; a models.json `hetzner` entry supplies one. nativeProvider: "hetzner", keyless: false, rateCooldownMs: MINUTE, quotaCooldownMs: HOUR, }, }; export const KILO_BASE_URL = "https://api.kilo.ai/api/gateway"; export const OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1"; const OPENROUTER_ANTHROPIC_BASE_URL = "https://openrouter.ai/api"; const GOOGLE_BASE_URL = "https://generativelanguage.googleapis.com/v1beta"; const GROQ_BASE_URL = "https://api.groq.com/openai/v1"; const MISTRAL_BASE_URL = "https://api.mistral.ai"; const NVIDIA_BASE_URL = "https://integrate.api.nvidia.com/v1"; // General endpoint only: Coding Plan endpoints (`/api/coding/paas/v4`, `/api/anthropic`) spend plan quota. const ZAI_BASE_URL = "https://api.z.ai/api/paas/v4"; const HETZNER_BASE_URL = "https://inference.hetzner.com/api/v1"; /** Backends whose catalog is refreshed from a live models endpoint. */ export type LiveBackend = "kilo" | "openrouter" | "nvidia" | "hetzner" | "zai"; export const LIVE_BACKENDS: readonly LiveBackend[] = [ "kilo", "openrouter", "nvidia", "hetzner", "zai", ]; /** Live catalog endpoints; only Hetzner's requires the backend key. */ export const DISCOVERY_URL: Record, string> = { kilo: `${KILO_BASE_URL}/models`, openrouter: `${OPENROUTER_BASE_URL}/models`, nvidia: `${NVIDIA_BASE_URL}/models`, hetzner: `${HETZNER_BASE_URL}/models`, }; export const DISCOVERY_NEEDS_KEY: Record< Exclude, boolean > = { kilo: false, openrouter: false, nvidia: false, hetzner: true, }; /** * Gratis-maintained free set. Discovery keeps a live model only when its id * matches one of these and its published pricing is zero. */ const VIRTUAL_MODELS: Partial> = { kilo: ["kilo-auto/free", "openrouter/free"], openrouter: ["openrouter/free"], }; /** Applied only when a catalog publishes no value. */ const FALLBACK_CONTEXT_WINDOW = 32_768; const FALLBACK_MAX_TOKENS = 8_192; /** `grts/auto` candidates: top-ranked models per backend with a usable context. */ export const AUTO_CANDIDATES_PER_BACKEND = 4; export const AUTO_MIN_CONTEXT_WINDOW = 131_072; const FREE_COST = { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }; const CAVEAT = "free tier, prompts may be logged"; const OPENROUTER_COMPAT = { supportsDeveloperRole: false, thinkingFormat: "openrouter", } as const; /** * Pi sends `reasoning: { effort: "none" }` for thinking off unless off is * unsupported; routers and free pools may land on models that reject that * ("Reasoning is mandatory"), so thinking off sends no reasoning setting. */ const NO_FORCED_OFF = { off: null } as const; /** vLLM-style OpenAI servers (NVIDIA NIM, Hetzner): plain max_tokens, no store, system role only. */ const VLLM_COMPAT = { supportsStore: false, supportsDeveloperRole: false, supportsReasoningEffort: false, maxTokensField: "max_tokens", supportsStrictMode: false, } as const; /** Pi's own Z.ai GLM compatibility settings. */ const ZAI_COMPAT = { supportsStore: false, supportsDeveloperRole: false, supportsReasoningEffort: false, maxTokensField: "max_tokens", thinkingFormat: "zai", supportsStrictMode: true, zaiToolStream: true, } as const; const GOOGLE_THINKING = { off: null, minimal: "minimal", low: "low", medium: "medium", high: "high", xhigh: null, max: null, } as const; /** Newer Gemini Flash models reject MINIMAL, the level Pi uses for thinking off. */ const GOOGLE_THINKING_NO_MINIMAL = { ...GOOGLE_THINKING, minimal: null, } as const; const GROQ_GPT_OSS_THINKING = { off: null, minimal: null, low: "low", medium: "medium", high: "high", xhigh: null, max: null, } as const; /** One curated or confirmed catalog entry before gratis naming and zero cost. */ export type ZaiEntry = Entry; interface Entry { id: string; name: string; api: Api; baseUrl: string; contextWindow: number; maxTokens: number; input: Input; reasoning: boolean; compat?: GratisModel["compat"]; thinkingLevelMap?: GratisModel["thinkingLevelMap"]; } export function gratisModel(backend: Backend, entry: Entry): GratisModel { const model: GratisModel = { id: `${backend}/${entry.id}`, name: `${BACKEND_INFO[backend].label} ${entry.name} (${CAVEAT})`, api: entry.api, provider: PROVIDER, baseUrl: entry.baseUrl, reasoning: entry.reasoning, input: entry.input, cost: { ...FREE_COST }, contextWindow: entry.contextWindow, maxTokens: entry.maxTokens, }; if (entry.compat) model.compat = entry.compat; if (entry.thinkingLevelMap) model.thinkingLevelMap = entry.thinkingLevelMap; return model; } export function backendOf(model: Pick): Backend | undefined { const prefix = model.id.slice(0, model.id.indexOf("/")); return (BACKENDS as readonly string[]).includes(prefix) ? (prefix as Backend) : undefined; } export function upstreamId(model: Pick): string { return model.id.slice(model.id.indexOf("/") + 1); } const google = ( id: string, name: string, thinkingLevelMap?: Entry["thinkingLevelMap"], ): Entry => ({ id, name, api: "google-generative-ai", baseUrl: GOOGLE_BASE_URL, contextWindow: 1_048_576, maxTokens: 65_536, input: ["text", "image"], reasoning: true, ...(thinkingLevelMap ? { thinkingLevelMap } : {}), }); const groq = ( id: string, name: string, maxTokens: number, thinkingLevelMap?: Entry["thinkingLevelMap"], ): Entry => ({ id, name, api: "openai-completions", baseUrl: GROQ_BASE_URL, contextWindow: 131_072, maxTokens, input: ["text"], reasoning: thinkingLevelMap !== undefined, compat: { supportsStrictMode: true }, ...(thinkingLevelMap ? { thinkingLevelMap } : {}), }); const mistral = ( id: string, name: string, contextWindow: number, maxTokens: number, input: Input, reasoning: boolean, ): Entry => ({ id, name, api: "mistral-conversations", baseUrl: MISTRAL_BASE_URL, contextWindow, maxTokens, input, reasoning, }); const openai = ( baseUrl: string, compat: Entry["compat"], id: string, name: string, contextWindow: number, maxTokens: number, input: Input, ): Entry => ({ id, name, api: "openai-completions", baseUrl, contextWindow, maxTokens, input, reasoning: true, compat, }); const nvidia = openai.bind(undefined, NVIDIA_BASE_URL, VLLM_COMPAT); const zai = openai.bind(undefined, ZAI_BASE_URL, ZAI_COMPAT); const hetzner = openai.bind(undefined, HETZNER_BASE_URL, VLLM_COMPAT); /** Curated static catalogs, ranked. Stale entries are fixed here, never by user config. */ const STATIC: Record, readonly Entry[]> = { // Only models Pi's NVIDIA catalog prices at 0; priced ones (Nemotron 3 Super/Ultra) stay out. nvidia: [ nvidia("moonshotai/kimi-k3", "Kimi K3", 1_048_576, 131_072, [ "text", "image", ]), nvidia("z-ai/glm-5.3", "GLM 5.3", 1_000_000, 131_072, ["text"]), nvidia("moonshotai/kimi-k2.6", "Kimi K2.6", 262_144, 262_144, [ "text", "image", ]), nvidia("z-ai/glm-5.3-flash", "GLM 5.3 Flash", 1_000_000, 131_072, [ "text", "image", ]), nvidia( "nvidia/nemotron-3.5-lightning-30b-a3b", "Nemotron 3.5 Lightning", 262_144, 262_144, ["text"], ), nvidia("openai/gpt-oss-20b", "GPT OSS 20B", 131_072, 32_768, ["text"]), ], // Exactly the models Z.ai prices Free; "flash" alone is not proof (glm-5.3-flash is paid). zai: [ zai("glm-4.7-flash", "GLM 4.7 Flash", 204_800, 131_072, ["text"]), zai("glm-4.5-flash", "GLM 4.5 Flash", 131_072, 98_304, ["text"]), zai("glm-4.6v-flash", "GLM 4.6V Flash", 128_000, 32_768, ["text", "image"]), ], // Fallback only: the keyed live models list replaces it when reachable. hetzner: [ hetzner("Qwen/Qwen3.6-35B-A3B-FP8", "Qwen 3.6 35B", 262_144, 65_536, [ "text", "image", ]), hetzner("Qwen3.8-27B", "Qwen 3.8 27B", 262_144, 65_536, ["text", "image"]), ], // Concrete ids only: `-latest` aliases move to models whose thinking levels differ. google: [ google("gemini-3.8-flash", "Gemini 3.8 Flash", GOOGLE_THINKING_NO_MINIMAL), google("gemini-3.7-flash", "Gemini 3.7 Flash", GOOGLE_THINKING_NO_MINIMAL), google("gemini-3.5-flash", "Gemini 3.5 Flash", GOOGLE_THINKING), google("gemini-3.5-flash-lite", "Gemini 3.5 Flash-Lite", GOOGLE_THINKING), ], // Fallback only: live discovery replaces it when reachable. openrouter: [ { id: "openrouter/free", name: "Free Models Router", api: "openai-completions", baseUrl: OPENROUTER_BASE_URL, contextWindow: 200_000, maxTokens: FALLBACK_MAX_TOKENS, input: ["text", "image"], reasoning: true, compat: OPENROUTER_COMPAT, thinkingLevelMap: NO_FORCED_OFF, }, ], groq: [ groq("openai/gpt-oss-120b", "GPT OSS 120B", 65_536, GROQ_GPT_OSS_THINKING), groq("llama-3.3-70b-versatile", "Llama 3.3 70B", 32_768), groq("openai/gpt-oss-20b", "GPT OSS 20B", 65_536, GROQ_GPT_OSS_THINKING), groq("llama-3.1-8b-instant", "Llama 3.1 8B", 131_072), ], mistral: [ mistral("devstral-latest", "Devstral", 262_144, 262_144, ["text"], false), mistral( "mistral-medium-latest", "Mistral Medium", 262_144, 262_144, ["text", "image"], true, ), mistral( "mistral-small-latest", "Mistral Small", 256_000, 256_000, ["text", "image"], true, ), mistral( "mistral-large-latest", "Mistral Large", 262_144, 262_144, ["text", "image"], false, ), mistral("codestral-latest", "Codestral", 256_000, 4_096, ["text"], false), ], }; export function staticModels(backend: Backend): GratisModel[] { if (backend === "kilo") return []; return STATIC[backend].map((entry) => gratisModel(backend, entry)); } const liveIds = (payload: unknown): string[] => { const data = (payload as { data?: unknown } | undefined)?.data; if (!Array.isArray(data)) throw new Error("catalog payload has no data"); return data.flatMap((raw) => typeof raw?.id === "string" ? [raw.id as string] : [], ); }; /** NVIDIA publishes bare ids: keep curated free models that are live. */ export function parseNvidiaCatalog(payload: unknown): GratisModel[] { const live = new Set(liveIds(payload)); return staticModels("nvidia").filter((model) => live.has(upstreamId(model))); } /** * Hetzner's list is definitive and every model is free while experimental. * Known ids keep curated metadata; new ids use the published context length. */ export function parseHetznerCatalog(payload: unknown): GratisModel[] { const data = (payload as { data?: unknown } | undefined)?.data; if (!Array.isArray(data)) throw new Error("catalog payload has no data"); const curated = new Map( staticModels("hetzner").map((model) => [upstreamId(model), model]), ); return data.flatMap((raw: { id?: unknown; max_model_len?: unknown }) => { if (typeof raw?.id !== "string") return []; const known = curated.get(raw.id); const contextWindow = positive(raw.max_model_len) ?? known?.contextWindow ?? FALLBACK_CONTEXT_WINDOW; if (known) return [{ ...known, contextWindow }]; return [ gratisModel( "hetzner", hetzner( raw.id, raw.id, contextWindow, Math.min(FALLBACK_MAX_TOKENS, contextWindow), ["text"], ), ), ]; }); } export function parseCatalog( backend: Exclude, payload: unknown, ): GratisModel[] { if (backend === "nvidia") return parseNvidiaCatalog(payload); if (backend === "hetzner") return parseHetznerCatalog(payload); return parseLiveCatalog(backend, payload); } /** Model ids the NVIDIA public list currently serves, for intersecting Pi's catalog. */ export function liveModelIds(payload: unknown): Set { return new Set(liveIds(payload)); } /** Z.ai's first-party pricing page, served as markdown. */ export const ZAI_PRICING_URL = "https://docs.z.ai/guides/overview/pricing.md"; /** models.dev confirms cost and limits for Z.ai ids gratis does not know. */ export const MODELS_DEV_URL = "https://models.dev/api.json"; export interface ZaiPricing { /** Lowercased model ids whose input and output are both listed as Free. */ free: string[]; /** Free ids listed under Vision Models. */ vision: Set; } /** * Parse the pricing page's model tables. Returns undefined when no model table * is found (format changed), so callers keep their last confirmed list. */ export function parseZaiPricing(markdown: string): ZaiPricing | undefined { let section = ""; let tables = 0; const free: string[] = []; const vision = new Set(); for (const line of markdown.split("\n")) { if (line.startsWith("#")) section = line; const cells = line.split("|").map((cell) => cell.trim()); // | Model | Input | Cached Input | Cached Input Storage | Output | if (cells.length !== 7) continue; const [, name = "", input, , , output] = cells; if (name === "Model" && input === "Input" && output === "Output") { tables++; continue; } const id = name.toLowerCase(); if (input !== "Free" || output !== "Free" || !/^glm-[a-z0-9.-]+$/.test(id)) continue; free.push(id); if (/vision/i.test(section)) vision.add(id); } return tables > 0 ? { free, vision } : undefined; } /** A models.dev `zai` entry confirming cost 0 on the general endpoint. */ export function modelsDevZaiEntry( payload: unknown, id: string, ): Entry | undefined { const provider = (payload as { zai?: { api?: unknown; models?: unknown } }) ?.zai; if (provider?.api !== ZAI_BASE_URL) return undefined; const raw = (provider.models as Record | undefined)?.[ id ]; if (raw?.cost?.input !== 0 || raw.cost?.output !== 0) return undefined; const inputs = strings(raw.modalities?.input); return zai( id, typeof raw.name === "string" ? raw.name : id, positive(raw.limit?.context) ?? FALLBACK_CONTEXT_WINDOW, positive(raw.limit?.output) ?? FALLBACK_MAX_TOKENS, inputs.includes("image") ? ["text", "image"] : ["text"], ); } interface LiveModelsDev { name?: unknown; cost?: { input?: unknown; output?: unknown }; limit?: { context?: unknown; output?: unknown }; modalities?: { input?: unknown }; } /** * Z.ai models the pricing page lists as Free: known ids keep curated * metadata; unknown ids need a models.dev confirmation (`confirmed`). */ export function zaiFreeModels( pricing: ZaiPricing, confirmed: ReadonlyMap, ): GratisModel[] { const known = new Map(STATIC.zai.map((entry) => [entry.id, entry])); return pricing.free.flatMap((id) => { const entry = known.get(id) ?? confirmed.get(id); return entry ? [gratisModel("zai", entry)] : []; }); } /** Z.ai free ids that need models.dev before they can be listed. */ export function unknownZaiIds(pricing: ZaiPricing): string[] { const known = new Set(STATIC.zai.map((entry) => entry.id)); return pricing.free.filter((id) => !known.has(id)); } /** Backends whose models come from Pi's runtime-refreshed built-in catalog. */ export type PiCatalogBackend = "google" | "groq" | "mistral" | "nvidia"; export const PI_CATALOG_BACKENDS: readonly PiCatalogBackend[] = [ "google", "groq", "mistral", "nvidia", ]; const GEMINI_FLASH = /^gemini-(\d+(?:\.\d+)?)-flash(-lite)?$/; const NOT_CHAT = /safeguard|guard|whisper|tts|embed|ocr|voxtral|moderation/i; const version = (id: string) => (/(\d+(?:\.\d+)?)/.exec(id)?.[1] ?? "0") .split(".") .map(Number) .reduce((sum, part, index) => sum + part / 1000 ** index, 0); /** Pi's free rule per backend; `live` narrows NVIDIA to ids its public list serves. */ function piFree( backend: PiCatalogBackend, model: Model, live: ReadonlySet | undefined, ): boolean { if (NOT_CHAT.test(model.id)) return false; if (backend === "google") return GEMINI_FLASH.test(model.id); if (backend === "mistral") return model.id.endsWith("-latest"); if (backend === "nvidia") return ( model.cost.input === 0 && model.cost.output === 0 && (live === undefined || live.has(model.id)) ); return true; } /** * Gratis models from Pi's built-in catalog for one backend, ranked: gratis's * known preferences first, then newer versions, then larger contexts. * Empty when Pi's catalog has no match, so callers fall back to the static list. */ export function fromPiCatalog( backend: PiCatalogBackend, catalog: readonly Model[], live?: ReadonlySet, ): GratisModel[] { // Gemini Flash versions are ordered, so newest wins; elsewhere ids carry no // comparable version and gratis's known preferences lead. const preference = backend === "google" ? [] : STATIC[backend].map((entry) => entry.id); const rank = (id: string) => { const index = preference.indexOf(id); return index === -1 ? preference.length : index; }; const flashLite = (id: string) => Number(id.endsWith("-lite")); return catalog .filter((model) => piFree(backend, model, live)) .sort( (a, b) => rank(a.id) - rank(b.id) || version(b.id) - version(a.id) || flashLite(a.id) - flashLite(b.id) || b.contextWindow - a.contextWindow, ) .map((model) => gratisModel(backend, { id: model.id, name: model.name, api: model.api, baseUrl: model.baseUrl, contextWindow: model.contextWindow, maxTokens: model.maxTokens, input: model.input, reasoning: model.reasoning, ...(model.compat ? { compat: model.compat } : {}), ...(model.thinkingLevelMap ? { thinkingLevelMap: model.thinkingLevelMap } : {}), }), ); } interface LiveModel { id?: unknown; name?: unknown; context_length?: unknown; pricing?: { prompt?: unknown; completion?: unknown }; isFree?: unknown; architecture?: { input_modalities?: unknown; output_modalities?: unknown }; top_provider?: { context_length?: unknown; max_completion_tokens?: unknown; }; supported_parameters?: unknown; } const positive = (value: unknown): number | undefined => typeof value === "number" && Number.isFinite(value) && value > 0 ? Math.floor(value) : undefined; const zeroPrice = (value: unknown): boolean => (typeof value === "string" || typeof value === "number") && Number(value) === 0; const strings = (value: unknown): string[] => Array.isArray(value) ? value.filter((item): item is string => typeof item === "string") : []; /** * Parse an OpenRouter-shaped `/models` payload into ranked free models: * virtual routers first, then upstream order. Anything unexpected is skipped, * so arbitrary churn and empty pools yield fewer models, never errors. */ export function parseLiveCatalog( backend: "kilo" | "openrouter", payload: unknown, ): GratisModel[] { const data = (payload as { data?: unknown } | undefined)?.data; if (!Array.isArray(data)) throw new Error("catalog payload has no data"); const virtual = VIRTUAL_MODELS[backend] ?? []; const models: GratisModel[] = []; for (const raw of data as LiveModel[]) { if (typeof raw?.id !== "string") continue; const id = raw.id; const isVirtualModel = virtual.includes(id); if (!isVirtualModel && !id.endsWith(":free")) continue; if (raw.isFree === false) continue; if (!zeroPrice(raw.pricing?.prompt) || !zeroPrice(raw.pricing?.completion)) continue; const params = strings(raw.supported_parameters); if (!params.includes("tools")) continue; const outputs = strings(raw.architecture?.output_modalities); if (outputs.length > 0 && !outputs.includes("text")) continue; const inputs = strings(raw.architecture?.input_modalities); const input: Input = inputs.includes("image") ? ["text", "image"] : ["text"]; const anthropic = backend === "openrouter" && id.startsWith("anthropic/"); models.push( gratisModel(backend, { id, name: typeof raw.name === "string" ? raw.name : id, api: anthropic ? "anthropic-messages" : "openai-completions", baseUrl: backend === "kilo" ? KILO_BASE_URL : anthropic ? OPENROUTER_ANTHROPIC_BASE_URL : OPENROUTER_BASE_URL, contextWindow: positive(raw.top_provider?.context_length) ?? positive(raw.context_length) ?? FALLBACK_CONTEXT_WINDOW, maxTokens: positive(raw.top_provider?.max_completion_tokens) ?? FALLBACK_MAX_TOKENS, input, reasoning: params.includes("reasoning"), ...(anthropic ? {} : { compat: OPENROUTER_COMPAT, ...(params.includes("reasoning") ? { thinkingLevelMap: NO_FORCED_OFF } : {}), }), }), ); } const rank = (model: GratisModel) => { const index = virtual.indexOf(upstreamId(model)); return index === -1 ? virtual.length : index; }; // Array sort is stable, so equal ranks keep upstream order. return models.sort((a, b) => rank(a) - rank(b)); }