import type { ExtensionAPI } from "@earendil-works/pi-coding-agent"; import { AUTO_ID, BACKENDS, type Backend, type GratisModel, } from "./catalog.ts"; import type { createRouter } from "./router.ts"; const MAX_STATUS_MODELS = 8; const SUBCOMMANDS = ["cool", "uncool"] as const; const DEFAULT_MANUAL_COOL_MINUTES = 60; const MAX_MANUAL_COOL_MINUTES = 24 * 60; export interface CommandDeps { router: ReturnType; /** Every registered grts model, `auto` first when present. */ models(): GratisModel[]; backendModels(backend: Backend): GratisModel[]; /** Resolved key; `null` when the backend is unusable. */ key(backend: Backend): string | undefined | null; } /** `/gratis`: status, plus manual cooling for `grts/auto`. */ export function registerGratisCommand( pi: ExtensionAPI, { router, models, backendModels, key }: CommandDeps, ): void { const routable = () => models() .map((model) => model.id) .filter((id) => id !== AUTO_ID); pi.registerCommand("gratis", { description: "Show grts status; cool [model] [minutes] skips a model in grts/auto; uncool [model] ends that", getArgumentCompletions: (prefix) => { const [subcommand, partial, ...rest] = prefix.split(" "); if (partial === undefined) { const matches = SUBCOMMANDS.filter((name) => name.startsWith(subcommand ?? ""), ); return matches.length ? matches.map((name) => ({ value: name, label: name })) : null; } if (rest.length > 0) return null; const ids = subcommand === "cool" ? routable() : subcommand === "uncool" ? router.status().manual.map(({ model }) => model) : []; const matches = ids.filter((id) => id.startsWith(partial)); return matches.length ? matches.map((id) => ({ value: `${subcommand} ${id}`, label: id })) : null; }, handler: async (args, ctx) => { const [subcommand, ...rest] = args.trim().split(/\s+/).filter(Boolean); if (subcommand === undefined) { ctx.ui.notify(statusText(), "info"); return; } const reply = runSubcommand(subcommand, rest); ctx.ui.notify(reply.text, reply.ok ? "info" : "warning"); }, }); function runSubcommand( subcommand: string, rest: string[], ): { ok: boolean; text: string } { if (subcommand === "uncool") { const [model] = rest; const ended = router.uncool(model); if (ended === 0) return { ok: false, text: model ? `${model} is not manually cooled` : "No model is manually cooled", }; return { ok: true, text: model ? `${model} is back in grts/auto` : `${ended} manually cooled ${ended === 1 ? "model is" : "models are"} back in grts/auto`, }; } if (subcommand !== "cool") return { ok: false, text: "Usage: /gratis [cool [model] [minutes] | uncool [model]]", }; const [named, minutesText] = rest; const last = router.status().last; const model = named ?? last?.candidate; if (!model) return { ok: false, text: "No grts model has answered yet; name one: /gratis cool ", }; if (!routable().includes(model)) return { ok: false, text: `${model} is not a current grts model` }; const minutes = minutesText === undefined ? DEFAULT_MANUAL_COOL_MINUTES : Number(minutesText); if ( !Number.isInteger(minutes) || minutes < 1 || minutes > MAX_MANUAL_COOL_MINUTES ) return { ok: false, text: `Minutes must be a whole number from 1 to ${MAX_MANUAL_COOL_MINUTES}`, }; router.cool(model, minutes * 60_000); const behind = named === undefined && last?.served && last.served !== model && last.candidate === model ? ` (it answered as ${last.served}; gratis can only skip ${model} itself)` : ""; return { ok: true, text: `grts/auto skips ${model} for ${minutes} min${behind}`, }; } function statusText(now = Date.now()): string { const ago = (at: number) => { const seconds = Math.max(0, Math.round((now - at) / 1000)); return seconds < 90 ? `${seconds}s ago` : `${Math.round(seconds / 60)}m ago`; }; const left = (until: number) => `${Math.max(1, Math.ceil((until - now) / 60_000))}m left`; const visible = BACKENDS.map((backend) => ({ backend, count: backendModels(backend).length, })); const shown = visible.filter(({ count }) => count > 0); const unkeyed = BACKENDS.filter((backend) => key(backend) === null); const empty = visible .filter(({ backend, count }) => count === 0 && key(backend) !== null) .map(({ backend }) => backend); const { last, cooling, manual, models: measured } = router.status(); const lines = [ `Backends: ${shown.map(({ backend, count }) => `${backend} ${count}`).join(", ") || "none"}`, ]; if (unkeyed.length > 0) lines.push(`No key: ${unkeyed.join(", ")}`); if (empty.length > 0) lines.push(`No models: ${empty.join(", ")}`); lines.push( `Cooling: ${cooling.map(({ backend, until }) => `${backend} (${left(until)})`).join(", ") || "none"}`, ); if (manual.length > 0) lines.push( `Manually cooled: ${manual.map(({ model, until }) => `${model} (${left(until)})`).join(", ")}`, ); if (!last) lines.push("Last: no grts request yet"); else { const hops = `${last.hops} ${last.hops === 1 ? "hop" : "hops"}`; lines.push( `Last: ${last.requested} → ${last.served ?? "no model"} · ${last.outcome} · ${hops} · ${ago(last.at)}`, ); if (last.notes.length > 0) lines.push(`Skipped: ${last.notes.join(", ")}`); } // Measured from real requests only; persisted across restarts. const shownModels = measured.slice(0, MAX_STATUS_MODELS); if (shownModels.length > 0) lines.push("Models:"); for (const model of shownModels) { const parts = [model.id]; if (model.firstTokenMs !== undefined) parts.push(`first token ${(model.firstTokenMs / 1000).toFixed(1)}s`); if (model.tokensPerSecond !== undefined) parts.push(`${Math.round(model.tokensPerSecond)} tok/s`); parts.push(`${Math.round(model.failureRate * 100)}% fail`); parts.push( `${model.samples} ${model.samples === 1 ? "hop" : "hops"}${model.demoted ? ", demoted" : ""}`, ); lines.push(` ${parts.join(" · ")}`); } return lines.join("\n"); } }