repositories / pi-ext
pi-ext
bugabingas pi extensions
owned by admin
extensions/gratis/src/command.ts
Rawimport type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
import {
AUTO_ID,
BACKENDS,
type Backend,
type GratisModel,
} from "./catalog.ts";
import type { createRouter } from "./router.ts";
const MAX_STATUS_MODELS = 8;
const SUBCOMMANDS = ["cool", "uncool"] as const;
const DEFAULT_MANUAL_COOL_MINUTES = 60;
const MAX_MANUAL_COOL_MINUTES = 24 * 60;
export interface CommandDeps {
router: ReturnType<typeof createRouter>;
/** Every registered grts model, `auto` first when present. */
models(): GratisModel[];
backendModels(backend: Backend): GratisModel[];
/** Resolved key; `null` when the backend is unusable. */
key(backend: Backend): string | undefined | null;
}
/** `/gratis`: status, plus manual cooling for `grts/auto`. */
export function registerGratisCommand(
pi: ExtensionAPI,
{ router, models, backendModels, key }: CommandDeps,
): void {
const routable = () =>
models()
.map((model) => model.id)
.filter((id) => id !== AUTO_ID);
pi.registerCommand("gratis", {
description:
"Show grts status; cool [model] [minutes] skips a model in grts/auto; uncool [model] ends that",
getArgumentCompletions: (prefix) => {
const [subcommand, partial, ...rest] = prefix.split(" ");
if (partial === undefined) {
const matches = SUBCOMMANDS.filter((name) =>
name.startsWith(subcommand ?? ""),
);
return matches.length
? matches.map((name) => ({ value: name, label: name }))
: null;
}
if (rest.length > 0) return null;
const ids =
subcommand === "cool"
? routable()
: subcommand === "uncool"
? router.status().manual.map(({ model }) => model)
: [];
const matches = ids.filter((id) => id.startsWith(partial));
return matches.length
? matches.map((id) => ({ value: `${subcommand} ${id}`, label: id }))
: null;
},
handler: async (args, ctx) => {
const [subcommand, ...rest] = args.trim().split(/\s+/).filter(Boolean);
if (subcommand === undefined) {
ctx.ui.notify(statusText(), "info");
return;
}
const reply = runSubcommand(subcommand, rest);
ctx.ui.notify(reply.text, reply.ok ? "info" : "warning");
},
});
function runSubcommand(
subcommand: string,
rest: string[],
): { ok: boolean; text: string } {
if (subcommand === "uncool") {
const [model] = rest;
const ended = router.uncool(model);
if (ended === 0)
return {
ok: false,
text: model
? `${model} is not manually cooled`
: "No model is manually cooled",
};
return {
ok: true,
text: model
? `${model} is back in grts/auto`
: `${ended} manually cooled ${ended === 1 ? "model is" : "models are"} back in grts/auto`,
};
}
if (subcommand !== "cool")
return {
ok: false,
text: "Usage: /gratis [cool [model] [minutes] | uncool [model]]",
};
const [named, minutesText] = rest;
const last = router.status().last;
const model = named ?? last?.candidate;
if (!model)
return {
ok: false,
text: "No grts model has answered yet; name one: /gratis cool <model>",
};
if (!routable().includes(model))
return { ok: false, text: `${model} is not a current grts model` };
const minutes =
minutesText === undefined
? DEFAULT_MANUAL_COOL_MINUTES
: Number(minutesText);
if (
!Number.isInteger(minutes) ||
minutes < 1 ||
minutes > MAX_MANUAL_COOL_MINUTES
)
return {
ok: false,
text: `Minutes must be a whole number from 1 to ${MAX_MANUAL_COOL_MINUTES}`,
};
router.cool(model, minutes * 60_000);
const behind =
named === undefined &&
last?.served &&
last.served !== model &&
last.candidate === model
? ` (it answered as ${last.served}; gratis can only skip ${model} itself)`
: "";
return {
ok: true,
text: `grts/auto skips ${model} for ${minutes} min${behind}`,
};
}
function statusText(now = Date.now()): string {
const ago = (at: number) => {
const seconds = Math.max(0, Math.round((now - at) / 1000));
return seconds < 90
? `${seconds}s ago`
: `${Math.round(seconds / 60)}m ago`;
};
const left = (until: number) =>
`${Math.max(1, Math.ceil((until - now) / 60_000))}m left`;
const visible = BACKENDS.map((backend) => ({
backend,
count: backendModels(backend).length,
}));
const shown = visible.filter(({ count }) => count > 0);
const unkeyed = BACKENDS.filter((backend) => key(backend) === null);
const empty = visible
.filter(({ backend, count }) => count === 0 && key(backend) !== null)
.map(({ backend }) => backend);
const { last, cooling, manual, models: measured } = router.status();
const lines = [
`Backends: ${shown.map(({ backend, count }) => `${backend} ${count}`).join(", ") || "none"}`,
];
if (unkeyed.length > 0) lines.push(`No key: ${unkeyed.join(", ")}`);
if (empty.length > 0) lines.push(`No models: ${empty.join(", ")}`);
lines.push(
`Cooling: ${cooling.map(({ backend, until }) => `${backend} (${left(until)})`).join(", ") || "none"}`,
);
if (manual.length > 0)
lines.push(
`Manually cooled: ${manual.map(({ model, until }) => `${model} (${left(until)})`).join(", ")}`,
);
if (!last) lines.push("Last: no grts request yet");
else {
const hops = `${last.hops} ${last.hops === 1 ? "hop" : "hops"}`;
lines.push(
`Last: ${last.requested} → ${last.served ?? "no model"} · ${last.outcome} · ${hops} · ${ago(last.at)}`,
);
if (last.notes.length > 0)
lines.push(`Skipped: ${last.notes.join(", ")}`);
}
// Measured from real requests only; persisted across restarts.
const shownModels = measured.slice(0, MAX_STATUS_MODELS);
if (shownModels.length > 0) lines.push("Models:");
for (const model of shownModels) {
const parts = [model.id];
if (model.firstTokenMs !== undefined)
parts.push(`first token ${(model.firstTokenMs / 1000).toFixed(1)}s`);
if (model.tokensPerSecond !== undefined)
parts.push(`${Math.round(model.tokensPerSecond)} tok/s`);
parts.push(`${Math.round(model.failureRate * 100)}% fail`);
parts.push(
`${model.samples} ${model.samples === 1 ? "hop" : "hops"}${model.demoted ? ", demoted" : ""}`,
);
lines.push(` ${parts.join(" · ")}`);
}
return lines.join("\n");
}
}