import { createServer, type IncomingHttpHeaders } from "node:http"; import type { AddressInfo } from "node:net"; export interface Recorded { method: string; path: string; headers: IncomingHttpHeaders; body: Record | undefined; } export type Reply = /** Never answers: models a backend that hangs under load. */ | { hang: true } | { status: number; text: string } | { status: number; json: unknown } | { sse: string[] } | { sseEvents: { event: string; data: unknown }[] }; export type Responder = (request: Recorded) => Reply | undefined; const ORIGINS: Record = { "https://api.kilo.ai": "/kilo", "https://openrouter.ai": "/openrouter", "https://generativelanguage.googleapis.com": "/google", "https://api.groq.com": "/groq", "https://api.mistral.ai": "/mistral", "https://integrate.api.nvidia.com": "/nvidia", "https://api.z.ai": "/zai", "https://inference.hetzner.com": "/hetzner", "https://docs.z.ai": "/zaidocs", "https://models.dev": "/modelsdev", }; /** Local stand-in for every gratis upstream, addressed by origin prefix. */ export async function startUpstream(respond: Responder) { const requests: Recorded[] = []; const server = createServer((req, res) => { const chunks: Buffer[] = []; req.on("data", (chunk: Buffer) => chunks.push(chunk)); req.on("end", () => { const text = Buffer.concat(chunks).toString("utf8"); const request: Recorded = { method: req.method ?? "GET", path: req.url ?? "/", headers: req.headers, body: text ? JSON.parse(text) : undefined, }; requests.push(request); const reply = respond(request) ?? { status: 404, json: { error: "no fake route" }, }; if ("hang" in reply) return; if ("text" in reply) { res.writeHead(reply.status, { "content-type": "text/markdown" }); res.end(reply.text); return; } if ("json" in reply) { res.writeHead(reply.status, { "content-type": "application/json" }); res.end(JSON.stringify(reply.json)); return; } res.writeHead(200, { "content-type": "text/event-stream" }); if ("sse" in reply) for (const data of reply.sse) res.write(`data: ${data}\n\n`); else for (const { event, data } of reply.sseEvents) res.write(`event: ${event}\ndata: ${JSON.stringify(data)}\n\n`); res.end(); }); }); await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); const base = `http://127.0.0.1:${(server.address() as AddressInfo).port}`; return { requests, resolveUrl(url: string): string { for (const [origin, prefix] of Object.entries(ORIGINS)) if (url.startsWith(origin)) return `${base}${prefix}${url.slice(origin.length)}`; return url; }, close: () => new Promise((resolve) => { // Hung requests would otherwise keep the server open. server.closeAllConnections(); server.close(() => resolve()); }), }; } export function liveModel( id: string, extra: Record = {}, ): Record { return { id, name: id, context_length: 262_144, pricing: { prompt: "0", completion: "0" }, architecture: { input_modalities: ["text"], output_modalities: ["text"] }, top_provider: { context_length: 262_144, max_completion_tokens: 32_768 }, supported_parameters: ["tools", "max_tokens"], ...extra, }; } /** `model` is what the upstream reports it ran, as routers like kilo-auto/free do. */ export function openAIText(text: string, model = "fake"): Reply { const chunk = (delta: unknown, finish: string | null) => JSON.stringify({ id: "chatcmpl-fake", object: "chat.completion.chunk", created: 0, model, choices: [{ index: 0, delta, finish_reason: finish }], ...(finish ? { usage: { prompt_tokens: 3, completion_tokens: 2, total_tokens: 5 }, } : {}), }); return { sse: [ chunk({ role: "assistant", content: text }, null), chunk({}, "stop"), "[DONE]", ], }; } export function googleText(text: string): Reply { return { sse: [ JSON.stringify({ candidates: [ { content: { parts: [{ text }], role: "model" }, finishReason: "STOP", index: 0, }, ], usageMetadata: { promptTokenCount: 3, candidatesTokenCount: 2, totalTokenCount: 5, }, }), ], }; } export function anthropicText(text: string): Reply { return { sseEvents: [ { event: "message_start", data: { type: "message_start", message: { id: "msg_fake", type: "message", role: "assistant", model: "fake", content: [], stop_reason: null, usage: { input_tokens: 3, output_tokens: 0 }, }, }, }, { event: "content_block_start", data: { type: "content_block_start", index: 0, content_block: { type: "text", text: "" }, }, }, { event: "content_block_delta", data: { type: "content_block_delta", index: 0, delta: { type: "text_delta", text }, }, }, { event: "content_block_stop", data: { type: "content_block_stop", index: 0 }, }, { event: "message_delta", data: { type: "message_delta", delta: { stop_reason: "end_turn" }, usage: { output_tokens: 2 }, }, }, { event: "message_stop", data: { type: "message_stop" } }, ], }; } export const rateLimited: Reply = { status: 429, json: { error: { message: "Rate limit exceeded: upstream provider saturated", code: 429, }, }, }; export const contextOverflow: Reply = { status: 400, json: { error: { message: "This endpoint's maximum context length is 262144 tokens. However, you requested about 300000 tokens.", code: 400, }, }, };