Luigit
repositories / pi-ext

pi-ext

bugabingas pi extensions

owned by admin

extensions/gratis/__tests__/fake-upstream.ts

Raw
import { createServer, type IncomingHttpHeaders } from "node:http";
import type { AddressInfo } from "node:net";

export interface Recorded {
	method: string;
	path: string;
	headers: IncomingHttpHeaders;
	body: Record<string, unknown> | undefined;
}

export type Reply =
	/** Never answers: models a backend that hangs under load. */
	| { hang: true }
	| { status: number; text: string }
	| { status: number; json: unknown }
	| { sse: string[] }
	| { sseEvents: { event: string; data: unknown }[] };

export type Responder = (request: Recorded) => Reply | undefined;

const ORIGINS: Record<string, string> = {
	"https://api.kilo.ai": "/kilo",
	"https://openrouter.ai": "/openrouter",
	"https://generativelanguage.googleapis.com": "/google",
	"https://api.groq.com": "/groq",
	"https://api.mistral.ai": "/mistral",
	"https://integrate.api.nvidia.com": "/nvidia",
	"https://api.z.ai": "/zai",
	"https://inference.hetzner.com": "/hetzner",
	"https://docs.z.ai": "/zaidocs",
	"https://models.dev": "/modelsdev",
};

/** Local stand-in for every gratis upstream, addressed by origin prefix. */
export async function startUpstream(respond: Responder) {
	const requests: Recorded[] = [];
	const server = createServer((req, res) => {
		const chunks: Buffer[] = [];
		req.on("data", (chunk: Buffer) => chunks.push(chunk));
		req.on("end", () => {
			const text = Buffer.concat(chunks).toString("utf8");
			const request: Recorded = {
				method: req.method ?? "GET",
				path: req.url ?? "/",
				headers: req.headers,
				body: text ? JSON.parse(text) : undefined,
			};
			requests.push(request);
			const reply = respond(request) ?? {
				status: 404,
				json: { error: "no fake route" },
			};
			if ("hang" in reply) return;
			if ("text" in reply) {
				res.writeHead(reply.status, { "content-type": "text/markdown" });
				res.end(reply.text);
				return;
			}
			if ("json" in reply) {
				res.writeHead(reply.status, { "content-type": "application/json" });
				res.end(JSON.stringify(reply.json));
				return;
			}
			res.writeHead(200, { "content-type": "text/event-stream" });
			if ("sse" in reply)
				for (const data of reply.sse) res.write(`data: ${data}\n\n`);
			else
				for (const { event, data } of reply.sseEvents)
					res.write(`event: ${event}\ndata: ${JSON.stringify(data)}\n\n`);
			res.end();
		});
	});
	await new Promise<void>((resolve) => server.listen(0, "127.0.0.1", resolve));
	const base = `http://127.0.0.1:${(server.address() as AddressInfo).port}`;
	return {
		requests,
		resolveUrl(url: string): string {
			for (const [origin, prefix] of Object.entries(ORIGINS))
				if (url.startsWith(origin))
					return `${base}${prefix}${url.slice(origin.length)}`;
			return url;
		},
		close: () =>
			new Promise<void>((resolve) => {
				// Hung requests would otherwise keep the server open.
				server.closeAllConnections();
				server.close(() => resolve());
			}),
	};
}

export function liveModel(
	id: string,
	extra: Record<string, unknown> = {},
): Record<string, unknown> {
	return {
		id,
		name: id,
		context_length: 262_144,
		pricing: { prompt: "0", completion: "0" },
		architecture: { input_modalities: ["text"], output_modalities: ["text"] },
		top_provider: { context_length: 262_144, max_completion_tokens: 32_768 },
		supported_parameters: ["tools", "max_tokens"],
		...extra,
	};
}

/** `model` is what the upstream reports it ran, as routers like kilo-auto/free do. */
export function openAIText(text: string, model = "fake"): Reply {
	const chunk = (delta: unknown, finish: string | null) =>
		JSON.stringify({
			id: "chatcmpl-fake",
			object: "chat.completion.chunk",
			created: 0,
			model,
			choices: [{ index: 0, delta, finish_reason: finish }],
			...(finish
				? {
						usage: { prompt_tokens: 3, completion_tokens: 2, total_tokens: 5 },
					}
				: {}),
		});
	return {
		sse: [
			chunk({ role: "assistant", content: text }, null),
			chunk({}, "stop"),
			"[DONE]",
		],
	};
}

export function googleText(text: string): Reply {
	return {
		sse: [
			JSON.stringify({
				candidates: [
					{
						content: { parts: [{ text }], role: "model" },
						finishReason: "STOP",
						index: 0,
					},
				],
				usageMetadata: {
					promptTokenCount: 3,
					candidatesTokenCount: 2,
					totalTokenCount: 5,
				},
			}),
		],
	};
}

export function anthropicText(text: string): Reply {
	return {
		sseEvents: [
			{
				event: "message_start",
				data: {
					type: "message_start",
					message: {
						id: "msg_fake",
						type: "message",
						role: "assistant",
						model: "fake",
						content: [],
						stop_reason: null,
						usage: { input_tokens: 3, output_tokens: 0 },
					},
				},
			},
			{
				event: "content_block_start",
				data: {
					type: "content_block_start",
					index: 0,
					content_block: { type: "text", text: "" },
				},
			},
			{
				event: "content_block_delta",
				data: {
					type: "content_block_delta",
					index: 0,
					delta: { type: "text_delta", text },
				},
			},
			{
				event: "content_block_stop",
				data: { type: "content_block_stop", index: 0 },
			},
			{
				event: "message_delta",
				data: {
					type: "message_delta",
					delta: { stop_reason: "end_turn" },
					usage: { output_tokens: 2 },
				},
			},
			{ event: "message_stop", data: { type: "message_stop" } },
		],
	};
}

export const rateLimited: Reply = {
	status: 429,
	json: {
		error: {
			message: "Rate limit exceeded: upstream provider saturated",
			code: 429,
		},
	},
};

export const contextOverflow: Reply = {
	status: 400,
	json: {
		error: {
			message:
				"This endpoint's maximum context length is 262144 tokens. However, you requested about 300000 tokens.",
			code: 400,
		},
	},
};