Luigit
repositories / pi-ext

pi-ext

bugabingas pi extensions

owned by admin

extensions/strata/filesystem.ts

Raw
import { constants } from "node:fs";
import { open, opendir, realpath, stat } from "node:fs/promises";
import path from "node:path";
import {
	type Tool,
	type ToolCall,
	Type,
	validateToolCall,
} from "@earendil-works/pi-ai";
import { isProtectedPath } from "./git.js";

const MAX_PATH_LENGTH = 4_096;
const MAX_FILE_BYTES = 2 * 1024 * 1024;
const MAX_READ_BYTES = 24 * 1024;
const MAX_READ_LINES = 2_000;
const MAX_GREP_MATCHES = 100;
const MAX_GREP_CONTEXT = 5;
const MAX_GREP_RESULT_BYTES = 48 * 1024;
const MAX_GREP_FILES = 1_024;
const MAX_GREP_DIRECTORIES = 256;
const MAX_GREP_TOTAL_BYTES = 8 * 1024 * 1024;
const MAX_GREP_DEPTH = 24;
const MAX_DIRECTORY_ENTRIES = 2_000;
const MAX_LS_ENTRIES = 500;
const MAX_LS_RESULT_BYTES = 48 * 1024;
const MAX_LINE_CHARACTERS = 500;
const FILE_READ_CHUNK_BYTES = 64 * 1024;

export const FILESYSTEM_TOOLS = [
	{
		name: "read",
		description:
			"Read a CURRENT FILESYSTEM regular text file. Relative paths resolve from the reviewed repository; absolute paths are allowed. Supports 1-indexed line offset/limit pagination. Output and file size are bounded.",
		parameters: Type.Object(
			{
				path: Type.String({ minLength: 1, maxLength: MAX_PATH_LENGTH }),
				offset: Type.Optional(
					Type.Integer({ minimum: 1, maximum: MAX_FILE_BYTES + 1 }),
				),
				limit: Type.Optional(
					Type.Integer({ minimum: 1, maximum: MAX_READ_LINES }),
				),
			},
			{ additionalProperties: false },
		),
	},
	{
		name: "grep",
		description:
			"Search CURRENT FILESYSTEM regular text files for a literal string. Relative paths resolve from the reviewed repository; absolute paths and ignored dependency directories such as node_modules are searched directly. Traversal, bytes, matches, context, and output are bounded.",
		parameters: Type.Object(
			{
				pattern: Type.String({ minLength: 1, maxLength: 256 }),
				path: Type.Optional(
					Type.String({ minLength: 1, maxLength: MAX_PATH_LENGTH }),
				),
				ignoreCase: Type.Optional(Type.Boolean()),
				context: Type.Optional(
					Type.Integer({ minimum: 0, maximum: MAX_GREP_CONTEXT }),
				),
				limit: Type.Optional(
					Type.Integer({ minimum: 1, maximum: MAX_GREP_MATCHES }),
				),
			},
			{ additionalProperties: false },
		),
	},
	{
		name: "ls",
		description:
			"List bounded CURRENT FILESYSTEM directory names. Relative paths resolve from the reviewed repository; absolute paths are allowed. Directories have a trailing slash. Listing exposes names only, never file contents.",
		parameters: Type.Object(
			{
				path: Type.Optional(
					Type.String({ minLength: 1, maxLength: MAX_PATH_LENGTH }),
				),
				limit: Type.Optional(
					Type.Integer({ minimum: 1, maximum: MAX_LS_ENTRIES }),
				),
			},
			{ additionalProperties: false },
		),
	},
] satisfies Tool[];

type ReadInput = { path: string; offset?: number; limit?: number };
type GrepInput = {
	pattern: string;
	path?: string;
	ignoreCase?: boolean;
	context?: number;
	limit?: number;
};
type LsInput = { path?: string; limit?: number };

export async function executeFilesystemTool(
	repositoryCwd: string,
	call: ToolCall,
	signal: AbortSignal,
): Promise<unknown> {
	signal.throwIfAborted();
	const input = validateToolCall(FILESYSTEM_TOOLS, call);
	if (call.name === "read")
		return readFilesystem(repositoryCwd, input as ReadInput, signal);
	if (call.name === "grep")
		return grepFilesystem(repositoryCwd, input as GrepInput, signal);
	if (call.name === "ls")
		return listFilesystem(repositoryCwd, input as LsInput, signal);
	throw new Error(`unknown filesystem tool: ${call.name}`);
}

export async function resolveAuthorizedPath(
	repositoryCwd: string,
	input: string,
	resolveRealpath: (value: string) => Promise<string> = realpath,
): Promise<{ requested: string; canonical: string }> {
	assertInputPath(input);
	assertPathParts(repositoryCwd);
	assertPathParts(input);
	const requested = path.resolve(repositoryCwd, input);
	assertPathParts(requested);
	let canonical: string;
	try {
		canonical = await resolveRealpath(requested);
	} catch (error) {
		throw new Error(`path does not exist or cannot be resolved: ${input}`, {
			cause: error,
		});
	}
	assertPathParts(canonical);
	return { requested, canonical };
}

async function readFilesystem(
	cwd: string,
	input: ReadInput,
	signal: AbortSignal,
): Promise<unknown> {
	const resolved = await resolveAuthorizedPath(cwd, input.path);
	const { buffer } = await readRegularText(
		resolved.canonical,
		signal,
		MAX_FILE_BYTES,
	);
	const text = decodeText(buffer, input.path);
	const lines = text.split("\n");
	const offset = input.offset ?? 1;
	if (offset > lines.length)
		throw new Error(
			`offset ${offset} is beyond end of file (${lines.length} lines)`,
		);
	const requestedLimit = input.limit ?? MAX_READ_LINES;
	const selected = lines.slice(offset - 1, offset - 1 + requestedLimit);
	const output: string[] = [];
	let bytes = 0;
	let lineTruncated = false;
	for (const line of selected) {
		const separator = output.length === 0 ? "" : "\n";
		const remaining = MAX_READ_BYTES - bytes - Buffer.byteLength(separator);
		if (remaining <= 0) break;
		const clipped = clipUtf8(line, remaining);
		output.push(clipped);
		bytes += Buffer.byteLength(separator + clipped);
		if (clipped !== line) {
			lineTruncated = true;
			break;
		}
	}
	const returnedLines = output.length;
	const nextOffset = offset - 1 + returnedLines + (lineTruncated ? 0 : 1);
	const hasMore = lineTruncated || offset - 1 + returnedLines < lines.length;
	return {
		evidence: "CURRENT FILESYSTEM",
		path: resolved.requested,
		canonicalPath: resolved.canonical,
		totalLines: lines.length,
		offset,
		returnedLines,
		content: output.join("\n"),
		truncated: hasMore,
		lineTruncated,
		...(hasMore && !lineTruncated ? { nextOffset } : {}),
	};
}

async function listFilesystem(
	cwd: string,
	input: LsInput,
	signal: AbortSignal,
): Promise<unknown> {
	const requestedPath = input.path ?? ".";
	const resolved = await resolveAuthorizedPath(cwd, requestedPath);
	signal.throwIfAborted();
	const info = await stat(resolved.canonical);
	if (!info.isDirectory()) throw new Error(`not a directory: ${requestedPath}`);
	const limit = input.limit ?? MAX_LS_ENTRIES;
	const names: string[] = [];
	let scanned = 0;
	let scanLimitReached = false;
	const directory = await opendir(resolved.canonical);
	try {
		for await (const entry of directory) {
			signal.throwIfAborted();
			if (scanned >= MAX_DIRECTORY_ENTRIES) {
				scanLimitReached = true;
				break;
			}
			scanned++;
			let suffix = "";
			try {
				const child = await resolveAuthorizedPath(
					resolved.canonical,
					entry.name,
				);
				if ((await stat(child.canonical)).isDirectory()) suffix = "/";
			} catch {
				// Names may be listed, but protected or inaccessible targets are not followed.
			}
			names.push(`${entry.name}${suffix}`);
		}
	} finally {
		await directory.close().catch(() => {});
	}
	names.sort((left, right) => left.localeCompare(right));
	const entries: string[] = [];
	let resultBytes = 0;
	let resultLimitReached = false;
	for (const name of names) {
		if (entries.length >= limit) break;
		const nextBytes = Buffer.byteLength(JSON.stringify(name), "utf8") + 1;
		if (resultBytes + nextBytes > MAX_LS_RESULT_BYTES) {
			resultLimitReached = true;
			break;
		}
		entries.push(name);
		resultBytes += nextBytes;
	}
	const limitReasons = [
		...(scanLimitReached ? ["directory traversal limit"] : []),
		...(resultLimitReached ? ["result byte limit"] : []),
	];
	return {
		evidence: "CURRENT FILESYSTEM",
		path: resolved.requested,
		entries,
		returned: entries.length,
		scanned,
		truncated:
			scanLimitReached || resultLimitReached || names.length > entries.length,
		limitReasons,
	};
}

async function grepFilesystem(
	cwd: string,
	input: GrepInput,
	signal: AbortSignal,
): Promise<unknown> {
	const requestedPath = input.path ?? ".";
	const root = await resolveAuthorizedPath(cwd, requestedPath);
	const rootInfo = await stat(root.canonical);
	const limit = input.limit ?? MAX_GREP_MATCHES;
	const context = input.context ?? 0;
	const needle = input.ignoreCase
		? input.pattern.toLocaleLowerCase("en-US")
		: input.pattern;
	const matches: Array<{
		path: string;
		line: number;
		text: string;
		before?: string[];
		after?: string[];
	}> = [];
	const visitedDirectories = new Set<string>();
	const queue: Array<{ canonical: string; display: string; depth: number }> =
		[];
	const files: Array<{ canonical: string; display: string }> = [];
	if (rootInfo.isDirectory())
		queue.push({ canonical: root.canonical, display: "", depth: 0 });
	else if (rootInfo.isFile())
		files.push({
			canonical: root.canonical,
			display: path.basename(root.requested),
		});
	else throw new Error(`not a regular file or directory: ${requestedPath}`);

	let directories = 0;
	let filesVisited = 0;
	let bytesVisited = 0;
	let totalMatches = 0;
	let skippedBinary = 0;
	let skippedProtectedOrInaccessible = 0;
	let linesTruncated = false;
	const limits = new Set<string>();

	while (queue.length > 0 || files.length > 0) {
		signal.throwIfAborted();
		while (files.length > 0) {
			const file = files.shift();
			if (!file) break;
			if (filesVisited >= MAX_GREP_FILES) {
				limits.add("file traversal limit");
				files.length = 0;
				queue.length = 0;
				break;
			}
			const remainingBytes = MAX_GREP_TOTAL_BYTES - bytesVisited;
			if (remainingBytes <= 0) {
				limits.add("total byte limit");
				files.length = 0;
				queue.length = 0;
				break;
			}
			const fileByteLimit = Math.min(MAX_FILE_BYTES, remainingBytes);
			filesVisited++;
			let buffer: Buffer;
			try {
				const read = await readRegularText(
					file.canonical,
					signal,
					fileByteLimit,
				);
				buffer = read.buffer;
				bytesVisited += read.bytesRead;
			} catch (error) {
				if (error instanceof FileByteLimitError) {
					bytesVisited += error.bytesRead;
					if (fileByteLimit < MAX_FILE_BYTES) {
						limits.add("total byte limit");
						files.length = 0;
						queue.length = 0;
						break;
					}
					limits.add("per-file byte limit");
					continue;
				}
				throw error;
			}
			let text: string;
			try {
				text = decodeText(buffer, file.display);
			} catch (error) {
				if (error instanceof BinaryFileError) {
					skippedBinary++;
					continue;
				}
				throw error;
			}
			const lines = text.split("\n");
			for (let index = 0; index < lines.length; index++) {
				signal.throwIfAborted();
				const line = lines[index] as string;
				const haystack = input.ignoreCase
					? line.toLocaleLowerCase("en-US")
					: line;
				if (!haystack.includes(needle)) continue;
				totalMatches++;
				if (matches.length >= limit) continue;
				const clipped = clipCharacters(line);
				const before = lines.slice(Math.max(0, index - context), index);
				const after = lines.slice(index + 1, index + context + 1);
				const clippedBefore = before.map(clipCharacters);
				const clippedAfter = after.map(clipCharacters);
				if (
					clipped !== line ||
					clippedBefore.some(
						(value, lineIndex) => value !== before[lineIndex],
					) ||
					clippedAfter.some((value, lineIndex) => value !== after[lineIndex])
				)
					linesTruncated = true;
				const candidate = {
					path: file.display || path.basename(file.canonical),
					line: index + 1,
					text: clipped,
					...(context > 0
						? { before: clippedBefore, after: clippedAfter }
						: {}),
				};
				if (
					Buffer.byteLength(JSON.stringify([...matches, candidate]), "utf8") >
					MAX_GREP_RESULT_BYTES
				) {
					limits.add("result byte limit");
					continue;
				}
				matches.push(candidate);
			}
		}

		const next = queue.shift();
		if (!next) continue;
		if (visitedDirectories.has(next.canonical)) continue;
		if (directories >= MAX_GREP_DIRECTORIES) {
			limits.add("directory traversal limit");
			queue.length = 0;
			continue;
		}
		if (next.depth > MAX_GREP_DEPTH) {
			limits.add("depth limit");
			continue;
		}
		visitedDirectories.add(next.canonical);
		directories++;
		const directory = await opendir(next.canonical);
		let entries = 0;
		try {
			for await (const entry of directory) {
				signal.throwIfAborted();
				if (entries >= MAX_DIRECTORY_ENTRIES) {
					limits.add("directory entry limit");
					break;
				}
				entries++;
				const display = next.display
					? `${next.display}/${entry.name}`
					: entry.name;
				let child: Awaited<ReturnType<typeof resolveAuthorizedPath>>;
				try {
					child = await resolveAuthorizedPath(next.canonical, entry.name);
				} catch {
					skippedProtectedOrInaccessible++;
					continue;
				}
				const childInfo = await stat(child.canonical);
				if (childInfo.isDirectory())
					queue.push({
						canonical: child.canonical,
						display,
						depth: next.depth + 1,
					});
				else if (childInfo.isFile())
					files.push({ canonical: child.canonical, display });
			}
		} finally {
			await directory.close().catch(() => {});
		}
	}

	return {
		evidence: "CURRENT FILESYSTEM",
		path: root.requested,
		literalPattern: input.pattern,
		matches,
		returnedMatches: matches.length,
		totalMatches,
		filesVisited,
		directoriesVisited: directories,
		bytesVisited,
		skippedBinary,
		skippedProtectedOrInaccessible,
		linesTruncated,
		truncated:
			linesTruncated || limits.size > 0 || totalMatches > matches.length,
		limitReasons: [...limits],
	};
}

async function readRegularText(
	canonicalPath: string,
	signal: AbortSignal,
	maximumBytes: number,
): Promise<{ buffer: Buffer; bytesRead: number }> {
	signal.throwIfAborted();
	const beforeCanonical = await canonicalizeAuthorized(canonicalPath);
	if (!sameCanonicalPath(canonicalPath, beforeCanonical))
		throw new Error(`file changed during authorization: ${canonicalPath}`);
	const before = await stat(beforeCanonical);
	if (!before.isFile()) throw new Error(`not a regular file: ${canonicalPath}`);

	let handle: Awaited<ReturnType<typeof open>>;
	try {
		handle = await open(
			canonicalPath,
			constants.O_RDONLY |
				(constants.O_NONBLOCK ?? 0) |
				(process.platform === "win32" ? 0 : (constants.O_NOFOLLOW ?? 0)),
		);
	} catch (error) {
		throw new Error(`file could not be safely opened: ${canonicalPath}`, {
			cause: error,
		});
	}
	try {
		const opened = await handle.stat();
		if (!opened.isFile())
			throw new Error(`not a regular file: ${canonicalPath}`);
		if (!sameFileIdentity(before, opened))
			throw new Error(`file changed during authorization: ${canonicalPath}`);

		const afterCanonical = await canonicalizeAuthorized(canonicalPath);
		if (!sameCanonicalPath(beforeCanonical, afterCanonical))
			throw new Error(`file changed during authorization: ${canonicalPath}`);
		const after = await stat(afterCanonical);
		if (!after.isFile() || !sameFileIdentity(opened, after))
			throw new Error(`file changed during authorization: ${canonicalPath}`);
		await authorizeDescriptorTarget(handle.fd, beforeCanonical);

		const chunks: Buffer[] = [];
		let bytesRead = 0;
		while (bytesRead <= maximumBytes) {
			signal.throwIfAborted();
			const chunk = Buffer.allocUnsafe(
				Math.min(FILE_READ_CHUNK_BYTES, maximumBytes + 1 - bytesRead),
			);
			const result = await handle.read(chunk, 0, chunk.length, null);
			signal.throwIfAborted();
			if (result.bytesRead === 0) break;
			chunks.push(chunk.subarray(0, result.bytesRead));
			bytesRead += result.bytesRead;
			if (bytesRead > maximumBytes)
				throw new FileByteLimitError(canonicalPath, maximumBytes, bytesRead);
		}
		return { buffer: Buffer.concat(chunks, bytesRead), bytesRead };
	} finally {
		await handle.close();
	}
}

async function canonicalizeAuthorized(value: string): Promise<string> {
	let canonical: string;
	try {
		canonical = await realpath(value);
	} catch (error) {
		throw new Error(`file changed during authorization: ${value}`, {
			cause: error,
		});
	}
	assertPathParts(canonical);
	return canonical;
}

async function authorizeDescriptorTarget(
	descriptor: number,
	expectedCanonical: string,
): Promise<void> {
	if (process.platform === "win32") return;
	const candidates = [
		...(process.platform === "linux" ? [`/proc/self/fd/${descriptor}`] : []),
		`/dev/fd/${descriptor}`,
	];
	for (const candidate of candidates) {
		let canonical: string;
		try {
			canonical = await realpath(candidate);
		} catch {
			continue;
		}
		if (sameCanonicalPath(candidate, canonical)) continue;
		assertPathParts(canonical);
		if (!sameCanonicalPath(expectedCanonical, canonical))
			throw new Error(
				`file changed during descriptor authorization: ${expectedCanonical}`,
			);
		return;
	}
}

function sameCanonicalPath(left: string, right: string): boolean {
	return process.platform === "win32"
		? left.toLocaleLowerCase("en-US") === right.toLocaleLowerCase("en-US")
		: left === right;
}

function sameFileIdentity(
	left: { dev: number | bigint; ino: number | bigint },
	right: { dev: number | bigint; ino: number | bigint },
): boolean {
	return left.dev === right.dev && left.ino === right.ino;
}

function decodeText(buffer: Buffer, displayPath: string): string {
	if (
		buffer.some(
			(byte) =>
				byte === 0 ||
				byte < 0x09 ||
				(byte > 0x0d && byte < 0x20) ||
				byte === 0x7f,
		)
	)
		throw new BinaryFileError(`binary file is not readable: ${displayPath}`);
	try {
		return new TextDecoder("utf-8", { fatal: true }).decode(buffer);
	} catch {
		throw new BinaryFileError(`file is not valid UTF-8 text: ${displayPath}`);
	}
}

function assertInputPath(value: string): void {
	if (
		!value ||
		value.length > MAX_PATH_LENGTH ||
		value.includes("\0") ||
		/^[A-Za-z][A-Za-z0-9+.-]*:\/\//u.test(value) ||
		/^file:/iu.test(value)
	)
		throw new Error(`invalid filesystem path: ${value}`);
}

function assertPathParts(value: string): void {
	const parts =
		process.platform === "win32" ? value.split(/[\\/]+/u) : value.split("/");
	const stack: string[] = [];
	for (const part of parts) {
		if (!part || part === ".") continue;
		if (part === "..") {
			stack.pop();
			continue;
		}
		stack.push(part);
		if (isProtectedPath(stack.join(path.sep), true))
			throw new Error(`access denied for protected path: ${value}`);
	}
}

function clipCharacters(value: string): string {
	return value.length <= MAX_LINE_CHARACTERS
		? value
		: `${value.slice(0, MAX_LINE_CHARACTERS)}…`;
}

function clipUtf8(value: string, maximum: number): string {
	if (Buffer.byteLength(value, "utf8") <= maximum) return value;
	let output = "";
	let bytes = 0;
	for (const character of value) {
		const characterBytes = Buffer.byteLength(character, "utf8");
		if (bytes + characterBytes > maximum) break;
		output += character;
		bytes += characterBytes;
	}
	return output;
}

class BinaryFileError extends Error {}

class FileByteLimitError extends Error {
	constructor(
		filePath: string,
		maximumBytes: number,
		readonly bytesRead: number,
	) {
		super(`file exceeds ${maximumBytes} byte limit: ${filePath}`);
	}
}