repositories / pi-ext
pi-ext
bugabingas pi extensions
owned by admin
extensions/web/index.ts
Raw/**
* Web Extension for pi
*
* Tool: web, with fetch, map, and search actions.
*
* Uses Node.js native fetch (HTTP/1.1 via undici) which helps avoid
* bot detection compared to HTTP/2 clients like curl.
*
* Content-Type routing:
* text/html → extract main content → pandoc + lua filter → markdown
* text/*, json, csv → return as-is
* application/pdf → pdftotext -layout
* docx, odt, epub → pandoc direct read
* image/* → identify -verbose (metadata)
* video/audio → ffprobe (metadata)
*/
import { mkdtemp, rm, writeFile } from "node:fs/promises";
import { tmpdir } from "node:os";
import { join } from "node:path";
import { StringEnum } from "@earendil-works/pi-ai";
import type {
ExtensionAPI,
ExtensionContext,
ToolDefinition,
} from "@earendil-works/pi-coding-agent";
import { getMarkdownTheme, keyHint } from "@earendil-works/pi-coding-agent";
import { Box, Markdown, Text } from "@earendil-works/pi-tui";
import { Type } from "typebox";
import {
DEFAULT_MAX_CHARS,
DEFAULT_TIMEOUT_MS,
MAX_VISIBLE_CHARS,
type SearchResult,
} from "./constants.js";
import { lazy } from "./lazy.js";
import { SETTING_READERS } from "./settings.ts";
import { checkToolAvailability, spawnToText, TOOL_HINTS } from "./spawn.js";
import {
createMarkers,
type SpotlightMarkers,
wrapUntrusted,
} from "./spotlight.js";
import { closeDebug, dbg, span } from "./src/debug.ts";
import { isStaleContextError } from "./src/pi-ext-stale-context.ts";
// ── Lazy implementation modules ───────────────────────────────────────────
// Loaded on first tool execution to keep extension startup fast.
// The shared promise also prevents jiti (pi's loader, module cache disabled)
// from handing concurrent tool calls a partially initialized module.
const loadFetch = lazy(() => import("./fetch.js"));
const loadExtract = lazy(() => import("./extract.js"));
const loadPandoc = lazy(() => import("./pandoc.js"));
const loadStore = lazy(() => import("./store.js"));
const loadLineMatch = lazy(() => import("./line-match.js"));
const loadMap = lazy(() => import("./map.js"));
const loadSearch = lazy(() => import("./search.js"));
// ── Shared formatters ────────────────────────────────────────────────────
function fmtSize(bytes: number): string {
return bytes >= 1_000_000
? `${(bytes / 1_048_576).toFixed(1)}MB`
: bytes >= 1_000
? `${(bytes / 1_024).toFixed(1)}KB`
: `${bytes}B`;
}
function fmtChars(n: number): string {
return n >= 10_000 ? `${(n / 1_000).toFixed(1)}K chars` : `${n} chars`;
}
// ── Temp file lifecycle ───────────────────────────────────────────────────
const tempDirs = new Set<string>();
async function createTempDir(prefix: string): Promise<string> {
const dir = await mkdtemp(join(tmpdir(), prefix));
tempDirs.add(dir);
return dir;
}
// ── Content-Type classification ──────────────────────────────────────────
type ContentKind =
| "html"
| "text"
| "json"
| "pdf"
| "office"
| "image"
| "media"
| "binary";
interface Classification {
kind: ContentKind;
officeFormat?: string;
}
interface ClassificationRule {
test: (ct: string) => boolean;
result: Classification;
}
const CLASSIFICATION_RULES: ClassificationRule[] = [
{
test: (ct) => ct.includes("text/html") || ct.includes("application/xhtml"),
result: { kind: "html" },
},
{ test: (ct) => ct.includes("application/json"), result: { kind: "json" } },
{
test: (ct) =>
ct.includes("application/xml") || ct.includes("application/yaml"),
result: { kind: "text" },
},
{
test: (ct) =>
ct.startsWith("text/") || ct.includes("csv") || ct.includes("tsv"),
result: { kind: "text" },
},
{ test: (ct) => ct.includes("application/pdf"), result: { kind: "pdf" } },
{
test: (ct) =>
ct.includes("officedocument.wordprocessingml") || ct.includes("msword"),
result: { kind: "office", officeFormat: "docx" },
},
{
test: (ct) => ct.includes("opendocument.text"),
result: { kind: "office", officeFormat: "odt" },
},
{
test: (ct) => ct.includes("application/epub"),
result: { kind: "office", officeFormat: "epub" },
},
{
test: (ct) =>
ct.includes("application/rtf") || ct.includes("text/richtext"),
result: { kind: "office", officeFormat: "rtf" },
},
{ test: (ct) => ct.startsWith("image/"), result: { kind: "image" } },
{
test: (ct) =>
ct.startsWith("video/") ||
ct.startsWith("audio/") ||
ct.includes("mpeg") ||
ct.includes("mp4") ||
ct.includes("webm") ||
ct.includes("ogg"),
result: { kind: "media" },
},
];
/** Strip parameters and lowercase a Content-Type header value. */
function stripCt(ct: string): string {
return ct.toLowerCase().split(";")[0].trim();
}
function classifyContentType(ct: string): Classification {
return (
CLASSIFICATION_RULES.find((r) => r.test(stripCt(ct)))?.result ?? {
kind: "binary",
}
);
}
// ── Extension from media type ────────────────────────────────────────────
const EXTENSION_MAP: Record<string, string> = {
pdf: "pdf",
docx: "docx",
msword: "doc",
odt: "odt",
epub: "epub",
rtf: "rtf",
png: "png",
gif: "gif",
webp: "webp",
jpeg: "jpg",
jpg: "jpg",
svg: "svg",
mp4: "mp4",
webm: "webm",
ogg: "ogg",
mp3: "mp3",
wav: "wav",
};
function extensionFromContentType(ct: string, url: string): string {
const lower = stripCt(ct);
for (const [needle, ext] of Object.entries(EXTENSION_MAP)) {
if (lower.includes(needle)) return ext;
}
if (lower.startsWith("image/")) return "bin";
if (lower.startsWith("video/")) return "mp4";
if (lower.startsWith("audio/")) return "mp3";
try {
const urlExt = new URL(url).pathname.match(/\.([^.]+)$/)?.[1];
if (urlExt) return urlExt.toLowerCase();
} catch {}
return "bin";
}
// ── Binary/temp helpers ──────────────────────────────────────────────────
async function saveBinaryToTemp(
data: Buffer,
extension: string,
): Promise<string> {
const dir = await createTempDir("pi-web-");
const tmpPath = join(dir, `download.${extension}`);
await writeFile(tmpPath, data);
return tmpPath;
}
async function pdfToText(
data: Buffer,
): Promise<{ text: string; tmpPath: string }> {
const hint = await checkToolAvailability("pdftotext", TOOL_HINTS.pdftotext);
if (hint) throw new Error(hint);
const dir = await createTempDir("pi-web-");
const tmpPath = join(dir, "input.pdf");
await writeFile(tmpPath, data);
const text = await spawnToText("pdftotext", ["-layout", tmpPath, "-"]);
return { text, tmpPath };
}
async function officeDocToText(
data: Buffer,
fromFormat: string,
): Promise<{ text: string; tmpPath: string }> {
const hint = await checkToolAvailability("pandoc", TOOL_HINTS.pandoc);
if (hint) throw new Error(hint);
const dir = await createTempDir("pi-web-");
const tmpPath = join(dir, `input.${fromFormat}`);
await writeFile(tmpPath, data);
const { runPandocFile } = await loadPandoc();
const text = await runPandocFile(tmpPath, fromFormat, "plain");
return { text, tmpPath };
}
// ── Metadata extraction ──────────────────────────────────────────────────
interface ImageSummary {
format: string;
width: number;
height: number;
colorspace: string;
depth: string;
fileSize: string;
type: string;
raw: string;
}
const EMPTY_IMAGE: ImageSummary = {
format: "",
width: 0,
height: 0,
colorspace: "",
depth: "",
fileSize: "",
type: "",
raw: "Image detected but identify (ImageMagick) not available or failed.",
};
async function imageMetadata(filePath: string): Promise<ImageSummary> {
try {
const raw = await spawnToText("identify", ["-verbose", filePath]);
return {
format: raw.match(/Format:\s*(\S+)/i)?.[1] ?? "",
width: parseInt(raw.match(/Geometry:\s*(\d+)x(\d+)/i)?.[1] ?? "0", 10),
height: parseInt(raw.match(/Geometry:\s*(\d+)x(\d+)/i)?.[2] ?? "0", 10),
colorspace: raw.match(/Colorspace:\s*(\S+)/i)?.[1] ?? "",
depth: raw.match(/Depth:\s*(\S+)/i)?.[1] ?? "",
fileSize: raw.match(/Filesize:\s*(\S+)/i)?.[1] ?? "",
type: raw.match(/Type:\s*(\S+)/i)?.[1] ?? "",
raw,
};
} catch {
return EMPTY_IMAGE;
}
}
interface MediaSummary {
duration?: string;
codec?: string;
width?: number;
height?: number;
formatName?: string;
bitRate?: string;
raw: string;
}
const EMPTY_MEDIA: MediaSummary = {
raw: "Media detected but ffprobe (ffmpeg) not available or failed.",
};
async function mediaMetadata(filePath: string): Promise<MediaSummary> {
try {
const raw = await spawnToText("ffprobe", [
"-v",
"quiet",
"-print_format",
"json",
"-show_format",
"-show_streams",
filePath,
]);
if (!raw) return { raw: "No media metadata available." };
const data = JSON.parse(raw);
const fmt = data.format ?? {};
const stream = data.streams?.[0] ?? {};
let duration: string | undefined;
if (fmt.duration) {
const secs = Math.floor(parseFloat(fmt.duration));
const h = Math.floor(secs / 3600);
const m = Math.floor((secs % 3600) / 60);
const s = secs % 60;
duration =
h > 0
? `${h}:${String(m).padStart(2, "0")}:${String(s).padStart(2, "0")}`
: `${m}:${String(s).padStart(2, "0")}`;
}
return {
duration,
codec: stream.codec_name ?? fmt.format_name?.split(",")[0],
width: stream.width,
height: stream.height,
formatName: fmt.format_name,
bitRate: fmt.bit_rate
? `${Math.round(parseInt(fmt.bit_rate, 10) / 1000)}kbps`
: undefined,
raw,
};
} catch {
return EMPTY_MEDIA;
}
}
// ── Content processors ───────────────────────────────────────────────────
/** Join truthy strings with a separator, falling back to a default. */
function joinParts(
parts: (string | false | undefined)[],
fallback: string,
sep = ", ",
): string {
const filtered = parts.filter(Boolean) as string[];
return filtered.length > 0 ? filtered.join(sep) : fallback;
}
interface ContentOutput {
output: string;
tmpPath?: string;
preview?: string;
meta?: Record<string, unknown>;
/** If set, signals an early-return with binary/image/media that skips the normal truncation path. */
earlyReturn?: {
content: Array<{
type: string;
text?: string;
data?: string;
mimeType?: string;
}>;
details: Record<string, unknown>;
};
}
function errorFetchResult(url: string, error: string) {
return {
content: [{ type: "text" as const, text: error }],
details: { url, kind: "error", error },
};
}
const PLATFORM_TOOL_DEPENDENCIES = [
{
cmd: "pandoc",
hint: TOOL_HINTS.pandoc,
level: "warning" as const,
purpose: "HTML and Office document conversion unavailable",
},
{
cmd: "pdftotext",
hint: TOOL_HINTS.pdftotext,
level: "warning" as const,
purpose: "PDF extraction unavailable",
},
];
async function notifyMissingPlatformTools(
ctx: ExtensionContext,
isCurrent: () => boolean,
): Promise<void> {
if (!ctx.hasUI) return;
const hints = await Promise.all(
PLATFORM_TOOL_DEPENDENCIES.map(async (dep) => ({
dep,
hint: await checkToolAvailability(dep.cmd, dep.hint),
})),
);
if (!isCurrent()) return;
try {
for (const { dep, hint } of hints)
if (hint) ctx.ui.notify(`web: ${dep.purpose}. ${hint}`, dep.level);
} catch (error) {
if (!isStaleContextError(error)) throw error;
}
}
async function processBinary(
body: Buffer,
contentType: string,
url: string,
headerLine: (s: string) => string,
): Promise<ContentOutput> {
const ext = extensionFromContentType(contentType, url);
const tmpPath = await saveBinaryToTemp(body, ext);
const text = `${headerLine("binary, not decoded")}\nBinary saved to: ${tmpPath}`;
return {
output: text,
tmpPath,
earlyReturn: {
content: [{ type: "text", text }],
details: {
url,
kind: "binary",
contentType,
binary: true,
tmpPath,
bodySize: body.length,
},
},
};
}
async function processImage(
body: Buffer,
contentType: string,
url: string,
headerLine: (s: string) => string,
): Promise<ContentOutput> {
const ext = extensionFromContentType(contentType, url);
const tmpPath = await saveBinaryToTemp(body, ext);
const meta = await imageMetadata(tmpPath);
const desc = joinParts(
[
meta.format,
!!meta.width && !!meta.height && `${meta.width}\u00D7${meta.height}`,
meta.type,
meta.colorspace,
!!meta.depth && `${meta.depth} depth`,
meta.fileSize,
],
"image",
);
return {
output: `${headerLine(`image inline (${desc})`)}\n${desc}`,
tmpPath,
preview: desc,
meta: meta as any,
earlyReturn: {
content: [
{
type: "text",
text: `${headerLine(`image inline (${desc})`)}\n${desc}`,
},
{ type: "image", data: body.toString("base64"), mimeType: contentType },
],
details: {
url,
kind: "image",
contentType,
chars: desc.length,
bodySize: body.length,
markdown: desc,
imageInfo: meta,
},
},
};
}
async function processMedia(
body: Buffer,
contentType: string,
url: string,
_maxChars: number,
headerLine: (s: string) => string,
): Promise<ContentOutput> {
const ext = extensionFromContentType(contentType, url);
const tmpPath = await saveBinaryToTemp(body, ext);
const meta = await mediaMetadata(tmpPath);
const desc = joinParts(
[
meta.formatName,
meta.codec,
meta.duration,
!!meta.width && !!meta.height && `${meta.width}\u00D7${meta.height}`,
meta.bitRate,
],
"media",
);
const fullOutput = `${headerLine(`ffprobe JSON (${desc})`)}\n${desc}\n\n${meta.raw}`;
return {
output: fullOutput,
tmpPath,
preview: desc,
meta: meta as any,
earlyReturn: {
content: [{ type: "text", text: `${url}\n\n${fullOutput}` }],
details: {
url,
kind: "media",
format: "text" as const,
chars: fullOutput.length,
contentType,
bodySize: body.length,
tmpPath,
mediaInfo: meta,
markdown: desc,
},
},
};
}
async function processPdf(
body: Buffer,
headerLine: (s: string) => string,
): Promise<ContentOutput> {
const { text, tmpPath } = await pdfToText(body);
return {
output: `${headerLine(`extracted text via pdftotext, ${fmtChars(text.length)}`)}\n\n${text}`,
tmpPath,
preview: text,
};
}
async function processOffice(
body: Buffer,
officeFormat: string,
headerLine: (s: string) => string,
): Promise<ContentOutput> {
const { text, tmpPath } = await officeDocToText(body, officeFormat);
return {
output: `${headerLine(`extracted text via pandoc from ${officeFormat.toUpperCase()}, ${fmtChars(text.length)}`)}\n\n${text}`,
tmpPath,
preview: text,
};
}
function processJson(
body: Buffer,
headerLine: (s: string) => string,
): ContentOutput {
const raw = body.toString("utf-8");
let formatted: string;
try {
formatted = JSON.stringify(JSON.parse(raw), null, 2);
} catch {
formatted = raw;
}
return {
output: `${headerLine(`formatted JSON, ${fmtChars(formatted.length)}`)}\n\n${formatted}`,
preview: formatted,
};
}
function processText(
body: Buffer,
ctShort: string,
headerLine: (s: string) => string,
): ContentOutput {
const text = body.toString("utf-8");
const sub = ctShort.includes("csv")
? "CSV"
: ctShort.includes("tsv")
? "TSV"
: "plain text";
return {
output: `${headerLine(`${sub}, ${fmtChars(text.length)}`)}\n\n${text}`,
preview: text,
};
}
async function processHtml(
body: Buffer,
format: string,
headerLine: (s: string) => string,
): Promise<ContentOutput> {
const hint = await checkToolAvailability("pandoc", TOOL_HINTS.pandoc);
if (hint) throw new Error(hint);
const [{ extractMainContent }, { runPandoc }] = await Promise.all([
loadExtract(),
loadPandoc(),
]);
const extracted = extractMainContent(body.toString("utf-8"));
let md: string;
try {
md = await runPandoc(extracted, format);
} catch {
md = await runPandoc(extracted, format, false);
}
const fmtLabel =
format === "text"
? "plain text"
: format === "json"
? "Pandoc AST JSON"
: format === "html"
? "extracted HTML"
: "markdown";
return {
output: `${headerLine(`${fmtLabel} via pandoc, ${fmtChars(md.length)}`)}\n\n${md}`,
preview: md,
};
}
interface ConvertInput {
kind: ContentKind;
body: Buffer;
contentType: string;
url: string;
format: string;
maxChars: number;
ctShort: string;
officeFormat?: string;
headerLine: (conversion: string) => string;
}
async function convertContent(input: ConvertInput): Promise<ContentOutput> {
switch (input.kind) {
case "binary":
return processBinary(
input.body,
input.contentType,
input.url,
input.headerLine,
);
case "image":
return processImage(
input.body,
input.contentType,
input.url,
input.headerLine,
);
case "media":
return processMedia(
input.body,
input.contentType,
input.url,
input.maxChars,
input.headerLine,
);
case "pdf":
return processPdf(input.body, input.headerLine);
case "office": {
if (!input.officeFormat)
throw new Error("Missing office document format");
return processOffice(input.body, input.officeFormat, input.headerLine);
}
case "json":
return processJson(input.body, input.headerLine);
case "text":
return processText(input.body, input.ctShort, input.headerLine);
case "html":
return processHtml(input.body, input.format, input.headerLine);
}
}
function truncateText(text: string, maxChars: number): string {
return text.length > maxChars
? text.slice(0, maxChars) + "\n… (truncated)"
: text;
}
function progressMessage(loaded: number, total: number): string {
const mb = (loaded / 1_048_576).toFixed(1);
if (total <= 0) return `⬇ ${mb}MB`;
const totalMb = (total / 1_048_576).toFixed(1);
const percent = Math.round((loaded / total) * 100);
return `⬇ ${mb}/${totalMb}MB (${percent}%)`;
}
function safeTerminalText(text: string): string {
return text
.replace(
/\x1B(?:[@-Z\\-_]|\[[0-?]*[ -/]*[@-~]|\][^\x07]*(?:\x07|\x1B\\))/g,
"",
)
.replace(/[\x00-\x08\x0B\x0C\x0E-\x1F\x7F-\x9F]/g, " ");
}
function safeDisplayText(text: string): string {
return safeTerminalText(text)
.replace(/[\t\r\n ]+/g, " ")
.trim();
}
/** Render expanded tool text: prefix + each line dimmed. Returns null if content is not text. */
function expandedText(
result: { content: Array<{ type: string; text?: string }> },
prefix: string,
theme: { fg: (...a: any[]) => string },
): string | null {
const content = result.content[0];
if (content?.type !== "text") return null;
let text = prefix;
for (const line of safeTerminalText(content.text ?? "").split("\n")) {
text += `\n${theme.fg("dim", line)}`;
}
return text;
}
function hintBox(
theme: { fg: (...a: any[]) => string },
style: "warning" | "error",
msg: string,
): Text {
return new Text(theme.fg(style, safeDisplayText(msg)), 0, 0);
}
function expandableError(
theme: { fg: (...a: any[]) => string },
msg: string,
expanded: boolean,
): Text {
const full = safeTerminalText(msg).trim();
if (expanded) return new Text(theme.fg("error", full), 0, 0);
const summary = safeDisplayText(full);
const clipped = summary.length > 160 ? `${summary.slice(0, 159)}…` : summary;
let text = theme.fg("error", clipped);
if (clipped !== summary)
text += `\n${theme.fg("muted", "↳")} ${keyHint("app.tools.expand", "to expand")}`;
return new Text(text, 0, 0);
}
/** Extract error message from a tool result, falling back to content text or a default. */
function errorMsg(result: any, fallback: string): string {
return (
(result.details as { error?: string } | undefined)?.error ||
(result.content[0]?.type === "text"
? (result.content[0].text ?? fallback)
: fallback)
);
}
type WebAction = Pick<
ToolDefinition,
"parameters" | "execute" | "renderCall" | "renderResult"
>;
export default function webExtension(pi: ExtensionAPI) {
// ── Session-scoped spotlight markers ───────────────────────────────────
const spotlightMarkers: SpotlightMarkers = createMarkers();
let generation = 0;
pi.on("session_start", (_event, ctx) => {
dbg?.("session.start");
const active = pi.getActiveTools();
if (!active.includes("web")) pi.setActiveTools([...active, "web"]);
const ticket = ++generation;
for (const read of SETTING_READERS) {
try {
read(ctx);
} catch (error) {
if (!(error instanceof Error)) throw error;
if (ctx.hasUI) ctx.ui.notify(`web: ${error.message}`, "warning");
}
}
void notifyMissingPlatformTools(ctx, () => ticket === generation).catch(
(error) => {
if (ticket === generation && ctx.hasUI)
ctx.ui.notify(
`web platform tool check failed: ${error instanceof Error ? error.message : String(error)}`,
"warning",
);
},
);
});
// ── Cleanup temp files on session shutdown ─────────────────────────────
pi.on("session_shutdown", async () => {
dbg?.("session.shutdown");
closeDebug();
generation++;
const ownedDirs = [...tempDirs];
tempDirs.clear();
await Promise.all(
ownedDirs.map((dir) =>
rm(dir, { recursive: true, force: true }).catch(() => {}),
),
);
});
const fetchAction = {
parameters: Type.Object({
url: Type.String({ description: "URL to fetch" }),
format: Type.Optional(
StringEnum(["markdown", "text", "html", "json", "raw"], {
description:
"Default markdown; json is Pandoc AST for HTML; raw skips conversion.",
}),
),
maxChars: Type.Optional(
Type.Number({
description: `Visible characters (default ${DEFAULT_MAX_CHARS}, max ${MAX_VISIBLE_CHARS}).`,
minimum: 100,
maximum: MAX_VISIBLE_CHARS,
}),
),
maxBytes: Type.Optional(
Type.Number({
description: "Maximum download bytes.",
minimum: 1_000,
maximum: 100_000_000,
}),
),
linesMatching: Type.Optional(
Type.Array(Type.String(), {
description: "Literal substrings; return matching lines only.",
}),
),
contextLines: Type.Optional(
Type.Number({
description: "Context lines per match (default 0, max 100).",
minimum: 0,
maximum: 100,
}),
),
caseSensitive: Type.Optional(
Type.Boolean({
description: "Case-sensitive matching (default false).",
}),
),
timeout: Type.Optional(
Type.Number({
description: `Timeout in ms (default ${DEFAULT_TIMEOUT_MS}).`,
minimum: 5000,
maximum: 600_000,
}),
),
}),
renderCall(args: any, theme) {
let text = theme.fg("toolTitle", theme.bold("\u2B07 web fetch "));
text += theme.fg("accent", safeDisplayText(args.url));
return new Text(text, 0, 0);
},
renderResult(result, { expanded, isPartial }, theme) {
if (isPartial) {
return hintBox(theme, "warning", "\u2B07 Fetching...");
}
const details = result.details as
| {
url?: string;
kind?: string;
format?: string;
chars?: number;
bodySize?: number;
contentType?: string;
tmpPath?: string;
preview?: string;
binary?: boolean;
officeFormat?: string;
imageInfo?: ImageSummary;
mediaInfo?: MediaSummary;
error?: string;
}
| undefined;
if ((result as any).isError || details?.kind === "error") {
const content = result.content[0];
const msg =
details?.error ||
(content?.type === "text" ? content.text : "Fetch failed");
return expandableError(theme, `\u2717 ${msg}`, expanded);
}
const kind = details?.kind ?? "text";
const ct = details?.contentType ?? "unknown";
const bodySize = details?.bodySize ?? 0;
const chars = details?.chars ?? 0;
const visibleChars =
(details as { visibleChars?: number } | undefined)?.visibleChars ??
chars;
const responseId = (details as { responseId?: string } | undefined)
?.responseId;
const isRaw = details?.format === "raw";
let header = theme.fg("success", "\u2713 ");
header += theme.fg("accent", fmtSize(bodySize));
header += theme.fg("dim", ` ${stripCt(ct)}`);
if (details?.imageInfo) {
header += theme.fg(
"dim",
` ${details.imageInfo.width}\u00D7${details.imageInfo.height}`,
);
}
if (details?.mediaInfo) {
const mi = details.mediaInfo;
const parts: string[] = [];
if (mi.duration) parts.push(mi.duration);
if (mi.width && mi.height) parts.push(`${mi.width}\u00D7${mi.height}`);
if (mi.codec) parts.push(mi.codec);
if (parts.length) header += theme.fg("dim", ` (${parts.join(", ")})`);
}
let llmDesc: string;
if (isRaw) {
llmDesc = `${fmtChars(visibleChars)} of raw content (no conversion)`;
} else {
switch (kind) {
case "html": {
const fmt = details?.format;
const typeLabel =
fmt === "text"
? "plain text"
: fmt === "json"
? "Pandoc AST JSON"
: fmt === "html"
? "extracted HTML"
: "markdown";
llmDesc = `${fmtChars(visibleChars)} of ${typeLabel} (full saved)`;
break;
}
case "text": {
const sub = ct.includes("csv")
? "CSV text"
: ct.includes("tsv")
? "TSV text"
: "plain text";
llmDesc = `${fmtChars(visibleChars)} of ${sub} (full saved)`;
break;
}
case "json":
llmDesc = `${fmtChars(visibleChars)} of formatted JSON (full saved)`;
break;
case "pdf":
llmDesc = `${fmtChars(visibleChars)} of extracted text (full saved)`;
break;
case "office":
llmDesc = `${fmtChars(visibleChars)} of extracted text (full saved)`;
break;
case "image":
llmDesc = `image inline (${details?.imageInfo?.format ?? ct.split("/")[1] ?? "?"}) + ${fmtChars(chars)} metadata text`;
break;
case "media": {
const dur = details?.mediaInfo?.duration;
llmDesc = `${fmtChars(visibleChars)} of ffprobe JSON metadata${dur ? ` (${dur})` : ""}`;
break;
}
case "binary":
llmDesc = "file path + content-type only (binary, not decoded)";
break;
default:
llmDesc = `${fmtChars(visibleChars)} of text (full saved)`;
}
}
header += `\n ${theme.fg("muted", "\u21B3 Agent sees:")} ${theme.fg("dim", llmDesc)}`;
if (responseId)
header += `\n ${theme.fg("muted", "\u21B3 Saved:")} ${theme.fg("dim", responseId)}`;
const content = result.content[0];
const hasExpandableText =
content?.type === "text" && !!content.text?.trim();
if (!expanded && hasExpandableText)
header += `\n ${theme.fg("muted", "\u21B3")} ${theme.fg("dim", keyHint("app.tools.expand", "to expand"))}`;
const mdTheme = getMarkdownTheme();
const box = new Box(1, 0, (t) => theme.bg("toolSuccessBg", t));
box.addChild(new Text(header, 1, 0));
if (kind === "image" && details?.imageInfo) {
const ii = details.imageInfo;
let metaMd = `**${ii.format || "Image"}** ${ii.width}\u00D7${ii.height}`;
if (ii.type) metaMd += ` \u00B7 ${ii.type}`;
metaMd += "\n";
if (ii.colorspace) metaMd += `\n- Color: ${ii.colorspace}`;
if (ii.depth) metaMd += `\n- Depth: ${ii.depth}`;
if (ii.fileSize) metaMd += `\n- File size: ${ii.fileSize}`;
box.addChild(new Markdown(metaMd, 1, 1, mdTheme));
return box;
}
if (kind === "media" && details?.mediaInfo) {
const mi = details.mediaInfo;
let metaMd = `**${mi.formatName || "Media"}**`;
if (mi.codec) metaMd += ` \u00B7 ${mi.codec}`;
metaMd += "\n";
if (mi.duration) metaMd += `\n- Duration: ${mi.duration}`;
if (mi.width && mi.height)
metaMd += `\n- Resolution: ${mi.width}\u00D7${mi.height}`;
if (mi.bitRate) metaMd += `\n- Bitrate: ${mi.bitRate}`;
box.addChild(new Markdown(metaMd, 1, 1, mdTheme));
return box;
}
if (expanded && details?.preview) {
box.addChild(
new Markdown(safeTerminalText(details.preview), 1, 1, mdTheme),
);
return box;
}
if (expanded && content?.type === "text") {
box.addChild(
new Markdown(safeTerminalText(content.text), 1, 1, mdTheme),
);
return box;
}
return box;
},
async execute(
_toolCallId: any,
params: any,
signal: any,
onUpdate: any,
ctx: any,
): Promise<any> {
const format = params.format ?? "markdown";
const maxChars = Math.min(
params.maxChars ?? DEFAULT_MAX_CHARS,
MAX_VISIBLE_CHARS,
);
const setStatus = (msg: string) => ctx?.ui?.setStatus("web", msg);
const clearStatus = () => ctx?.ui?.setStatus("web", undefined);
onUpdate?.({
content: [
{ type: "text" as const, text: `Fetching ${params.url}\u2026` },
],
details: {},
});
const onProgress = (loaded: number, total: number) =>
setStatus(progressMessage(loaded, total));
// ── Fetch ────────────────────────────────────────────────────
let body: Buffer;
let contentType: string;
let finalUrl = params.url;
try {
const { fetchUrlStream } = await loadFetch();
const result = await fetchUrlStream(params.url, onProgress, signal, {
maxBytes: params.maxBytes,
timeoutMs: params.timeout,
});
body = result.body;
contentType = result.contentType;
finalUrl = result.finalUrl;
} catch (err) {
clearStatus();
return errorFetchResult(
params.url,
`Error fetching ${params.url}: ${(err as Error).message}`,
);
}
// ── Route by content type ────────────────────────────────────
const { kind, officeFormat } = classifyContentType(contentType);
onUpdate?.({
content: [
{ type: "text" as const, text: `Converting ${kind} content\u2026` },
],
details: {},
});
const ctShort = stripCt(contentType);
const bodyLen = body.length;
const headerLine = (conversion: string) =>
`[${ctShort} | ${fmtSize(bodyLen)} | \u2192 ${conversion}]`;
try {
const processed =
format === "raw"
? { output: body.toString("utf-8") }
: await convertContent({
kind,
body,
contentType,
url: finalUrl,
format,
maxChars,
ctShort,
officeFormat,
headerLine,
});
const fullOutput = processed.output;
const [{ saveJsonResult }, { filterLineMatches }] = await Promise.all([
loadStore(),
loadLineMatch(),
]);
const stored = await saveJsonResult("web", ctx?.cwd ?? process.cwd(), {
url: params.url,
finalUrl,
kind,
format,
contentType,
bodySize: bodyLen,
output: fullOutput,
...(processed.tmpPath && { tmpPath: processed.tmpPath }),
...(officeFormat && { officeFormat }),
});
const matchSource =
"preview" in processed && processed.preview
? processed.preview
: fullOutput;
const matched = params.linesMatching?.length
? filterLineMatches(matchSource, {
needles: params.linesMatching,
contextLines: params.contextLines,
caseSensitive: params.caseSensitive,
maxChars,
})
: undefined;
const visible = matched
? matched.text ||
`No lines matched: ${params.linesMatching.join(", ")}`
: truncateText(fullOutput, maxChars);
const outputOmitted = matched ? true : fullOutput.length > maxChars;
const recovery = outputOmitted
? `\nfullOutputPath: ${stored.fullOutputPath}`
: "";
clearStatus();
return {
content: [
{
type: "text",
text: wrapUntrusted(
`${finalUrl}\n\n${visible}\n\nresponseId: ${stored.responseId}${recovery}`,
spotlightMarkers,
),
},
],
details: {
url: params.url,
finalUrl,
kind,
format,
chars: fullOutput.length,
visibleChars: visible.length,
truncated: visible.length < fullOutput.length,
contentType,
bodySize: bodyLen,
responseId: stored.responseId,
fullOutputPath: stored.fullOutputPath,
storedBytes: stored.byteLength,
...(processed.tmpPath && { tmpPath: processed.tmpPath }),
...(officeFormat && { officeFormat }),
...(processed.meta && { meta: processed.meta }),
...(matched && {
lineMatches: {
matchCount: matched.matchCount,
truncated: matched.truncated,
},
}),
},
};
} catch (err) {
clearStatus();
return errorFetchResult(params.url, (err as Error).message);
}
},
} satisfies WebAction;
const mapAction = {
parameters: Type.Object({
url: Type.String({ description: "Site URL" }),
maxUrls: Type.Optional(
Type.Number({
description: "URL limit (default 1000).",
minimum: 1,
maximum: 10_000,
}),
),
maxSitemaps: Type.Optional(
Type.Number({
description: "Sitemap-file limit (default 20).",
minimum: 1,
maximum: 100,
}),
),
maxChars: Type.Optional(
Type.Number({
description: `Visible characters (default 8000, max ${MAX_VISIBLE_CHARS}).`,
minimum: 100,
maximum: MAX_VISIBLE_CHARS,
}),
),
}),
renderCall(args: any, theme) {
let text = theme.fg("toolTitle", theme.bold("🗺 web map "));
text += theme.fg("accent", safeDisplayText(args.url));
return new Text(text, 0, 0);
},
renderResult(result, { expanded, isPartial }, theme) {
if (isPartial)
return hintBox(theme, "warning", "🗺 Mapping site metadata…");
const details = result.details as
| {
urlCount?: number;
sitemapCount?: number;
responseId?: string;
fullOutputPath?: string;
}
| undefined;
if ((result.details as { kind?: string } | undefined)?.kind === "error") {
const content = result.content[0];
const msg = content?.type === "text" ? content.text : "Map failed";
return expandableError(theme, `✗ ${msg}`, expanded);
}
let text = theme.fg("success", "✓ ");
text += theme.fg("accent", `${details?.urlCount ?? 0} URL(s)`);
text += theme.fg(
"dim",
` · ${details?.sitemapCount ?? 0} sitemap candidate(s)`,
);
if (details?.responseId)
text += theme.fg("dim", ` · saved ${details.responseId}`);
const content = result.content[0];
const hasDetails =
(content?.type === "text" && !!content.text?.trim()) ||
!!details?.fullOutputPath;
if (expanded) {
if (details?.fullOutputPath)
text += `\n ${theme.fg("muted", "full:")} ${theme.fg("dim", details.fullOutputPath)}`;
const rendered = expandedText(result, text, theme);
if (rendered !== null) text = rendered;
} else if (hasDetails) {
text += `\n ${theme.fg("muted", "↳")} ${keyHint("app.tools.expand", "to expand")}`;
}
return new Text(text, 0, 0);
},
async execute(_toolCallId, params: any, signal, onUpdate, ctx) {
onUpdate?.({
content: [{ type: "text", text: `Mapping ${params.url}…` }],
details: {},
});
try {
const [{ discoverSiteMap }, { saveJsonResult }] = await Promise.all([
loadMap(),
loadStore(),
]);
const map = await discoverSiteMap(
params.url,
{ maxUrls: params.maxUrls, maxSitemaps: params.maxSitemaps },
signal,
);
const stored = await saveJsonResult(
"web",
ctx?.cwd ?? process.cwd(),
map,
);
const listed = map.urls
.slice(0, 50)
.map((entry) => `- ${entry.url}`)
.join("\n");
const fullListing = `Mapped ${map.urls.length} URL(s) from ${map.sitemaps.length} sitemap candidate(s).\n\n${listed}`;
const maxChars = params.maxChars ?? 8_000;
const body = truncateText(fullListing, maxChars);
const partialReasons: string[] = [];
if (map.urls.length > 50)
partialReasons.push(`showing at most 50 of ${map.urls.length} URLs`);
if (fullListing.length > maxChars)
partialReasons.push(`visible text capped at ${maxChars} characters`);
const recovery = partialReasons.length
? `\n\nPartial URL listing: ${partialReasons.join("; ")}.\nfullOutputPath: ${stored.fullOutputPath}`
: "";
return {
content: [
{
type: "text",
text: wrapUntrusted(
`${body}\n\nresponseId: ${stored.responseId}${recovery}`,
spotlightMarkers,
),
},
],
details: {
url: params.url,
kind: "map",
urlCount: map.urls.length,
sitemapCount: map.sitemaps.length,
responseId: stored.responseId,
fullOutputPath: stored.fullOutputPath,
},
};
} catch (err) {
return errorFetchResult(params.url, (err as Error).message);
}
},
} satisfies WebAction;
const searchAction = {
parameters: Type.Object({
query: Type.String({ description: "Search query" }),
count: Type.Optional(
Type.Number({
description: "Result count (default 5).",
minimum: 1,
maximum: 20,
}),
),
format: Type.Optional(
StringEnum(["text", "json"], {
description: "Default: text.",
}),
),
timeout: Type.Optional(
Type.Number({
description: `Timeout in ms (default ${DEFAULT_TIMEOUT_MS}).`,
minimum: 5000,
maximum: 600_000,
}),
),
}),
renderCall(args: any, theme) {
let text = theme.fg("toolTitle", theme.bold("\uD83D\uDD0D web search "));
text += theme.fg("accent", `"${safeDisplayText(args.query)}"`);
return new Text(text, 0, 0);
},
renderResult(result, { expanded, isPartial }, theme) {
if (isPartial) {
return hintBox(theme, "warning", "Searching...");
}
if ((result as any).isError || (result.details as any)?.error) {
return expandableError(
theme,
`\u2717 ${errorMsg(result, "Search failed")}`,
expanded,
);
}
const details = result.details as
| { query?: string; results?: SearchResult[]; source?: string }
| undefined;
if (expanded) {
const rendered = expandedText(result, "", theme);
if (rendered !== null) return new Text(rendered, 0, 0);
}
const resultCount = details?.results?.length ?? 0;
let text = theme.fg(
"success",
`\u2713 ${resultCount} result${resultCount === 1 ? "" : "s"}`,
);
if (details?.source)
text += ` ${theme.fg("muted", `(${details.source})`)}`;
if (!expanded && resultCount > 0)
text += ` ${theme.fg("muted", `(${keyHint("app.tools.expand", "to expand")})`)}`;
return new Text(text, 0, 0);
},
async execute(
_toolCallId: any,
params: any,
signal: any,
onUpdate: any,
ctx: any,
): Promise<any> {
const count = Math.min(params.count ?? 5, 20);
const format = params.format ?? "text";
const timeout = params.timeout ?? DEFAULT_TIMEOUT_MS;
onUpdate?.({
content: [
{
type: "text" as const,
text: `Searching for "${params.query}"\u2026`,
},
],
details: {},
});
const { searchWeb } = await loadSearch();
const { results, source } = await searchWeb({
query: params.query,
count,
timeout,
signal,
ctx,
});
if (!results || results.length === 0) {
return {
content: [
{ type: "text", text: `No results found for: ${params.query}` },
],
details: { query: params.query, results: [], source },
};
}
const sliced = results.slice(0, count);
if (format === "json") {
return {
content: [
{
type: "text",
text: wrapUntrusted(
JSON.stringify(
{
query: params.query,
count: sliced.length,
results: sliced,
source,
},
null,
2,
),
spotlightMarkers,
),
},
],
details: { query: params.query, results: sliced, source },
};
}
const lines: string[] = [
`Search: ${params.query}`,
`Results: ${sliced.length}${results.length > sliced.length ? ` of ${results.length}` : ""} (${source})`,
"",
];
for (let i = 0; i < sliced.length; i++) {
const r = sliced[i];
lines.push(`[${i + 1}] ${r.title}`);
lines.push(` ${r.url}`);
if (r.snippet) lines.push(` ${r.snippet}`);
if (r.date) lines.push(` Date: ${r.date}`);
lines.push("");
}
return {
content: [
{
type: "text",
text: wrapUntrusted(lines.join("\n"), spotlightMarkers),
},
],
details: { query: params.query, results: sliced, source },
};
},
} satisfies WebAction;
const actions = { fetch: fetchAction, map: mapAction, search: searchAction };
pi.registerTool({
name: "web",
label: "web",
description:
"Fetch HTTP(S) pages, documents, or media metadata; map site URLs from robots.txt, sitemaps, and llms.txt without page bodies; search for titles, URLs, and snippets.",
promptSnippet: "Fetch web content, discover site URLs, or search the web.",
parameters: Type.Object(
{
...fetchAction.parameters.properties,
...mapAction.parameters.properties,
...searchAction.parameters.properties,
action: StringEnum(["fetch", "map", "search"] as const),
url: Type.Optional(
Type.String({ description: "Required for fetch/map." }),
),
query: Type.Optional(
Type.String({ description: "Required for search." }),
),
format: Type.Optional(
StringEnum(["markdown", "text", "html", "json", "raw"], {
description:
"Fetch: default markdown; json is Pandoc AST for HTML; raw skips conversion. Search: text (default) or json.",
}),
),
maxChars: Type.Optional(
Type.Number({
description: `Fetch: default ${DEFAULT_MAX_CHARS}, max ${MAX_VISIBLE_CHARS}. Map: default 8000, max ${MAX_VISIBLE_CHARS}.`,
minimum: 100,
maximum: MAX_VISIBLE_CHARS,
}),
),
},
{ additionalProperties: false },
),
async execute(id, { action, ...params }, signal, onUpdate, ctx) {
if (!Object.hasOwn(actions, action)) {
dbg?.("action.invalid", { type: action });
throw new Error("Expected action: fetch, map, or search");
}
const handler = actions[action];
// Keep action-specific required fields, limits, and formats without
// duplicating their schemas in every model request.
const { Check, Errors } = await import("typebox/value");
const schema = { ...handler.parameters, additionalProperties: false };
if (!Check(schema, params)) {
dbg?.("action.invalid", { type: action });
const details = [...Errors(schema, params)]
.map((error) => {
const path = "path" in error ? error.path : "arguments";
const value =
"value" in error ? JSON.stringify(error.value) : "unknown";
return `${path} ${error.message} (received ${value})`;
})
.join("; ");
throw new Error(`Invalid web ${action} arguments: ${details}`);
}
const finish = span?.("action.execute", { type: action });
try {
const result = await handler.execute(id, params, signal, onUpdate, ctx);
finish?.();
return result;
} catch (error) {
finish?.("error");
throw error;
}
},
renderCall(args, theme) {
if (Object.hasOwn(actions, args.action))
return actions[args.action].renderCall(args, theme);
return new Text(theme.fg("toolTitle", theme.bold("web")), 0, 0);
},
renderResult(result, options, theme, context) {
if (context.isError)
return expandableError(
theme,
errorMsg(result, "Web request failed"),
options.expanded,
);
const action = context.args.action as keyof typeof actions;
if (Object.hasOwn(actions, action))
return actions[action].renderResult(result, options, theme);
return hintBox(theme, "warning", "Waiting for web action...");
},
});
}