/** * Web Extension for pi * * Tool: web, with fetch, map, and search actions. * * Uses Node.js native fetch (HTTP/1.1 via undici) which helps avoid * bot detection compared to HTTP/2 clients like curl. * * Content-Type routing: * text/html → extract main content → pandoc + lua filter → markdown * text/*, json, csv → return as-is * application/pdf → pdftotext -layout * docx, odt, epub → pandoc direct read * image/* → identify -verbose (metadata) * video/audio → ffprobe (metadata) */ import { mkdtemp, rm, writeFile } from "node:fs/promises"; import { tmpdir } from "node:os"; import { join } from "node:path"; import { StringEnum } from "@earendil-works/pi-ai"; import type { ExtensionAPI, ExtensionContext, ToolDefinition, } from "@earendil-works/pi-coding-agent"; import { getMarkdownTheme, keyHint } from "@earendil-works/pi-coding-agent"; import { Box, Markdown, Text } from "@earendil-works/pi-tui"; import { Type } from "typebox"; import { DEFAULT_MAX_CHARS, DEFAULT_TIMEOUT_MS, MAX_VISIBLE_CHARS, type SearchResult, } from "./constants.js"; import { lazy } from "./lazy.js"; import { SETTING_READERS } from "./settings.ts"; import { checkToolAvailability, spawnToText, TOOL_HINTS } from "./spawn.js"; import { createMarkers, type SpotlightMarkers, wrapUntrusted, } from "./spotlight.js"; import { closeDebug, dbg, span } from "./src/debug.ts"; import { isStaleContextError } from "./src/pi-ext-stale-context.ts"; // ── Lazy implementation modules ─────────────────────────────────────────── // Loaded on first tool execution to keep extension startup fast. // The shared promise also prevents jiti (pi's loader, module cache disabled) // from handing concurrent tool calls a partially initialized module. const loadFetch = lazy(() => import("./fetch.js")); const loadExtract = lazy(() => import("./extract.js")); const loadPandoc = lazy(() => import("./pandoc.js")); const loadStore = lazy(() => import("./store.js")); const loadLineMatch = lazy(() => import("./line-match.js")); const loadMap = lazy(() => import("./map.js")); const loadSearch = lazy(() => import("./search.js")); // ── Shared formatters ──────────────────────────────────────────────────── function fmtSize(bytes: number): string { return bytes >= 1_000_000 ? `${(bytes / 1_048_576).toFixed(1)}MB` : bytes >= 1_000 ? `${(bytes / 1_024).toFixed(1)}KB` : `${bytes}B`; } function fmtChars(n: number): string { return n >= 10_000 ? `${(n / 1_000).toFixed(1)}K chars` : `${n} chars`; } // ── Temp file lifecycle ─────────────────────────────────────────────────── const tempDirs = new Set(); async function createTempDir(prefix: string): Promise { const dir = await mkdtemp(join(tmpdir(), prefix)); tempDirs.add(dir); return dir; } // ── Content-Type classification ────────────────────────────────────────── type ContentKind = | "html" | "text" | "json" | "pdf" | "office" | "image" | "media" | "binary"; interface Classification { kind: ContentKind; officeFormat?: string; } interface ClassificationRule { test: (ct: string) => boolean; result: Classification; } const CLASSIFICATION_RULES: ClassificationRule[] = [ { test: (ct) => ct.includes("text/html") || ct.includes("application/xhtml"), result: { kind: "html" }, }, { test: (ct) => ct.includes("application/json"), result: { kind: "json" } }, { test: (ct) => ct.includes("application/xml") || ct.includes("application/yaml"), result: { kind: "text" }, }, { test: (ct) => ct.startsWith("text/") || ct.includes("csv") || ct.includes("tsv"), result: { kind: "text" }, }, { test: (ct) => ct.includes("application/pdf"), result: { kind: "pdf" } }, { test: (ct) => ct.includes("officedocument.wordprocessingml") || ct.includes("msword"), result: { kind: "office", officeFormat: "docx" }, }, { test: (ct) => ct.includes("opendocument.text"), result: { kind: "office", officeFormat: "odt" }, }, { test: (ct) => ct.includes("application/epub"), result: { kind: "office", officeFormat: "epub" }, }, { test: (ct) => ct.includes("application/rtf") || ct.includes("text/richtext"), result: { kind: "office", officeFormat: "rtf" }, }, { test: (ct) => ct.startsWith("image/"), result: { kind: "image" } }, { test: (ct) => ct.startsWith("video/") || ct.startsWith("audio/") || ct.includes("mpeg") || ct.includes("mp4") || ct.includes("webm") || ct.includes("ogg"), result: { kind: "media" }, }, ]; /** Strip parameters and lowercase a Content-Type header value. */ function stripCt(ct: string): string { return ct.toLowerCase().split(";")[0].trim(); } function classifyContentType(ct: string): Classification { return ( CLASSIFICATION_RULES.find((r) => r.test(stripCt(ct)))?.result ?? { kind: "binary", } ); } // ── Extension from media type ──────────────────────────────────────────── const EXTENSION_MAP: Record = { pdf: "pdf", docx: "docx", msword: "doc", odt: "odt", epub: "epub", rtf: "rtf", png: "png", gif: "gif", webp: "webp", jpeg: "jpg", jpg: "jpg", svg: "svg", mp4: "mp4", webm: "webm", ogg: "ogg", mp3: "mp3", wav: "wav", }; function extensionFromContentType(ct: string, url: string): string { const lower = stripCt(ct); for (const [needle, ext] of Object.entries(EXTENSION_MAP)) { if (lower.includes(needle)) return ext; } if (lower.startsWith("image/")) return "bin"; if (lower.startsWith("video/")) return "mp4"; if (lower.startsWith("audio/")) return "mp3"; try { const urlExt = new URL(url).pathname.match(/\.([^.]+)$/)?.[1]; if (urlExt) return urlExt.toLowerCase(); } catch {} return "bin"; } // ── Binary/temp helpers ────────────────────────────────────────────────── async function saveBinaryToTemp( data: Buffer, extension: string, ): Promise { const dir = await createTempDir("pi-web-"); const tmpPath = join(dir, `download.${extension}`); await writeFile(tmpPath, data); return tmpPath; } async function pdfToText( data: Buffer, ): Promise<{ text: string; tmpPath: string }> { const hint = await checkToolAvailability("pdftotext", TOOL_HINTS.pdftotext); if (hint) throw new Error(hint); const dir = await createTempDir("pi-web-"); const tmpPath = join(dir, "input.pdf"); await writeFile(tmpPath, data); const text = await spawnToText("pdftotext", ["-layout", tmpPath, "-"]); return { text, tmpPath }; } async function officeDocToText( data: Buffer, fromFormat: string, ): Promise<{ text: string; tmpPath: string }> { const hint = await checkToolAvailability("pandoc", TOOL_HINTS.pandoc); if (hint) throw new Error(hint); const dir = await createTempDir("pi-web-"); const tmpPath = join(dir, `input.${fromFormat}`); await writeFile(tmpPath, data); const { runPandocFile } = await loadPandoc(); const text = await runPandocFile(tmpPath, fromFormat, "plain"); return { text, tmpPath }; } // ── Metadata extraction ────────────────────────────────────────────────── interface ImageSummary { format: string; width: number; height: number; colorspace: string; depth: string; fileSize: string; type: string; raw: string; } const EMPTY_IMAGE: ImageSummary = { format: "", width: 0, height: 0, colorspace: "", depth: "", fileSize: "", type: "", raw: "Image detected but identify (ImageMagick) not available or failed.", }; async function imageMetadata(filePath: string): Promise { try { const raw = await spawnToText("identify", ["-verbose", filePath]); return { format: raw.match(/Format:\s*(\S+)/i)?.[1] ?? "", width: parseInt(raw.match(/Geometry:\s*(\d+)x(\d+)/i)?.[1] ?? "0", 10), height: parseInt(raw.match(/Geometry:\s*(\d+)x(\d+)/i)?.[2] ?? "0", 10), colorspace: raw.match(/Colorspace:\s*(\S+)/i)?.[1] ?? "", depth: raw.match(/Depth:\s*(\S+)/i)?.[1] ?? "", fileSize: raw.match(/Filesize:\s*(\S+)/i)?.[1] ?? "", type: raw.match(/Type:\s*(\S+)/i)?.[1] ?? "", raw, }; } catch { return EMPTY_IMAGE; } } interface MediaSummary { duration?: string; codec?: string; width?: number; height?: number; formatName?: string; bitRate?: string; raw: string; } const EMPTY_MEDIA: MediaSummary = { raw: "Media detected but ffprobe (ffmpeg) not available or failed.", }; async function mediaMetadata(filePath: string): Promise { try { const raw = await spawnToText("ffprobe", [ "-v", "quiet", "-print_format", "json", "-show_format", "-show_streams", filePath, ]); if (!raw) return { raw: "No media metadata available." }; const data = JSON.parse(raw); const fmt = data.format ?? {}; const stream = data.streams?.[0] ?? {}; let duration: string | undefined; if (fmt.duration) { const secs = Math.floor(parseFloat(fmt.duration)); const h = Math.floor(secs / 3600); const m = Math.floor((secs % 3600) / 60); const s = secs % 60; duration = h > 0 ? `${h}:${String(m).padStart(2, "0")}:${String(s).padStart(2, "0")}` : `${m}:${String(s).padStart(2, "0")}`; } return { duration, codec: stream.codec_name ?? fmt.format_name?.split(",")[0], width: stream.width, height: stream.height, formatName: fmt.format_name, bitRate: fmt.bit_rate ? `${Math.round(parseInt(fmt.bit_rate, 10) / 1000)}kbps` : undefined, raw, }; } catch { return EMPTY_MEDIA; } } // ── Content processors ─────────────────────────────────────────────────── /** Join truthy strings with a separator, falling back to a default. */ function joinParts( parts: (string | false | undefined)[], fallback: string, sep = ", ", ): string { const filtered = parts.filter(Boolean) as string[]; return filtered.length > 0 ? filtered.join(sep) : fallback; } interface ContentOutput { output: string; tmpPath?: string; preview?: string; meta?: Record; /** If set, signals an early-return with binary/image/media that skips the normal truncation path. */ earlyReturn?: { content: Array<{ type: string; text?: string; data?: string; mimeType?: string; }>; details: Record; }; } function errorFetchResult(url: string, error: string) { return { content: [{ type: "text" as const, text: error }], details: { url, kind: "error", error }, }; } const PLATFORM_TOOL_DEPENDENCIES = [ { cmd: "pandoc", hint: TOOL_HINTS.pandoc, level: "warning" as const, purpose: "HTML and Office document conversion unavailable", }, { cmd: "pdftotext", hint: TOOL_HINTS.pdftotext, level: "warning" as const, purpose: "PDF extraction unavailable", }, ]; async function notifyMissingPlatformTools( ctx: ExtensionContext, isCurrent: () => boolean, ): Promise { if (!ctx.hasUI) return; const hints = await Promise.all( PLATFORM_TOOL_DEPENDENCIES.map(async (dep) => ({ dep, hint: await checkToolAvailability(dep.cmd, dep.hint), })), ); if (!isCurrent()) return; try { for (const { dep, hint } of hints) if (hint) ctx.ui.notify(`web: ${dep.purpose}. ${hint}`, dep.level); } catch (error) { if (!isStaleContextError(error)) throw error; } } async function processBinary( body: Buffer, contentType: string, url: string, headerLine: (s: string) => string, ): Promise { const ext = extensionFromContentType(contentType, url); const tmpPath = await saveBinaryToTemp(body, ext); const text = `${headerLine("binary, not decoded")}\nBinary saved to: ${tmpPath}`; return { output: text, tmpPath, earlyReturn: { content: [{ type: "text", text }], details: { url, kind: "binary", contentType, binary: true, tmpPath, bodySize: body.length, }, }, }; } async function processImage( body: Buffer, contentType: string, url: string, headerLine: (s: string) => string, ): Promise { const ext = extensionFromContentType(contentType, url); const tmpPath = await saveBinaryToTemp(body, ext); const meta = await imageMetadata(tmpPath); const desc = joinParts( [ meta.format, !!meta.width && !!meta.height && `${meta.width}\u00D7${meta.height}`, meta.type, meta.colorspace, !!meta.depth && `${meta.depth} depth`, meta.fileSize, ], "image", ); return { output: `${headerLine(`image inline (${desc})`)}\n${desc}`, tmpPath, preview: desc, meta: meta as any, earlyReturn: { content: [ { type: "text", text: `${headerLine(`image inline (${desc})`)}\n${desc}`, }, { type: "image", data: body.toString("base64"), mimeType: contentType }, ], details: { url, kind: "image", contentType, chars: desc.length, bodySize: body.length, markdown: desc, imageInfo: meta, }, }, }; } async function processMedia( body: Buffer, contentType: string, url: string, _maxChars: number, headerLine: (s: string) => string, ): Promise { const ext = extensionFromContentType(contentType, url); const tmpPath = await saveBinaryToTemp(body, ext); const meta = await mediaMetadata(tmpPath); const desc = joinParts( [ meta.formatName, meta.codec, meta.duration, !!meta.width && !!meta.height && `${meta.width}\u00D7${meta.height}`, meta.bitRate, ], "media", ); const fullOutput = `${headerLine(`ffprobe JSON (${desc})`)}\n${desc}\n\n${meta.raw}`; return { output: fullOutput, tmpPath, preview: desc, meta: meta as any, earlyReturn: { content: [{ type: "text", text: `${url}\n\n${fullOutput}` }], details: { url, kind: "media", format: "text" as const, chars: fullOutput.length, contentType, bodySize: body.length, tmpPath, mediaInfo: meta, markdown: desc, }, }, }; } async function processPdf( body: Buffer, headerLine: (s: string) => string, ): Promise { const { text, tmpPath } = await pdfToText(body); return { output: `${headerLine(`extracted text via pdftotext, ${fmtChars(text.length)}`)}\n\n${text}`, tmpPath, preview: text, }; } async function processOffice( body: Buffer, officeFormat: string, headerLine: (s: string) => string, ): Promise { const { text, tmpPath } = await officeDocToText(body, officeFormat); return { output: `${headerLine(`extracted text via pandoc from ${officeFormat.toUpperCase()}, ${fmtChars(text.length)}`)}\n\n${text}`, tmpPath, preview: text, }; } function processJson( body: Buffer, headerLine: (s: string) => string, ): ContentOutput { const raw = body.toString("utf-8"); let formatted: string; try { formatted = JSON.stringify(JSON.parse(raw), null, 2); } catch { formatted = raw; } return { output: `${headerLine(`formatted JSON, ${fmtChars(formatted.length)}`)}\n\n${formatted}`, preview: formatted, }; } function processText( body: Buffer, ctShort: string, headerLine: (s: string) => string, ): ContentOutput { const text = body.toString("utf-8"); const sub = ctShort.includes("csv") ? "CSV" : ctShort.includes("tsv") ? "TSV" : "plain text"; return { output: `${headerLine(`${sub}, ${fmtChars(text.length)}`)}\n\n${text}`, preview: text, }; } async function processHtml( body: Buffer, format: string, headerLine: (s: string) => string, ): Promise { const hint = await checkToolAvailability("pandoc", TOOL_HINTS.pandoc); if (hint) throw new Error(hint); const [{ extractMainContent }, { runPandoc }] = await Promise.all([ loadExtract(), loadPandoc(), ]); const extracted = extractMainContent(body.toString("utf-8")); let md: string; try { md = await runPandoc(extracted, format); } catch { md = await runPandoc(extracted, format, false); } const fmtLabel = format === "text" ? "plain text" : format === "json" ? "Pandoc AST JSON" : format === "html" ? "extracted HTML" : "markdown"; return { output: `${headerLine(`${fmtLabel} via pandoc, ${fmtChars(md.length)}`)}\n\n${md}`, preview: md, }; } interface ConvertInput { kind: ContentKind; body: Buffer; contentType: string; url: string; format: string; maxChars: number; ctShort: string; officeFormat?: string; headerLine: (conversion: string) => string; } async function convertContent(input: ConvertInput): Promise { switch (input.kind) { case "binary": return processBinary( input.body, input.contentType, input.url, input.headerLine, ); case "image": return processImage( input.body, input.contentType, input.url, input.headerLine, ); case "media": return processMedia( input.body, input.contentType, input.url, input.maxChars, input.headerLine, ); case "pdf": return processPdf(input.body, input.headerLine); case "office": { if (!input.officeFormat) throw new Error("Missing office document format"); return processOffice(input.body, input.officeFormat, input.headerLine); } case "json": return processJson(input.body, input.headerLine); case "text": return processText(input.body, input.ctShort, input.headerLine); case "html": return processHtml(input.body, input.format, input.headerLine); } } function truncateText(text: string, maxChars: number): string { return text.length > maxChars ? text.slice(0, maxChars) + "\n… (truncated)" : text; } function progressMessage(loaded: number, total: number): string { const mb = (loaded / 1_048_576).toFixed(1); if (total <= 0) return `⬇ ${mb}MB`; const totalMb = (total / 1_048_576).toFixed(1); const percent = Math.round((loaded / total) * 100); return `⬇ ${mb}/${totalMb}MB (${percent}%)`; } function safeTerminalText(text: string): string { return text .replace( /\x1B(?:[@-Z\\-_]|\[[0-?]*[ -/]*[@-~]|\][^\x07]*(?:\x07|\x1B\\))/g, "", ) .replace(/[\x00-\x08\x0B\x0C\x0E-\x1F\x7F-\x9F]/g, " "); } function safeDisplayText(text: string): string { return safeTerminalText(text) .replace(/[\t\r\n ]+/g, " ") .trim(); } /** Render expanded tool text: prefix + each line dimmed. Returns null if content is not text. */ function expandedText( result: { content: Array<{ type: string; text?: string }> }, prefix: string, theme: { fg: (...a: any[]) => string }, ): string | null { const content = result.content[0]; if (content?.type !== "text") return null; let text = prefix; for (const line of safeTerminalText(content.text ?? "").split("\n")) { text += `\n${theme.fg("dim", line)}`; } return text; } function hintBox( theme: { fg: (...a: any[]) => string }, style: "warning" | "error", msg: string, ): Text { return new Text(theme.fg(style, safeDisplayText(msg)), 0, 0); } function expandableError( theme: { fg: (...a: any[]) => string }, msg: string, expanded: boolean, ): Text { const full = safeTerminalText(msg).trim(); if (expanded) return new Text(theme.fg("error", full), 0, 0); const summary = safeDisplayText(full); const clipped = summary.length > 160 ? `${summary.slice(0, 159)}…` : summary; let text = theme.fg("error", clipped); if (clipped !== summary) text += `\n${theme.fg("muted", "↳")} ${keyHint("app.tools.expand", "to expand")}`; return new Text(text, 0, 0); } /** Extract error message from a tool result, falling back to content text or a default. */ function errorMsg(result: any, fallback: string): string { return ( (result.details as { error?: string } | undefined)?.error || (result.content[0]?.type === "text" ? (result.content[0].text ?? fallback) : fallback) ); } type WebAction = Pick< ToolDefinition, "parameters" | "execute" | "renderCall" | "renderResult" >; export default function webExtension(pi: ExtensionAPI) { // ── Session-scoped spotlight markers ─────────────────────────────────── const spotlightMarkers: SpotlightMarkers = createMarkers(); let generation = 0; pi.on("session_start", (_event, ctx) => { dbg?.("session.start"); const active = pi.getActiveTools(); if (!active.includes("web")) pi.setActiveTools([...active, "web"]); const ticket = ++generation; for (const read of SETTING_READERS) { try { read(ctx); } catch (error) { if (!(error instanceof Error)) throw error; if (ctx.hasUI) ctx.ui.notify(`web: ${error.message}`, "warning"); } } void notifyMissingPlatformTools(ctx, () => ticket === generation).catch( (error) => { if (ticket === generation && ctx.hasUI) ctx.ui.notify( `web platform tool check failed: ${error instanceof Error ? error.message : String(error)}`, "warning", ); }, ); }); // ── Cleanup temp files on session shutdown ───────────────────────────── pi.on("session_shutdown", async () => { dbg?.("session.shutdown"); closeDebug(); generation++; const ownedDirs = [...tempDirs]; tempDirs.clear(); await Promise.all( ownedDirs.map((dir) => rm(dir, { recursive: true, force: true }).catch(() => {}), ), ); }); const fetchAction = { parameters: Type.Object({ url: Type.String({ description: "URL to fetch" }), format: Type.Optional( StringEnum(["markdown", "text", "html", "json", "raw"], { description: "Default markdown; json is Pandoc AST for HTML; raw skips conversion.", }), ), maxChars: Type.Optional( Type.Number({ description: `Visible characters (default ${DEFAULT_MAX_CHARS}, max ${MAX_VISIBLE_CHARS}).`, minimum: 100, maximum: MAX_VISIBLE_CHARS, }), ), maxBytes: Type.Optional( Type.Number({ description: "Maximum download bytes.", minimum: 1_000, maximum: 100_000_000, }), ), linesMatching: Type.Optional( Type.Array(Type.String(), { description: "Literal substrings; return matching lines only.", }), ), contextLines: Type.Optional( Type.Number({ description: "Context lines per match (default 0, max 100).", minimum: 0, maximum: 100, }), ), caseSensitive: Type.Optional( Type.Boolean({ description: "Case-sensitive matching (default false).", }), ), timeout: Type.Optional( Type.Number({ description: `Timeout in ms (default ${DEFAULT_TIMEOUT_MS}).`, minimum: 5000, maximum: 600_000, }), ), }), renderCall(args: any, theme) { let text = theme.fg("toolTitle", theme.bold("\u2B07 web fetch ")); text += theme.fg("accent", safeDisplayText(args.url)); return new Text(text, 0, 0); }, renderResult(result, { expanded, isPartial }, theme) { if (isPartial) { return hintBox(theme, "warning", "\u2B07 Fetching..."); } const details = result.details as | { url?: string; kind?: string; format?: string; chars?: number; bodySize?: number; contentType?: string; tmpPath?: string; preview?: string; binary?: boolean; officeFormat?: string; imageInfo?: ImageSummary; mediaInfo?: MediaSummary; error?: string; } | undefined; if ((result as any).isError || details?.kind === "error") { const content = result.content[0]; const msg = details?.error || (content?.type === "text" ? content.text : "Fetch failed"); return expandableError(theme, `\u2717 ${msg}`, expanded); } const kind = details?.kind ?? "text"; const ct = details?.contentType ?? "unknown"; const bodySize = details?.bodySize ?? 0; const chars = details?.chars ?? 0; const visibleChars = (details as { visibleChars?: number } | undefined)?.visibleChars ?? chars; const responseId = (details as { responseId?: string } | undefined) ?.responseId; const isRaw = details?.format === "raw"; let header = theme.fg("success", "\u2713 "); header += theme.fg("accent", fmtSize(bodySize)); header += theme.fg("dim", ` ${stripCt(ct)}`); if (details?.imageInfo) { header += theme.fg( "dim", ` ${details.imageInfo.width}\u00D7${details.imageInfo.height}`, ); } if (details?.mediaInfo) { const mi = details.mediaInfo; const parts: string[] = []; if (mi.duration) parts.push(mi.duration); if (mi.width && mi.height) parts.push(`${mi.width}\u00D7${mi.height}`); if (mi.codec) parts.push(mi.codec); if (parts.length) header += theme.fg("dim", ` (${parts.join(", ")})`); } let llmDesc: string; if (isRaw) { llmDesc = `${fmtChars(visibleChars)} of raw content (no conversion)`; } else { switch (kind) { case "html": { const fmt = details?.format; const typeLabel = fmt === "text" ? "plain text" : fmt === "json" ? "Pandoc AST JSON" : fmt === "html" ? "extracted HTML" : "markdown"; llmDesc = `${fmtChars(visibleChars)} of ${typeLabel} (full saved)`; break; } case "text": { const sub = ct.includes("csv") ? "CSV text" : ct.includes("tsv") ? "TSV text" : "plain text"; llmDesc = `${fmtChars(visibleChars)} of ${sub} (full saved)`; break; } case "json": llmDesc = `${fmtChars(visibleChars)} of formatted JSON (full saved)`; break; case "pdf": llmDesc = `${fmtChars(visibleChars)} of extracted text (full saved)`; break; case "office": llmDesc = `${fmtChars(visibleChars)} of extracted text (full saved)`; break; case "image": llmDesc = `image inline (${details?.imageInfo?.format ?? ct.split("/")[1] ?? "?"}) + ${fmtChars(chars)} metadata text`; break; case "media": { const dur = details?.mediaInfo?.duration; llmDesc = `${fmtChars(visibleChars)} of ffprobe JSON metadata${dur ? ` (${dur})` : ""}`; break; } case "binary": llmDesc = "file path + content-type only (binary, not decoded)"; break; default: llmDesc = `${fmtChars(visibleChars)} of text (full saved)`; } } header += `\n ${theme.fg("muted", "\u21B3 Agent sees:")} ${theme.fg("dim", llmDesc)}`; if (responseId) header += `\n ${theme.fg("muted", "\u21B3 Saved:")} ${theme.fg("dim", responseId)}`; const content = result.content[0]; const hasExpandableText = content?.type === "text" && !!content.text?.trim(); if (!expanded && hasExpandableText) header += `\n ${theme.fg("muted", "\u21B3")} ${theme.fg("dim", keyHint("app.tools.expand", "to expand"))}`; const mdTheme = getMarkdownTheme(); const box = new Box(1, 0, (t) => theme.bg("toolSuccessBg", t)); box.addChild(new Text(header, 1, 0)); if (kind === "image" && details?.imageInfo) { const ii = details.imageInfo; let metaMd = `**${ii.format || "Image"}** ${ii.width}\u00D7${ii.height}`; if (ii.type) metaMd += ` \u00B7 ${ii.type}`; metaMd += "\n"; if (ii.colorspace) metaMd += `\n- Color: ${ii.colorspace}`; if (ii.depth) metaMd += `\n- Depth: ${ii.depth}`; if (ii.fileSize) metaMd += `\n- File size: ${ii.fileSize}`; box.addChild(new Markdown(metaMd, 1, 1, mdTheme)); return box; } if (kind === "media" && details?.mediaInfo) { const mi = details.mediaInfo; let metaMd = `**${mi.formatName || "Media"}**`; if (mi.codec) metaMd += ` \u00B7 ${mi.codec}`; metaMd += "\n"; if (mi.duration) metaMd += `\n- Duration: ${mi.duration}`; if (mi.width && mi.height) metaMd += `\n- Resolution: ${mi.width}\u00D7${mi.height}`; if (mi.bitRate) metaMd += `\n- Bitrate: ${mi.bitRate}`; box.addChild(new Markdown(metaMd, 1, 1, mdTheme)); return box; } if (expanded && details?.preview) { box.addChild( new Markdown(safeTerminalText(details.preview), 1, 1, mdTheme), ); return box; } if (expanded && content?.type === "text") { box.addChild( new Markdown(safeTerminalText(content.text), 1, 1, mdTheme), ); return box; } return box; }, async execute( _toolCallId: any, params: any, signal: any, onUpdate: any, ctx: any, ): Promise { const format = params.format ?? "markdown"; const maxChars = Math.min( params.maxChars ?? DEFAULT_MAX_CHARS, MAX_VISIBLE_CHARS, ); const setStatus = (msg: string) => ctx?.ui?.setStatus("web", msg); const clearStatus = () => ctx?.ui?.setStatus("web", undefined); onUpdate?.({ content: [ { type: "text" as const, text: `Fetching ${params.url}\u2026` }, ], details: {}, }); const onProgress = (loaded: number, total: number) => setStatus(progressMessage(loaded, total)); // ── Fetch ──────────────────────────────────────────────────── let body: Buffer; let contentType: string; let finalUrl = params.url; try { const { fetchUrlStream } = await loadFetch(); const result = await fetchUrlStream(params.url, onProgress, signal, { maxBytes: params.maxBytes, timeoutMs: params.timeout, }); body = result.body; contentType = result.contentType; finalUrl = result.finalUrl; } catch (err) { clearStatus(); return errorFetchResult( params.url, `Error fetching ${params.url}: ${(err as Error).message}`, ); } // ── Route by content type ──────────────────────────────────── const { kind, officeFormat } = classifyContentType(contentType); onUpdate?.({ content: [ { type: "text" as const, text: `Converting ${kind} content\u2026` }, ], details: {}, }); const ctShort = stripCt(contentType); const bodyLen = body.length; const headerLine = (conversion: string) => `[${ctShort} | ${fmtSize(bodyLen)} | \u2192 ${conversion}]`; try { const processed = format === "raw" ? { output: body.toString("utf-8") } : await convertContent({ kind, body, contentType, url: finalUrl, format, maxChars, ctShort, officeFormat, headerLine, }); const fullOutput = processed.output; const [{ saveJsonResult }, { filterLineMatches }] = await Promise.all([ loadStore(), loadLineMatch(), ]); const stored = await saveJsonResult("web", ctx?.cwd ?? process.cwd(), { url: params.url, finalUrl, kind, format, contentType, bodySize: bodyLen, output: fullOutput, ...(processed.tmpPath && { tmpPath: processed.tmpPath }), ...(officeFormat && { officeFormat }), }); const matchSource = "preview" in processed && processed.preview ? processed.preview : fullOutput; const matched = params.linesMatching?.length ? filterLineMatches(matchSource, { needles: params.linesMatching, contextLines: params.contextLines, caseSensitive: params.caseSensitive, maxChars, }) : undefined; const visible = matched ? matched.text || `No lines matched: ${params.linesMatching.join(", ")}` : truncateText(fullOutput, maxChars); const outputOmitted = matched ? true : fullOutput.length > maxChars; const recovery = outputOmitted ? `\nfullOutputPath: ${stored.fullOutputPath}` : ""; clearStatus(); return { content: [ { type: "text", text: wrapUntrusted( `${finalUrl}\n\n${visible}\n\nresponseId: ${stored.responseId}${recovery}`, spotlightMarkers, ), }, ], details: { url: params.url, finalUrl, kind, format, chars: fullOutput.length, visibleChars: visible.length, truncated: visible.length < fullOutput.length, contentType, bodySize: bodyLen, responseId: stored.responseId, fullOutputPath: stored.fullOutputPath, storedBytes: stored.byteLength, ...(processed.tmpPath && { tmpPath: processed.tmpPath }), ...(officeFormat && { officeFormat }), ...(processed.meta && { meta: processed.meta }), ...(matched && { lineMatches: { matchCount: matched.matchCount, truncated: matched.truncated, }, }), }, }; } catch (err) { clearStatus(); return errorFetchResult(params.url, (err as Error).message); } }, } satisfies WebAction; const mapAction = { parameters: Type.Object({ url: Type.String({ description: "Site URL" }), maxUrls: Type.Optional( Type.Number({ description: "URL limit (default 1000).", minimum: 1, maximum: 10_000, }), ), maxSitemaps: Type.Optional( Type.Number({ description: "Sitemap-file limit (default 20).", minimum: 1, maximum: 100, }), ), maxChars: Type.Optional( Type.Number({ description: `Visible characters (default 8000, max ${MAX_VISIBLE_CHARS}).`, minimum: 100, maximum: MAX_VISIBLE_CHARS, }), ), }), renderCall(args: any, theme) { let text = theme.fg("toolTitle", theme.bold("🗺 web map ")); text += theme.fg("accent", safeDisplayText(args.url)); return new Text(text, 0, 0); }, renderResult(result, { expanded, isPartial }, theme) { if (isPartial) return hintBox(theme, "warning", "🗺 Mapping site metadata…"); const details = result.details as | { urlCount?: number; sitemapCount?: number; responseId?: string; fullOutputPath?: string; } | undefined; if ((result.details as { kind?: string } | undefined)?.kind === "error") { const content = result.content[0]; const msg = content?.type === "text" ? content.text : "Map failed"; return expandableError(theme, `✗ ${msg}`, expanded); } let text = theme.fg("success", "✓ "); text += theme.fg("accent", `${details?.urlCount ?? 0} URL(s)`); text += theme.fg( "dim", ` · ${details?.sitemapCount ?? 0} sitemap candidate(s)`, ); if (details?.responseId) text += theme.fg("dim", ` · saved ${details.responseId}`); const content = result.content[0]; const hasDetails = (content?.type === "text" && !!content.text?.trim()) || !!details?.fullOutputPath; if (expanded) { if (details?.fullOutputPath) text += `\n ${theme.fg("muted", "full:")} ${theme.fg("dim", details.fullOutputPath)}`; const rendered = expandedText(result, text, theme); if (rendered !== null) text = rendered; } else if (hasDetails) { text += `\n ${theme.fg("muted", "↳")} ${keyHint("app.tools.expand", "to expand")}`; } return new Text(text, 0, 0); }, async execute(_toolCallId, params: any, signal, onUpdate, ctx) { onUpdate?.({ content: [{ type: "text", text: `Mapping ${params.url}…` }], details: {}, }); try { const [{ discoverSiteMap }, { saveJsonResult }] = await Promise.all([ loadMap(), loadStore(), ]); const map = await discoverSiteMap( params.url, { maxUrls: params.maxUrls, maxSitemaps: params.maxSitemaps }, signal, ); const stored = await saveJsonResult( "web", ctx?.cwd ?? process.cwd(), map, ); const listed = map.urls .slice(0, 50) .map((entry) => `- ${entry.url}`) .join("\n"); const fullListing = `Mapped ${map.urls.length} URL(s) from ${map.sitemaps.length} sitemap candidate(s).\n\n${listed}`; const maxChars = params.maxChars ?? 8_000; const body = truncateText(fullListing, maxChars); const partialReasons: string[] = []; if (map.urls.length > 50) partialReasons.push(`showing at most 50 of ${map.urls.length} URLs`); if (fullListing.length > maxChars) partialReasons.push(`visible text capped at ${maxChars} characters`); const recovery = partialReasons.length ? `\n\nPartial URL listing: ${partialReasons.join("; ")}.\nfullOutputPath: ${stored.fullOutputPath}` : ""; return { content: [ { type: "text", text: wrapUntrusted( `${body}\n\nresponseId: ${stored.responseId}${recovery}`, spotlightMarkers, ), }, ], details: { url: params.url, kind: "map", urlCount: map.urls.length, sitemapCount: map.sitemaps.length, responseId: stored.responseId, fullOutputPath: stored.fullOutputPath, }, }; } catch (err) { return errorFetchResult(params.url, (err as Error).message); } }, } satisfies WebAction; const searchAction = { parameters: Type.Object({ query: Type.String({ description: "Search query" }), count: Type.Optional( Type.Number({ description: "Result count (default 5).", minimum: 1, maximum: 20, }), ), format: Type.Optional( StringEnum(["text", "json"], { description: "Default: text.", }), ), timeout: Type.Optional( Type.Number({ description: `Timeout in ms (default ${DEFAULT_TIMEOUT_MS}).`, minimum: 5000, maximum: 600_000, }), ), }), renderCall(args: any, theme) { let text = theme.fg("toolTitle", theme.bold("\uD83D\uDD0D web search ")); text += theme.fg("accent", `"${safeDisplayText(args.query)}"`); return new Text(text, 0, 0); }, renderResult(result, { expanded, isPartial }, theme) { if (isPartial) { return hintBox(theme, "warning", "Searching..."); } if ((result as any).isError || (result.details as any)?.error) { return expandableError( theme, `\u2717 ${errorMsg(result, "Search failed")}`, expanded, ); } const details = result.details as | { query?: string; results?: SearchResult[]; source?: string } | undefined; if (expanded) { const rendered = expandedText(result, "", theme); if (rendered !== null) return new Text(rendered, 0, 0); } const resultCount = details?.results?.length ?? 0; let text = theme.fg( "success", `\u2713 ${resultCount} result${resultCount === 1 ? "" : "s"}`, ); if (details?.source) text += ` ${theme.fg("muted", `(${details.source})`)}`; if (!expanded && resultCount > 0) text += ` ${theme.fg("muted", `(${keyHint("app.tools.expand", "to expand")})`)}`; return new Text(text, 0, 0); }, async execute( _toolCallId: any, params: any, signal: any, onUpdate: any, ctx: any, ): Promise { const count = Math.min(params.count ?? 5, 20); const format = params.format ?? "text"; const timeout = params.timeout ?? DEFAULT_TIMEOUT_MS; onUpdate?.({ content: [ { type: "text" as const, text: `Searching for "${params.query}"\u2026`, }, ], details: {}, }); const { searchWeb } = await loadSearch(); const { results, source } = await searchWeb({ query: params.query, count, timeout, signal, ctx, }); if (!results || results.length === 0) { return { content: [ { type: "text", text: `No results found for: ${params.query}` }, ], details: { query: params.query, results: [], source }, }; } const sliced = results.slice(0, count); if (format === "json") { return { content: [ { type: "text", text: wrapUntrusted( JSON.stringify( { query: params.query, count: sliced.length, results: sliced, source, }, null, 2, ), spotlightMarkers, ), }, ], details: { query: params.query, results: sliced, source }, }; } const lines: string[] = [ `Search: ${params.query}`, `Results: ${sliced.length}${results.length > sliced.length ? ` of ${results.length}` : ""} (${source})`, "", ]; for (let i = 0; i < sliced.length; i++) { const r = sliced[i]; lines.push(`[${i + 1}] ${r.title}`); lines.push(` ${r.url}`); if (r.snippet) lines.push(` ${r.snippet}`); if (r.date) lines.push(` Date: ${r.date}`); lines.push(""); } return { content: [ { type: "text", text: wrapUntrusted(lines.join("\n"), spotlightMarkers), }, ], details: { query: params.query, results: sliced, source }, }; }, } satisfies WebAction; const actions = { fetch: fetchAction, map: mapAction, search: searchAction }; pi.registerTool({ name: "web", label: "web", description: "Fetch HTTP(S) pages, documents, or media metadata; map site URLs from robots.txt, sitemaps, and llms.txt without page bodies; search for titles, URLs, and snippets.", promptSnippet: "Fetch web content, discover site URLs, or search the web.", parameters: Type.Object( { ...fetchAction.parameters.properties, ...mapAction.parameters.properties, ...searchAction.parameters.properties, action: StringEnum(["fetch", "map", "search"] as const), url: Type.Optional( Type.String({ description: "Required for fetch/map." }), ), query: Type.Optional( Type.String({ description: "Required for search." }), ), format: Type.Optional( StringEnum(["markdown", "text", "html", "json", "raw"], { description: "Fetch: default markdown; json is Pandoc AST for HTML; raw skips conversion. Search: text (default) or json.", }), ), maxChars: Type.Optional( Type.Number({ description: `Fetch: default ${DEFAULT_MAX_CHARS}, max ${MAX_VISIBLE_CHARS}. Map: default 8000, max ${MAX_VISIBLE_CHARS}.`, minimum: 100, maximum: MAX_VISIBLE_CHARS, }), ), }, { additionalProperties: false }, ), async execute(id, { action, ...params }, signal, onUpdate, ctx) { if (!Object.hasOwn(actions, action)) { dbg?.("action.invalid", { type: action }); throw new Error("Expected action: fetch, map, or search"); } const handler = actions[action]; // Keep action-specific required fields, limits, and formats without // duplicating their schemas in every model request. const { Check, Errors } = await import("typebox/value"); const schema = { ...handler.parameters, additionalProperties: false }; if (!Check(schema, params)) { dbg?.("action.invalid", { type: action }); const details = [...Errors(schema, params)] .map((error) => { const path = "path" in error ? error.path : "arguments"; const value = "value" in error ? JSON.stringify(error.value) : "unknown"; return `${path} ${error.message} (received ${value})`; }) .join("; "); throw new Error(`Invalid web ${action} arguments: ${details}`); } const finish = span?.("action.execute", { type: action }); try { const result = await handler.execute(id, params, signal, onUpdate, ctx); finish?.(); return result; } catch (error) { finish?.("error"); throw error; } }, renderCall(args, theme) { if (Object.hasOwn(actions, args.action)) return actions[args.action].renderCall(args, theme); return new Text(theme.fg("toolTitle", theme.bold("web")), 0, 0); }, renderResult(result, options, theme, context) { if (context.isError) return expandableError( theme, errorMsg(result, "Web request failed"), options.expanded, ); const action = context.args.action as keyof typeof actions; if (Object.hasOwn(actions, action)) return actions[action].renderResult(result, options, theme); return hintBox(theme, "warning", "Waiting for web action..."); }, }); }