repositories / pi-ext
pi-ext
bugabingas pi extensions
owned by admin
extensions/web/extract.ts
Raw// HTML semantic container extraction
// Extracts the main content area before feeding to pandoc
// Uses balanced tag counting for nested elements
// ── Prompt injection hardening ──────────────────────────────────────────
// Strip known injection vectors from HTML before content extraction.
// These are deterministic, zero-cost filters — no ML, no arms race.
// They remove delivery mechanisms, not payload content.
//
// Performance: merged 10+ sequential .replace() passes into 5 grouped passes.
// Each .replace() on a 100KB string allocates a new 100KB string, so fewer
// passes = fewer allocations. Block-element removals share one regex via
// alternation since they all follow the same <tag...>...</tag> pattern.
// [perf] Single regex for all block-level elements that need removal.
// Matches <script>, <style>, <iframe>, <object>, <template>, <noscript>
// and their content. Using alternation avoids 6 separate .replace() calls.
const BLOCK_REMOVE_RE =
/<(?:script|style|iframe|object|template|noscript)[\s>][\s\S]*?<\/(?:script|style|iframe|object|template|noscript)>/gi;
// [perf] Pre-compiled regex for hidden input detection, reused in callback.
const HIDDEN_TYPE_RE = /type=["']hidden["']/i;
// [perf] Visual hide patterns compiled once, not per-call.
const VISUAL_HIDE_PATTERNS: RegExp[] = [
/display\s*:\s*none/i,
/visibility\s*:\s*hidden/i,
/opacity\s*:\s*0(?!\.)/i,
/font-size\s*:\s*0/i,
/color\s*:\s*(white|#fff(?:fff)?|transparent)\s*(?:[;"']|$)/i,
/left\s*:\s*-\d{4,}/i,
];
export function sanitizeInjectionVectors(html: string): string {
// Pass 1: Remove HTML comments first (cheap, reduces input for later passes)
let out = html.replace(/<!--[\s\S]*?-->/g, "");
// Pass 2: Remove all block-level dangerous elements in one sweep
out = out.replace(BLOCK_REMOVE_RE, "");
// Pass 3: Remove void/inline dangerous elements + hidden inputs
// - <embed> (void, no closing tag)
// - <input type="hidden"> (the #1 injection vector per BrowseSafe)
// - <input value="..."> where type="hidden" appears anywhere in the tag
// - <meta> with long content attributes
out = out.replace(/<embed[^>]*>/gi, "");
out = out.replace(/<input[^>]+type=["']hidden["'][^>]*>/gi, "");
out = out.replace(/<input[^>]+value=["'][^"']*["'][^>]*>/gi, (m) =>
HIDDEN_TYPE_RE.test(m) ? "" : m,
);
out = out.replace(/<meta[^>]+content=["'][^"']{50,}["'][^>]*>/gi, "");
// Pass 4: Remove elements with hidden attribute or visually-hidden styles
out = out.replace(
/<(\w+)([^>]*\shidden(?:\s[^>]*>|>|\/>))[\s\S]*?<\/\1>/gi,
"",
);
out = out.replace(
/<(\w+)([^>]*style=["'])([^"']*)(["'][^>]*)>([\s\S]*?)<\/\1>/gi,
(_match, _tag, _before, styleVal, _after, _content) => {
return VISUAL_HIDE_PATTERNS.some((p) => p.test(styleVal)) ? "" : _match;
},
);
// Pass 5: Strip aria-label, alt, title from non-image elements
out = out.replace(
/<([^\s>]+)([^>]*?)\b(aria-label|alt|title)=["'][^"']*["']([^>]*?)>/gi,
(_match, tag, before, _attr, after) => {
// Keep alt/aria-label on images, areas, inputs
if (/^(img|area|input)$/i.test(tag)) return _match;
return `<${tag}${before}${after}>`;
},
);
return out;
}
export function extractMainContent(html: string): string {
html = sanitizeInjectionVectors(html);
// Try <article> first (most semantic, usually cleanest)
const article = matchTag(html, "article");
if (article !== null) return article;
// Try <div id="mw-content-text"> (Wikipedia article body)
// Before <main> because Wikipedia's <main> includes nav chrome
const mwContent = matchId(html, "mw-content-text", "div");
if (mwContent) return mwContent;
// Try <div id="docContent"> (PostgreSQL/Sphinx documentation)
const docContent = matchId(html, "docContent", "div");
if (docContent) return docContent;
// Try <div id="content"> (generic, common)
const contentDiv = matchId(html, "content", "div");
if (contentDiv) return contentDiv;
// Try <main>
const main = matchTag(html, "main");
if (main) return main;
// Try role="main" on any element
const roleMain = matchRoleMain(html);
if (roleMain) return roleMain;
// Fallback: full page
return html;
}
function matchTag(html: string, tag: string): string | null {
const openRe = new RegExp(`<${tag}(?:\\s[^>]*)?>`, "i");
const openMatch = html.match(openRe);
if (!openMatch) return null;
if (openMatch.index === undefined) return null;
const startIdx = openMatch.index + openMatch[0].length;
return extractBalanced(html, startIdx, tag);
}
export function extractBalanced(
html: string,
startIdx: number,
tagName: string,
): string | null {
let depth = 1;
let pos = startIdx;
const openTag = `<${tagName}`;
const closeTag = `</${tagName}`;
while (depth > 0 && pos < html.length) {
const nextOpen = html.indexOf(openTag, pos);
const nextClose = html.indexOf(closeTag, pos);
if (nextClose === -1) return null;
if (nextOpen !== -1 && nextOpen < nextClose) {
const tagEndIdx = html.indexOf(">", nextOpen);
if (tagEndIdx === -1) return null;
if (html[tagEndIdx - 1] === "/") {
pos = tagEndIdx + 1;
continue;
}
const afterOpen = html[nextOpen + openTag.length];
if (afterOpen === ">" || afterOpen === " ") {
depth++;
pos = tagEndIdx + 1;
continue;
}
}
depth--;
if (depth === 0) {
return html.slice(startIdx, nextClose);
}
pos = html.indexOf(">", nextClose) + 1;
}
return null;
}
function matchRoleMain(html: string): string | null {
const re = /<(\w+)([^>]*\srole=["']main["'][^>]*)>/i;
const m = html.match(re);
if (!m) return null;
const tag = m[1];
const startIdx = html.indexOf(m[0]) + m[0].length;
return extractBalanced(html, startIdx, tag);
}
function matchId(html: string, id: string, tag: string): string | null {
const re = new RegExp(`<${tag}[^>]*\\bid=["']${id}["'][^>]*>`, "i");
const m = html.match(re);
if (!m) return null;
const startIdx = html.indexOf(m[0]) + m[0].length;
return extractBalanced(html, startIdx, tag);
}