Luigit
repositories / pi-ext

pi-ext

bugabingas pi extensions

owned by admin

extensions/web/extract.ts

Raw
// HTML semantic container extraction
// Extracts the main content area before feeding to pandoc
// Uses balanced tag counting for nested elements

// ── Prompt injection hardening ──────────────────────────────────────────
// Strip known injection vectors from HTML before content extraction.
// These are deterministic, zero-cost filters — no ML, no arms race.
// They remove delivery mechanisms, not payload content.
//
// Performance: merged 10+ sequential .replace() passes into 5 grouped passes.
// Each .replace() on a 100KB string allocates a new 100KB string, so fewer
// passes = fewer allocations. Block-element removals share one regex via
// alternation since they all follow the same <tag...>...</tag> pattern.

// [perf] Single regex for all block-level elements that need removal.
// Matches <script>, <style>, <iframe>, <object>, <template>, <noscript>
// and their content. Using alternation avoids 6 separate .replace() calls.
const BLOCK_REMOVE_RE =
	/<(?:script|style|iframe|object|template|noscript)[\s>][\s\S]*?<\/(?:script|style|iframe|object|template|noscript)>/gi;

// [perf] Pre-compiled regex for hidden input detection, reused in callback.
const HIDDEN_TYPE_RE = /type=["']hidden["']/i;

// [perf] Visual hide patterns compiled once, not per-call.
const VISUAL_HIDE_PATTERNS: RegExp[] = [
	/display\s*:\s*none/i,
	/visibility\s*:\s*hidden/i,
	/opacity\s*:\s*0(?!\.)/i,
	/font-size\s*:\s*0/i,
	/color\s*:\s*(white|#fff(?:fff)?|transparent)\s*(?:[;"']|$)/i,
	/left\s*:\s*-\d{4,}/i,
];

export function sanitizeInjectionVectors(html: string): string {
	// Pass 1: Remove HTML comments first (cheap, reduces input for later passes)
	let out = html.replace(/<!--[\s\S]*?-->/g, "");

	// Pass 2: Remove all block-level dangerous elements in one sweep
	out = out.replace(BLOCK_REMOVE_RE, "");

	// Pass 3: Remove void/inline dangerous elements + hidden inputs
	// - <embed> (void, no closing tag)
	// - <input type="hidden"> (the #1 injection vector per BrowseSafe)
	// - <input value="..."> where type="hidden" appears anywhere in the tag
	// - <meta> with long content attributes
	out = out.replace(/<embed[^>]*>/gi, "");
	out = out.replace(/<input[^>]+type=["']hidden["'][^>]*>/gi, "");
	out = out.replace(/<input[^>]+value=["'][^"']*["'][^>]*>/gi, (m) =>
		HIDDEN_TYPE_RE.test(m) ? "" : m,
	);
	out = out.replace(/<meta[^>]+content=["'][^"']{50,}["'][^>]*>/gi, "");

	// Pass 4: Remove elements with hidden attribute or visually-hidden styles
	out = out.replace(
		/<(\w+)([^>]*\shidden(?:\s[^>]*>|>|\/>))[\s\S]*?<\/\1>/gi,
		"",
	);
	out = out.replace(
		/<(\w+)([^>]*style=["'])([^"']*)(["'][^>]*)>([\s\S]*?)<\/\1>/gi,
		(_match, _tag, _before, styleVal, _after, _content) => {
			return VISUAL_HIDE_PATTERNS.some((p) => p.test(styleVal)) ? "" : _match;
		},
	);

	// Pass 5: Strip aria-label, alt, title from non-image elements
	out = out.replace(
		/<([^\s>]+)([^>]*?)\b(aria-label|alt|title)=["'][^"']*["']([^>]*?)>/gi,
		(_match, tag, before, _attr, after) => {
			// Keep alt/aria-label on images, areas, inputs
			if (/^(img|area|input)$/i.test(tag)) return _match;
			return `<${tag}${before}${after}>`;
		},
	);

	return out;
}

export function extractMainContent(html: string): string {
	html = sanitizeInjectionVectors(html);

	// Try <article> first (most semantic, usually cleanest)
	const article = matchTag(html, "article");
	if (article !== null) return article;

	// Try <div id="mw-content-text"> (Wikipedia article body)
	// Before <main> because Wikipedia's <main> includes nav chrome
	const mwContent = matchId(html, "mw-content-text", "div");
	if (mwContent) return mwContent;

	// Try <div id="docContent"> (PostgreSQL/Sphinx documentation)
	const docContent = matchId(html, "docContent", "div");
	if (docContent) return docContent;

	// Try <div id="content"> (generic, common)
	const contentDiv = matchId(html, "content", "div");
	if (contentDiv) return contentDiv;

	// Try <main>
	const main = matchTag(html, "main");
	if (main) return main;

	// Try role="main" on any element
	const roleMain = matchRoleMain(html);
	if (roleMain) return roleMain;

	// Fallback: full page
	return html;
}

function matchTag(html: string, tag: string): string | null {
	const openRe = new RegExp(`<${tag}(?:\\s[^>]*)?>`, "i");
	const openMatch = html.match(openRe);
	if (!openMatch) return null;

	if (openMatch.index === undefined) return null;
	const startIdx = openMatch.index + openMatch[0].length;
	return extractBalanced(html, startIdx, tag);
}

export function extractBalanced(
	html: string,
	startIdx: number,
	tagName: string,
): string | null {
	let depth = 1;
	let pos = startIdx;

	const openTag = `<${tagName}`;
	const closeTag = `</${tagName}`;

	while (depth > 0 && pos < html.length) {
		const nextOpen = html.indexOf(openTag, pos);
		const nextClose = html.indexOf(closeTag, pos);

		if (nextClose === -1) return null;

		if (nextOpen !== -1 && nextOpen < nextClose) {
			const tagEndIdx = html.indexOf(">", nextOpen);
			if (tagEndIdx === -1) return null;

			if (html[tagEndIdx - 1] === "/") {
				pos = tagEndIdx + 1;
				continue;
			}

			const afterOpen = html[nextOpen + openTag.length];
			if (afterOpen === ">" || afterOpen === " ") {
				depth++;
				pos = tagEndIdx + 1;
				continue;
			}
		}

		depth--;
		if (depth === 0) {
			return html.slice(startIdx, nextClose);
		}
		pos = html.indexOf(">", nextClose) + 1;
	}
	return null;
}

function matchRoleMain(html: string): string | null {
	const re = /<(\w+)([^>]*\srole=["']main["'][^>]*)>/i;
	const m = html.match(re);
	if (!m) return null;
	const tag = m[1];
	const startIdx = html.indexOf(m[0]) + m[0].length;
	return extractBalanced(html, startIdx, tag);
}

function matchId(html: string, id: string, tag: string): string | null {
	const re = new RegExp(`<${tag}[^>]*\\bid=["']${id}["'][^>]*>`, "i");
	const m = html.match(re);
	if (!m) return null;
	const startIdx = html.indexOf(m[0]) + m[0].length;
	return extractBalanced(html, startIdx, tag);
}