Luigit
repositories / pi-ext

pi-ext

bugabingas pi extensions

owned by admin

extensions/web/ddg.ts

Raw
// DuckDuckGo HTML search parser with rate throttling

import type { SearchResult } from "./constants.js";

let lastSearchTs = 0;
const MIN_INTERVAL_MS = 2000; // 2s between DDG requests

export function throttleWait(): number {
	const now = Date.now();
	const elapsed = now - lastSearchTs;
	return Math.max(0, MIN_INTERVAL_MS - elapsed);
}

export function markSearchTime(): void {
	lastSearchTs = Date.now();
}

function decodeDDGUrl(rawUrl: string): string | null {
	if (!rawUrl) return null;

	if (rawUrl.includes("uddg=")) {
		const match = rawUrl.match(/uddg=([^&]+)/);
		if (match) return decodeURIComponent(match[1]);
	}

	if (rawUrl.startsWith("//")) return "https:" + rawUrl;
	if (rawUrl.startsWith("/")) return "https://duckduckgo.com" + rawUrl;

	return rawUrl;
}

// [perf] Single-pass HTML entity decoder.
// Replaces 12 chained .replace() calls with one regex pass + map lookup.
// Named entities are O(1) via Map; numeric entities use parseInt.
const NAMED_ENTITIES: Record<string, string> = {
	"&amp": "&",
	"&lt": "<",
	"&gt": ">",
	"&quot": '"',
	"&apos": "'",
	"&nbsp": " ",
};
const ENTITY_RE = /&(?:amp|lt|gt|quot|apos|nbsp);?|&#(?:\d+|x[0-9a-f]+);/gi;

function decodeHTMLEntities(str: string): string {
	return str
		.replace(ENTITY_RE, (entity) => {
			const key = entity.replace(/;$/, "").toLowerCase();
			const named = NAMED_ENTITIES[key];
			if (named !== undefined) return named;
			const hex = key.startsWith("&#x");
			const code = parseInt(key.slice(hex ? 3 : 2), hex ? 16 : 10);
			return code > 0 && code <= 0x10ffff && !(code >= 0xd800 && code <= 0xdfff)
				? String.fromCodePoint(code)
				: "�";
		})
		.trim();
}

export interface ParseDiagnostics {
	captcha?: true;
	resultsFound: number;
	resultsSkipped: number;
	rawBlockCount: number;
}

export function parseDdgResults(html: string): {
	results: SearchResult[];
	diagnostics: ParseDiagnostics;
	error?: string;
} {
	const diagnostics: ParseDiagnostics = {
		resultsFound: 0,
		resultsSkipped: 0,
		rawBlockCount: 0,
	};
	// Detect captcha
	if (
		html.includes("anomaly-modal") ||
		html.includes("captcha") ||
		html.includes("botdetection")
	) {
		return {
			results: [],
			diagnostics: { ...diagnostics, captcha: true },
			error:
				"DuckDuckGo is showing a captcha. Try again later or use a different search method.",
		};
	}

	const results: SearchResult[] = [];
	const seenUrls = new Set<string>();

	// Match the outer result class, not nested result__extras before the snippet.
	const RESULT_BLOCK_RE =
		/<div\b[^>]*class="result(?:\s[^"]*)?"[^>]*>([\s\S]*?)(?=<div\b[^>]*class="result(?:\s[^"]*)?"|$)/gi;
	const LINK_RE =
		/<a[^>]*class="result__a"[^>]*href="([^"]+)"[^>]*>([\s\S]*?)<\/a>/i;
	const SNIPPET_RE = /class="result__snippet"[^>]*>([\s\S]*?)<\/a>/i;

	let blockMatch: RegExpExecArray | null;
	// biome-ignore lint/suspicious/noAssignInExpressions: standard regex exec loop
	while ((blockMatch = RESULT_BLOCK_RE.exec(html)) !== null) {
		diagnostics.rawBlockCount++;
		const block = blockMatch[1];

		const linkMatch = block.match(LINK_RE);
		if (!linkMatch) continue;

		const rawUrl = linkMatch[1];
		const rawTitle = linkMatch[2].replace(/<[^>]+>/g, "").trim();
		const url = decodeDDGUrl(rawUrl);

		if (!url || seenUrls.has(url)) {
			if (url) diagnostics.resultsSkipped++;
			continue;
		}
		seenUrls.add(url);
		diagnostics.resultsFound++;

		const snippetMatch = block.match(SNIPPET_RE);
		const snippet = snippetMatch
			? decodeHTMLEntities(snippetMatch[1].replace(/<[^>]+>/g, "").trim())
			: "";

		results.push({
			title: decodeHTMLEntities(rawTitle),
			url,
			snippet,
		});
	}

	return { results, diagnostics };
}