Luigit
repositories / pi-ext

pi-ext

bugabingas pi extensions

owned by admin

extensions/web/map.ts

Raw
import { fetchUrlStream } from "./fetch.js";

export interface MapUrlEntry {
	url: string;
	source: "sitemap" | "llms";
	sourceUrl: string;
}

export interface SiteMapResult {
	seedUrl: string;
	urls: MapUrlEntry[];
	sitemaps: string[];
	tree: Record<string, string[]>;
}

export interface DiscoverSiteMapOptions {
	maxUrls?: number;
	maxSitemaps?: number;
	fetchText?: (
		url: string,
		signal?: AbortSignal,
	) => Promise<string | undefined>;
}

export async function discoverSiteMap(
	seed: string,
	options: DiscoverSiteMapOptions = {},
	signal?: AbortSignal,
): Promise<SiteMapResult> {
	const seedUrl = new URL(seed).toString();
	const fetchText = options.fetchText ?? defaultFetchText;
	const urls = new Map<string, MapUrlEntry>();
	const sitemapSet = new Set<string>();
	const robotsUrl = siteUrl(seedUrl, "/robots.txt");
	const robots = await fetchText(robotsUrl, signal).catch(() => undefined);
	for (const sitemap of parseRobotsSitemaps(robots ?? ""))
		sitemapSet.add(sitemap);
	sitemapSet.add(siteUrl(seedUrl, "/sitemap.xml"));

	const sitemapQueue = [...sitemapSet];
	for (
		let i = 0;
		i < sitemapQueue.length && i < (options.maxSitemaps ?? 20);
		i++
	) {
		const sitemapUrl = sitemapQueue[i];
		const xml = await fetchText(sitemapUrl, signal).catch(() => undefined);
		if (!xml) continue;
		for (const nested of parseSitemapIndexes(xml)) {
			if (!sitemapSet.has(nested)) {
				sitemapSet.add(nested);
				sitemapQueue.push(nested);
			}
		}
		for (const url of parseSitemapXml(xml)) {
			urls.set(url, { url, source: "sitemap", sourceUrl: sitemapUrl });
			if (urls.size >= (options.maxUrls ?? 1_000)) break;
		}
	}

	const llmsUrl = siteUrl(seedUrl, "/llms.txt");
	const llms = await fetchText(llmsUrl, signal).catch(() => undefined);
	for (const url of parseLlmsLinks(llms ?? "", llmsUrl)) {
		urls.set(url, { url, source: "llms", sourceUrl: llmsUrl });
		if (urls.size >= (options.maxUrls ?? 1_000)) break;
	}

	const entries = [...urls.values()].toSorted((a, b) =>
		a.url.localeCompare(b.url),
	);
	return {
		seedUrl,
		urls: entries,
		sitemaps: [...sitemapSet],
		tree: buildTree(entries),
	};
}

export function parseRobotsSitemaps(text: string): string[] {
	return text
		.split(/\r?\n/u)
		.map((line) => line.match(/^\s*sitemap:\s*(\S+)\s*$/iu)?.[1])
		.filter((x): x is string => Boolean(x));
}

export function parseSitemapXml(xml: string): string[] {
	return locs(xml).filter(
		(url) => !url.endsWith(".xml") && !url.endsWith(".xml.gz"),
	);
}

function parseSitemapIndexes(xml: string): string[] {
	return locs(xml).filter(
		(url) => url.endsWith(".xml") || url.endsWith(".xml.gz"),
	);
}

export function parseLlmsLinks(text: string, baseUrl: string): string[] {
	const out: string[] = [];
	for (const match of text.matchAll(/\[[^\]]*\]\(([^)]+)\)/gu)) {
		const raw = match[1]?.trim();
		if (!raw) continue;
		try {
			out.push(new URL(raw, baseUrl).toString());
		} catch {}
	}
	return out;
}

async function defaultFetchText(
	url: string,
	signal?: AbortSignal,
): Promise<string | undefined> {
	const result = await fetchUrlStream(url, undefined, signal, {
		maxBytes: 2 * 1024 * 1024,
	});
	return result.body.toString("utf8");
}

function locs(xml: string): string[] {
	return [...xml.matchAll(/<loc>\s*([^<]+?)\s*<\/loc>/giu)].map((x) =>
		decodeXml(x[1]),
	);
}

function decodeXml(text: string): string {
	return text
		.replaceAll("&amp;", "&")
		.replaceAll("&lt;", "<")
		.replaceAll("&gt;", ">")
		.replaceAll("&quot;", '"')
		.replaceAll("&apos;", "'");
}

function siteUrl(seed: string, pathname: string): string {
	const url = new URL(seed);
	url.pathname = pathname;
	url.search = "";
	url.hash = "";
	return url.toString();
}

function buildTree(entries: MapUrlEntry[]): Record<string, string[]> {
	const tree: Record<string, string[]> = {};
	for (const entry of entries) {
		const section = new URL(entry.url).pathname.split("/").find(Boolean) ?? "/";
		const sectionUrls = tree[section] ?? [];
		sectionUrls.push(entry.url);
		tree[section] = sectionUrls;
	}
	return tree;
}