import { fetchUrlStream } from "./fetch.js"; export interface MapUrlEntry { url: string; source: "sitemap" | "llms"; sourceUrl: string; } export interface SiteMapResult { seedUrl: string; urls: MapUrlEntry[]; sitemaps: string[]; tree: Record; } export interface DiscoverSiteMapOptions { maxUrls?: number; maxSitemaps?: number; fetchText?: ( url: string, signal?: AbortSignal, ) => Promise; } export async function discoverSiteMap( seed: string, options: DiscoverSiteMapOptions = {}, signal?: AbortSignal, ): Promise { const seedUrl = new URL(seed).toString(); const fetchText = options.fetchText ?? defaultFetchText; const urls = new Map(); const sitemapSet = new Set(); const robotsUrl = siteUrl(seedUrl, "/robots.txt"); const robots = await fetchText(robotsUrl, signal).catch(() => undefined); for (const sitemap of parseRobotsSitemaps(robots ?? "")) sitemapSet.add(sitemap); sitemapSet.add(siteUrl(seedUrl, "/sitemap.xml")); const sitemapQueue = [...sitemapSet]; for ( let i = 0; i < sitemapQueue.length && i < (options.maxSitemaps ?? 20); i++ ) { const sitemapUrl = sitemapQueue[i]; const xml = await fetchText(sitemapUrl, signal).catch(() => undefined); if (!xml) continue; for (const nested of parseSitemapIndexes(xml)) { if (!sitemapSet.has(nested)) { sitemapSet.add(nested); sitemapQueue.push(nested); } } for (const url of parseSitemapXml(xml)) { urls.set(url, { url, source: "sitemap", sourceUrl: sitemapUrl }); if (urls.size >= (options.maxUrls ?? 1_000)) break; } } const llmsUrl = siteUrl(seedUrl, "/llms.txt"); const llms = await fetchText(llmsUrl, signal).catch(() => undefined); for (const url of parseLlmsLinks(llms ?? "", llmsUrl)) { urls.set(url, { url, source: "llms", sourceUrl: llmsUrl }); if (urls.size >= (options.maxUrls ?? 1_000)) break; } const entries = [...urls.values()].toSorted((a, b) => a.url.localeCompare(b.url), ); return { seedUrl, urls: entries, sitemaps: [...sitemapSet], tree: buildTree(entries), }; } export function parseRobotsSitemaps(text: string): string[] { return text .split(/\r?\n/u) .map((line) => line.match(/^\s*sitemap:\s*(\S+)\s*$/iu)?.[1]) .filter((x): x is string => Boolean(x)); } export function parseSitemapXml(xml: string): string[] { return locs(xml).filter( (url) => !url.endsWith(".xml") && !url.endsWith(".xml.gz"), ); } function parseSitemapIndexes(xml: string): string[] { return locs(xml).filter( (url) => url.endsWith(".xml") || url.endsWith(".xml.gz"), ); } export function parseLlmsLinks(text: string, baseUrl: string): string[] { const out: string[] = []; for (const match of text.matchAll(/\[[^\]]*\]\(([^)]+)\)/gu)) { const raw = match[1]?.trim(); if (!raw) continue; try { out.push(new URL(raw, baseUrl).toString()); } catch {} } return out; } async function defaultFetchText( url: string, signal?: AbortSignal, ): Promise { const result = await fetchUrlStream(url, undefined, signal, { maxBytes: 2 * 1024 * 1024, }); return result.body.toString("utf8"); } function locs(xml: string): string[] { return [...xml.matchAll(/\s*([^<]+?)\s*<\/loc>/giu)].map((x) => decodeXml(x[1]), ); } function decodeXml(text: string): string { return text .replaceAll("&", "&") .replaceAll("<", "<") .replaceAll(">", ">") .replaceAll(""", '"') .replaceAll("'", "'"); } function siteUrl(seed: string, pathname: string): string { const url = new URL(seed); url.pathname = pathname; url.search = ""; url.hash = ""; return url.toString(); } function buildTree(entries: MapUrlEntry[]): Record { const tree: Record = {}; for (const entry of entries) { const section = new URL(entry.url).pathname.split("/").find(Boolean) ?? "/"; const sectionUrls = tree[section] ?? []; sectionUrls.push(entry.url); tree[section] = sectionUrls; } return tree; }