repositories / pi-ext
pi-ext
bugabingas pi extensions
owned by admin
extensions/web/map.ts
Rawimport { fetchUrlStream } from "./fetch.js";
export interface MapUrlEntry {
url: string;
source: "sitemap" | "llms";
sourceUrl: string;
}
export interface SiteMapResult {
seedUrl: string;
urls: MapUrlEntry[];
sitemaps: string[];
tree: Record<string, string[]>;
}
export interface DiscoverSiteMapOptions {
maxUrls?: number;
maxSitemaps?: number;
fetchText?: (
url: string,
signal?: AbortSignal,
) => Promise<string | undefined>;
}
export async function discoverSiteMap(
seed: string,
options: DiscoverSiteMapOptions = {},
signal?: AbortSignal,
): Promise<SiteMapResult> {
const seedUrl = new URL(seed).toString();
const fetchText = options.fetchText ?? defaultFetchText;
const urls = new Map<string, MapUrlEntry>();
const sitemapSet = new Set<string>();
const robotsUrl = siteUrl(seedUrl, "/robots.txt");
const robots = await fetchText(robotsUrl, signal).catch(() => undefined);
for (const sitemap of parseRobotsSitemaps(robots ?? ""))
sitemapSet.add(sitemap);
sitemapSet.add(siteUrl(seedUrl, "/sitemap.xml"));
const sitemapQueue = [...sitemapSet];
for (
let i = 0;
i < sitemapQueue.length && i < (options.maxSitemaps ?? 20);
i++
) {
const sitemapUrl = sitemapQueue[i];
const xml = await fetchText(sitemapUrl, signal).catch(() => undefined);
if (!xml) continue;
for (const nested of parseSitemapIndexes(xml)) {
if (!sitemapSet.has(nested)) {
sitemapSet.add(nested);
sitemapQueue.push(nested);
}
}
for (const url of parseSitemapXml(xml)) {
urls.set(url, { url, source: "sitemap", sourceUrl: sitemapUrl });
if (urls.size >= (options.maxUrls ?? 1_000)) break;
}
}
const llmsUrl = siteUrl(seedUrl, "/llms.txt");
const llms = await fetchText(llmsUrl, signal).catch(() => undefined);
for (const url of parseLlmsLinks(llms ?? "", llmsUrl)) {
urls.set(url, { url, source: "llms", sourceUrl: llmsUrl });
if (urls.size >= (options.maxUrls ?? 1_000)) break;
}
const entries = [...urls.values()].toSorted((a, b) =>
a.url.localeCompare(b.url),
);
return {
seedUrl,
urls: entries,
sitemaps: [...sitemapSet],
tree: buildTree(entries),
};
}
export function parseRobotsSitemaps(text: string): string[] {
return text
.split(/\r?\n/u)
.map((line) => line.match(/^\s*sitemap:\s*(\S+)\s*$/iu)?.[1])
.filter((x): x is string => Boolean(x));
}
export function parseSitemapXml(xml: string): string[] {
return locs(xml).filter(
(url) => !url.endsWith(".xml") && !url.endsWith(".xml.gz"),
);
}
function parseSitemapIndexes(xml: string): string[] {
return locs(xml).filter(
(url) => url.endsWith(".xml") || url.endsWith(".xml.gz"),
);
}
export function parseLlmsLinks(text: string, baseUrl: string): string[] {
const out: string[] = [];
for (const match of text.matchAll(/\[[^\]]*\]\(([^)]+)\)/gu)) {
const raw = match[1]?.trim();
if (!raw) continue;
try {
out.push(new URL(raw, baseUrl).toString());
} catch {}
}
return out;
}
async function defaultFetchText(
url: string,
signal?: AbortSignal,
): Promise<string | undefined> {
const result = await fetchUrlStream(url, undefined, signal, {
maxBytes: 2 * 1024 * 1024,
});
return result.body.toString("utf8");
}
function locs(xml: string): string[] {
return [...xml.matchAll(/<loc>\s*([^<]+?)\s*<\/loc>/giu)].map((x) =>
decodeXml(x[1]),
);
}
function decodeXml(text: string): string {
return text
.replaceAll("&", "&")
.replaceAll("<", "<")
.replaceAll(">", ">")
.replaceAll(""", '"')
.replaceAll("'", "'");
}
function siteUrl(seed: string, pathname: string): string {
const url = new URL(seed);
url.pathname = pathname;
url.search = "";
url.hash = "";
return url.toString();
}
function buildTree(entries: MapUrlEntry[]): Record<string, string[]> {
const tree: Record<string, string[]> = {};
for (const entry of entries) {
const section = new URL(entry.url).pathname.split("/").find(Boolean) ?? "/";
const sectionUrls = tree[section] ?? [];
sectionUrls.push(entry.url);
tree[section] = sectionUrls;
}
return tree;
}