Luigit
repositories / pi-ext

pi-ext

bugabingas pi extensions

owned by admin

extensions/web/pandoc.ts

Raw
// Pandoc HTML→markdown conversion with LLM-friendly content extraction
// Uses a lua filter to strip nav/sidebar/ads and fix code block languages
// Then strips residual raw-HTML tags from commonmark output

import { dirname, join } from "node:path";
import { fileURLToPath } from "node:url";
import { spawnToText } from "./spawn.js";

const __dirname = dirname(fileURLToPath(import.meta.url));
const LUA_FILTER = join(__dirname, "llm-extract.lua");

// [perf] Single-pass residual tag cleanup.
// Merged 4 chained .replace() into 2 passes to reduce intermediate strings.
const RESIDUAL_DIV_RE = /<div[^>]*>\n?|<\/div>\n?/gi;
const RESIDUAL_SPAN_RE = /<span[^>]*>([\s\S]*?)<\/span>/gi;

export function stripResidualTags(md: string): string {
	return md
		.replace(RESIDUAL_DIV_RE, "")
		.replace(RESIDUAL_SPAN_RE, "$1")
		.replace(/\n{3,}/g, "\n\n");
}

// Strip problematic class attributes and fix table structures for pipe table conversion
export function cleanTableClasses(html: string): string {
	return html
		.replace(/class="[^"]*\[[^\]]*\][^"]*"/g, (match) => {
			const cleaned = match.replace(/\[[^\]]*\]/g, "");
			return cleaned === 'class=""' ? "" : cleaned;
		})
		.replace(/<th[^>]*>\s*<\/th>/g, "")
		.replace(/<\/?strong>/g, (m) => m.replace(/strong/, "b"))
		.replace(
			/<pre[^>]*><code[^>]*>([\s\S]*?)<\/code><\/pre>/gi,
			(_match, content) => {
				return (
					"<code>" +
					content.replace(/\s+/g, " ").trim().replace(/`/g, "\\`") +
					"</code>"
				);
			},
		);
}

export async function runPandoc(
	html: string,
	format: string,
	useFilter = true,
): Promise<string> {
	const toFormat =
		format === "text"
			? "plain"
			: format === "html"
				? "html"
				: format === "json"
					? "json"
					: "markdown-raw_html";

	const args = ["--from=html", "--to=" + toFormat, "--wrap=none"];

	if (useFilter) {
		args.push("--lua-filter=" + LUA_FILTER);
	}

	const stdout = await spawnToText("pandoc", args, cleanTableClasses(html));
	return format === "html" || format === "json"
		? stdout
		: stripResidualTags(stdout);
}

// Convert non-HTML formats via pandoc (docx, odt, epub, rtf, etc.)
export function runPandocFile(
	filePath: string,
	fromFormat: string,
	toFormat: string,
): Promise<string> {
	return spawnToText("pandoc", [
		"--from=" + fromFormat,
		"--to=" + toFormat,
		"--wrap=none",
		filePath,
	]);
}