// Pandoc HTML→markdown conversion with LLM-friendly content extraction // Uses a lua filter to strip nav/sidebar/ads and fix code block languages // Then strips residual raw-HTML tags from commonmark output import { dirname, join } from "node:path"; import { fileURLToPath } from "node:url"; import { spawnToText } from "./spawn.js"; const __dirname = dirname(fileURLToPath(import.meta.url)); const LUA_FILTER = join(__dirname, "llm-extract.lua"); // [perf] Single-pass residual tag cleanup. // Merged 4 chained .replace() into 2 passes to reduce intermediate strings. const RESIDUAL_DIV_RE = /
]*>]*>([\s\S]*?)<\/code><\/pre>/gi,
(_match, content) => {
return (
"" +
content.replace(/\s+/g, " ").trim().replace(/`/g, "\\`") +
""
);
},
);
}
export async function runPandoc(
html: string,
format: string,
useFilter = true,
): Promise {
const toFormat =
format === "text"
? "plain"
: format === "html"
? "html"
: format === "json"
? "json"
: "markdown-raw_html";
const args = ["--from=html", "--to=" + toFormat, "--wrap=none"];
if (useFilter) {
args.push("--lua-filter=" + LUA_FILTER);
}
const stdout = await spawnToText("pandoc", args, cleanTableClasses(html));
return format === "html" || format === "json"
? stdout
: stripResidualTags(stdout);
}
// Convert non-HTML formats via pandoc (docx, odt, epub, rtf, etc.)
export function runPandocFile(
filePath: string,
fromFormat: string,
toFormat: string,
): Promise {
return spawnToText("pandoc", [
"--from=" + fromFormat,
"--to=" + toFormat,
"--wrap=none",
filePath,
]);
}