repositories / pi-ext
pi-ext
bugabingas pi extensions
owned by admin
extensions/web/pandoc.ts
Raw// Pandoc HTML→markdown conversion with LLM-friendly content extraction
// Uses a lua filter to strip nav/sidebar/ads and fix code block languages
// Then strips residual raw-HTML tags from commonmark output
import { dirname, join } from "node:path";
import { fileURLToPath } from "node:url";
import { spawnToText } from "./spawn.js";
const __dirname = dirname(fileURLToPath(import.meta.url));
const LUA_FILTER = join(__dirname, "llm-extract.lua");
// [perf] Single-pass residual tag cleanup.
// Merged 4 chained .replace() into 2 passes to reduce intermediate strings.
const RESIDUAL_DIV_RE = /<div[^>]*>\n?|<\/div>\n?/gi;
const RESIDUAL_SPAN_RE = /<span[^>]*>([\s\S]*?)<\/span>/gi;
export function stripResidualTags(md: string): string {
return md
.replace(RESIDUAL_DIV_RE, "")
.replace(RESIDUAL_SPAN_RE, "$1")
.replace(/\n{3,}/g, "\n\n");
}
// Strip problematic class attributes and fix table structures for pipe table conversion
export function cleanTableClasses(html: string): string {
return html
.replace(/class="[^"]*\[[^\]]*\][^"]*"/g, (match) => {
const cleaned = match.replace(/\[[^\]]*\]/g, "");
return cleaned === 'class=""' ? "" : cleaned;
})
.replace(/<th[^>]*>\s*<\/th>/g, "")
.replace(/<\/?strong>/g, (m) => m.replace(/strong/, "b"))
.replace(
/<pre[^>]*><code[^>]*>([\s\S]*?)<\/code><\/pre>/gi,
(_match, content) => {
return (
"<code>" +
content.replace(/\s+/g, " ").trim().replace(/`/g, "\\`") +
"</code>"
);
},
);
}
export async function runPandoc(
html: string,
format: string,
useFilter = true,
): Promise<string> {
const toFormat =
format === "text"
? "plain"
: format === "html"
? "html"
: format === "json"
? "json"
: "markdown-raw_html";
const args = ["--from=html", "--to=" + toFormat, "--wrap=none"];
if (useFilter) {
args.push("--lua-filter=" + LUA_FILTER);
}
const stdout = await spawnToText("pandoc", args, cleanTableClasses(html));
return format === "html" || format === "json"
? stdout
: stripResidualTags(stdout);
}
// Convert non-HTML formats via pandoc (docx, odt, epub, rtf, etc.)
export function runPandocFile(
filePath: string,
fromFormat: string,
toFormat: string,
): Promise<string> {
return spawnToText("pandoc", [
"--from=" + fromFormat,
"--to=" + toFormat,
"--wrap=none",
filePath,
]);
}