Luigit
repositories / pi-ext

pi-ext

bugabingas pi extensions

owned by admin

extensions/web/llm-extract.lua

Raw
-- llm-extract.lua
--
-- Pandoc lua filter that converts HTML into LLM-friendly markdown.
-- Run: pandoc -f html --lua-filter=llm-extract.lua -t commonmark --wrap=none
-- What it does:
--   1. Strips non-content blocks: nav, sidebar, footer, ads, copy buttons,
--      breadcrumbs, pagination, cookie banners, interlanguage links
--   2. Unwraps layout divs: container, row, col, theme wrappers
--   3. Fixes code block languages: "language-zig" class → ```zig
--   4. Removes base64 inline images
--   5. Removes links with hreflang (interlanguage)
--   6. Preserves: headings, paragraphs, code blocks, links, lists

-- ── Exact class matches to strip ──────────────────────────────────────────
local strip_exact = {
  -- Navigation
  sidebar = true, nav = true, navigation = true, menu = true,
  breadcrumb = true, breadcrumbs = true, toc = true,
  -- Chrome
  header = true, footer = true, aside = true,
  -- Ads/promo
  ad = true, ads = true, promo = true,
  -- Social
  social = true, share = true,
  -- Comments
  comments = true, comment = true,
  -- Related
  related = true, recommend = true,
  -- Popups
  cookie = true, banner = true, popup = true, modal = true,
  overlay = true, toolbar = true, widget = true,
  -- Code copy buttons (Docusaurus, Prism, etc.)
  buttonGroup = true, ["buttonGroup__atx"] = true,
  copyButtonIcons = true, copyButtonIcon = true,
  -- Pagination
  ["pagination-nav"] = true, ["pagination-nav__sublabel"] = true,
  ["pagination-nav__label"] = true,
  -- Edit/footer meta
  ["theme-doc-footer-edit-meta-row"] = true,
  lastUpdated = true,
  -- MDN
  baseline = true, ["baseline-indicator"] = true,
  ["reference-toc"] = true, ["reference-layout__toc"] = true,
  ["learn-more"] = true, ["feedback-link"] = true,
  ["content-feedback"] = true, ["content-feedback--buttons"] = true,
  ["article-footer"] = true, ["article-footer__inner"] = true,
  ["article-footer__links"] = true, ["article-footer__svg-container"] = true,
  ["example-header"] = true, ["language-name"] = true,
  ["interactive-example"] = true,
  ["notecard"] = true, ["example-bad"] = true,
  -- Wikipedia
  ["interlanguage-link-target"] = true,
  -- Metadata/info boxes (Wikipedia ambox, see also, etc.)
  mbox = true, ambox = true, metadata = true,
  -- Wikipedia infobox/navbox
  infobox = true, vevent = true, navbox = true,
  ["navbox-inner"] = true, hlist = true, nowraplinks = true,
  ["box-Multiple_issues"] = true, ["box-Primary_sources"] = true,
  ["box-Unreliable_sources"] = true, ["box-Promotional"] = true,
  -- Table cells used for metadata/notice
  ["mbox-text"] = true, ["mbox-image"] = true,
  ["mbox-details"] = true,
  -- Generic
  status = true, ["status-title"] = true,
  -- Documentation navigation chrome
  ["navheader"] = true, ["navfooter"] = true,
  ["docs-version"] = true,
  -- Copy/share/page-action chrome
  ["copy-page-split"] = true, ["copy-page-panel"] = true,
  ["markdownBlockTitle"] = true, ["markdown-block-title"] = true,
  ["copy-page-main-label"] = true, ["copy-page-main-btn"] = true,
  ["copy-page-toggle-btn"] = true, ["copy-page-menu"] = true,
  ["copy-icon"] = true, ["check-icon"] = true,
  ["copy-page-chevron"] = true, ["copy-page-toggle"] = true,
  -- Sidebar/collapsible navigation (details/summary)
  details = true, summary = true,
  -- Mobile-only / responsive visibility
  ["lg:hidden"] = true, ["md:hidden"] = true, ["sm:hidden"] = true,
  -- Wikipedia Vector chrome
  ["mw-body-header"] = true,
  ["vector-dropdown"] = true,
  ["after-portlet"] = true,
  ["mw-editsection"] = true,
  toctogglecheckbox = true,
  toctitle = true,
  ["vector-page-titlebar"] = true,
  ["no-font-mode-scale"] = true,
}

-- ── Prefix patterns to strip (class starts with these) ────────────────────
-- Separate from exact matches so the intent is explicit.
local strip_prefixes = {
  "navheader",
  "navfooter",
  "copyButton",
  "buttonGroup",
  "pagination-nav",
  "copy-page",
  "breadcrumb",
  "infobox",
  "navbox",
  "mbox",
}

-- ARIA roles to remove
local strip_roles = {
  navigation = true,
  banner = true,
  contentinfo = true,
  complementary = true,
}

-- TOC/chrome paragraph text patterns
local strip_text_patterns = {
  ["^On this page$"] = true,
  ["^In this article$"] = true,
  ["^Copy page$"] = true,
  ["^Copy pageCopy$"] = true,
  ["^Table of contents$"] = true,
  ["^Copy as Markdown$"] = true,
  ["^Copy$Jump to heading.*$"] = true,
  ["^Jump to heading.*$"] = true,
  ["^API Reference$"] = true,
  ["^Hooks$"] = true,
}

-- Extract language from "language-xxx" or "brush: lang" class
local function get_lang(classes)
  for i, cls in ipairs(classes) do
    local lang = cls:match("^language%-(.+)$")
    if lang then return lang end
    -- Sphinx/Prism "brush: js" style (brush: and js are SEPARATE classes)
    if cls == "brush:" and classes[i+1] then return classes[i+1] end
  end
  return nil
end

-- Check if element should be removed
local function should_strip(el)
  for _, cls in ipairs(el.classes) do
    if strip_exact[cls] then return true end
    -- Check prefix patterns
    for _, prefix in ipairs(strip_prefixes) do
      if cls:find("^" .. prefix .. "[%-_]") then return true end
    end
  end
  if el.attributes.role and strip_roles[el.attributes.role] then return true end
  return false
end

-- ── Filters ────────────────────────────────────────────────────────────────

-- Propagate language classes from wrapper divs to child CodeBlocks,
-- then unwrap or strip. Also handles <details> TOC stripping via sibling analysis.
function Div(el)
  if should_strip(el) then
    return {}
  end

  -- Strip breadcrumb nav divs that contain only OrderedList or BulletList
  local has_breadcrumb_class = false
  for _, cls in ipairs(el.classes) do
    if cls == "breadcrumb" or cls == "breadcrumbs" then
      has_breadcrumb_class = true; break
    end
  end
  if has_breadcrumb_class and #el.content == 1 then
    local inner = el.content[1]
    if inner.t == "OrderedList" or inner.t == "BulletList" then
      return {}
    end
  end

  -- If this div has a language-xxx class, push it down to child CodeBlocks
  local lang = get_lang(el.classes)
  if lang then
    for _, block in ipairs(el.content) do
      if block.t == "CodeBlock" then
        if not get_lang(block.classes) then
          table.insert(block.classes, 1, "language-" .. lang)
        end
      end
    end
  end

  -- <details>/<summary> is parsed as Plain + BulletList by pandoc.
  -- Strip TOC: Plain(text) followed by BulletList(all # links), and text is TOC header.
  local new_content = {}
  local i = 1
  while i <= #el.content do
    local block = el.content[i]
    -- Check for TOC pattern: Plain/Para("On this page") + BulletList(all # links)
    if (block.t == "Plain" or block.t == "Para") and i + 1 <= #el.content then
      local next_block = el.content[i + 1]
      local block_text = pandoc.utils.stringify(block)
      local is_toc_header = false
      for pattern in pairs(strip_text_patterns) do
        if block_text:match(pattern) then is_toc_header = true; break end
      end
      local next_is_toc_list = false
      if next_block.t == "BulletList" then
        local all_hash = true
        for _, item in ipairs(next_block.content) do
          for _, blk in ipairs(item.content or {}) do
            for _, inline in ipairs(blk.content or {}) do
              if inline.t == "Link" then
                local href = inline.target or ""
                if href:sub(1,1) ~= "#" then all_hash = false end
              else
                all_hash = false
              end
            end
          end
        end
        next_is_toc_list = all_hash and #next_block.content > 0
      end
      if is_toc_header and next_is_toc_list then
        -- Skip both (strip them)
        i = i + 2
      else
        table.insert(new_content, block)
        i = i + 1
      end
    else
      table.insert(new_content, block)
      i = i + 1
    end
  end

  el.content = new_content
  return el.content
end

-- Strip XML processing instructions (<?> etc.) and nav chrome HTML
function RawBlock(el)
  if el.format == "html" and el.text:match("^%s*<%?") then return {} end
  -- Strip nav header/footer HTML blocks that contain only navigation links
  if el.format == "html" then
    local text = el.text
    if text:match('class="[^"]*navheader[^"]*"') or
       text:match('class="[^"]*navfooter[^"]*"') or
       text:match('id="[^"]*navheader[^"]*"') or
       text:match('id="[^"]*navfooter[^"]*"') then
      return {}
    end
  end
  return el
end

-- Fix code blocks: extract language from class, drop noise
function CodeBlock(el)
  local lang = get_lang(el.classes)
  el.classes = lang and { lang } or {}
  el.attributes = {}
  el.identifier = ""
  return el
end

-- Unwrap spans, remove copy buttons and heading anchor decorations
function Span(el)
  for _, cls in ipairs(el.classes) do
    if cls:find("copyButton") or cls:find("buttonGroup") then
      return {}
    end
    if cls == "header-anchor" or cls == "anchor-end" or cls == "sr-only" then
      return {}
    end
  end
  return el.content
end

-- Remove base64 inline images (SVGs embedded in HTML)
function Image(el)
  if el.src and el.src:find("^data:") then return {} end
  el.attributes = {}
  el.classes = {}
  return el
end

-- Remove interlanguage links (hreflang attribute = not content)
function Link(el)
  if el.attributes.hreflang then return {} end
  local target = el.target[1] or ""
  -- Strip heading anchor links (self-referencing section links, or header-anchor class)
  if target:match("^#") or el.classes["header-anchor"] then
    return {}
  end
  el.attributes = {}
  el.classes = {}
  return el
end

-- Strip TOC/Chrome paragraphs and processing instruction markers
function Plain(el)
  local text = pandoc.utils.stringify(el)
  if text:match("^%s*$") then return {} end
  if text:match("^%s*<%s*%?%s*>%s*$") then return {} end
  for pattern in pairs(strip_text_patterns) do
    if text:match(pattern) then return {} end
  end
  return el
end

function Para(el)
  local text = pandoc.utils.stringify(el)
  if text:match("^%s*$") then return {} end
  if text:match("^%s*<%s*%?%s*>%s*$") then return {} end
  for pattern in pairs(strip_text_patterns) do
    if text:match(pattern) then return {} end
  end
  return el
end

-- Strip BulletList where ALL links are internal (#anchor) — TOC lists
-- Only strip if there are 3+ items to avoid removing legitimate short lists
function BulletList(el)
  local all_internal = true
  local link_count = 0
  for _, item in ipairs(el.content) do
    for _, blk in ipairs(item.content or {}) do
      for _, inline in ipairs(blk.content or {}) do
        if inline.t == "Link" then
          link_count = link_count + 1
          local href = inline.c[2][1] or ""
          if href:sub(1,1) ~= "#" then
            all_internal = false
          end
        end
      end
    end
  end
  if all_internal and link_count >= 3 then return {} end
  return el
end

-- Strip OrderedList navigation (breadcrumbs, numbered nav menus)
-- Only strip if all items are links AND list is small (≤3)
function OrderedList(el)
  local all_link_items = true
  for _, item_list in ipairs(el.content) do
    local has_link = false
    for _, blk in ipairs(item_list) do
      if blk.content then
        for _, inline in ipairs(blk.content) do
          if inline.t == "Link" then has_link = true end
        end
      end
    end
    if not has_link then all_link_items = false; break end
  end
  if all_link_items and #el.content > 0 and #el.content <= 3 then return {} end
  return el
end

-- Strip TOC/navigation headings (usually preceding in-article TOCs)
local strip_heading_patterns = {
  ["^In this article$"] = true,
  ["^On this page$"] = true,
  ["^Contents$"] = true,
  ["^Jump to.*$"] = true,
  ["^Table of contents$"] = true,
}

function Header(el)
  local text = pandoc.utils.stringify(el)
  for pattern in pairs(strip_heading_patterns) do
    if text:match(pattern) then return {} end
  end
  -- Unwrap heading anchor links: keep link text, remove href
  local new_content = {}
  for _, item in ipairs(el.content) do
    if item.t == "Span" then
      local all_anchor_links = true
      local new_span_content = {}
      for _, inline in ipairs(item.content) do
        if inline.t == "Link" and (inline.target or ""):match("^#") then
          for _, c in ipairs(inline.content) do
            table.insert(new_span_content, c)
          end
        else
          all_anchor_links = false
          table.insert(new_span_content, inline)
        end
      end
      if not all_anchor_links then
        item.content = new_span_content
        table.insert(new_content, item)
      end
    elseif item.t == "Link" and (item.target or ""):match("^#") then
      for _, c in ipairs(item.content) do
        table.insert(new_content, c)
      end
    else
      table.insert(new_content, item)
    end
  end
  el.content = new_content
  return el
end

-- Remove metadata/infobox tables (Wikipedia ambox, etc.)
function Table(el)
  for _, cls in ipairs(el.head.classes) do
    if strip_exact[cls] then return {} end
  end
  for _, row in ipairs(el.head.rows) do
    for _, cell in ipairs(row.cells) do
      for _, cls in ipairs(cell.classes) do
        if strip_exact[cls] then return {} end
      end
    end
  end
  for _, cls in ipairs(el.attr.classes) do
    if strip_exact[cls] then return {} end
  end
  return el
end

return {
  Div = Div,
  Table = Table,
  BulletList = BulletList,
  OrderedList = OrderedList,
  Header = Header,
  RawBlock = RawBlock,
  CodeBlock = CodeBlock,
  Span = Span,
  Image = Image,
  Link = Link,
  Plain = Plain,
  Para = Para,
}