import DOMPurify from "isomorphic-dompurify"; import { MAX_MD_CACHE } from "../constants.ts"; // GFM parity with the previous `marked` configuration: tables, strikethrough // and task lists are on by default in Bun.markdown; autolinks are not. const MD_OPTIONS: Bun.markdown.Options = { autolinks: true }; const mdCache = new Map(); export interface MarkdownContext { repo: string; ref: string; /** Directory of the markdown file relative to repo root, e.g. "" or "docs/subdir" */ dir: string; } /** * Resolves a markdown href to a repo-root-relative path for rewriting. * Returns null if the href should not be rewritten (protocol-absolute or anchor). * * - Protocol-absolute (http://, mailto:, data:, …): returns null * - Anchor (#section): returns null * - Root-relative (/subdir/img.png): strips leading slash → "subdir/img.png" * - Path-relative (./img.png, ../img.png, subdir/img.png): resolved against dir */ export function resolveMarkdownHref(dir: string, href: string): string | null { if (/^[a-zA-Z][a-zA-Z\d+\-.]*:/.test(href)) return null; // protocol-absolute if (href.startsWith("#")) return null; // anchor if (href.startsWith("/")) return href.slice(1); // root-relative // Path-relative: resolve against current directory using URL API const base = new URL(`http://x/${dir ? `${dir}/` : ""}`); return new URL(href, base).pathname.slice(1); // strip leading / } /** * Point relative links and images at the repo's blob/raw endpoints. * * Done as a post-pass over the rendered HTML rather than inside the parser: * Bun.markdown has no per-element renderer override for `html()` output, and * HTMLRewriter only touches the two attributes we care about. */ function rewriteRepoUrls(html: string, ctx: MarkdownContext): string { const rewrite = (attr: "href" | "src", route: "blob" | "raw") => ({ element(el: HTMLRewriterTypes.Element) { const value = el.getAttribute(attr); if (value === null) return; const resolved = resolveMarkdownHref(ctx.dir, value); if (resolved === null) return; el.setAttribute( attr, `/${ctx.repo}/${route}/${ctx.ref}/${resolved}`, ); }, }); return new HTMLRewriter() .on("a[href]", rewrite("href", "blob")) .on("img[src]", rewrite("src", "raw")) .transform(html); } export function renderMarkdown( md: string, cacheKey?: string, ctx?: MarkdownContext, ): string { if (cacheKey) { const cached = mdCache.get(cacheKey); if (cached) return cached; } let raw = Bun.markdown.html(md, MD_OPTIONS); if (ctx) raw = rewriteRepoUrls(raw, ctx); const result = DOMPurify.sanitize(raw, { ADD_TAGS: ["details", "summary"], ADD_ATTR: ["class"], }); if (cacheKey) { if (mdCache.size >= MAX_MD_CACHE) mdCache.delete(mdCache.keys().next().value!); mdCache.set(cacheKey, result); } return result; } // Sentinels used to hand structure from a child callback up to its parent, // which is the only place that knows the numbering (list) or the separator // (table row). Both are stripped before the result is returned. const ITEM = "\u0000"; const CELL = "\u0001"; /** Convert markdown to plaintext, preserving structure (list prefixes, headings, etc.) */ export function markdownToPlaintext(md: string): string { const text = Bun.markdown.render(md, { heading: (c) => `${c}\n\n`, paragraph: (c) => `${c}\n\n`, code: (c) => `${c}\n\n`, blockquote: (c) => `${c .trim() .split("\n") .map((l) => `> ${l}`) .join("\n")}\n\n`, listItem: (c, meta) => { const task = meta?.checked === undefined ? "" : meta.checked ? "[x] " : "[ ] "; return `${ITEM}${task}${c.trim().replace(/\n+/g, " ")}\n`; }, list: (c, meta) => { const start = meta.start ?? 1; const items = c.split(ITEM).slice(1); const body = items .map( (item, i) => (meta.ordered ? `${start + i}. ` : "- ") + item, ) .join(""); return `${body}\n`; }, th: (c) => `${c}${CELL}`, td: (c) => `${c}${CELL}`, tr: (c) => `${c.split(CELL).slice(0, -1).join(" | ")}\n`, table: (c) => `${c}\n`, hr: () => "---\n\n", html: () => "", link: (c) => c, image: (c, meta) => { const label = (meta.title || c || "").trim(); return label ? `[Image: ${label}]` : "[Image]"; }, codespan: (c) => `\`${c}\``, }); return ( text // The `html` callback only fires for block-level raw HTML. Inline // spans (`x`) reach the output verbatim, so drop tag-shaped // runs here — this string is escaped and shown as a preview. .replace(/<\/?[a-zA-Z][^>]*>/g, "") .replace(/\n{3,}/g, "\n\n") .trim() ); }