markdown.ts
| 1 | import DOMPurify from "isomorphic-dompurify"; |
| 2 | import { MAX_MD_CACHE } from "../constants.ts"; |
| 3 | |
| 4 | // GFM parity with the previous `marked` configuration: tables, strikethrough |
| 5 | // and task lists are on by default in Bun.markdown; autolinks are not. |
| 6 | const MD_OPTIONS: Bun.markdown.Options = { autolinks: true }; |
| 7 | |
| 8 | const mdCache = new Map<string, string>(); |
| 9 | |
| 10 | export interface MarkdownContext { |
| 11 | repo: string; |
| 12 | ref: string; |
| 13 | /** Directory of the markdown file relative to repo root, e.g. "" or "docs/subdir" */ |
| 14 | dir: string; |
| 15 | } |
| 16 | |
| 17 | /** |
| 18 | * Resolves a markdown href to a repo-root-relative path for rewriting. |
| 19 | * Returns null if the href should not be rewritten (protocol-absolute or anchor). |
| 20 | * |
| 21 | * - Protocol-absolute (http://, mailto:, data:, …): returns null |
| 22 | * - Anchor (#section): returns null |
| 23 | * - Root-relative (/subdir/img.png): strips leading slash → "subdir/img.png" |
| 24 | * - Path-relative (./img.png, ../img.png, subdir/img.png): resolved against dir |
| 25 | */ |
| 26 | export function resolveMarkdownHref(dir: string, href: string): string | null { |
| 27 | if (/^[a-zA-Z][a-zA-Z\d+\-.]*:/.test(href)) return null; // protocol-absolute |
| 28 | if (href.startsWith("#")) return null; // anchor |
| 29 | if (href.startsWith("/")) return href.slice(1); // root-relative |
| 30 | // Path-relative: resolve against current directory using URL API |
| 31 | const base = new URL(`http://x/${dir ? `${dir}/` : ""}`); |
| 32 | return new URL(href, base).pathname.slice(1); // strip leading / |
| 33 | } |
| 34 | |
| 35 | /** |
| 36 | * Point relative links and images at the repo's blob/raw endpoints. |
| 37 | * |
| 38 | * Done as a post-pass over the rendered HTML rather than inside the parser: |
| 39 | * Bun.markdown has no per-element renderer override for `html()` output, and |
| 40 | * HTMLRewriter only touches the two attributes we care about. |
| 41 | */ |
| 42 | function rewriteRepoUrls(html: string, ctx: MarkdownContext): string { |
| 43 | const rewrite = (attr: "href" | "src", route: "blob" | "raw") => ({ |
| 44 | element(el: HTMLRewriterTypes.Element) { |
| 45 | const value = el.getAttribute(attr); |
| 46 | if (value === null) return; |
| 47 | const resolved = resolveMarkdownHref(ctx.dir, value); |
| 48 | if (resolved === null) return; |
| 49 | el.setAttribute( |
| 50 | attr, |
| 51 | `/${ctx.repo}/${route}/${ctx.ref}/${resolved}`, |
| 52 | ); |
| 53 | }, |
| 54 | }); |
| 55 | return new HTMLRewriter() |
| 56 | .on("a[href]", rewrite("href", "blob")) |
| 57 | .on("img[src]", rewrite("src", "raw")) |
| 58 | .transform(html); |
| 59 | } |
| 60 | |
| 61 | export function renderMarkdown( |
| 62 | md: string, |
| 63 | cacheKey?: string, |
| 64 | ctx?: MarkdownContext, |
| 65 | ): string { |
| 66 | if (cacheKey) { |
| 67 | const cached = mdCache.get(cacheKey); |
| 68 | if (cached) return cached; |
| 69 | } |
| 70 | let raw = Bun.markdown.html(md, MD_OPTIONS); |
| 71 | if (ctx) raw = rewriteRepoUrls(raw, ctx); |
| 72 | const result = DOMPurify.sanitize(raw, { |
| 73 | ADD_TAGS: ["details", "summary"], |
| 74 | ADD_ATTR: ["class"], |
| 75 | }); |
| 76 | if (cacheKey) { |
| 77 | if (mdCache.size >= MAX_MD_CACHE) |
| 78 | mdCache.delete(mdCache.keys().next().value!); |
| 79 | mdCache.set(cacheKey, result); |
| 80 | } |
| 81 | return result; |
| 82 | } |
| 83 | |
| 84 | // Sentinels used to hand structure from a child callback up to its parent, |
| 85 | // which is the only place that knows the numbering (list) or the separator |
| 86 | // (table row). Both are stripped before the result is returned. |
| 87 | const ITEM = "\u0000"; |
| 88 | const CELL = "\u0001"; |
| 89 | |
| 90 | /** Convert markdown to plaintext, preserving structure (list prefixes, headings, etc.) */ |
| 91 | export function markdownToPlaintext(md: string): string { |
| 92 | const text = Bun.markdown.render(md, { |
| 93 | heading: (c) => `${c}\n\n`, |
| 94 | paragraph: (c) => `${c}\n\n`, |
| 95 | code: (c) => `${c}\n\n`, |
| 96 | blockquote: (c) => |
| 97 | `${c |
| 98 | .trim() |
| 99 | .split("\n") |
| 100 | .map((l) => `> ${l}`) |
| 101 | .join("\n")}\n\n`, |
| 102 | listItem: (c, meta) => { |
| 103 | const task = |
| 104 | meta?.checked === undefined |
| 105 | ? "" |
| 106 | : meta.checked |
| 107 | ? "[x] " |
| 108 | : "[ ] "; |
| 109 | return `${ITEM}${task}${c.trim().replace(/\n+/g, " ")}\n`; |
| 110 | }, |
| 111 | list: (c, meta) => { |
| 112 | const start = meta.start ?? 1; |
| 113 | const items = c.split(ITEM).slice(1); |
| 114 | const body = items |
| 115 | .map( |
| 116 | (item, i) => |
| 117 | (meta.ordered ? `${start + i}. ` : "- ") + item, |
| 118 | ) |
| 119 | .join(""); |
| 120 | return `${body}\n`; |
| 121 | }, |
| 122 | th: (c) => `${c}${CELL}`, |
| 123 | td: (c) => `${c}${CELL}`, |
| 124 | tr: (c) => `${c.split(CELL).slice(0, -1).join(" | ")}\n`, |
| 125 | table: (c) => `${c}\n`, |
| 126 | hr: () => "---\n\n", |
| 127 | html: () => "", |
| 128 | link: (c) => c, |
| 129 | image: (c, meta) => { |
| 130 | const label = (meta.title || c || "").trim(); |
| 131 | return label ? `[Image: ${label}]` : "[Image]"; |
| 132 | }, |
| 133 | codespan: (c) => `\`${c}\``, |
| 134 | }); |
| 135 | return ( |
| 136 | text |
| 137 | // The `html` callback only fires for block-level raw HTML. Inline |
| 138 | // spans (`<b>x</b>`) reach the output verbatim, so drop tag-shaped |
| 139 | // runs here — this string is escaped and shown as a preview. |
| 140 | .replace(/<\/?[a-zA-Z][^>]*>/g, "") |
| 141 | .replace(/\n{3,}/g, "\n\n") |
| 142 | .trim() |
| 143 | ); |
| 144 | } |
| 145 |