import path from "node:path"; import type { BundledLanguage } from "shiki"; import config from "../config.ts"; import { git } from "./git.ts"; import { detectLang, ensureLang, getHighlighter } from "./highlight.ts"; // ─── Types ─────────────────────────────────────────────────────────────────── export type DiffStatus = | "added" | "deleted" | "modified" | "renamed" | "copied"; export interface RenderedRow { type: "add" | "del" | "context"; oldLine: number | null; newLine: number | null; html: string; } export interface RenderedHunk { header: string; rows: RenderedRow[]; } export interface RenderedDiffFile { oldPath: string; newPath: string; status: DiffStatus; added: number; removed: number; isBinary: boolean; /** Byte sizes for binary files. Populated from GIT binary patch literals or blob lookups. */ binarySize?: { before: number; after: number }; hunks: RenderedHunk[]; } // ─── Parser ────────────────────────────────────────────────────────────────── interface ParsedLine { type: "add" | "del" | "context"; content: string; } interface ParsedHunk { header: string; oldStart: number; newStart: number; lines: ParsedLine[]; } export interface ParsedFile { oldPath: string; newPath: string; status: DiffStatus; added: number; removed: number; isBinary: boolean; binaryNewSize?: number; binaryOldSize?: number; hunks: ParsedHunk[]; } export async function parseDiff( raw: string, repoName?: string, ): Promise { const files: ParsedFile[] = []; const allLines = raw.split("\n"); let i = 0; while (i < allLines.length) { if (!allLines[i]!.startsWith("diff --git ")) { i++; continue; } const m = allLines[i]!.match(/^diff --git a\/(.+) b\/(.+)$/); const fallback = m ? (m[2] ?? "") : ""; const file: ParsedFile = { oldPath: fallback, newPath: fallback, status: "modified", added: 0, removed: 0, isBinary: false, hunks: [], }; i++; let oldBlob = ""; let newBlob = ""; while (i < allLines.length) { const line = allLines[i]!; if (line.startsWith("diff --git ") || line.startsWith("@@ ")) break; if (line.startsWith("new file")) file.status = "added"; else if (line.startsWith("deleted file")) file.status = "deleted"; else if (line.startsWith("rename from ")) { file.status = "renamed"; file.oldPath = line.slice(12); } else if (line.startsWith("rename to ")) { file.newPath = line.slice(10); } else if (line.startsWith("copy from ")) { file.status = "copied"; file.oldPath = line.slice(10); } else if (line.startsWith("copy to ")) { file.newPath = line.slice(8); } else if (line.startsWith("--- ") && line !== "--- /dev/null") file.oldPath = line.slice(6); else if (line.startsWith("+++ ") && line !== "+++ /dev/null") file.newPath = line.slice(6); else if (line.startsWith("index ")) { const idxm = line.match(/^index ([0-9a-f]+)\.\.([0-9a-f]+)/i); if (idxm) { oldBlob = idxm[1]!; newBlob = idxm[2]!; } } else if (line.startsWith("Binary files ")) { file.isBinary = true; if (repoName) { const [oldSize, newSize] = await Promise.all([ git.blobSize(repoName, oldBlob), git.blobSize(repoName, newBlob), ]); file.binaryOldSize = oldSize; file.binaryNewSize = newSize; } } else if (line === "GIT binary patch") { file.isBinary = true; } else if (file.isBinary && line.startsWith("literal ")) { const size = parseInt(line.slice(8), 10); if (file.binaryNewSize === undefined) file.binaryNewSize = size; else if (file.binaryOldSize === undefined) file.binaryOldSize = size; } i++; } while (i < allLines.length && allLines[i]!.startsWith("@@ ")) { const hm = allLines[i]!.match( /@@ -(\d+)(?:,\d+)? \+(\d+)(?:,\d+)? @@/, ); const hunk: ParsedHunk = { header: allLines[i]!, oldStart: hm ? parseInt(hm[1]!, 10) : 1, newStart: hm ? parseInt(hm[2]!, 10) : 1, lines: [], }; i++; while ( i < allLines.length && !allLines[i]!.startsWith("@@ ") && !allLines[i]!.startsWith("diff --git ") ) { const l = allLines[i]!; if (l.startsWith("+")) { hunk.lines.push({ type: "add", content: l.slice(1) }); file.added++; } else if (l.startsWith("-")) { hunk.lines.push({ type: "del", content: l.slice(1) }); file.removed++; } else if (l.startsWith(" ")) { hunk.lines.push({ type: "context", content: l.slice(1) }); } // skip "\ No newline at end of file" i++; } file.hunks.push(hunk); } files.push(file); } return files; } function _escapeHtml(s: string): string { return s.replace(/&/g, "&").replace(//g, ">"); } // biome-ignore lint/suspicious/noControlCharactersInRegex: intentional control char rendering const CTRL_RE = /[\x00-\x08\x0b\x0c\x0d\x0e-\x1f]/g; function escapeHtmlAndCtrl(s: string): string { return s .replace(/&/g, "&") .replace(//g, ">") .replace(CTRL_RE, (ch) => { const label = `^${String.fromCharCode(ch.charCodeAt(0) + 64)}`; return `${label}`; }); } // ─── Highlight a hunk ──────────────────────────────────────────────────────── async function highlightHunk( hunk: ParsedHunk, lang: string, ): Promise { if (hunk.lines.length === 0) return []; // Strip trailing \r before joining with \n: Shiki normalises \r\n as a // single newline and silently drops the \r from every line except the last. const contents = hunk.lines.map((l) => l.content); const trailingCR = contents.map((c) => c.endsWith("\r")); const stripped = contents.map((c, i) => trailingCR[i] ? c.slice(0, -1) : c, ); const code = stripped.join("\n"); try { const h = await getHighlighter(); const tokensByLine = h.codeToTokensWithThemes(code, { lang: lang as BundledLanguage, themes: { light: "github-light", dark: "github-dark" }, }); const lines = tokensByLine.map((lineTokens, i) => { const html = lineTokens .map((token) => { const style = Object.entries(token.variants) .map( ([theme, v]) => `--shiki-${theme}:${v.color ?? "inherit"}`, ) .join(";"); return `${escapeHtmlAndCtrl(token.content)}`; }) .join(""); return trailingCR[i] ? `${html}^M` : html; }); while (lines.length < hunk.lines.length) lines.push(""); return lines; } catch { return hunk.lines.map((l) => escapeHtmlAndCtrl(l.content)); } } // ─── Build rendered rows ────────────────────────────────────────────────────── function buildRows(hunk: ParsedHunk, highlighted: string[]): RenderedRow[] { const rows: RenderedRow[] = []; let oldLine = hunk.oldStart; let newLine = hunk.newStart; for (let i = 0; i < hunk.lines.length; i++) { const { type } = hunk.lines[i]!; const html = highlighted[i] ?? ""; if (type === "context") { rows.push({ type, oldLine: oldLine++, newLine: newLine++, html }); } else if (type === "add") { rows.push({ type, oldLine: null, newLine: newLine++, html }); } else { rows.push({ type, oldLine: oldLine++, newLine: null, html }); } } return rows; } // ─── Highlight a single parsed file ────────────────────────────────────────── export async function highlightFile( file: ParsedFile, ): Promise { const displayPath = file.newPath || file.oldPath; const totalBytes = file.hunks.reduce( (sum, h) => sum + h.lines.reduce((s, l) => s + l.content.length, 0), 0, ); const tooBig = totalBytes > config.INLINE_MAX_BYTES; // Load the grammar once per file, before the per-hunk try/catch that // would otherwise swallow the "language not loaded" error and emit // unhighlighted lines. Skipped entirely when we are not highlighting. const lang = tooBig ? "text" : await ensureLang(detectLang(path.basename(displayPath))); const highlightedHunks = tooBig ? file.hunks.map((hunk) => hunk.lines.map((l) => escapeHtmlAndCtrl(l.content)), ) : await Promise.all( file.hunks.map((hunk) => highlightHunk(hunk, lang)), ); const hunks: RenderedHunk[] = file.hunks.map((hunk, i) => ({ header: hunk.header, rows: buildRows(hunk, highlightedHunks[i]!), })); let binarySize: RenderedDiffFile["binarySize"]; if (file.isBinary && file.binaryNewSize !== undefined) { binarySize = { after: file.binaryNewSize, before: file.binaryOldSize ?? 0, }; } return { oldPath: file.oldPath, newPath: file.newPath, status: file.status, added: file.added, removed: file.removed, isBinary: file.isBinary, binarySize, hunks, }; }