highlight.go
| 1 | // Package highlight renders syntax-highlighted file and diff HTML with chroma. |
| 2 | // |
| 3 | // Chroma runs in-process and is safe for concurrent use, so there is no |
| 4 | // worker pool. |
| 5 | package highlight |
| 6 | |
| 7 | import ( |
| 8 | "bytes" |
| 9 | "fmt" |
| 10 | "html" |
| 11 | "path/filepath" |
| 12 | "strings" |
| 13 | |
| 14 | "github.com/alecthomas/chroma/v2" |
| 15 | "github.com/alecthomas/chroma/v2/lexers" |
| 16 | "github.com/gabriel-vasile/mimetype" |
| 17 | |
| 18 | "hearthforge/internal/util" |
| 19 | ) |
| 20 | |
| 21 | // BinaryDetectBytes is how much of a file is scanned for a NUL byte. |
| 22 | const BinaryDetectBytes = 8000 |
| 23 | |
| 24 | const ( |
| 25 | maxFileCache = 500 |
| 26 | maxDiffCache = 500 |
| 27 | maxFileCacheBytes = 64 << 20 |
| 28 | maxDiffCacheBytes = 64 << 20 |
| 29 | ) |
| 30 | |
| 31 | // ClassPrefix keeps chroma token classes out of the app's own class namespace. |
| 32 | const ClassPrefix = "ch-" |
| 33 | |
| 34 | // Highlighter holds the bounded render caches. Create one at startup. |
| 35 | type Highlighter struct { |
| 36 | inlineMaxBytes int64 |
| 37 | files *util.Cache[string, FileView] |
| 38 | diffs *util.Cache[string, []RenderedDiffFile] |
| 39 | } |
| 40 | |
| 41 | // New returns a Highlighter. inlineMaxBytes is config.InlineMaxBytes. |
| 42 | func New(inlineMaxBytes int64) *Highlighter { |
| 43 | return &Highlighter{ |
| 44 | inlineMaxBytes: inlineMaxBytes, |
| 45 | files: util.NewSizedCache[string](maxFileCache, maxFileCacheBytes, 0, fileViewSize), |
| 46 | diffs: util.NewSizedCache[string](maxDiffCache, maxDiffCacheBytes, 0, diffSize), |
| 47 | } |
| 48 | } |
| 49 | |
| 50 | func fileViewSize(v FileView) int64 { return int64(len(v.HTML)) } |
| 51 | |
| 52 | // diffSize approximates the memory of a rendered diff by its string bytes. |
| 53 | func diffSize(files []RenderedDiffFile) int64 { |
| 54 | var n int |
| 55 | for _, f := range files { |
| 56 | n += len(f.OldPath) + len(f.NewPath) |
| 57 | for _, h := range f.Hunks { |
| 58 | n += len(h.Header) |
| 59 | for _, r := range h.Rows { |
| 60 | n += len(r.HTML) + len(r.Type) |
| 61 | } |
| 62 | } |
| 63 | } |
| 64 | return int64(n) |
| 65 | } |
| 66 | |
| 67 | // HasBinaryContent reports whether the first BinaryDetectBytes contain a NUL. |
| 68 | // Git uses the same rule. Other control characters stay text. |
| 69 | func HasBinaryContent(content []byte) bool { |
| 70 | if len(content) > BinaryDetectBytes { |
| 71 | content = content[:BinaryDetectBytes] |
| 72 | } |
| 73 | return bytes.IndexByte(content, 0) >= 0 |
| 74 | } |
| 75 | |
| 76 | // DetectLang returns the chroma lexer name for a path, or "" when none matches. |
| 77 | // Chroma matches on filename globs, so Dockerfile and Makefile work too. |
| 78 | func DetectLang(path string) string { |
| 79 | lexer := lexers.Match(filepath.Base(path)) |
| 80 | if lexer == nil { |
| 81 | return "" |
| 82 | } |
| 83 | return lexer.Config().Name |
| 84 | } |
| 85 | |
| 86 | // FileView describes how a blob should be shown. |
| 87 | // Type is one of "inline", "download", "binary" or "media". |
| 88 | type FileView struct { |
| 89 | Type string |
| 90 | HTML string |
| 91 | Lines int |
| 92 | Size int64 |
| 93 | MimeType string |
| 94 | } |
| 95 | |
| 96 | // ServeFile classifies a blob and renders inline files as a line-numbered table. |
| 97 | // cacheKey may be empty to skip caching. |
| 98 | func (h *Highlighter) ServeFile(content []byte, filename, cacheKey string) FileView { |
| 99 | if v, ok := h.files.Get(cacheKey); ok { |
| 100 | return v |
| 101 | } |
| 102 | size := int64(len(content)) |
| 103 | |
| 104 | mt := mimetype.Detect(content) |
| 105 | switch strings.SplitN(mt.String(), "/", 2)[0] { |
| 106 | case "image", "audio", "video": |
| 107 | return FileView{Type: "media", MimeType: mt.String(), Size: size} |
| 108 | } |
| 109 | if HasBinaryContent(content) { |
| 110 | return FileView{Type: "binary", Size: size} |
| 111 | } |
| 112 | if size > h.inlineMaxBytes { |
| 113 | return FileView{Type: "download", Size: size} |
| 114 | } |
| 115 | |
| 116 | text := string(content) |
| 117 | view := FileView{ |
| 118 | Type: "inline", |
| 119 | HTML: blobTable(text, DetectLang(filename)), |
| 120 | Lines: strings.Count(text, "\n") + 1, |
| 121 | Size: size, |
| 122 | } |
| 123 | if cacheKey != "" { |
| 124 | h.files.Set(cacheKey, view) |
| 125 | } |
| 126 | return view |
| 127 | } |
| 128 | |
| 129 | // splitLines splits source text into display lines, dropping the trailing |
| 130 | // empty entry a final newline produces. |
| 131 | func splitLines(text string) []string { |
| 132 | lines := strings.Split(text, "\n") |
| 133 | if n := len(lines); n > 0 && lines[n-1] == "" { |
| 134 | lines = lines[:n-1] |
| 135 | } |
| 136 | return lines |
| 137 | } |
| 138 | |
| 139 | func blobTable(text, lang string) string { |
| 140 | // One row per split entry, so a file ending in a newline gets a final |
| 141 | // empty row. The previous highlighter did the same, so the line numbers |
| 142 | // still match. |
| 143 | // Chroma rewrites CRLF and a lone CR to LF before tokenising. Split the |
| 144 | // same way, or a file with a bare CR gets shifted line numbers. |
| 145 | src := strings.Split(strings.ReplaceAll(strings.ReplaceAll(text, "\r\n", "\n"), "\r", "\n"), "\n") |
| 146 | rendered := highlightLines(text, lang, html.EscapeString) |
| 147 | var b strings.Builder |
| 148 | b.WriteString(`<table class="blob-table"><tbody>`) |
| 149 | for i := range src { |
| 150 | line := html.EscapeString(src[i]) |
| 151 | if i < len(rendered) { |
| 152 | line = rendered[i] |
| 153 | } |
| 154 | n := i + 1 |
| 155 | fmt.Fprintf(&b, `<tr id="L%d"><td class="blob-ln"><a href="#L%d">%d</a></td><td class="blob-code">%s</td></tr>`, n, n, n, line) |
| 156 | } |
| 157 | b.WriteString("</tbody></table>") |
| 158 | return b.String() |
| 159 | } |
| 160 | |
| 161 | // tokenClass maps a token type to its CSS class, walking up to the parent type |
| 162 | // the way chroma's own HTML formatter does. It returns "" for unstyled tokens. |
| 163 | func tokenClass(t chroma.TokenType) string { |
| 164 | for t != 0 { |
| 165 | cls, ok := chroma.StandardTypes[t] |
| 166 | if ok { |
| 167 | if cls == "" { |
| 168 | return "" |
| 169 | } |
| 170 | return ClassPrefix + cls |
| 171 | } |
| 172 | t = t.Parent() |
| 173 | } |
| 174 | return "" |
| 175 | } |
| 176 | |
| 177 | // highlightLines tokenises code and returns one HTML fragment per source line. |
| 178 | // escape converts raw token text to HTML. It returns nil when lang is unknown |
| 179 | // or tokenising fails, and the caller falls back to plain escaped text. |
| 180 | func highlightLines(code, lang string, escape func(string) string) []string { |
| 181 | if lang == "" { |
| 182 | return nil |
| 183 | } |
| 184 | lexer := lexers.Get(lang) |
| 185 | if lexer == nil { |
| 186 | return nil |
| 187 | } |
| 188 | iter, err := chroma.Coalesce(lexer).Tokenise(nil, code) |
| 189 | if err != nil { |
| 190 | return nil |
| 191 | } |
| 192 | tokenLines := chroma.SplitTokensIntoLines(iter.Tokens()) |
| 193 | out := make([]string, len(tokenLines)) |
| 194 | for i, tokens := range tokenLines { |
| 195 | var b strings.Builder |
| 196 | for _, tok := range tokens { |
| 197 | value := strings.TrimSuffix(tok.Value, "\n") |
| 198 | if value == "" { |
| 199 | continue |
| 200 | } |
| 201 | cls := tokenClass(tok.Type) |
| 202 | if cls == "" { |
| 203 | b.WriteString(escape(value)) |
| 204 | continue |
| 205 | } |
| 206 | fmt.Fprintf(&b, `<span class="%s">%s</span>`, cls, escape(value)) |
| 207 | } |
| 208 | out[i] = b.String() |
| 209 | } |
| 210 | return out |
| 211 | } |
| 212 | |
| 213 | // Code returns highlighted HTML for a code block body. lang is a chroma lexer |
| 214 | // name or alias, e.g. "js". Unknown languages come back HTML-escaped. |
| 215 | func Code(code, lang string) string { |
| 216 | src := splitLines(code) |
| 217 | rendered := highlightLines(code, lang, html.EscapeString) |
| 218 | out := make([]string, len(src)) |
| 219 | for i := range src { |
| 220 | out[i] = html.EscapeString(src[i]) |
| 221 | if i < len(rendered) { |
| 222 | out[i] = rendered[i] |
| 223 | } |
| 224 | } |
| 225 | result := strings.Join(out, "\n") |
| 226 | if strings.HasSuffix(code, "\n") { |
| 227 | result += "\n" |
| 228 | } |
| 229 | return result |
| 230 | } |
| 231 |