highlight.go
⎇
Raw
1// Package highlight renders syntax-highlighted file and diff HTML with chroma.
2//
3// Chroma runs in-process and is safe for concurrent use, so there is no
4// worker pool.
5package highlight
6
7import (
8 "bytes"
9 "fmt"
10 "html"
11 "path/filepath"
12 "strings"
13
14 "github.com/alecthomas/chroma/v2"
15 "github.com/alecthomas/chroma/v2/lexers"
16 "github.com/gabriel-vasile/mimetype"
17
18 "hearthforge/internal/util"
19)
20
21// BinaryDetectBytes is how much of a file is scanned for a NUL byte.
22const BinaryDetectBytes = 8000
23
24const (
25 maxFileCache = 500
26 maxDiffCache = 500
27)
28
29// classPrefix keeps chroma token classes out of the app's own class namespace.
30const classPrefix = "ch-"
31
32// Highlighter holds the bounded render caches. Create one at startup.
33type Highlighter struct {
34 inlineMaxBytes int64
35 files *util.Cache[string, FileView]
36 diffs *util.Cache[string, []RenderedDiffFile]
37}
38
39// New returns a Highlighter. inlineMaxBytes is config.InlineMaxBytes.
40func New(inlineMaxBytes int64) *Highlighter {
41 return &Highlighter{
42 inlineMaxBytes: inlineMaxBytes,
43 files: util.NewCache[string, FileView](maxFileCache, 0),
44 diffs: util.NewCache[string, []RenderedDiffFile](maxDiffCache, 0),
45 }
46}
47
48// HasBinaryContent reports whether the first BinaryDetectBytes contain a NUL.
49// Git uses the same rule. Other control characters stay text.
50func HasBinaryContent(content []byte) bool {
51 if len(content) > BinaryDetectBytes {
52 content = content[:BinaryDetectBytes]
53 }
54 return bytes.IndexByte(content, 0) >= 0
55}
56
57// DetectLang returns the chroma lexer name for a path, or "" when none matches.
58// Chroma matches on filename globs, so Dockerfile and Makefile work too.
59func DetectLang(path string) string {
60 lexer := lexers.Match(filepath.Base(path))
61 if lexer == nil {
62 return ""
63 }
64 return lexer.Config().Name
65}
66
67// FileView describes how a blob should be shown.
68// Type is one of "inline", "download", "binary" or "media".
69type FileView struct {
70 Type string
71 HTML string
72 Lines int
73 Size int64
74 MimeType string
75}
76
77// ServeFile classifies a blob and renders inline files as a line-numbered table.
78// cacheKey may be empty to skip caching.
79func (h *Highlighter) ServeFile(content []byte, filename, cacheKey string) FileView {
80 if v, ok := h.files.Get(cacheKey); ok {
81 return v
82 }
83 size := int64(len(content))
84
85 mt := mimetype.Detect(content)
86 switch strings.SplitN(mt.String(), "/", 2)[0] {
87 case "image", "audio", "video":
88 return FileView{Type: "media", MimeType: mt.String(), Size: size}
89 }
90 if HasBinaryContent(content) {
91 return FileView{Type: "binary", Size: size}
92 }
93 if size > h.inlineMaxBytes {
94 return FileView{Type: "download", Size: size}
95 }
96
97 text := string(content)
98 view := FileView{
99 Type: "inline",
100 HTML: blobTable(text, DetectLang(filename)),
101 Lines: strings.Count(text, "\n") + 1,
102 Size: size,
103 }
104 if cacheKey != "" {
105 h.files.Set(cacheKey, view)
106 }
107 return view
108}
109
110// splitLines splits source text into display lines, dropping the trailing
111// empty entry a final newline produces.
112func splitLines(text string) []string {
113 lines := strings.Split(text, "\n")
114 if n := len(lines); n > 0 && lines[n-1] == "" {
115 lines = lines[:n-1]
116 }
117 return lines
118}
119
120func blobTable(text, lang string) string {
121 // One row per split entry, so a file ending in a newline gets a final
122 // empty row. The previous highlighter did the same, so the line numbers
123 // still match.
124 // Chroma rewrites CRLF and a lone CR to LF before tokenising. Split the
125 // same way, or a file with a bare CR gets shifted line numbers.
126 src := strings.Split(strings.ReplaceAll(strings.ReplaceAll(text, "\r\n", "\n"), "\r", "\n"), "\n")
127 rendered := highlightLines(text, lang, html.EscapeString)
128 var b strings.Builder
129 b.WriteString(`<table class="blob-table"><tbody>`)
130 for i := range src {
131 line := html.EscapeString(src[i])
132 if i < len(rendered) {
133 line = rendered[i]
134 }
135 n := i + 1
136 fmt.Fprintf(&b, `<tr id="L%d"><td class="blob-ln"><a href="#L%d">%d</a></td><td class="blob-code">%s</td></tr>`, n, n, n, line)
137 }
138 b.WriteString("</tbody></table>")
139 return b.String()
140}
141
142// tokenClass maps a token type to its CSS class, walking up to the parent type
143// the way chroma's own HTML formatter does. It returns "" for unstyled tokens.
144func tokenClass(t chroma.TokenType) string {
145 for t != 0 {
146 cls, ok := chroma.StandardTypes[t]
147 if ok {
148 if cls == "" {
149 return ""
150 }
151 return classPrefix + cls
152 }
153 t = t.Parent()
154 }
155 return ""
156}
157
158// highlightLines tokenises code and returns one HTML fragment per source line.
159// escape converts raw token text to HTML. It returns nil when lang is unknown
160// or tokenising fails, and the caller falls back to plain escaped text.
161func highlightLines(code, lang string, escape func(string) string) []string {
162 if lang == "" {
163 return nil
164 }
165 lexer := lexers.Get(lang)
166 if lexer == nil {
167 return nil
168 }
169 iter, err := chroma.Coalesce(lexer).Tokenise(nil, code)
170 if err != nil {
171 return nil
172 }
173 tokenLines := chroma.SplitTokensIntoLines(iter.Tokens())
174 out := make([]string, len(tokenLines))
175 for i, tokens := range tokenLines {
176 var b strings.Builder
177 for _, tok := range tokens {
178 value := strings.TrimSuffix(tok.Value, "\n")
179 if value == "" {
180 continue
181 }
182 cls := tokenClass(tok.Type)
183 if cls == "" {
184 b.WriteString(escape(value))
185 continue
186 }
187 fmt.Fprintf(&b, `<span class="%s">%s</span>`, cls, escape(value))
188 }
189 out[i] = b.String()
190 }
191 return out
192}
193
194// Code returns highlighted HTML for a code block body. lang is a chroma lexer
195// name or alias, e.g. "js". Unknown languages come back HTML-escaped.
196func Code(code, lang string) string {
197 src := splitLines(code)
198 rendered := highlightLines(code, lang, html.EscapeString)
199 out := make([]string, len(src))
200 for i := range src {
201 out[i] = html.EscapeString(src[i])
202 if i < len(rendered) {
203 out[i] = rendered[i]
204 }
205 }
206 result := strings.Join(out, "\n")
207 if strings.HasSuffix(code, "\n") {
208 result += "\n"
209 }
210 return result
211}
212