// Package highlight renders syntax-highlighted file and diff HTML with chroma. // // Chroma runs in-process and is safe for concurrent use, so there is no // worker pool. package highlight import ( "bytes" "fmt" "html" "path/filepath" "strings" "github.com/alecthomas/chroma/v2" "github.com/alecthomas/chroma/v2/lexers" "github.com/gabriel-vasile/mimetype" "hearthforge/internal/util" ) // BinaryDetectBytes is how much of a file is scanned for a NUL byte. const BinaryDetectBytes = 8000 const ( maxFileCache = 500 maxDiffCache = 500 ) // classPrefix keeps chroma token classes out of the app's own class namespace. const classPrefix = "ch-" // Highlighter holds the bounded render caches. Create one at startup. type Highlighter struct { inlineMaxBytes int64 files *util.Cache[string, FileView] diffs *util.Cache[string, []RenderedDiffFile] } // New returns a Highlighter. inlineMaxBytes is config.InlineMaxBytes. func New(inlineMaxBytes int64) *Highlighter { return &Highlighter{ inlineMaxBytes: inlineMaxBytes, files: util.NewCache[string, FileView](maxFileCache, 0), diffs: util.NewCache[string, []RenderedDiffFile](maxDiffCache, 0), } } // HasBinaryContent reports whether the first BinaryDetectBytes contain a NUL. // Git uses the same rule. Other control characters stay text. func HasBinaryContent(content []byte) bool { if len(content) > BinaryDetectBytes { content = content[:BinaryDetectBytes] } return bytes.IndexByte(content, 0) >= 0 } // DetectLang returns the chroma lexer name for a path, or "" when none matches. // Chroma matches on filename globs, so Dockerfile and Makefile work too. func DetectLang(path string) string { lexer := lexers.Match(filepath.Base(path)) if lexer == nil { return "" } return lexer.Config().Name } // FileView describes how a blob should be shown. // Type is one of "inline", "download", "binary" or "media". type FileView struct { Type string HTML string Lines int Size int64 MimeType string } // ServeFile classifies a blob and renders inline files as a line-numbered table. // cacheKey may be empty to skip caching. func (h *Highlighter) ServeFile(content []byte, filename, cacheKey string) FileView { if v, ok := h.files.Get(cacheKey); ok { return v } size := int64(len(content)) mt := mimetype.Detect(content) switch strings.SplitN(mt.String(), "/", 2)[0] { case "image", "audio", "video": return FileView{Type: "media", MimeType: mt.String(), Size: size} } if HasBinaryContent(content) { return FileView{Type: "binary", Size: size} } if size > h.inlineMaxBytes { return FileView{Type: "download", Size: size} } text := string(content) view := FileView{ Type: "inline", HTML: blobTable(text, DetectLang(filename)), Lines: strings.Count(text, "\n") + 1, Size: size, } if cacheKey != "" { h.files.Set(cacheKey, view) } return view } // splitLines splits source text into display lines, dropping the trailing // empty entry a final newline produces. func splitLines(text string) []string { lines := strings.Split(text, "\n") if n := len(lines); n > 0 && lines[n-1] == "" { lines = lines[:n-1] } return lines } func blobTable(text, lang string) string { // One row per split entry, so a file ending in a newline gets a final // empty row. The previous highlighter did the same, so the line numbers // still match. // Chroma rewrites CRLF and a lone CR to LF before tokenising. Split the // same way, or a file with a bare CR gets shifted line numbers. src := strings.Split(strings.ReplaceAll(strings.ReplaceAll(text, "\r\n", "\n"), "\r", "\n"), "\n") rendered := highlightLines(text, lang, html.EscapeString) var b strings.Builder b.WriteString(``) for i := range src { line := html.EscapeString(src[i]) if i < len(rendered) { line = rendered[i] } n := i + 1 fmt.Fprintf(&b, ``, n, n, n, line) } b.WriteString("
%d%s
") return b.String() } // tokenClass maps a token type to its CSS class, walking up to the parent type // the way chroma's own HTML formatter does. It returns "" for unstyled tokens. func tokenClass(t chroma.TokenType) string { for t != 0 { cls, ok := chroma.StandardTypes[t] if ok { if cls == "" { return "" } return classPrefix + cls } t = t.Parent() } return "" } // highlightLines tokenises code and returns one HTML fragment per source line. // escape converts raw token text to HTML. It returns nil when lang is unknown // or tokenising fails, and the caller falls back to plain escaped text. func highlightLines(code, lang string, escape func(string) string) []string { if lang == "" { return nil } lexer := lexers.Get(lang) if lexer == nil { return nil } iter, err := chroma.Coalesce(lexer).Tokenise(nil, code) if err != nil { return nil } tokenLines := chroma.SplitTokensIntoLines(iter.Tokens()) out := make([]string, len(tokenLines)) for i, tokens := range tokenLines { var b strings.Builder for _, tok := range tokens { value := strings.TrimSuffix(tok.Value, "\n") if value == "" { continue } cls := tokenClass(tok.Type) if cls == "" { b.WriteString(escape(value)) continue } fmt.Fprintf(&b, `%s`, cls, escape(value)) } out[i] = b.String() } return out } // Code returns highlighted HTML for a code block body. lang is a chroma lexer // name or alias, e.g. "js". Unknown languages come back HTML-escaped. func Code(code, lang string) string { src := splitLines(code) rendered := highlightLines(code, lang, html.EscapeString) out := make([]string, len(src)) for i := range src { out[i] = html.EscapeString(src[i]) if i < len(rendered) { out[i] = rendered[i] } } result := strings.Join(out, "\n") if strings.HasSuffix(code, "\n") { result += "\n" } return result }