package highlight
import (
"fmt"
"path/filepath"
"regexp"
"strconv"
"strings"
)
// DiffStatus is how git changed a file in a diff.
type DiffStatus string
const (
StatusAdded DiffStatus = "added"
StatusDeleted DiffStatus = "deleted"
StatusModified DiffStatus = "modified"
StatusRenamed DiffStatus = "renamed"
StatusCopied DiffStatus = "copied"
)
// NoLine marks a diff row that has no number on one side.
const NoLine = -1
// RenderedRow is one line of a hunk with its highlighted HTML.
// OldLine and NewLine are NoLine when the row has no number on that side.
// A real number can be 0, because a hunk header may start at 0.
type RenderedRow struct {
Type string // add | del | context
OldLine int
NewLine int
HTML string
}
// RenderedHunk is one @@ block.
type RenderedHunk struct {
Header string
Rows []RenderedRow
}
// RenderedDiffFile is one file of a diff, ready for the view.
type RenderedDiffFile struct {
OldPath string
NewPath string
Status DiffStatus
Added int
Removed int
IsBinary bool
BinaryFrom int64 // byte size before, only for binary files
BinaryTo int64 // byte size after, only for binary files
HasBinarySize bool
Hunks []RenderedHunk
}
type parsedLine struct {
typ string
content string
}
type parsedHunk struct {
header string
oldStart int
newStart int
lines []parsedLine
}
// ParsedFile is the raw parse result, before highlighting.
type ParsedFile struct {
OldPath string
NewPath string
Status DiffStatus
Added int
Removed int
IsBinary bool
BinaryOldSize int64
BinaryNewSize int64
hasOldSize bool
hasNewSize bool
oldBlob string // set for "Binary files ..." lines, which carry no size
newBlob string
hunks []parsedHunk
}
var (
diffGitRE = regexp.MustCompile(`^diff --git a/(.+) b/(.+)$`)
indexRE = regexp.MustCompile(`(?i)^index ([0-9a-f]+)\.\.([0-9a-f]+)`)
hunkRE = regexp.MustCompile(`@@ -(\d+)(?:,\d+)? \+(\d+)(?:,\d+)? @@`)
)
// unquotePath undoes git's C-style quoting of paths with non-ASCII or
// special bytes. Go's string escapes are a superset of git's.
func unquotePath(s string) string {
if strings.HasPrefix(s, `"`) {
if u, err := strconv.Unquote(s); err == nil {
return u
}
}
return s
}
// trimDiffPath strips the "--- " or "+++ " marker, the quoting and the "a/"
// or "b/" path prefix. It never slices, so a short or truncated line is safe.
// Git appends a TAB to names that contain a space.
func trimDiffPath(line, marker, prefix string) string {
return strings.TrimPrefix(unquotePath(strings.TrimSuffix(strings.TrimPrefix(line, marker), "\t")), prefix)
}
// diffGitNewPath returns the "b/" path of a "diff --git" line, or "".
func diffGitNewPath(line string) string {
if strings.HasSuffix(line, `"`) {
// A quoted name escapes every inner quote, so ` "` starts the last token.
if i := strings.LastIndex(line, ` "`); i >= 0 {
return strings.TrimPrefix(unquotePath(line[i+1:]), "b/")
}
}
if m := diffGitRE.FindStringSubmatch(line); m != nil {
return m[2]
}
return ""
}
// parseDiff parses `git diff` output into per-file structures.
//
// blobSizes resolves blob SHAs to byte sizes in one batch, for "Binary files
// ..." lines that carry no size. It may be nil.
func parseDiff(raw string, blobSizes func(shas []string) map[string]int64) []ParsedFile {
var files []ParsedFile
all := strings.Split(raw, "\n")
i := 0
for i < len(all) {
if !strings.HasPrefix(all[i], "diff --git ") {
i++
continue
}
fallback := diffGitNewPath(all[i])
file := ParsedFile{OldPath: fallback, NewPath: fallback, Status: StatusModified}
i++
oldBlob, newBlob := "", ""
for i < len(all) {
line := all[i]
if strings.HasPrefix(line, "diff --git ") || strings.HasPrefix(line, "@@ ") {
break
}
switch {
case strings.HasPrefix(line, "new file"):
file.Status = StatusAdded
case strings.HasPrefix(line, "deleted file"):
file.Status = StatusDeleted
case strings.HasPrefix(line, "rename from "):
file.Status = StatusRenamed
file.OldPath = unquotePath(line[12:])
case strings.HasPrefix(line, "rename to "):
file.NewPath = unquotePath(line[10:])
case strings.HasPrefix(line, "copy from "):
file.Status = StatusCopied
file.OldPath = unquotePath(line[10:])
case strings.HasPrefix(line, "copy to "):
file.NewPath = unquotePath(line[8:])
case strings.HasPrefix(line, "--- ") && line != "--- /dev/null":
file.OldPath = trimDiffPath(line, "--- ", "a/")
case strings.HasPrefix(line, "+++ ") && line != "+++ /dev/null":
file.NewPath = trimDiffPath(line, "+++ ", "b/")
case strings.HasPrefix(line, "index "):
if m := indexRE.FindStringSubmatch(line); m != nil {
oldBlob, newBlob = m[1], m[2]
}
case strings.HasPrefix(line, "Binary files "):
file.IsBinary = true
file.oldBlob, file.newBlob = oldBlob, newBlob
case line == "GIT binary patch":
file.IsBinary = true
case file.IsBinary && strings.HasPrefix(line, "literal "):
size, _ := strconv.ParseInt(strings.TrimSpace(line[8:]), 10, 64)
if !file.hasNewSize {
file.BinaryNewSize, file.hasNewSize = size, true
} else if !file.hasOldSize {
file.BinaryOldSize, file.hasOldSize = size, true
}
}
i++
}
for i < len(all) && strings.HasPrefix(all[i], "@@ ") {
hunk := parsedHunk{header: all[i], oldStart: 1, newStart: 1}
if m := hunkRE.FindStringSubmatch(all[i]); m != nil {
hunk.oldStart, _ = strconv.Atoi(m[1])
hunk.newStart, _ = strconv.Atoi(m[2])
}
i++
for i < len(all) && !strings.HasPrefix(all[i], "@@ ") && !strings.HasPrefix(all[i], "diff --git ") {
l := all[i]
switch {
case strings.HasPrefix(l, "+"):
hunk.lines = append(hunk.lines, parsedLine{"add", l[1:]})
file.Added++
case strings.HasPrefix(l, "-"):
hunk.lines = append(hunk.lines, parsedLine{"del", l[1:]})
file.Removed++
case strings.HasPrefix(l, " "):
hunk.lines = append(hunk.lines, parsedLine{"context", l[1:]})
}
// "\ No newline at end of file" is skipped.
i++
}
file.hunks = append(file.hunks, hunk)
}
files = append(files, file)
}
if blobSizes != nil {
fillBlobSizes(files, blobSizes)
}
return files
}
// fillBlobSizes sets the sizes of "Binary files ..." entries. A blob that
// does not resolve, like the all-zero id of an added file, counts as 0.
func fillBlobSizes(files []ParsedFile, blobSizes func(shas []string) map[string]int64) {
var shas []string
for _, f := range files {
for _, sha := range []string{f.oldBlob, f.newBlob} {
if strings.Trim(sha, "0") != "" {
shas = append(shas, sha)
}
}
}
if len(shas) == 0 {
return
}
sizes := blobSizes(shas)
for i := range files {
f := &files[i]
if f.oldBlob == "" && f.newBlob == "" {
continue
}
f.BinaryOldSize, f.hasOldSize = sizes[f.oldBlob], true
f.BinaryNewSize, f.hasNewSize = sizes[f.newBlob], true
}
}
// escapeHTMLAndCtrl escapes HTML and renders C0 control characters as caret
// notation in a visible span. TAB, LF and DEL are left alone.
func escapeHTMLAndCtrl(s string) string {
var b strings.Builder
for _, r := range s {
switch {
case r == '&':
b.WriteString("&")
case r == '<':
b.WriteString("<")
case r == '>':
b.WriteString(">")
case r < 0x20 && r != '\t' && r != '\n':
fmt.Fprintf(&b, `^%c`, byte(r)+64)
default:
b.WriteRune(r)
}
}
return b.String()
}
// highlightHunk returns the HTML for every line of a hunk.
//
// Trailing CR is stripped before joining, then re-appended as a ^M marker.
// A lexer would otherwise treat the CR as ordinary whitespace inside a token.
func highlightHunk(hunk parsedHunk, lang string) []string {
if len(hunk.lines) == 0 {
return nil
}
stripped := make([]string, len(hunk.lines))
trailingCR := make([]bool, len(hunk.lines))
for i, l := range hunk.lines {
trailingCR[i] = strings.HasSuffix(l.content, "\r")
stripped[i] = strings.TrimSuffix(l.content, "\r")
}
// Chroma's lexer rewrites a lone CR to LF, which would split one source
// line into two and misalign every later line. Such lines are rare, so the
// whole hunk falls back to plain escaped text.
innerCR := false
for _, s := range stripped {
if strings.Contains(s, "\r") {
innerCR = true
break
}
}
var out []string
if !innerCR {
out = highlightLines(strings.Join(stripped, "\n"), lang, escapeHTMLAndCtrl)
}
for len(out) < len(hunk.lines) {
out = append(out, "")
}
for i := range hunk.lines {
if out[i] == "" && stripped[i] != "" {
out[i] = escapeHTMLAndCtrl(stripped[i])
}
if trailingCR[i] {
out[i] += `^M`
}
}
return out[:len(hunk.lines)]
}
func buildRows(hunk parsedHunk, highlighted []string) []RenderedRow {
rows := make([]RenderedRow, 0, len(hunk.lines))
oldLine, newLine := hunk.oldStart, hunk.newStart
for i, l := range hunk.lines {
html := ""
if i < len(highlighted) {
html = highlighted[i]
}
row := RenderedRow{Type: l.typ, HTML: html, OldLine: NoLine, NewLine: NoLine}
switch l.typ {
case "context":
row.OldLine, row.NewLine = oldLine, newLine
oldLine++
newLine++
case "add":
row.NewLine = newLine
newLine++
default:
row.OldLine = oldLine
oldLine++
}
rows = append(rows, row)
}
return rows
}
// highlightFile renders one parsed file. Files larger than inlineMaxBytes are
// rendered without highlighting.
func (h *Highlighter) highlightFile(file ParsedFile) RenderedDiffFile {
displayPath := file.NewPath
if displayPath == "" {
displayPath = file.OldPath
}
var totalBytes int64
for _, hunk := range file.hunks {
for _, l := range hunk.lines {
totalBytes += int64(len(l.content))
}
}
lang := ""
if totalBytes <= h.inlineMaxBytes {
lang = DetectLang(filepath.Base(displayPath))
}
hunks := make([]RenderedHunk, len(file.hunks))
for i, hunk := range file.hunks {
hunks[i] = RenderedHunk{Header: hunk.header, Rows: buildRows(hunk, highlightHunk(hunk, lang))}
}
return RenderedDiffFile{
OldPath: file.OldPath,
NewPath: file.NewPath,
Status: file.Status,
Added: file.Added,
Removed: file.Removed,
IsBinary: file.IsBinary,
BinaryFrom: file.BinaryOldSize,
BinaryTo: file.BinaryNewSize,
HasBinarySize: file.IsBinary && file.hasNewSize,
Hunks: hunks,
}
}
// PrepareDiff parses and highlights a whole diff, with caching.
// cacheKey may be empty to skip caching.
func (h *Highlighter) PrepareDiff(raw, cacheKey string, blobSizes func(shas []string) map[string]int64) []RenderedDiffFile {
if v, ok := h.diffs.Get(cacheKey); ok {
return v
}
parsed := parseDiff(raw, blobSizes)
out := make([]RenderedDiffFile, len(parsed))
for i, f := range parsed {
out[i] = h.highlightFile(f)
}
if cacheKey != "" {
h.diffs.Set(cacheKey, out)
}
return out
}