diff.go
⎇
Raw
1package highlight
2
3import (
4 "fmt"
5 "path/filepath"
6 "regexp"
7 "strconv"
8 "strings"
9)
10
11// DiffStatus is how git changed a file in a diff.
12type DiffStatus string
13
14const (
15 StatusAdded DiffStatus = "added"
16 StatusDeleted DiffStatus = "deleted"
17 StatusModified DiffStatus = "modified"
18 StatusRenamed DiffStatus = "renamed"
19 StatusCopied DiffStatus = "copied"
20)
21
22// NoLine marks a diff row that has no number on one side.
23const NoLine = -1
24
25// RenderedRow is one line of a hunk with its highlighted HTML.
26// OldLine and NewLine are NoLine when the row has no number on that side.
27// A real number can be 0, because a hunk header may start at 0.
28type RenderedRow struct {
29 Type string // add | del | context
30 OldLine int
31 NewLine int
32 HTML string
33}
34
35// RenderedHunk is one @@ block.
36type RenderedHunk struct {
37 Header string
38 Rows []RenderedRow
39}
40
41// RenderedDiffFile is one file of a diff, ready for the view.
42type RenderedDiffFile struct {
43 OldPath string
44 NewPath string
45 Status DiffStatus
46 Added int
47 Removed int
48 IsBinary bool
49 BinaryFrom int64 // byte size before, only for binary files
50 BinaryTo int64 // byte size after, only for binary files
51 HasBinarySize bool
52 Hunks []RenderedHunk
53}
54
55type parsedLine struct {
56 typ string
57 content string
58}
59
60type parsedHunk struct {
61 header string
62 oldStart int
63 newStart int
64 lines []parsedLine
65}
66
67// ParsedFile is the raw parse result, before highlighting.
68type ParsedFile struct {
69 OldPath string
70 NewPath string
71 Status DiffStatus
72 Added int
73 Removed int
74 IsBinary bool
75 BinaryOldSize int64
76 BinaryNewSize int64
77 hasOldSize bool
78 hasNewSize bool
79 hunks []parsedHunk
80}
81
82var (
83 diffGitRE = regexp.MustCompile(`^diff --git a/(.+) b/(.+)$`)
84 indexRE = regexp.MustCompile(`(?i)^index ([0-9a-f]+)\.\.([0-9a-f]+)`)
85 hunkRE = regexp.MustCompile(`@@ -(\d+)(?:,\d+)? \+(\d+)(?:,\d+)? @@`)
86)
87
88// trimDiffPath strips the "--- " or "+++ " marker and the "a/" or "b/" path
89// prefix. It never slices, so a short or truncated line is safe.
90func trimDiffPath(line, marker, prefix string) string {
91 return strings.TrimPrefix(strings.TrimPrefix(line, marker), prefix)
92}
93
94// parseDiff parses `git diff` output into per-file structures.
95//
96// blobSize resolves a blob SHA to its byte size, for "Binary files ..." lines
97// that carry no size. It may be nil.
98func parseDiff(raw string, blobSize func(sha string) int64) []ParsedFile {
99 var files []ParsedFile
100 all := strings.Split(raw, "\n")
101 i := 0
102
103 for i < len(all) {
104 if !strings.HasPrefix(all[i], "diff --git ") {
105 i++
106 continue
107 }
108 fallback := ""
109 if m := diffGitRE.FindStringSubmatch(all[i]); m != nil {
110 fallback = m[2]
111 }
112 file := ParsedFile{OldPath: fallback, NewPath: fallback, Status: StatusModified}
113 i++
114
115 oldBlob, newBlob := "", ""
116 for i < len(all) {
117 line := all[i]
118 if strings.HasPrefix(line, "diff --git ") || strings.HasPrefix(line, "@@ ") {
119 break
120 }
121 switch {
122 case strings.HasPrefix(line, "new file"):
123 file.Status = StatusAdded
124 case strings.HasPrefix(line, "deleted file"):
125 file.Status = StatusDeleted
126 case strings.HasPrefix(line, "rename from "):
127 file.Status = StatusRenamed
128 file.OldPath = line[12:]
129 case strings.HasPrefix(line, "rename to "):
130 file.NewPath = line[10:]
131 case strings.HasPrefix(line, "copy from "):
132 file.Status = StatusCopied
133 file.OldPath = line[10:]
134 case strings.HasPrefix(line, "copy to "):
135 file.NewPath = line[8:]
136 case strings.HasPrefix(line, "--- ") && line != "--- /dev/null":
137 file.OldPath = trimDiffPath(line, "--- ", "a/")
138 case strings.HasPrefix(line, "+++ ") && line != "+++ /dev/null":
139 file.NewPath = trimDiffPath(line, "+++ ", "b/")
140 case strings.HasPrefix(line, "index "):
141 if m := indexRE.FindStringSubmatch(line); m != nil {
142 oldBlob, newBlob = m[1], m[2]
143 }
144 case strings.HasPrefix(line, "Binary files "):
145 file.IsBinary = true
146 if blobSize != nil {
147 file.BinaryOldSize, file.hasOldSize = blobSize(oldBlob), true
148 file.BinaryNewSize, file.hasNewSize = blobSize(newBlob), true
149 }
150 case line == "GIT binary patch":
151 file.IsBinary = true
152 case file.IsBinary && strings.HasPrefix(line, "literal "):
153 size, _ := strconv.ParseInt(strings.TrimSpace(line[8:]), 10, 64)
154 if !file.hasNewSize {
155 file.BinaryNewSize, file.hasNewSize = size, true
156 } else if !file.hasOldSize {
157 file.BinaryOldSize, file.hasOldSize = size, true
158 }
159 }
160 i++
161 }
162
163 for i < len(all) && strings.HasPrefix(all[i], "@@ ") {
164 hunk := parsedHunk{header: all[i], oldStart: 1, newStart: 1}
165 if m := hunkRE.FindStringSubmatch(all[i]); m != nil {
166 hunk.oldStart, _ = strconv.Atoi(m[1])
167 hunk.newStart, _ = strconv.Atoi(m[2])
168 }
169 i++
170 for i < len(all) && !strings.HasPrefix(all[i], "@@ ") && !strings.HasPrefix(all[i], "diff --git ") {
171 l := all[i]
172 switch {
173 case strings.HasPrefix(l, "+"):
174 hunk.lines = append(hunk.lines, parsedLine{"add", l[1:]})
175 file.Added++
176 case strings.HasPrefix(l, "-"):
177 hunk.lines = append(hunk.lines, parsedLine{"del", l[1:]})
178 file.Removed++
179 case strings.HasPrefix(l, " "):
180 hunk.lines = append(hunk.lines, parsedLine{"context", l[1:]})
181 }
182 // "\ No newline at end of file" is skipped.
183 i++
184 }
185 file.hunks = append(file.hunks, hunk)
186 }
187 files = append(files, file)
188 }
189 return files
190}
191
192// escapeHTMLAndCtrl escapes HTML and renders C0 control characters as caret
193// notation in a visible span. TAB, LF and DEL are left alone.
194func escapeHTMLAndCtrl(s string) string {
195 var b strings.Builder
196 for _, r := range s {
197 switch {
198 case r == '&':
199 b.WriteString("&amp;")
200 case r == '<':
201 b.WriteString("&lt;")
202 case r == '>':
203 b.WriteString("&gt;")
204 case r < 0x20 && r != '\t' && r != '\n':
205 fmt.Fprintf(&b, `<span class="diff-ctrl">^%c</span>`, byte(r)+64)
206 default:
207 b.WriteRune(r)
208 }
209 }
210 return b.String()
211}
212
213// highlightHunk returns the HTML for every line of a hunk.
214//
215// Trailing CR is stripped before joining, then re-appended as a ^M marker.
216// A lexer would otherwise treat the CR as ordinary whitespace inside a token.
217func highlightHunk(hunk parsedHunk, lang string) []string {
218 if len(hunk.lines) == 0 {
219 return nil
220 }
221 stripped := make([]string, len(hunk.lines))
222 trailingCR := make([]bool, len(hunk.lines))
223 for i, l := range hunk.lines {
224 trailingCR[i] = strings.HasSuffix(l.content, "\r")
225 stripped[i] = strings.TrimSuffix(l.content, "\r")
226 }
227 // Chroma's lexer rewrites a lone CR to LF, which would split one source
228 // line into two and misalign every later line. Such lines are rare, so the
229 // whole hunk falls back to plain escaped text.
230 innerCR := false
231 for _, s := range stripped {
232 if strings.Contains(s, "\r") {
233 innerCR = true
234 break
235 }
236 }
237 var out []string
238 if !innerCR {
239 out = highlightLines(strings.Join(stripped, "\n"), lang, escapeHTMLAndCtrl)
240 }
241 for len(out) < len(hunk.lines) {
242 out = append(out, "")
243 }
244 for i := range hunk.lines {
245 if out[i] == "" && stripped[i] != "" {
246 out[i] = escapeHTMLAndCtrl(stripped[i])
247 }
248 if trailingCR[i] {
249 out[i] += `<span class="diff-ctrl">^M</span>`
250 }
251 }
252 return out[:len(hunk.lines)]
253}
254
255func buildRows(hunk parsedHunk, highlighted []string) []RenderedRow {
256 rows := make([]RenderedRow, 0, len(hunk.lines))
257 oldLine, newLine := hunk.oldStart, hunk.newStart
258 for i, l := range hunk.lines {
259 html := ""
260 if i < len(highlighted) {
261 html = highlighted[i]
262 }
263 row := RenderedRow{Type: l.typ, HTML: html, OldLine: NoLine, NewLine: NoLine}
264 switch l.typ {
265 case "context":
266 row.OldLine, row.NewLine = oldLine, newLine
267 oldLine++
268 newLine++
269 case "add":
270 row.NewLine = newLine
271 newLine++
272 default:
273 row.OldLine = oldLine
274 oldLine++
275 }
276 rows = append(rows, row)
277 }
278 return rows
279}
280
281// highlightFile renders one parsed file. Files larger than inlineMaxBytes are
282// rendered without highlighting.
283func (h *Highlighter) highlightFile(file ParsedFile) RenderedDiffFile {
284 displayPath := file.NewPath
285 if displayPath == "" {
286 displayPath = file.OldPath
287 }
288 var totalBytes int64
289 for _, hunk := range file.hunks {
290 for _, l := range hunk.lines {
291 totalBytes += int64(len(l.content))
292 }
293 }
294 lang := ""
295 if totalBytes <= h.inlineMaxBytes {
296 lang = DetectLang(filepath.Base(displayPath))
297 }
298
299 hunks := make([]RenderedHunk, len(file.hunks))
300 for i, hunk := range file.hunks {
301 hunks[i] = RenderedHunk{Header: hunk.header, Rows: buildRows(hunk, highlightHunk(hunk, lang))}
302 }
303 return RenderedDiffFile{
304 OldPath: file.OldPath,
305 NewPath: file.NewPath,
306 Status: file.Status,
307 Added: file.Added,
308 Removed: file.Removed,
309 IsBinary: file.IsBinary,
310 BinaryFrom: file.BinaryOldSize,
311 BinaryTo: file.BinaryNewSize,
312 HasBinarySize: file.IsBinary && file.hasNewSize,
313 Hunks: hunks,
314 }
315}
316
317// PrepareDiff parses and highlights a whole diff, with caching.
318// cacheKey may be empty to skip caching.
319func (h *Highlighter) PrepareDiff(raw, cacheKey string, blobSize func(sha string) int64) []RenderedDiffFile {
320 if v, ok := h.diffs.Get(cacheKey); ok {
321 return v
322 }
323 parsed := parseDiff(raw, blobSize)
324 out := make([]RenderedDiffFile, len(parsed))
325 for i, f := range parsed {
326 out[i] = h.highlightFile(f)
327 }
328 if cacheKey != "" {
329 h.diffs.Set(cacheKey, out)
330 }
331 return out
332}
333