diff.go
⎇
Raw
1package highlight
2
3import (
4 "fmt"
5 "path/filepath"
6 "regexp"
7 "strconv"
8 "strings"
9)
10
11// DiffStatus is how git changed a file in a diff.
12type DiffStatus string
13
14const (
15 StatusAdded DiffStatus = "added"
16 StatusDeleted DiffStatus = "deleted"
17 StatusModified DiffStatus = "modified"
18 StatusRenamed DiffStatus = "renamed"
19 StatusCopied DiffStatus = "copied"
20)
21
22// NoLine marks a diff row that has no number on one side.
23const NoLine = -1
24
25// RenderedRow is one line of a hunk with its highlighted HTML.
26// OldLine and NewLine are NoLine when the row has no number on that side.
27// A real number can be 0, because a hunk header may start at 0.
28type RenderedRow struct {
29 Type string // add | del | context
30 OldLine int
31 NewLine int
32 HTML string
33}
34
35// RenderedHunk is one @@ block.
36type RenderedHunk struct {
37 Header string
38 Rows []RenderedRow
39}
40
41// RenderedDiffFile is one file of a diff, ready for the view.
42type RenderedDiffFile struct {
43 OldPath string
44 NewPath string
45 Status DiffStatus
46 Added int
47 Removed int
48 IsBinary bool
49 BinaryFrom int64 // byte size before, only for binary files
50 BinaryTo int64 // byte size after, only for binary files
51 HasBinarySize bool
52 Hunks []RenderedHunk
53}
54
55type parsedLine struct {
56 typ string
57 content string
58}
59
60type parsedHunk struct {
61 header string
62 oldStart int
63 newStart int
64 lines []parsedLine
65}
66
67// ParsedFile is the raw parse result, before highlighting.
68type ParsedFile struct {
69 OldPath string
70 NewPath string
71 Status DiffStatus
72 Added int
73 Removed int
74 IsBinary bool
75 BinaryOldSize int64
76 BinaryNewSize int64
77 hasOldSize bool
78 hasNewSize bool
79 oldBlob string // set for "Binary files ..." lines, which carry no size
80 newBlob string
81 hunks []parsedHunk
82}
83
84var (
85 diffGitRE = regexp.MustCompile(`^diff --git a/(.+) b/(.+)$`)
86 indexRE = regexp.MustCompile(`(?i)^index ([0-9a-f]+)\.\.([0-9a-f]+)`)
87 hunkRE = regexp.MustCompile(`@@ -(\d+)(?:,\d+)? \+(\d+)(?:,\d+)? @@`)
88)
89
90// unquotePath undoes git's C-style quoting of paths with non-ASCII or
91// special bytes. Go's string escapes are a superset of git's.
92func unquotePath(s string) string {
93 if strings.HasPrefix(s, `"`) {
94 if u, err := strconv.Unquote(s); err == nil {
95 return u
96 }
97 }
98 return s
99}
100
101// trimDiffPath strips the "--- " or "+++ " marker, the quoting and the "a/"
102// or "b/" path prefix. It never slices, so a short or truncated line is safe.
103// Git appends a TAB to names that contain a space.
104func trimDiffPath(line, marker, prefix string) string {
105 return strings.TrimPrefix(unquotePath(strings.TrimSuffix(strings.TrimPrefix(line, marker), "\t")), prefix)
106}
107
108// diffGitNewPath returns the "b/" path of a "diff --git" line, or "".
109func diffGitNewPath(line string) string {
110 if strings.HasSuffix(line, `"`) {
111 // A quoted name escapes every inner quote, so ` "` starts the last token.
112 if i := strings.LastIndex(line, ` "`); i >= 0 {
113 return strings.TrimPrefix(unquotePath(line[i+1:]), "b/")
114 }
115 }
116 if m := diffGitRE.FindStringSubmatch(line); m != nil {
117 return m[2]
118 }
119 return ""
120}
121
122// parseDiff parses `git diff` output into per-file structures.
123//
124// blobSizes resolves blob SHAs to byte sizes in one batch, for "Binary files
125// ..." lines that carry no size. It may be nil.
126func parseDiff(raw string, blobSizes func(shas []string) map[string]int64) []ParsedFile {
127 var files []ParsedFile
128 all := strings.Split(raw, "\n")
129 i := 0
130
131 for i < len(all) {
132 if !strings.HasPrefix(all[i], "diff --git ") {
133 i++
134 continue
135 }
136 fallback := diffGitNewPath(all[i])
137 file := ParsedFile{OldPath: fallback, NewPath: fallback, Status: StatusModified}
138 i++
139
140 oldBlob, newBlob := "", ""
141 for i < len(all) {
142 line := all[i]
143 if strings.HasPrefix(line, "diff --git ") || strings.HasPrefix(line, "@@ ") {
144 break
145 }
146 switch {
147 case strings.HasPrefix(line, "new file"):
148 file.Status = StatusAdded
149 case strings.HasPrefix(line, "deleted file"):
150 file.Status = StatusDeleted
151 case strings.HasPrefix(line, "rename from "):
152 file.Status = StatusRenamed
153 file.OldPath = unquotePath(line[12:])
154 case strings.HasPrefix(line, "rename to "):
155 file.NewPath = unquotePath(line[10:])
156 case strings.HasPrefix(line, "copy from "):
157 file.Status = StatusCopied
158 file.OldPath = unquotePath(line[10:])
159 case strings.HasPrefix(line, "copy to "):
160 file.NewPath = unquotePath(line[8:])
161 case strings.HasPrefix(line, "--- ") && line != "--- /dev/null":
162 file.OldPath = trimDiffPath(line, "--- ", "a/")
163 case strings.HasPrefix(line, "+++ ") && line != "+++ /dev/null":
164 file.NewPath = trimDiffPath(line, "+++ ", "b/")
165 case strings.HasPrefix(line, "index "):
166 if m := indexRE.FindStringSubmatch(line); m != nil {
167 oldBlob, newBlob = m[1], m[2]
168 }
169 case strings.HasPrefix(line, "Binary files "):
170 file.IsBinary = true
171 file.oldBlob, file.newBlob = oldBlob, newBlob
172 case line == "GIT binary patch":
173 file.IsBinary = true
174 case file.IsBinary && strings.HasPrefix(line, "literal "):
175 size, _ := strconv.ParseInt(strings.TrimSpace(line[8:]), 10, 64)
176 if !file.hasNewSize {
177 file.BinaryNewSize, file.hasNewSize = size, true
178 } else if !file.hasOldSize {
179 file.BinaryOldSize, file.hasOldSize = size, true
180 }
181 }
182 i++
183 }
184
185 for i < len(all) && strings.HasPrefix(all[i], "@@ ") {
186 hunk := parsedHunk{header: all[i], oldStart: 1, newStart: 1}
187 if m := hunkRE.FindStringSubmatch(all[i]); m != nil {
188 hunk.oldStart, _ = strconv.Atoi(m[1])
189 hunk.newStart, _ = strconv.Atoi(m[2])
190 }
191 i++
192 for i < len(all) && !strings.HasPrefix(all[i], "@@ ") && !strings.HasPrefix(all[i], "diff --git ") {
193 l := all[i]
194 switch {
195 case strings.HasPrefix(l, "+"):
196 hunk.lines = append(hunk.lines, parsedLine{"add", l[1:]})
197 file.Added++
198 case strings.HasPrefix(l, "-"):
199 hunk.lines = append(hunk.lines, parsedLine{"del", l[1:]})
200 file.Removed++
201 case strings.HasPrefix(l, " "):
202 hunk.lines = append(hunk.lines, parsedLine{"context", l[1:]})
203 }
204 // "\ No newline at end of file" is skipped.
205 i++
206 }
207 file.hunks = append(file.hunks, hunk)
208 }
209 files = append(files, file)
210 }
211 if blobSizes != nil {
212 fillBlobSizes(files, blobSizes)
213 }
214 return files
215}
216
217// fillBlobSizes sets the sizes of "Binary files ..." entries. A blob that
218// does not resolve, like the all-zero id of an added file, counts as 0.
219func fillBlobSizes(files []ParsedFile, blobSizes func(shas []string) map[string]int64) {
220 var shas []string
221 for _, f := range files {
222 for _, sha := range []string{f.oldBlob, f.newBlob} {
223 if strings.Trim(sha, "0") != "" {
224 shas = append(shas, sha)
225 }
226 }
227 }
228 if len(shas) == 0 {
229 return
230 }
231 sizes := blobSizes(shas)
232 for i := range files {
233 f := &files[i]
234 if f.oldBlob == "" && f.newBlob == "" {
235 continue
236 }
237 f.BinaryOldSize, f.hasOldSize = sizes[f.oldBlob], true
238 f.BinaryNewSize, f.hasNewSize = sizes[f.newBlob], true
239 }
240}
241
242// escapeHTMLAndCtrl escapes HTML and renders C0 control characters as caret
243// notation in a visible span. TAB, LF and DEL are left alone.
244func escapeHTMLAndCtrl(s string) string {
245 var b strings.Builder
246 for _, r := range s {
247 switch {
248 case r == '&':
249 b.WriteString("&amp;")
250 case r == '<':
251 b.WriteString("&lt;")
252 case r == '>':
253 b.WriteString("&gt;")
254 case r < 0x20 && r != '\t' && r != '\n':
255 fmt.Fprintf(&b, `<span class="diff-ctrl">^%c</span>`, byte(r)+64)
256 default:
257 b.WriteRune(r)
258 }
259 }
260 return b.String()
261}
262
263// highlightHunk returns the HTML for every line of a hunk.
264//
265// Trailing CR is stripped before joining, then re-appended as a ^M marker.
266// A lexer would otherwise treat the CR as ordinary whitespace inside a token.
267func highlightHunk(hunk parsedHunk, lang string) []string {
268 if len(hunk.lines) == 0 {
269 return nil
270 }
271 stripped := make([]string, len(hunk.lines))
272 trailingCR := make([]bool, len(hunk.lines))
273 for i, l := range hunk.lines {
274 trailingCR[i] = strings.HasSuffix(l.content, "\r")
275 stripped[i] = strings.TrimSuffix(l.content, "\r")
276 }
277 // Chroma's lexer rewrites a lone CR to LF, which would split one source
278 // line into two and misalign every later line. Such lines are rare, so the
279 // whole hunk falls back to plain escaped text.
280 innerCR := false
281 for _, s := range stripped {
282 if strings.Contains(s, "\r") {
283 innerCR = true
284 break
285 }
286 }
287 var out []string
288 if !innerCR {
289 out = highlightLines(strings.Join(stripped, "\n"), lang, escapeHTMLAndCtrl)
290 }
291 for len(out) < len(hunk.lines) {
292 out = append(out, "")
293 }
294 for i := range hunk.lines {
295 if out[i] == "" && stripped[i] != "" {
296 out[i] = escapeHTMLAndCtrl(stripped[i])
297 }
298 if trailingCR[i] {
299 out[i] += `<span class="diff-ctrl">^M</span>`
300 }
301 }
302 return out[:len(hunk.lines)]
303}
304
305func buildRows(hunk parsedHunk, highlighted []string) []RenderedRow {
306 rows := make([]RenderedRow, 0, len(hunk.lines))
307 oldLine, newLine := hunk.oldStart, hunk.newStart
308 for i, l := range hunk.lines {
309 html := ""
310 if i < len(highlighted) {
311 html = highlighted[i]
312 }
313 row := RenderedRow{Type: l.typ, HTML: html, OldLine: NoLine, NewLine: NoLine}
314 switch l.typ {
315 case "context":
316 row.OldLine, row.NewLine = oldLine, newLine
317 oldLine++
318 newLine++
319 case "add":
320 row.NewLine = newLine
321 newLine++
322 default:
323 row.OldLine = oldLine
324 oldLine++
325 }
326 rows = append(rows, row)
327 }
328 return rows
329}
330
331// highlightFile renders one parsed file. Files larger than inlineMaxBytes are
332// rendered without highlighting.
333func (h *Highlighter) highlightFile(file ParsedFile) RenderedDiffFile {
334 displayPath := file.NewPath
335 if displayPath == "" {
336 displayPath = file.OldPath
337 }
338 var totalBytes int64
339 for _, hunk := range file.hunks {
340 for _, l := range hunk.lines {
341 totalBytes += int64(len(l.content))
342 }
343 }
344 lang := ""
345 if totalBytes <= h.inlineMaxBytes {
346 lang = DetectLang(filepath.Base(displayPath))
347 }
348
349 hunks := make([]RenderedHunk, len(file.hunks))
350 for i, hunk := range file.hunks {
351 hunks[i] = RenderedHunk{Header: hunk.header, Rows: buildRows(hunk, highlightHunk(hunk, lang))}
352 }
353 return RenderedDiffFile{
354 OldPath: file.OldPath,
355 NewPath: file.NewPath,
356 Status: file.Status,
357 Added: file.Added,
358 Removed: file.Removed,
359 IsBinary: file.IsBinary,
360 BinaryFrom: file.BinaryOldSize,
361 BinaryTo: file.BinaryNewSize,
362 HasBinarySize: file.IsBinary && file.hasNewSize,
363 Hunks: hunks,
364 }
365}
366
367// PrepareDiff parses and highlights a whole diff, with caching.
368// cacheKey may be empty to skip caching.
369func (h *Highlighter) PrepareDiff(raw, cacheKey string, blobSizes func(shas []string) map[string]int64) []RenderedDiffFile {
370 if v, ok := h.diffs.Get(cacheKey); ok {
371 return v
372 }
373 parsed := parseDiff(raw, blobSizes)
374 out := make([]RenderedDiffFile, len(parsed))
375 for i, f := range parsed {
376 out[i] = h.highlightFile(f)
377 }
378 if cacheKey != "" {
379 h.diffs.Set(cacheKey, out)
380 }
381 return out
382}
383