internal/httpd/diff.go
304 lines · 9259 bytes
1package httpd
2
3import (
4 "bytes"
5 "html/template"
6 "regexp"
7 "strconv"
8 "strings"
9
10 "github.com/alecthomas/chroma/v2"
11 "github.com/alecthomas/chroma/v2/formatters/html"
12 "github.com/alecthomas/chroma/v2/lexers"
13 "github.com/alecthomas/chroma/v2/styles"
14)
15
16type diffLine struct {
17 Class string // meta | hunk | add | del | ctx
18 Text string // the raw diff line, marker included
19 Content string // the line without its +/- marker
20 Code template.HTML // Content highlighted; empty when the type is unknown
21 Path string // file this line belongs to
22 NewLine int64 // line number in the new file (0 when absent)
23 OldLine int64 // line number in the old file (0 when absent)
24 Threads []diffThread
25 Compose bool // render the new-thread form under this line
26}
27
28// diffFile is one file's worth of a unified diff: the header lines are
29// consumed into the fields here, so the template renders a section rather
30// than replaying "diff --git" at the reader.
31type diffFile struct {
32 Path string // new path; the old one for a delete
33 OldPath string // set only on a rename
34 Status string // added | deleted | renamed | modified
35 Adds int
36 Dels int
37 Binary bool
38 Lines []diffLine
39 Threads int // threads anchored in this file, so it can stay unfolded
40 Open bool // rendered unfolded: small files, and anything under review
41}
42
43type diffStat struct{ Files, Adds, Dels int }
44
45var hunkPat = regexp.MustCompile(`^@@ -(\d+)(?:,\d+)? \+(\d+)(?:,\d+)? @@`)
46
47// parseDiff splits a unified diff into per-file sections, tracking old and
48// new line numbers so review threads can anchor inline.
49func parseDiff(patch string) []diffFile {
50 var files []diffFile
51 var cur *diffFile
52 var oldN, newN int64
53
54 // Paths arrive both in "diff --git a/x b/y" and in the ---/+++ pair.
55 // The latter is authoritative (it survives quoting oddities), so the
56 // git line only opens the section.
57 start := func() *diffFile {
58 files = append(files, diffFile{Status: "modified"})
59 return &files[len(files)-1]
60 }
61
62 for _, l := range strings.Split(patch, "\n") {
63 switch {
64 case strings.HasPrefix(l, "diff --git "):
65 cur = start()
66 if a, b, ok := gitHeaderPaths(l); ok {
67 cur.OldPath, cur.Path = a, b
68 }
69 continue
70 case cur == nil:
71 continue // preamble before the first file
72 case strings.HasPrefix(l, "new file mode"):
73 cur.Status = "added"
74 continue
75 case strings.HasPrefix(l, "deleted file mode"):
76 cur.Status = "deleted"
77 continue
78 case strings.HasPrefix(l, "rename from "):
79 cur.Status, cur.OldPath = "renamed", strings.TrimPrefix(l, "rename from ")
80 continue
81 case strings.HasPrefix(l, "rename to "):
82 cur.Status, cur.Path = "renamed", strings.TrimPrefix(l, "rename to ")
83 continue
84 case strings.HasPrefix(l, "Binary files "), strings.HasPrefix(l, "GIT binary patch"):
85 cur.Binary = true
86 continue
87 case strings.HasPrefix(l, "--- "):
88 if p := strings.TrimPrefix(l, "--- "); p != "/dev/null" {
89 cur.OldPath = strings.TrimPrefix(p, "a/")
90 }
91 continue
92 case strings.HasPrefix(l, "+++ "):
93 if p := strings.TrimPrefix(l, "+++ "); p != "/dev/null" {
94 cur.Path = strings.TrimPrefix(p, "b/")
95 }
96 continue
97 case strings.HasPrefix(l, "index "), strings.HasPrefix(l, "old mode "),
98 strings.HasPrefix(l, "new mode "), strings.HasPrefix(l, "similarity index "),
99 strings.HasPrefix(l, "dissimilarity index "):
100 continue
101 }
102
103 d := diffLine{Text: l, Content: l, Path: cur.Path}
104 switch {
105 case strings.HasPrefix(l, "@@"):
106 d.Class, d.Path = "hunk", ""
107 if m := hunkPat.FindStringSubmatch(l); m != nil {
108 oldN, _ = strconv.ParseInt(m[1], 10, 64)
109 newN, _ = strconv.ParseInt(m[2], 10, 64)
110 }
111 case strings.HasPrefix(l, "+"):
112 d.Class, d.Content, d.NewLine = "add", l[1:], newN
113 newN++
114 cur.Adds++
115 case strings.HasPrefix(l, "-"):
116 d.Class, d.Content, d.OldLine = "del", l[1:], oldN
117 oldN++
118 cur.Dels++
119 case l == `\ No newline at end of file`:
120 d.Class, d.Path = "meta", ""
121 case l == "":
122 continue // trailing newline from the split
123 default:
124 d.Class, d.Content, d.OldLine, d.NewLine = "ctx", l[1:], oldN, newN
125 oldN++
126 newN++
127 }
128 cur.Lines = append(cur.Lines, d)
129 }
130
131 for i := range files {
132 if files[i].Path == "" {
133 files[i].Path = files[i].OldPath
134 }
135 if files[i].Status == "renamed" && files[i].OldPath == files[i].Path {
136 files[i].Status = "modified"
137 }
138 highlightFile(&files[i])
139 // Big files fold shut so a large diff is navigable; anything
140 // carrying review threads stays open regardless.
141 files[i].Open = len(files[i].Lines) <= 300
142 }
143 return files
144}
145
146// gitHeaderPaths pulls both paths out of a "diff --git a/x b/y" line. Paths
147// with spaces make this ambiguous in general; git quotes those, and the
148// ---/+++ lines correct us either way.
149func gitHeaderPaths(l string) (string, string, bool) {
150 rest := strings.TrimPrefix(l, "diff --git ")
151 i := strings.Index(rest, " b/")
152 if !strings.HasPrefix(rest, "a/") || i < 0 {
153 return "", "", false
154 }
155 return rest[2:i], rest[i+3:], true
156}
157
158// diffFormatter is the blob formatter without line numbers: the diff
159// supplies its own gutters. PreventSurroundingPre also drops chroma's
160// per-line <span class="line"> wrapper, which the generated CSS gives
161// display:flex — inside a diff row that breaks the +/- marker onto a line
162// of its own.
163var diffFormatter = html.New(html.WithClasses(true), html.PreventSurroundingPre(true))
164
165// highlightFile syntax-highlights a file's diff content one hunk at a time,
166// each side separately. A hunk's context+deletions are contiguous lines of
167// the old file and its context+additions are contiguous lines of the new
168// one, so each side lexes as real code — highlighting line by line instead
169// would break every multi-line string and block comment.
170func highlightFile(f *diffFile) {
171 if f.Binary || len(f.Lines) == 0 {
172 return
173 }
174 lexer := lexers.Match(f.Path)
175 if lexer == nil {
176 return // unknown type: plain text reads fine, and guessing is worse
177 }
178 for start := 0; start < len(f.Lines); {
179 if f.Lines[start].Class == "hunk" || f.Lines[start].Class == "meta" {
180 start++
181 continue
182 }
183 end := start
184 for end < len(f.Lines) && f.Lines[end].Class != "hunk" && f.Lines[end].Class != "meta" {
185 end++
186 }
187 hunk := f.Lines[start:end]
188 assign(hunk, "del", highlightLines(lexer, sideText(hunk, "del")))
189 assign(hunk, "add", highlightLines(lexer, sideText(hunk, "add")))
190 start = end
191 }
192}
193
194// sideText joins one side of a hunk: context plus the given change class.
195// Content, not Text: the marker is already off it. Taking it off Text here
196// meant stripping "+" and then "-", which ate the dash of an added line that
197// begins with one.
198func sideText(hunk []diffLine, class string) string {
199 var b strings.Builder
200 for _, l := range hunk {
201 if l.Class == "ctx" || l.Class == class {
202 b.WriteString(l.Content)
203 b.WriteByte('\n')
204 }
205 }
206 return b.String()
207}
208
209// assign hands highlighted lines back to the diff lines they came from.
210// Context lines take whichever side ran last; both sides hold identical
211// text there, so the result is the same either way.
212func assign(hunk []diffLine, class string, out []template.HTML) {
213 i := 0
214 for j := range hunk {
215 if hunk[j].Class != "ctx" && hunk[j].Class != class {
216 continue
217 }
218 if i < len(out) {
219 hunk[j].Code = out[i]
220 }
221 i++
222 }
223}
224
225// highlightLines formats source and splits the result back into lines.
226// chroma emits tokens that may span newlines, so the split happens on the
227// rendered HTML with tags reopened per line.
228func highlightLines(lexer chroma.Lexer, src string) []template.HTML {
229 if src == "" {
230 return nil
231 }
232 it, err := lexer.Tokenise(nil, src)
233 if err != nil {
234 return nil
235 }
236 var buf bytes.Buffer
237 if err := diffFormatter.Format(&buf, styles.Get(lightStyle), it); err != nil {
238 return nil
239 }
240 body := strings.TrimSuffix(buf.String(), "\n")
241
242 var out []template.HTML
243 for _, line := range splitHighlighted(body) {
244 out = append(out, template.HTML(line))
245 }
246 return out
247}
248
249// splitHighlighted breaks formatted HTML on newlines that sit outside a
250// tag, closing and reopening the spans that straddle the break so every
251// line is balanced markup on its own.
252func splitHighlighted(body string) []string {
253 var lines []string
254 var open []string
255 var cur strings.Builder
256 for i := 0; i < len(body); {
257 switch body[i] {
258 case '<':
259 j := strings.IndexByte(body[i:], '>')
260 if j < 0 {
261 cur.WriteString(body[i:])
262 i = len(body)
263 continue
264 }
265 tag := body[i : i+j+1]
266 if strings.HasPrefix(tag, "</") {
267 if len(open) > 0 {
268 open = open[:len(open)-1]
269 }
270 } else if !strings.HasSuffix(tag, "/>") {
271 open = append(open, tag)
272 }
273 cur.WriteString(tag)
274 i += j + 1
275 case '\n':
276 for range open {
277 cur.WriteString("</span>")
278 }
279 lines = append(lines, cur.String())
280 cur.Reset()
281 for _, t := range open {
282 cur.WriteString(t)
283 }
284 i++
285 default:
286 cur.WriteByte(body[i])
287 i++
288 }
289 }
290 if cur.Len() > 0 {
291 lines = append(lines, cur.String())
292 }
293 return lines
294}
295
296// statOf totals a parsed diff for the summary line.
297func statOf(files []diffFile) diffStat {
298 st := diffStat{Files: len(files)}
299 for _, f := range files {
300 st.Adds += f.Adds
301 st.Dels += f.Dels
302 }
303 return st
304}