internal/httpd/diff.go
305 lines · 9109 bytes
1package httpd
2
3import (
4 "bytes"
5 "html/template"
6 "regexp"
7 "strconv"
8 "strings"
9
10 "github.com/alecthomas/chroma/v2"
11 "github.com/alecthomas/chroma/v2/formatters/html"
12 "github.com/alecthomas/chroma/v2/lexers"
13 "github.com/alecthomas/chroma/v2/styles"
14)
15
16type diffLine struct {
17 Class string // meta | hunk | add | del | ctx
18 Text string // the raw diff line, marker included
19 Content string // the line without its +/- marker
20 Code template.HTML // Content highlighted; empty when the type is unknown
21 Path string // file this line belongs to
22 NewLine int64 // line number in the new file (0 when absent)
23 OldLine int64 // line number in the old file (0 when absent)
24 Threads []diffThread
25}
26
27// diffFile is one file's worth of a unified diff: the header lines are
28// consumed into the fields here, so the template renders a section rather
29// than replaying "diff --git" at the reader.
30type diffFile struct {
31 Path string // new path; the old one for a delete
32 OldPath string // set only on a rename
33 Status string // added | deleted | renamed | modified
34 Adds int
35 Dels int
36 Binary bool
37 Lines []diffLine
38 Threads int // threads anchored in this file, so it can stay unfolded
39 Open bool // rendered unfolded: small files, and anything under review
40}
41
42type diffStat struct{ Files, Adds, Dels int }
43
44var hunkPat = regexp.MustCompile(`^@@ -(\d+)(?:,\d+)? \+(\d+)(?:,\d+)? @@`)
45
46// parseDiff splits a unified diff into per-file sections, tracking old and
47// new line numbers so review threads can anchor inline.
48func parseDiff(patch string) []diffFile {
49 var files []diffFile
50 var cur *diffFile
51 var oldN, newN int64
52
53 // Paths arrive both in "diff --git a/x b/y" and in the ---/+++ pair.
54 // The latter is authoritative (it survives quoting oddities), so the
55 // git line only opens the section.
56 start := func() *diffFile {
57 files = append(files, diffFile{Status: "modified"})
58 return &files[len(files)-1]
59 }
60
61 for _, l := range strings.Split(patch, "\n") {
62 switch {
63 case strings.HasPrefix(l, "diff --git "):
64 cur = start()
65 if a, b, ok := gitHeaderPaths(l); ok {
66 cur.OldPath, cur.Path = a, b
67 }
68 continue
69 case cur == nil:
70 continue // preamble before the first file
71 case strings.HasPrefix(l, "new file mode"):
72 cur.Status = "added"
73 continue
74 case strings.HasPrefix(l, "deleted file mode"):
75 cur.Status = "deleted"
76 continue
77 case strings.HasPrefix(l, "rename from "):
78 cur.Status, cur.OldPath = "renamed", strings.TrimPrefix(l, "rename from ")
79 continue
80 case strings.HasPrefix(l, "rename to "):
81 cur.Status, cur.Path = "renamed", strings.TrimPrefix(l, "rename to ")
82 continue
83 case strings.HasPrefix(l, "Binary files "), strings.HasPrefix(l, "GIT binary patch"):
84 cur.Binary = true
85 continue
86 case strings.HasPrefix(l, "--- "):
87 if p := strings.TrimPrefix(l, "--- "); p != "/dev/null" {
88 cur.OldPath = strings.TrimPrefix(p, "a/")
89 }
90 continue
91 case strings.HasPrefix(l, "+++ "):
92 if p := strings.TrimPrefix(l, "+++ "); p != "/dev/null" {
93 cur.Path = strings.TrimPrefix(p, "b/")
94 }
95 continue
96 case strings.HasPrefix(l, "index "), strings.HasPrefix(l, "old mode "),
97 strings.HasPrefix(l, "new mode "), strings.HasPrefix(l, "similarity index "),
98 strings.HasPrefix(l, "dissimilarity index "):
99 continue
100 }
101
102 d := diffLine{Text: l, Content: l, Path: cur.Path}
103 switch {
104 case strings.HasPrefix(l, "@@"):
105 d.Class, d.Path = "hunk", ""
106 if m := hunkPat.FindStringSubmatch(l); m != nil {
107 oldN, _ = strconv.ParseInt(m[1], 10, 64)
108 newN, _ = strconv.ParseInt(m[2], 10, 64)
109 }
110 case strings.HasPrefix(l, "+"):
111 d.Class, d.Content, d.NewLine = "add", l[1:], newN
112 newN++
113 cur.Adds++
114 case strings.HasPrefix(l, "-"):
115 d.Class, d.Content, d.OldLine = "del", l[1:], oldN
116 oldN++
117 cur.Dels++
118 case l == `\ No newline at end of file`:
119 d.Class, d.Path = "meta", ""
120 case l == "":
121 continue // trailing newline from the split
122 default:
123 d.Class, d.Content, d.OldLine, d.NewLine = "ctx", l[1:], oldN, newN
124 oldN++
125 newN++
126 }
127 cur.Lines = append(cur.Lines, d)
128 }
129
130 for i := range files {
131 if files[i].Path == "" {
132 files[i].Path = files[i].OldPath
133 }
134 if files[i].Status == "renamed" && files[i].OldPath == files[i].Path {
135 files[i].Status = "modified"
136 }
137 highlightFile(&files[i])
138 // Big files fold shut so a large diff is navigable; anything
139 // carrying review threads stays open regardless.
140 files[i].Open = len(files[i].Lines) <= 300
141 }
142 return files
143}
144
145// gitHeaderPaths pulls both paths out of a "diff --git a/x b/y" line. Paths
146// with spaces make this ambiguous in general; git quotes those, and the
147// ---/+++ lines correct us either way.
148func gitHeaderPaths(l string) (string, string, bool) {
149 rest := strings.TrimPrefix(l, "diff --git ")
150 i := strings.Index(rest, " b/")
151 if !strings.HasPrefix(rest, "a/") || i < 0 {
152 return "", "", false
153 }
154 return rest[2:i], rest[i+3:], true
155}
156
157// diffFormatter is the blob formatter without line numbers: the diff
158// supplies its own gutters.
159var diffFormatter = html.New(html.WithClasses(true))
160
161// highlightFile syntax-highlights a file's diff content one hunk at a time,
162// each side separately. A hunk's context+deletions are contiguous lines of
163// the old file and its context+additions are contiguous lines of the new
164// one, so each side lexes as real code — highlighting line by line instead
165// would break every multi-line string and block comment.
166func highlightFile(f *diffFile) {
167 if f.Binary || len(f.Lines) == 0 {
168 return
169 }
170 lexer := lexers.Match(f.Path)
171 if lexer == nil {
172 return // unknown type: plain text reads fine, and guessing is worse
173 }
174 for start := 0; start < len(f.Lines); {
175 if f.Lines[start].Class == "hunk" || f.Lines[start].Class == "meta" {
176 start++
177 continue
178 }
179 end := start
180 for end < len(f.Lines) && f.Lines[end].Class != "hunk" && f.Lines[end].Class != "meta" {
181 end++
182 }
183 hunk := f.Lines[start:end]
184 assign(hunk, "del", highlightLines(lexer, sideText(hunk, "del")))
185 assign(hunk, "add", highlightLines(lexer, sideText(hunk, "add")))
186 start = end
187 }
188}
189
190// sideText joins one side of a hunk: context plus the given change class.
191func sideText(hunk []diffLine, class string) string {
192 var b strings.Builder
193 for _, l := range hunk {
194 if l.Class == "ctx" || l.Class == class {
195 b.WriteString(strings.TrimPrefix(strings.TrimPrefix(l.Text, "+"), "-"))
196 b.WriteByte('\n')
197 }
198 }
199 return b.String()
200}
201
202// assign hands highlighted lines back to the diff lines they came from.
203// Context lines take whichever side ran last; both sides hold identical
204// text there, so the result is the same either way.
205func assign(hunk []diffLine, class string, out []template.HTML) {
206 i := 0
207 for j := range hunk {
208 if hunk[j].Class != "ctx" && hunk[j].Class != class {
209 continue
210 }
211 if i < len(out) {
212 hunk[j].Code = out[i]
213 }
214 i++
215 }
216}
217
218// highlightLines formats source and splits the result back into lines.
219// chroma emits tokens that may span newlines, so the split happens on the
220// rendered HTML with tags reopened per line.
221func highlightLines(lexer chroma.Lexer, src string) []template.HTML {
222 if src == "" {
223 return nil
224 }
225 it, err := lexer.Tokenise(nil, src)
226 if err != nil {
227 return nil
228 }
229 var buf bytes.Buffer
230 if err := diffFormatter.Format(&buf, styles.Get(lightStyle), it); err != nil {
231 return nil
232 }
233 body := buf.String()
234 // Strip the wrapper chroma puts around the whole block.
235 if i := strings.Index(body, "<code"); i >= 0 {
236 if j := strings.IndexByte(body[i:], '>'); j >= 0 {
237 body = body[i+j+1:]
238 }
239 }
240 body = strings.TrimSuffix(strings.TrimSuffix(body, "</pre>"), "</code>")
241 body = strings.TrimSuffix(body, "\n")
242
243 var out []template.HTML
244 for _, line := range splitHighlighted(body) {
245 out = append(out, template.HTML(line))
246 }
247 return out
248}
249
250// splitHighlighted breaks formatted HTML on newlines that sit outside a
251// tag, closing and reopening the spans that straddle the break so every
252// line is balanced markup on its own.
253func splitHighlighted(body string) []string {
254 var lines []string
255 var open []string
256 var cur strings.Builder
257 for i := 0; i < len(body); {
258 switch body[i] {
259 case '<':
260 j := strings.IndexByte(body[i:], '>')
261 if j < 0 {
262 cur.WriteString(body[i:])
263 i = len(body)
264 continue
265 }
266 tag := body[i : i+j+1]
267 if strings.HasPrefix(tag, "</") {
268 if len(open) > 0 {
269 open = open[:len(open)-1]
270 }
271 } else if !strings.HasSuffix(tag, "/>") {
272 open = append(open, tag)
273 }
274 cur.WriteString(tag)
275 i += j + 1
276 case '\n':
277 for range open {
278 cur.WriteString("</span>")
279 }
280 lines = append(lines, cur.String())
281 cur.Reset()
282 for _, t := range open {
283 cur.WriteString(t)
284 }
285 i++
286 default:
287 cur.WriteByte(body[i])
288 i++
289 }
290 }
291 if cur.Len() > 0 {
292 lines = append(lines, cur.String())
293 }
294 return lines
295}
296
297// statOf totals a parsed diff for the summary line.
298func statOf(files []diffFile) diffStat {
299 st := diffStat{Files: len(files)}
300 for _, f := range files {
301 st.Adds += f.Adds
302 st.Dels += f.Dels
303 }
304 return st
305}