internal/httpd/diff.go

456 lines · 13321 bytes

  1package httpd
  2
  3import (
  4	"bytes"
  5	"html/template"
  6	"net/http"
  7	"net/url"
  8	"regexp"
  9	"strconv"
 10	"strings"
 11
 12	"github.com/alecthomas/chroma/v2"
 13	"github.com/alecthomas/chroma/v2/formatters/html"
 14	"github.com/alecthomas/chroma/v2/lexers"
 15	"github.com/alecthomas/chroma/v2/styles"
 16)
 17
 18type diffLine struct {
 19	Class   string        // meta | hunk | add | del | ctx
 20	Text    string        // the raw diff line, marker included
 21	Content string        // the line without its +/- marker
 22	Code    template.HTML // Content highlighted; empty when the type is unknown
 23	Path    string        // file this line belongs to
 24	NewLine int64         // line number in the new file (0 when absent)
 25	OldLine int64         // line number in the old file (0 when absent)
 26	Threads []diffThread
 27	Compose bool // render the new-thread form under this line
 28}
 29
 30// diffFile is one file's worth of a unified diff: the header lines are
 31// consumed into the fields here, so the template renders a section rather
 32// than replaying "diff --git" at the reader.
 33type diffFile struct {
 34	Path    string // new path; the old one for a delete
 35	OldPath string // set only on a rename
 36	Status  string // added | deleted | renamed | modified
 37	Adds    int
 38	Dels    int
 39	Binary  bool
 40	Lines   []diffLine
 41	Rows    []splitRow // the split layout's rows; empty in the unified layout
 42	Threads int        // threads anchored in this file, so it can stay unfolded
 43	Open    bool       // rendered unfolded: small files, and anything under review
 44}
 45
 46type diffStat struct{ Files, Adds, Dels int }
 47
 48var hunkPat = regexp.MustCompile(`^@@ -(\d+)(?:,\d+)? \+(\d+)(?:,\d+)? @@`)
 49
 50// parseDiff splits a unified diff into per-file sections, tracking old and
 51// new line numbers so review threads can anchor inline.
 52func parseDiff(patch string) []diffFile {
 53	var files []diffFile
 54	var cur *diffFile
 55	var oldN, newN int64
 56
 57	// Paths arrive both in "diff --git a/x b/y" and in the ---/+++ pair.
 58	// The latter is authoritative (it survives quoting oddities), so the
 59	// git line only opens the section.
 60	start := func() *diffFile {
 61		files = append(files, diffFile{Status: "modified"})
 62		return &files[len(files)-1]
 63	}
 64
 65	for _, l := range strings.Split(patch, "\n") {
 66		switch {
 67		case strings.HasPrefix(l, "diff --git "):
 68			cur = start()
 69			if a, b, ok := gitHeaderPaths(l); ok {
 70				cur.OldPath, cur.Path = a, b
 71			}
 72			continue
 73		case cur == nil:
 74			continue // preamble before the first file
 75		case strings.HasPrefix(l, "new file mode"):
 76			cur.Status = "added"
 77			continue
 78		case strings.HasPrefix(l, "deleted file mode"):
 79			cur.Status = "deleted"
 80			continue
 81		case strings.HasPrefix(l, "rename from "):
 82			cur.Status, cur.OldPath = "renamed", strings.TrimPrefix(l, "rename from ")
 83			continue
 84		case strings.HasPrefix(l, "rename to "):
 85			cur.Status, cur.Path = "renamed", strings.TrimPrefix(l, "rename to ")
 86			continue
 87		case strings.HasPrefix(l, "Binary files "), strings.HasPrefix(l, "GIT binary patch"):
 88			cur.Binary = true
 89			continue
 90		case strings.HasPrefix(l, "--- "):
 91			if p := strings.TrimPrefix(l, "--- "); p != "/dev/null" {
 92				cur.OldPath = strings.TrimPrefix(p, "a/")
 93			}
 94			continue
 95		case strings.HasPrefix(l, "+++ "):
 96			if p := strings.TrimPrefix(l, "+++ "); p != "/dev/null" {
 97				cur.Path = strings.TrimPrefix(p, "b/")
 98			}
 99			continue
100		case strings.HasPrefix(l, "index "), strings.HasPrefix(l, "old mode "),
101			strings.HasPrefix(l, "new mode "), strings.HasPrefix(l, "similarity index "),
102			strings.HasPrefix(l, "dissimilarity index "):
103			continue
104		}
105
106		d := diffLine{Text: l, Content: l, Path: cur.Path}
107		switch {
108		case strings.HasPrefix(l, "@@"):
109			d.Class, d.Path = "hunk", ""
110			if m := hunkPat.FindStringSubmatch(l); m != nil {
111				oldN, _ = strconv.ParseInt(m[1], 10, 64)
112				newN, _ = strconv.ParseInt(m[2], 10, 64)
113			}
114		case strings.HasPrefix(l, "+"):
115			d.Class, d.Content, d.NewLine = "add", l[1:], newN
116			newN++
117			cur.Adds++
118		case strings.HasPrefix(l, "-"):
119			d.Class, d.Content, d.OldLine = "del", l[1:], oldN
120			oldN++
121			cur.Dels++
122		case l == `\ No newline at end of file`:
123			d.Class, d.Path = "meta", ""
124		case l == "":
125			continue // trailing newline from the split
126		default:
127			d.Class, d.Content, d.OldLine, d.NewLine = "ctx", l[1:], oldN, newN
128			oldN++
129			newN++
130		}
131		cur.Lines = append(cur.Lines, d)
132	}
133
134	for i := range files {
135		if files[i].Path == "" {
136			files[i].Path = files[i].OldPath
137		}
138		if files[i].Status == "renamed" && files[i].OldPath == files[i].Path {
139			files[i].Status = "modified"
140		}
141		highlightFile(&files[i])
142		// Big files fold shut so a large diff is navigable; anything
143		// carrying review threads stays open regardless.
144		files[i].Open = len(files[i].Lines) <= 300
145	}
146	return files
147}
148
149// gitHeaderPaths pulls both paths out of a "diff --git a/x b/y" line. Paths
150// with spaces make this ambiguous in general; git quotes those, and the
151// ---/+++ lines correct us either way.
152func gitHeaderPaths(l string) (string, string, bool) {
153	rest := strings.TrimPrefix(l, "diff --git ")
154	i := strings.Index(rest, " b/")
155	if !strings.HasPrefix(rest, "a/") || i < 0 {
156		return "", "", false
157	}
158	return rest[2:i], rest[i+3:], true
159}
160
161// diffFormatter is the blob formatter without line numbers: the diff
162// supplies its own gutters. PreventSurroundingPre also drops chroma's
163// per-line <span class="line"> wrapper, which the generated CSS gives
164// display:flex — inside a diff row that breaks the +/- marker onto a line
165// of its own.
166var diffFormatter = html.New(html.WithClasses(true), html.PreventSurroundingPre(true))
167
168// highlightFile syntax-highlights a file's diff content one hunk at a time,
169// each side separately. A hunk's context+deletions are contiguous lines of
170// the old file and its context+additions are contiguous lines of the new
171// one, so each side lexes as real code — highlighting line by line instead
172// would break every multi-line string and block comment.
173func highlightFile(f *diffFile) {
174	if f.Binary || len(f.Lines) == 0 {
175		return
176	}
177	lexer := lexers.Match(f.Path)
178	if lexer == nil {
179		return // unknown type: plain text reads fine, and guessing is worse
180	}
181	for start := 0; start < len(f.Lines); {
182		if f.Lines[start].Class == "hunk" || f.Lines[start].Class == "meta" {
183			start++
184			continue
185		}
186		end := start
187		for end < len(f.Lines) && f.Lines[end].Class != "hunk" && f.Lines[end].Class != "meta" {
188			end++
189		}
190		hunk := f.Lines[start:end]
191		assign(hunk, "del", highlightLines(lexer, sideText(hunk, "del")))
192		assign(hunk, "add", highlightLines(lexer, sideText(hunk, "add")))
193		start = end
194	}
195}
196
197// sideText joins one side of a hunk: context plus the given change class.
198// Content, not Text: the marker is already off it. Taking it off Text here
199// meant stripping "+" and then "-", which ate the dash of an added line that
200// begins with one.
201func sideText(hunk []diffLine, class string) string {
202	var b strings.Builder
203	for _, l := range hunk {
204		if l.Class == "ctx" || l.Class == class {
205			b.WriteString(l.Content)
206			b.WriteByte('\n')
207		}
208	}
209	return b.String()
210}
211
212// assign hands highlighted lines back to the diff lines they came from.
213// Context lines take whichever side ran last; both sides hold identical
214// text there, so the result is the same either way.
215func assign(hunk []diffLine, class string, out []template.HTML) {
216	i := 0
217	for j := range hunk {
218		if hunk[j].Class != "ctx" && hunk[j].Class != class {
219			continue
220		}
221		if i < len(out) {
222			hunk[j].Code = out[i]
223		}
224		i++
225	}
226}
227
228// highlightLines formats source and splits the result back into lines.
229// chroma emits tokens that may span newlines, so the split happens on the
230// rendered HTML with tags reopened per line.
231func highlightLines(lexer chroma.Lexer, src string) []template.HTML {
232	if src == "" {
233		return nil
234	}
235	it, err := lexer.Tokenise(nil, src)
236	if err != nil {
237		return nil
238	}
239	var buf bytes.Buffer
240	if err := diffFormatter.Format(&buf, styles.Get(lightStyle), it); err != nil {
241		return nil
242	}
243	body := strings.TrimSuffix(buf.String(), "\n")
244
245	var out []template.HTML
246	for _, line := range splitHighlighted(body) {
247		out = append(out, template.HTML(line))
248	}
249	return out
250}
251
252// splitHighlighted breaks formatted HTML on newlines that sit outside a
253// tag, closing and reopening the spans that straddle the break so every
254// line is balanced markup on its own.
255func splitHighlighted(body string) []string {
256	var lines []string
257	var open []string
258	var cur strings.Builder
259	for i := 0; i < len(body); {
260		switch body[i] {
261		case '<':
262			j := strings.IndexByte(body[i:], '>')
263			if j < 0 {
264				cur.WriteString(body[i:])
265				i = len(body)
266				continue
267			}
268			tag := body[i : i+j+1]
269			if strings.HasPrefix(tag, "</") {
270				if len(open) > 0 {
271					open = open[:len(open)-1]
272				}
273			} else if !strings.HasSuffix(tag, "/>") {
274				open = append(open, tag)
275			}
276			cur.WriteString(tag)
277			i += j + 1
278		case '\n':
279			for range open {
280				cur.WriteString("</span>")
281			}
282			lines = append(lines, cur.String())
283			cur.Reset()
284			for _, t := range open {
285				cur.WriteString(t)
286			}
287			i++
288		default:
289			cur.WriteByte(body[i])
290			i++
291		}
292	}
293	if cur.Len() > 0 {
294		lines = append(lines, cur.String())
295	}
296	return lines
297}
298
299// statOf totals a parsed diff for the summary line.
300func statOf(files []diffFile) diffStat {
301	st := diffStat{Files: len(files)}
302	for _, f := range files {
303		st.Adds += f.Adds
304		st.Dels += f.Dels
305	}
306	return st
307}
308
309// splitRow is one row of the side-by-side layout: a hunk or meta line
310// spanning both columns, or a pair of lines. In a run of deletions
311// followed by additions the two are zipped, and the shorter side is left
312// empty. Old and New point into the file's Lines.
313type splitRow struct {
314	Kind    string // hunk | meta | pair
315	Text    string
316	Old     *diffLine
317	New     *diffLine
318	Threads []diffThread
319	OldNote string    // "\ No newline" marker belonging to the old side
320	NewNote string    // and to the new side
321	Compose *diffLine // the line whose new-thread form opens under this row
322}
323
324// diffLayout is the layout a diff page renders in and the links that
325// switch it.
326type diffLayout struct {
327	Split      bool
328	UnifiedURL string
329	SplitURL   string
330	Carry      string // "split" or "unified" when the request chose it, so links keep it
331}
332
333// splitFiles fills each file's Rows. It runs after threads and compose
334// forms are attached to the lines.
335func splitFiles(files []diffFile) {
336	for f := range files {
337		lines := files[f].Lines
338		var rows []splitRow
339		for i := 0; i < len(lines); {
340			ln := &lines[i]
341			switch ln.Class {
342			case "hunk", "meta":
343				if ln.Class == "meta" && ln.Path == "" && len(rows) > 0 && strings.HasPrefix(ln.Text, `\`) {
344					// a marker after a context line: neither side ends in a newline
345					if last := &rows[len(rows)-1]; last.Old != nil && last.Old == last.New {
346						last.OldNote, last.NewNote = ln.Text, ln.Text
347						i++
348						continue
349					}
350				}
351				rows = append(rows, splitRow{Kind: ln.Class, Text: ln.Text})
352				i++
353			case "ctx":
354				r := splitRow{Kind: "pair", Old: ln, New: ln, Threads: ln.Threads}
355				if ln.Compose {
356					r.Compose = ln
357				}
358				rows = append(rows, r)
359				i++
360			default:
361				var dels, adds []*diffLine
362				marker := func() string {
363					if i < len(lines) && lines[i].Class == "meta" && strings.HasPrefix(lines[i].Text, `\`) {
364						i++
365						return lines[i-1].Text
366					}
367					return ""
368				}
369				var oldNote, newNote string
370				for i < len(lines) && lines[i].Class == "del" {
371					dels = append(dels, &lines[i])
372					i++
373				}
374				if len(dels) > 0 {
375					oldNote = marker()
376				}
377				for i < len(lines) && lines[i].Class == "add" {
378					adds = append(adds, &lines[i])
379					i++
380				}
381				if len(adds) > 0 {
382					newNote = marker()
383				}
384				if len(dels)+len(adds) == 0 {
385					i++ // an unknown class: skip rather than loop
386					continue
387				}
388				first := len(rows)
389				for k := 0; k < len(dels) || k < len(adds); k++ {
390					r := splitRow{Kind: "pair"}
391					for _, l := range []*diffLine{pick(dels, k), pick(adds, k)} {
392						if l == nil {
393							continue
394						}
395						if l.Class == "del" {
396							r.Old = l
397						} else {
398							r.New = l
399						}
400						r.Threads = append(r.Threads, l.Threads...)
401						if l.Compose {
402							r.Compose = l
403						}
404					}
405					rows = append(rows, r)
406				}
407				if len(dels) > 0 {
408					rows[first+len(dels)-1].OldNote = oldNote
409				}
410				if len(adds) > 0 {
411					rows[first+len(adds)-1].NewNote = newNote
412				}
413			}
414		}
415		files[f].Rows = rows
416	}
417}
418
419func pick(s []*diffLine, i int) *diffLine {
420	if i < len(s) {
421		return s[i]
422	}
423	return nil
424}
425
426// diffLayoutFor resolves the layout for a request: ?layout= wins, then the
427// signed-in account's setting, then unified. The two switch links keep
428// every other query parameter.
429func (s *Server) diffLayoutFor(r *http.Request) diffLayout {
430	q := r.URL.Query()
431	l := diffLayout{}
432	switch q.Get("layout") {
433	case "split":
434		l.Split, l.Carry = true, "split"
435	case "unified":
436		l.Carry = "unified"
437	default:
438		if s.cfg.Web.Mode == "accounts" {
439			if u := s.viewer(r); u.ID != 0 {
440				if v, err := s.st.DiffLayout(u.ID); err == nil {
441					l.Split = v == "split"
442				}
443			}
444		}
445	}
446	link := func(v string) string {
447		c := url.Values{}
448		for k, vs := range q {
449			c[k] = vs
450		}
451		c.Set("layout", v)
452		return r.URL.Path + "?" + c.Encode()
453	}
454	l.UnifiedURL, l.SplitURL = link("unified"), link("split")
455	return l
456}