internal/httpd/diff.go

6821a6f76082b1e10ff899ff51021b11695c4ad6
gitbay/internal/httpd/diff.go history · blame · raw

304 lines · 9259 bytes

  1package httpd
  2
  3import (
  4	"bytes"
  5	"html/template"
  6	"regexp"
  7	"strconv"
  8	"strings"
  9
 10	"github.com/alecthomas/chroma/v2"
 11	"github.com/alecthomas/chroma/v2/formatters/html"
 12	"github.com/alecthomas/chroma/v2/lexers"
 13	"github.com/alecthomas/chroma/v2/styles"
 14)
 15
 16type diffLine struct {
 17	Class   string        // meta | hunk | add | del | ctx
 18	Text    string        // the raw diff line, marker included
 19	Content string        // the line without its +/- marker
 20	Code    template.HTML // Content highlighted; empty when the type is unknown
 21	Path    string        // file this line belongs to
 22	NewLine int64         // line number in the new file (0 when absent)
 23	OldLine int64         // line number in the old file (0 when absent)
 24	Threads []diffThread
 25	Compose bool // render the new-thread form under this line
 26}
 27
 28// diffFile is one file's worth of a unified diff: the header lines are
 29// consumed into the fields here, so the template renders a section rather
 30// than replaying "diff --git" at the reader.
 31type diffFile struct {
 32	Path    string // new path; the old one for a delete
 33	OldPath string // set only on a rename
 34	Status  string // added | deleted | renamed | modified
 35	Adds    int
 36	Dels    int
 37	Binary  bool
 38	Lines   []diffLine
 39	Threads int  // threads anchored in this file, so it can stay unfolded
 40	Open    bool // rendered unfolded: small files, and anything under review
 41}
 42
 43type diffStat struct{ Files, Adds, Dels int }
 44
 45var hunkPat = regexp.MustCompile(`^@@ -(\d+)(?:,\d+)? \+(\d+)(?:,\d+)? @@`)
 46
 47// parseDiff splits a unified diff into per-file sections, tracking old and
 48// new line numbers so review threads can anchor inline.
 49func parseDiff(patch string) []diffFile {
 50	var files []diffFile
 51	var cur *diffFile
 52	var oldN, newN int64
 53
 54	// Paths arrive both in "diff --git a/x b/y" and in the ---/+++ pair.
 55	// The latter is authoritative (it survives quoting oddities), so the
 56	// git line only opens the section.
 57	start := func() *diffFile {
 58		files = append(files, diffFile{Status: "modified"})
 59		return &files[len(files)-1]
 60	}
 61
 62	for _, l := range strings.Split(patch, "\n") {
 63		switch {
 64		case strings.HasPrefix(l, "diff --git "):
 65			cur = start()
 66			if a, b, ok := gitHeaderPaths(l); ok {
 67				cur.OldPath, cur.Path = a, b
 68			}
 69			continue
 70		case cur == nil:
 71			continue // preamble before the first file
 72		case strings.HasPrefix(l, "new file mode"):
 73			cur.Status = "added"
 74			continue
 75		case strings.HasPrefix(l, "deleted file mode"):
 76			cur.Status = "deleted"
 77			continue
 78		case strings.HasPrefix(l, "rename from "):
 79			cur.Status, cur.OldPath = "renamed", strings.TrimPrefix(l, "rename from ")
 80			continue
 81		case strings.HasPrefix(l, "rename to "):
 82			cur.Status, cur.Path = "renamed", strings.TrimPrefix(l, "rename to ")
 83			continue
 84		case strings.HasPrefix(l, "Binary files "), strings.HasPrefix(l, "GIT binary patch"):
 85			cur.Binary = true
 86			continue
 87		case strings.HasPrefix(l, "--- "):
 88			if p := strings.TrimPrefix(l, "--- "); p != "/dev/null" {
 89				cur.OldPath = strings.TrimPrefix(p, "a/")
 90			}
 91			continue
 92		case strings.HasPrefix(l, "+++ "):
 93			if p := strings.TrimPrefix(l, "+++ "); p != "/dev/null" {
 94				cur.Path = strings.TrimPrefix(p, "b/")
 95			}
 96			continue
 97		case strings.HasPrefix(l, "index "), strings.HasPrefix(l, "old mode "),
 98			strings.HasPrefix(l, "new mode "), strings.HasPrefix(l, "similarity index "),
 99			strings.HasPrefix(l, "dissimilarity index "):
100			continue
101		}
102
103		d := diffLine{Text: l, Content: l, Path: cur.Path}
104		switch {
105		case strings.HasPrefix(l, "@@"):
106			d.Class, d.Path = "hunk", ""
107			if m := hunkPat.FindStringSubmatch(l); m != nil {
108				oldN, _ = strconv.ParseInt(m[1], 10, 64)
109				newN, _ = strconv.ParseInt(m[2], 10, 64)
110			}
111		case strings.HasPrefix(l, "+"):
112			d.Class, d.Content, d.NewLine = "add", l[1:], newN
113			newN++
114			cur.Adds++
115		case strings.HasPrefix(l, "-"):
116			d.Class, d.Content, d.OldLine = "del", l[1:], oldN
117			oldN++
118			cur.Dels++
119		case l == `\ No newline at end of file`:
120			d.Class, d.Path = "meta", ""
121		case l == "":
122			continue // trailing newline from the split
123		default:
124			d.Class, d.Content, d.OldLine, d.NewLine = "ctx", l[1:], oldN, newN
125			oldN++
126			newN++
127		}
128		cur.Lines = append(cur.Lines, d)
129	}
130
131	for i := range files {
132		if files[i].Path == "" {
133			files[i].Path = files[i].OldPath
134		}
135		if files[i].Status == "renamed" && files[i].OldPath == files[i].Path {
136			files[i].Status = "modified"
137		}
138		highlightFile(&files[i])
139		// Big files fold shut so a large diff is navigable; anything
140		// carrying review threads stays open regardless.
141		files[i].Open = len(files[i].Lines) <= 300
142	}
143	return files
144}
145
146// gitHeaderPaths pulls both paths out of a "diff --git a/x b/y" line. Paths
147// with spaces make this ambiguous in general; git quotes those, and the
148// ---/+++ lines correct us either way.
149func gitHeaderPaths(l string) (string, string, bool) {
150	rest := strings.TrimPrefix(l, "diff --git ")
151	i := strings.Index(rest, " b/")
152	if !strings.HasPrefix(rest, "a/") || i < 0 {
153		return "", "", false
154	}
155	return rest[2:i], rest[i+3:], true
156}
157
158// diffFormatter is the blob formatter without line numbers: the diff
159// supplies its own gutters. PreventSurroundingPre also drops chroma's
160// per-line <span class="line"> wrapper, which the generated CSS gives
161// display:flex — inside a diff row that breaks the +/- marker onto a line
162// of its own.
163var diffFormatter = html.New(html.WithClasses(true), html.PreventSurroundingPre(true))
164
165// highlightFile syntax-highlights a file's diff content one hunk at a time,
166// each side separately. A hunk's context+deletions are contiguous lines of
167// the old file and its context+additions are contiguous lines of the new
168// one, so each side lexes as real code — highlighting line by line instead
169// would break every multi-line string and block comment.
170func highlightFile(f *diffFile) {
171	if f.Binary || len(f.Lines) == 0 {
172		return
173	}
174	lexer := lexers.Match(f.Path)
175	if lexer == nil {
176		return // unknown type: plain text reads fine, and guessing is worse
177	}
178	for start := 0; start < len(f.Lines); {
179		if f.Lines[start].Class == "hunk" || f.Lines[start].Class == "meta" {
180			start++
181			continue
182		}
183		end := start
184		for end < len(f.Lines) && f.Lines[end].Class != "hunk" && f.Lines[end].Class != "meta" {
185			end++
186		}
187		hunk := f.Lines[start:end]
188		assign(hunk, "del", highlightLines(lexer, sideText(hunk, "del")))
189		assign(hunk, "add", highlightLines(lexer, sideText(hunk, "add")))
190		start = end
191	}
192}
193
194// sideText joins one side of a hunk: context plus the given change class.
195// Content, not Text: the marker is already off it. Taking it off Text here
196// meant stripping "+" and then "-", which ate the dash of an added line that
197// begins with one.
198func sideText(hunk []diffLine, class string) string {
199	var b strings.Builder
200	for _, l := range hunk {
201		if l.Class == "ctx" || l.Class == class {
202			b.WriteString(l.Content)
203			b.WriteByte('\n')
204		}
205	}
206	return b.String()
207}
208
209// assign hands highlighted lines back to the diff lines they came from.
210// Context lines take whichever side ran last; both sides hold identical
211// text there, so the result is the same either way.
212func assign(hunk []diffLine, class string, out []template.HTML) {
213	i := 0
214	for j := range hunk {
215		if hunk[j].Class != "ctx" && hunk[j].Class != class {
216			continue
217		}
218		if i < len(out) {
219			hunk[j].Code = out[i]
220		}
221		i++
222	}
223}
224
225// highlightLines formats source and splits the result back into lines.
226// chroma emits tokens that may span newlines, so the split happens on the
227// rendered HTML with tags reopened per line.
228func highlightLines(lexer chroma.Lexer, src string) []template.HTML {
229	if src == "" {
230		return nil
231	}
232	it, err := lexer.Tokenise(nil, src)
233	if err != nil {
234		return nil
235	}
236	var buf bytes.Buffer
237	if err := diffFormatter.Format(&buf, styles.Get(lightStyle), it); err != nil {
238		return nil
239	}
240	body := strings.TrimSuffix(buf.String(), "\n")
241
242	var out []template.HTML
243	for _, line := range splitHighlighted(body) {
244		out = append(out, template.HTML(line))
245	}
246	return out
247}
248
249// splitHighlighted breaks formatted HTML on newlines that sit outside a
250// tag, closing and reopening the spans that straddle the break so every
251// line is balanced markup on its own.
252func splitHighlighted(body string) []string {
253	var lines []string
254	var open []string
255	var cur strings.Builder
256	for i := 0; i < len(body); {
257		switch body[i] {
258		case '<':
259			j := strings.IndexByte(body[i:], '>')
260			if j < 0 {
261				cur.WriteString(body[i:])
262				i = len(body)
263				continue
264			}
265			tag := body[i : i+j+1]
266			if strings.HasPrefix(tag, "</") {
267				if len(open) > 0 {
268					open = open[:len(open)-1]
269				}
270			} else if !strings.HasSuffix(tag, "/>") {
271				open = append(open, tag)
272			}
273			cur.WriteString(tag)
274			i += j + 1
275		case '\n':
276			for range open {
277				cur.WriteString("</span>")
278			}
279			lines = append(lines, cur.String())
280			cur.Reset()
281			for _, t := range open {
282				cur.WriteString(t)
283			}
284			i++
285		default:
286			cur.WriteByte(body[i])
287			i++
288		}
289	}
290	if cur.Len() > 0 {
291		lines = append(lines, cur.String())
292	}
293	return lines
294}
295
296// statOf totals a parsed diff for the summary line.
297func statOf(files []diffFile) diffStat {
298	st := diffStat{Files: len(files)}
299	for _, f := range files {
300		st.Adds += f.Adds
301		st.Dels += f.Dels
302	}
303	return st
304}