internal/httpd/diff.go

eec36526d2721d98ac0d9446ba4e2975d16b5d8b
gitbay/internal/httpd/diff.go history · blame · raw

305 lines · 9109 bytes

  1package httpd
  2
  3import (
  4	"bytes"
  5	"html/template"
  6	"regexp"
  7	"strconv"
  8	"strings"
  9
 10	"github.com/alecthomas/chroma/v2"
 11	"github.com/alecthomas/chroma/v2/formatters/html"
 12	"github.com/alecthomas/chroma/v2/lexers"
 13	"github.com/alecthomas/chroma/v2/styles"
 14)
 15
 16type diffLine struct {
 17	Class   string        // meta | hunk | add | del | ctx
 18	Text    string        // the raw diff line, marker included
 19	Content string        // the line without its +/- marker
 20	Code    template.HTML // Content highlighted; empty when the type is unknown
 21	Path    string        // file this line belongs to
 22	NewLine int64         // line number in the new file (0 when absent)
 23	OldLine int64         // line number in the old file (0 when absent)
 24	Threads []diffThread
 25}
 26
 27// diffFile is one file's worth of a unified diff: the header lines are
 28// consumed into the fields here, so the template renders a section rather
 29// than replaying "diff --git" at the reader.
 30type diffFile struct {
 31	Path    string // new path; the old one for a delete
 32	OldPath string // set only on a rename
 33	Status  string // added | deleted | renamed | modified
 34	Adds    int
 35	Dels    int
 36	Binary  bool
 37	Lines   []diffLine
 38	Threads int  // threads anchored in this file, so it can stay unfolded
 39	Open    bool // rendered unfolded: small files, and anything under review
 40}
 41
 42type diffStat struct{ Files, Adds, Dels int }
 43
 44var hunkPat = regexp.MustCompile(`^@@ -(\d+)(?:,\d+)? \+(\d+)(?:,\d+)? @@`)
 45
 46// parseDiff splits a unified diff into per-file sections, tracking old and
 47// new line numbers so review threads can anchor inline.
 48func parseDiff(patch string) []diffFile {
 49	var files []diffFile
 50	var cur *diffFile
 51	var oldN, newN int64
 52
 53	// Paths arrive both in "diff --git a/x b/y" and in the ---/+++ pair.
 54	// The latter is authoritative (it survives quoting oddities), so the
 55	// git line only opens the section.
 56	start := func() *diffFile {
 57		files = append(files, diffFile{Status: "modified"})
 58		return &files[len(files)-1]
 59	}
 60
 61	for _, l := range strings.Split(patch, "\n") {
 62		switch {
 63		case strings.HasPrefix(l, "diff --git "):
 64			cur = start()
 65			if a, b, ok := gitHeaderPaths(l); ok {
 66				cur.OldPath, cur.Path = a, b
 67			}
 68			continue
 69		case cur == nil:
 70			continue // preamble before the first file
 71		case strings.HasPrefix(l, "new file mode"):
 72			cur.Status = "added"
 73			continue
 74		case strings.HasPrefix(l, "deleted file mode"):
 75			cur.Status = "deleted"
 76			continue
 77		case strings.HasPrefix(l, "rename from "):
 78			cur.Status, cur.OldPath = "renamed", strings.TrimPrefix(l, "rename from ")
 79			continue
 80		case strings.HasPrefix(l, "rename to "):
 81			cur.Status, cur.Path = "renamed", strings.TrimPrefix(l, "rename to ")
 82			continue
 83		case strings.HasPrefix(l, "Binary files "), strings.HasPrefix(l, "GIT binary patch"):
 84			cur.Binary = true
 85			continue
 86		case strings.HasPrefix(l, "--- "):
 87			if p := strings.TrimPrefix(l, "--- "); p != "/dev/null" {
 88				cur.OldPath = strings.TrimPrefix(p, "a/")
 89			}
 90			continue
 91		case strings.HasPrefix(l, "+++ "):
 92			if p := strings.TrimPrefix(l, "+++ "); p != "/dev/null" {
 93				cur.Path = strings.TrimPrefix(p, "b/")
 94			}
 95			continue
 96		case strings.HasPrefix(l, "index "), strings.HasPrefix(l, "old mode "),
 97			strings.HasPrefix(l, "new mode "), strings.HasPrefix(l, "similarity index "),
 98			strings.HasPrefix(l, "dissimilarity index "):
 99			continue
100		}
101
102		d := diffLine{Text: l, Content: l, Path: cur.Path}
103		switch {
104		case strings.HasPrefix(l, "@@"):
105			d.Class, d.Path = "hunk", ""
106			if m := hunkPat.FindStringSubmatch(l); m != nil {
107				oldN, _ = strconv.ParseInt(m[1], 10, 64)
108				newN, _ = strconv.ParseInt(m[2], 10, 64)
109			}
110		case strings.HasPrefix(l, "+"):
111			d.Class, d.Content, d.NewLine = "add", l[1:], newN
112			newN++
113			cur.Adds++
114		case strings.HasPrefix(l, "-"):
115			d.Class, d.Content, d.OldLine = "del", l[1:], oldN
116			oldN++
117			cur.Dels++
118		case l == `\ No newline at end of file`:
119			d.Class, d.Path = "meta", ""
120		case l == "":
121			continue // trailing newline from the split
122		default:
123			d.Class, d.Content, d.OldLine, d.NewLine = "ctx", l[1:], oldN, newN
124			oldN++
125			newN++
126		}
127		cur.Lines = append(cur.Lines, d)
128	}
129
130	for i := range files {
131		if files[i].Path == "" {
132			files[i].Path = files[i].OldPath
133		}
134		if files[i].Status == "renamed" && files[i].OldPath == files[i].Path {
135			files[i].Status = "modified"
136		}
137		highlightFile(&files[i])
138		// Big files fold shut so a large diff is navigable; anything
139		// carrying review threads stays open regardless.
140		files[i].Open = len(files[i].Lines) <= 300
141	}
142	return files
143}
144
145// gitHeaderPaths pulls both paths out of a "diff --git a/x b/y" line. Paths
146// with spaces make this ambiguous in general; git quotes those, and the
147// ---/+++ lines correct us either way.
148func gitHeaderPaths(l string) (string, string, bool) {
149	rest := strings.TrimPrefix(l, "diff --git ")
150	i := strings.Index(rest, " b/")
151	if !strings.HasPrefix(rest, "a/") || i < 0 {
152		return "", "", false
153	}
154	return rest[2:i], rest[i+3:], true
155}
156
157// diffFormatter is the blob formatter without line numbers: the diff
158// supplies its own gutters.
159var diffFormatter = html.New(html.WithClasses(true))
160
161// highlightFile syntax-highlights a file's diff content one hunk at a time,
162// each side separately. A hunk's context+deletions are contiguous lines of
163// the old file and its context+additions are contiguous lines of the new
164// one, so each side lexes as real code — highlighting line by line instead
165// would break every multi-line string and block comment.
166func highlightFile(f *diffFile) {
167	if f.Binary || len(f.Lines) == 0 {
168		return
169	}
170	lexer := lexers.Match(f.Path)
171	if lexer == nil {
172		return // unknown type: plain text reads fine, and guessing is worse
173	}
174	for start := 0; start < len(f.Lines); {
175		if f.Lines[start].Class == "hunk" || f.Lines[start].Class == "meta" {
176			start++
177			continue
178		}
179		end := start
180		for end < len(f.Lines) && f.Lines[end].Class != "hunk" && f.Lines[end].Class != "meta" {
181			end++
182		}
183		hunk := f.Lines[start:end]
184		assign(hunk, "del", highlightLines(lexer, sideText(hunk, "del")))
185		assign(hunk, "add", highlightLines(lexer, sideText(hunk, "add")))
186		start = end
187	}
188}
189
190// sideText joins one side of a hunk: context plus the given change class.
191func sideText(hunk []diffLine, class string) string {
192	var b strings.Builder
193	for _, l := range hunk {
194		if l.Class == "ctx" || l.Class == class {
195			b.WriteString(strings.TrimPrefix(strings.TrimPrefix(l.Text, "+"), "-"))
196			b.WriteByte('\n')
197		}
198	}
199	return b.String()
200}
201
202// assign hands highlighted lines back to the diff lines they came from.
203// Context lines take whichever side ran last; both sides hold identical
204// text there, so the result is the same either way.
205func assign(hunk []diffLine, class string, out []template.HTML) {
206	i := 0
207	for j := range hunk {
208		if hunk[j].Class != "ctx" && hunk[j].Class != class {
209			continue
210		}
211		if i < len(out) {
212			hunk[j].Code = out[i]
213		}
214		i++
215	}
216}
217
218// highlightLines formats source and splits the result back into lines.
219// chroma emits tokens that may span newlines, so the split happens on the
220// rendered HTML with tags reopened per line.
221func highlightLines(lexer chroma.Lexer, src string) []template.HTML {
222	if src == "" {
223		return nil
224	}
225	it, err := lexer.Tokenise(nil, src)
226	if err != nil {
227		return nil
228	}
229	var buf bytes.Buffer
230	if err := diffFormatter.Format(&buf, styles.Get("friendly"), it); err != nil {
231		return nil
232	}
233	body := buf.String()
234	// Strip the wrapper chroma puts around the whole block.
235	if i := strings.Index(body, "<code"); i >= 0 {
236		if j := strings.IndexByte(body[i:], '>'); j >= 0 {
237			body = body[i+j+1:]
238		}
239	}
240	body = strings.TrimSuffix(strings.TrimSuffix(body, "</pre>"), "</code>")
241	body = strings.TrimSuffix(body, "\n")
242
243	var out []template.HTML
244	for _, line := range splitHighlighted(body) {
245		out = append(out, template.HTML(line))
246	}
247	return out
248}
249
250// splitHighlighted breaks formatted HTML on newlines that sit outside a
251// tag, closing and reopening the spans that straddle the break so every
252// line is balanced markup on its own.
253func splitHighlighted(body string) []string {
254	var lines []string
255	var open []string
256	var cur strings.Builder
257	for i := 0; i < len(body); {
258		switch body[i] {
259		case '<':
260			j := strings.IndexByte(body[i:], '>')
261			if j < 0 {
262				cur.WriteString(body[i:])
263				i = len(body)
264				continue
265			}
266			tag := body[i : i+j+1]
267			if strings.HasPrefix(tag, "</") {
268				if len(open) > 0 {
269					open = open[:len(open)-1]
270				}
271			} else if !strings.HasSuffix(tag, "/>") {
272				open = append(open, tag)
273			}
274			cur.WriteString(tag)
275			i += j + 1
276		case '\n':
277			for range open {
278				cur.WriteString("</span>")
279			}
280			lines = append(lines, cur.String())
281			cur.Reset()
282			for _, t := range open {
283				cur.WriteString(t)
284			}
285			i++
286		default:
287			cur.WriteByte(body[i])
288			i++
289		}
290	}
291	if cur.Len() > 0 {
292		lines = append(lines, cur.String())
293	}
294	return lines
295}
296
297// statOf totals a parsed diff for the summary line.
298func statOf(files []diffFile) diffStat {
299	st := diffStat{Files: len(files)}
300	for _, f := range files {
301		st.Adds += f.Adds
302		st.Dels += f.Dels
303	}
304	return st
305}