internal/httpd/math.go

e6cd75b5f28bacf51620bb531320c30fd4e66bfd
gitbay/internal/httpd/math.go history · blame · raw

436 lines · 13097 bytes

41 symbols in this file
  1package httpd
  2
  3import (
  4	"bytes"
  5	"crypto/rand"
  6	"encoding/hex"
  7	"html/template"
  8	"regexp"
  9	"sort"
 10	"strconv"
 11	"strings"
 12
 13	"github.com/microcosm-cc/bluemonday"
 14	"github.com/niklasfasching/go-org/org"
 15	"github.com/yuin/goldmark"
 16	"github.com/yuin/goldmark/ast"
 17	"github.com/yuin/goldmark/parser"
 18	"github.com/yuin/goldmark/renderer"
 19	"github.com/yuin/goldmark/text"
 20	"github.com/yuin/goldmark/util"
 21
 22	"gitbay.org/gitbay/internal/texmath"
 23)
 24
 25// TeX math renders server-side as MathML, which browsers display natively,
 26// so the page needs no script and the CSP does not change (#294).
 27//
 28// The converter is internal/texmath rather than a library. The pure-Go
 29// options were treeblood (MIT), which writes \color and \class arguments
 30// into attributes unescaped, expands \def macros without a bound (a 180-byte
 31// input produced 3 MB), did not finish 5000 nested braces in 30 seconds and
 32// logs to stderr; goldmark-mathml, which runs Temml in a JavaScript VM; and
 33// converters inside large typesetting modules. texmath covers a documented
 34// subset, refuses everything else, and bounds input size and nesting.
 35//
 36// MathML reaches a page only from the converter. ugcPolicy, which cleans
 37// user-authored HTML, admits none of it; mathPolicy admits exactly what
 38// texmath emits and cleans each converted expression.
 39
 40// mathPolicy admits the MathML texmath emits: its elements, and each
 41// attribute only on its element and only with the values it writes.
 42var mathPolicy = func() *bluemonday.Policy {
 43	p := bluemonday.NewPolicy()
 44	p.AllowElements(texmath.Elements...)
 45	p.AllowNoAttrs().OnElements(texmath.Elements...)
 46	for element, attrs := range texmath.Attrs {
 47		for name, value := range attrs {
 48			pattern := `^` + regexp.QuoteMeta(value) + `$`
 49			if value == "<length>" {
 50				pattern = `^-?[0-9]+(\.[0-9]+)?em$`
 51			}
 52			p.AllowAttrs(name).Matching(regexp.MustCompile(pattern)).OnElements(element)
 53		}
 54	}
 55	return p
 56}()
 57
 58// mathHTML renders one expression, or escapes its source when the converter
 59// refuses it.
 60func mathHTML(tex, source string, display bool) (string, bool) {
 61	out, err := texmath.Convert(tex, display)
 62	if err != nil {
 63		return template.HTMLEscapeString(source), false
 64	}
 65	return mathPolicy.Sanitize(out), true
 66}
 67
 68// Per document, math stops rendering after this many expressions or this
 69// much TeX; later delimiters stay literal text.
 70const (
 71	maxMathExprs = 1000
 72	maxMathBytes = 256 << 10
 73)
 74
 75type mathBudget struct{ n, bytes int }
 76
 77func (b *mathBudget) take(size int) bool {
 78	if b.n >= maxMathExprs || b.bytes+size > maxMathBytes {
 79		return false
 80	}
 81	b.n++
 82	b.bytes += size
 83	return true
 84}
 85
 86// Markdown: $…$ inline and $$…$$ display, inline or as a block.
 87
 88var (
 89	kindMath      = ast.NewNodeKind("Math")
 90	kindMathBlock = ast.NewNodeKind("MathBlock")
 91
 92	mathBudgetKey = parser.NewContextKey()
 93	mathLineKey   = parser.NewContextKey()
 94)
 95
 96func budgetFor(pc parser.Context) *mathBudget {
 97	b, _ := pc.Get(mathBudgetKey).(*mathBudget)
 98	if b == nil {
 99		b = &mathBudget{}
100		pc.Set(mathBudgetKey, b)
101	}
102	return b
103}
104
105type mathInline struct {
106	ast.BaseInline
107	tex     string
108	display bool
109}
110
111func (n *mathInline) Kind() ast.NodeKind { return kindMath }
112func (n *mathInline) Dump(src []byte, level int) {
113	ast.DumpHelper(n, src, level, map[string]string{"TeX": n.tex}, nil)
114}
115
116type mathBlock struct {
117	ast.BaseBlock
118	tex    []byte
119	source []byte
120	closed bool
121}
122
123func (n *mathBlock) Kind() ast.NodeKind { return kindMathBlock }
124func (n *mathBlock) Dump(src []byte, level int) {
125	ast.DumpHelper(n, src, level, map[string]string{"TeX": string(n.tex)}, nil)
126}
127
128func isMathSpace(c byte) bool { return c == ' ' || c == '\t' || c == '\n' || c == '\r' }
129
130// mathLine is the valid closing dollars of one line, as source offsets,
131// found in one pass so each opener on the line looks its closer up rather
132// than rescanning the rest of the line.
133type mathLine struct {
134	stop    int
135	closers []int
136}
137
138// closers scans line, which starts at an opening $, for dollars that can
139// close inline math: a non-space before, no digit after, not escaped.
140func closers(line []byte, base int) []int {
141	var out []int
142	for i := 1; i < len(line); i++ {
143		switch line[i] {
144		case '\\':
145			i++
146		case '$':
147			if isMathSpace(line[i-1]) || i+1 < len(line) && line[i+1] >= '0' && line[i+1] <= '9' {
148				continue
149			}
150			out = append(out, base+i)
151		}
152	}
153	return out
154}
155
156// mathInlineParser follows pandoc's rule so prices stay prose: the opening
157// $ has a non-space after it, and the closing $ a non-space before it and no
158// digit after it. "$5 and $10" is text. A backslash escapes the next byte.
159type mathInlineParser struct{}
160
161func (mathInlineParser) Trigger() []byte { return []byte{'$'} }
162
163func (mathInlineParser) Parse(parent ast.Node, block text.Reader, pc parser.Context) ast.Node {
164	line, seg := block.PeekLine()
165	if len(line) >= 2 && line[1] == '$' {
166		// Scanning stops at the next $$, where the next attempt starts,
167		// so each byte is scanned at most twice.
168		body := line[2:]
169		for i := 0; i+1 < len(body); i++ {
170			if body[i] == '\\' {
171				i++
172				continue
173			}
174			if body[i] == '$' && body[i+1] == '$' {
175				if i == 0 || !budgetFor(pc).take(i) {
176					break
177				}
178				block.Advance(i + 4)
179				return &mathInline{tex: string(body[:i]), display: true}
180			}
181		}
182		// Consume both dollars so the second does not open inline math.
183		block.Advance(2)
184		return ast.NewTextSegment(seg.WithStop(seg.Start + 2))
185	}
186	if len(line) < 2 || isMathSpace(line[1]) {
187		return nil
188	}
189	start := seg.Start - seg.Padding
190	cache, _ := pc.Get(mathLineKey).(*mathLine)
191	if cache == nil || cache.stop != seg.Stop {
192		cache = &mathLine{stop: seg.Stop, closers: closers(line, start)}
193		pc.Set(mathLineKey, cache)
194	}
195	k := sort.SearchInts(cache.closers, start+2)
196	if k == len(cache.closers) {
197		return nil
198	}
199	end := cache.closers[k] - start
200	if !budgetFor(pc).take(end - 1) {
201		return nil
202	}
203	block.Advance(end + 1)
204	return &mathInline{tex: string(line[1:end])}
205}
206
207// mathBlockParser opens on a line that is $$ alone, when a line ending in
208// $$ follows before a blank line, or on a line that is $$…$$ whole.
209// Anything else stays paragraph text.
210type mathBlockParser struct{}
211
212func (mathBlockParser) Trigger() []byte { return []byte{'$'} }
213
214func (mathBlockParser) Open(parent ast.Node, reader text.Reader, pc parser.Context) (ast.Node, parser.State) {
215	line, seg := reader.PeekLine()
216	pos := pc.BlockOffset()
217	if pos < 0 || !bytes.HasPrefix(line[pos:], []byte("$$")) {
218		return nil, parser.NoChildren
219	}
220	rest := util.TrimRightSpace(line[pos+2:])
221	n := &mathBlock{}
222	switch {
223	case len(rest) == 0:
224		// The lookahead stops at the first line ending in $$, which the
225		// block then consumes, so no line is looked at twice by it.
226		// It reads the source rather than moving the reader, whose
227		// SetPosition keeps the line it last peeked.
228		size, found := 0, false
229		src := reader.Source()
230		for i := seg.Stop; i < len(src); {
231			end := len(src)
232			if j := bytes.IndexByte(src[i:], '\n'); j >= 0 {
233				end = i + j + 1
234			}
235			next := src[i:end]
236			if util.IsBlank(next) {
237				break
238			}
239			size += len(next)
240			if bytes.HasSuffix(util.TrimRightSpace(next), []byte("$$")) {
241				found = true
242				break
243			}
244			i = end
245		}
246		if !found || !budgetFor(pc).take(size) {
247			return nil, parser.NoChildren
248		}
249		n.source = append(n.source, "$$\n"...)
250	case len(rest) > 2 && bytes.HasSuffix(rest, []byte("$$")):
251		if !budgetFor(pc).take(len(rest) - 2) {
252			return nil, parser.NoChildren
253		}
254		n.tex, n.closed = rest[:len(rest)-2], true
255		n.source = append([]byte("$$"), rest...)
256	default:
257		return nil, parser.NoChildren
258	}
259	reader.AdvanceToEOL()
260	return n, parser.NoChildren
261}
262
263func (mathBlockParser) Continue(node ast.Node, reader text.Reader, pc parser.Context) parser.State {
264	n := node.(*mathBlock)
265	if n.closed {
266		return parser.Close
267	}
268	line, _ := reader.PeekLine()
269	if line == nil || util.IsBlank(line) {
270		return parser.Close
271	}
272	trimmed := util.TrimRightSpace(line)
273	if bytes.HasSuffix(trimmed, []byte("$$")) {
274		n.tex = append(n.tex, trimmed[:len(trimmed)-2]...)
275		n.source = append(n.source, trimmed...)
276		n.closed = true
277		reader.AdvanceToEOL()
278		return parser.Close
279	}
280	n.tex = append(n.tex, line...)
281	n.source = append(n.source, line...)
282	reader.AdvanceToEOL()
283	return parser.Continue | parser.NoChildren
284}
285
286func (mathBlockParser) Close(ast.Node, text.Reader, parser.Context) {}
287func (mathBlockParser) CanInterruptParagraph() bool                 { return true }
288func (mathBlockParser) CanAcceptIndentedLine() bool                 { return false }
289
290type mathRenderer struct{}
291
292func (mathRenderer) RegisterFuncs(reg renderer.NodeRendererFuncRegisterer) {
293	reg.Register(kindMath, func(w util.BufWriter, _ []byte, node ast.Node, entering bool) (ast.WalkStatus, error) {
294		if entering {
295			n := node.(*mathInline)
296			delim := "$"
297			if n.display {
298				delim = "$$"
299			}
300			out, _ := mathHTML(n.tex, delim+n.tex+delim, n.display)
301			_, _ = w.WriteString(out)
302		}
303		return ast.WalkSkipChildren, nil
304	})
305	reg.Register(kindMathBlock, func(w util.BufWriter, _ []byte, node ast.Node, entering bool) (ast.WalkStatus, error) {
306		if !entering {
307			return ast.WalkContinue, nil
308		}
309		n := node.(*mathBlock)
310		source := string(n.source)
311		out, ok := mathHTML(string(n.tex), source, true)
312		if !ok || !n.closed {
313			// A container that ended before the closing line leaves the
314			// block unclosed; show what it held.
315			out = "<pre>" + template.HTMLEscapeString(source) + "</pre>"
316		}
317		_, _ = w.WriteString(out + "\n")
318		return ast.WalkSkipChildren, nil
319	})
320}
321
322type mathExtension struct{}
323
324func (mathExtension) Extend(m goldmark.Markdown) {
325	m.Parser().AddOptions(
326		parser.WithBlockParsers(util.Prioritized(mathBlockParser{}, 701)),
327		parser.WithInlineParsers(util.Prioritized(mathInlineParser{}, 501)))
328	m.Renderer().AddOptions(renderer.WithNodeRenderers(util.Prioritized(mathRenderer{}, 500)))
329}
330
331// Org: go-org already parses $…$, $$…$$, \(…\), \[…\] and \begin{…}…\end{…}
332// as LaTeX fragments, and \begin{…} on its own lines as a LaTeX block; it
333// writes them back out as text. These render them instead.
334//
335// The whole org document goes through ugcPolicy, which admits no MathML, so
336// the writer leaves a placeholder for each expression and fill puts the
337// mathPolicy-cleaned MathML back after sanitizing. The placeholder carries a
338// random per-render prefix, so a document cannot spell one.
339
340type mathSlots struct {
341	prefix string
342	html   []string
343	source []string
344	budget mathBudget
345}
346
347func newMathSlots() *mathSlots {
348	var b [12]byte
349	_, _ = rand.Read(b[:])
350	return &mathSlots{prefix: "gitbaymath" + hex.EncodeToString(b[:]) + "n"}
351}
352
353func (m *mathSlots) put(html, source string) string {
354	m.html = append(m.html, html)
355	m.source = append(m.source, source)
356	return m.prefix + strconv.Itoa(len(m.html)-1) + "z"
357}
358
359// fill replaces each placeholder in sanitized HTML with its MathML, or with
360// its escaped source where the placeholder landed inside a tag (an
361// attribute value). Sanitized output escapes < and > everywhere but in
362// tags, so the last of them seen says whether the text is inside one.
363func (m *mathSlots) fill(doc string) string {
364	if len(m.html) == 0 {
365		return doc
366	}
367	var b strings.Builder
368	inTag := false
369	for {
370		i := strings.Index(doc, m.prefix)
371		if i < 0 {
372			b.WriteString(doc)
373			return b.String()
374		}
375		before := doc[:i]
376		if j := strings.LastIndexAny(before, "<>"); j >= 0 {
377			inTag = before[j] == '<'
378		}
379		b.WriteString(before)
380		doc = doc[i+len(m.prefix):]
381		end := strings.IndexByte(doc, 'z')
382		if end < 0 {
383			b.WriteString(m.prefix)
384			continue
385		}
386		k, err := strconv.Atoi(doc[:end])
387		if err != nil || k < 0 || k >= len(m.html) {
388			b.WriteString(m.prefix)
389			continue
390		}
391		doc = doc[end+1:]
392		if inTag {
393			b.WriteString(template.HTMLEscapeString(m.source[k]))
394		} else {
395			b.WriteString(m.html[k])
396		}
397	}
398}
399
400// writeMath writes an expression's placeholder, or its source as text when
401// the converter refuses it or the document's budget is spent.
402func (w *orgWriter) writeMath(tex, source string, display bool) {
403	if w.math.budget.take(len(tex)) {
404		if out, ok := mathHTML(tex, source, display); ok {
405			w.WriteString(w.math.put(out, source))
406			return
407		}
408	}
409	w.WriteText(org.Text{Content: source, IsRaw: true})
410}
411
412func (w *orgWriter) WriteLatexFragment(l org.LatexFragment) {
413	tex := org.String(l.Content...)
414	source := l.OpeningPair + tex + l.ClosingPair
415	// go-org takes any $…$; org's own rule keeps "$5 and $10" prose.
416	if l.OpeningPair == "$" && (tex == "" || isMathSpace(tex[0]) || isMathSpace(tex[len(tex)-1])) {
417		w.WriteText(org.Text{Content: source, IsRaw: true})
418		return
419	}
420	display := l.OpeningPair != "$" && l.OpeningPair != `\(`
421	if strings.HasPrefix(l.OpeningPair, `\begin{`) {
422		tex = source
423	}
424	w.writeMath(tex, source, display)
425}
426
427func (w *orgWriter) WriteLatexBlock(b org.LatexBlock) {
428	tex := org.String(b.Content...)
429	if w.math.budget.take(len(tex)) {
430		if out, ok := mathHTML(tex, tex, true); ok {
431			w.WriteString(w.math.put(out, tex) + "\n")
432			return
433		}
434	}
435	w.WriteString("<pre>" + template.HTMLEscapeString(tex) + "</pre>\n")
436}