| @@ -2,8 +2,12 @@ package httpd |
| 2 | 2 | |
| 3 | 3 | import ( |
| 4 | 4 | "bytes" |
| 5 | "crypto/rand" |
| 6 | "encoding/hex" |
| 5 | 7 | "html/template" |
| 6 | 8 | "regexp" |
| 9 | "sort" |
| 10 | "strconv" |
| 7 | 11 | "strings" |
| 8 | 12 | |
| 9 | 13 | "github.com/microcosm-cc/bluemonday" |
| @@ -28,21 +32,15 @@ import ( |
| 28 | 32 | // logs to stderr; goldmark-mathml, which runs Temml in a JavaScript VM; and |
| 29 | 33 | // converters inside large typesetting modules. texmath covers a documented |
| 30 | 34 | // subset, refuses everything else, and bounds input size and nesting. |
| 35 | // |
| 36 | // MathML reaches a page only from the converter. ugcPolicy, which cleans |
| 37 | // user-authored HTML, admits none of it; mathPolicy admits exactly what |
| 38 | // texmath emits and cleans each converted expression. |
| 31 | 39 | |
| 32 | | // mathHTML renders one expression, or escapes its source when the converter |
| 33 | | // refuses it. The result passes through ugcPolicy even on the markdown path, |
| 34 | | // so the policy is the one statement of what math may emit. |
| 35 | | func mathHTML(tex, source string, display bool) (string, bool) { |
| 36 | | out, err := texmath.Convert(tex, display) |
| 37 | | if err != nil { |
| 38 | | return template.HTMLEscapeString(source), false |
| 39 | | } |
| 40 | | return ugcPolicy.Sanitize(out), true |
| 41 | | } |
| 42 | | |
| 43 | | // allowMath admits exactly the MathML texmath emits: its elements, and each |
| 40 | // mathPolicy admits the MathML texmath emits: its elements, and each |
| 44 | 41 | // attribute only on its element and only with the values it writes. |
| 45 | | func allowMath(p *bluemonday.Policy) { |
| 42 | var mathPolicy = func() *bluemonday.Policy { |
| 43 | p := bluemonday.NewPolicy() |
| 46 | 44 | p.AllowElements(texmath.Elements...) |
| 47 | 45 | p.AllowNoAttrs().OnElements(texmath.Elements...) |
| 48 | 46 | for element, attrs := range texmath.Attrs { |
| @@ -54,6 +52,35 @@ func allowMath(p *bluemonday.Policy) { |
| 54 | 52 | p.AllowAttrs(name).Matching(regexp.MustCompile(pattern)).OnElements(element) |
| 55 | 53 | } |
| 56 | 54 | } |
| 55 | return p |
| 56 | }() |
| 57 | |
| 58 | // mathHTML renders one expression, or escapes its source when the converter |
| 59 | // refuses it. |
| 60 | func mathHTML(tex, source string, display bool) (string, bool) { |
| 61 | out, err := texmath.Convert(tex, display) |
| 62 | if err != nil { |
| 63 | return template.HTMLEscapeString(source), false |
| 64 | } |
| 65 | return mathPolicy.Sanitize(out), true |
| 66 | } |
| 67 | |
| 68 | // Per document, math stops rendering after this many expressions or this |
| 69 | // much TeX; later delimiters stay literal text. |
| 70 | const ( |
| 71 | maxMathExprs = 1000 |
| 72 | maxMathBytes = 256 << 10 |
| 73 | ) |
| 74 | |
| 75 | type mathBudget struct{ n, bytes int } |
| 76 | |
| 77 | func (b *mathBudget) take(size int) bool { |
| 78 | if b.n >= maxMathExprs || b.bytes+size > maxMathBytes { |
| 79 | return false |
| 80 | } |
| 81 | b.n++ |
| 82 | b.bytes += size |
| 83 | return true |
| 57 | 84 | } |
| 58 | 85 | |
| 59 | 86 | // Markdown: $…$ inline and $$…$$ display, inline or as a block. |
| @@ -61,8 +88,20 @@ func allowMath(p *bluemonday.Policy) { |
| 61 | 88 | var ( |
| 62 | 89 | kindMath = ast.NewNodeKind("Math") |
| 63 | 90 | kindMathBlock = ast.NewNodeKind("MathBlock") |
| 91 | |
| 92 | mathBudgetKey = parser.NewContextKey() |
| 93 | mathLineKey = parser.NewContextKey() |
| 64 | 94 | ) |
| 65 | 95 | |
| 96 | func budgetFor(pc parser.Context) *mathBudget { |
| 97 | b, _ := pc.Get(mathBudgetKey).(*mathBudget) |
| 98 | if b == nil { |
| 99 | b = &mathBudget{} |
| 100 | pc.Set(mathBudgetKey, b) |
| 101 | } |
| 102 | return b |
| 103 | } |
| 104 | |
| 66 | 105 | type mathInline struct { |
| 67 | 106 | ast.BaseInline |
| 68 | 107 | tex string |
| @@ -77,6 +116,7 @@ func (n *mathInline) Dump(src []byte, level int) { |
| 77 | 116 | type mathBlock struct { |
| 78 | 117 | ast.BaseBlock |
| 79 | 118 | tex []byte |
| 119 | source []byte |
| 80 | 120 | closed bool |
| 81 | 121 | } |
| 82 | 122 | |
| @@ -87,6 +127,32 @@ func (n *mathBlock) Dump(src []byte, level int) { |
| 87 | 127 | |
| 88 | 128 | func isMathSpace(c byte) bool { return c == ' ' || c == '\t' || c == '\n' || c == '\r' } |
| 89 | 129 | |
| 130 | // mathLine is the valid closing dollars of one line, as source offsets, |
| 131 | // found in one pass so each opener on the line looks its closer up rather |
| 132 | // than rescanning the rest of the line. |
| 133 | type mathLine struct { |
| 134 | stop int |
| 135 | closers []int |
| 136 | } |
| 137 | |
| 138 | // closers scans line, which starts at an opening $, for dollars that can |
| 139 | // close inline math: a non-space before, no digit after, not escaped. |
| 140 | func closers(line []byte, base int) []int { |
| 141 | var out []int |
| 142 | for i := 1; i < len(line); i++ { |
| 143 | switch line[i] { |
| 144 | case '\\': |
| 145 | i++ |
| 146 | case '$': |
| 147 | if isMathSpace(line[i-1]) || i+1 < len(line) && line[i+1] >= '0' && line[i+1] <= '9' { |
| 148 | continue |
| 149 | } |
| 150 | out = append(out, base+i) |
| 151 | } |
| 152 | } |
| 153 | return out |
| 154 | } |
| 155 | |
| 90 | 156 | // mathInlineParser follows pandoc's rule so prices stay prose: the opening |
| 91 | 157 | // $ has a non-space after it, and the closing $ a non-space before it and no |
| 92 | 158 | // digit after it. "$5 and $10" is text. A backslash escapes the next byte. |
| @@ -97,6 +163,8 @@ func (mathInlineParser) Trigger() []byte { return []byte{'$'} } |
| 97 | 163 | func (mathInlineParser) Parse(parent ast.Node, block text.Reader, pc parser.Context) ast.Node { |
| 98 | 164 | line, seg := block.PeekLine() |
| 99 | 165 | if len(line) >= 2 && line[1] == '$' { |
| 166 | // Scanning stops at the next $$, where the next attempt starts, |
| 167 | // so each byte is scanned at most twice. |
| 100 | 168 | body := line[2:] |
| 101 | 169 | for i := 0; i+1 < len(body); i++ { |
| 102 | 170 | if body[i] == '\\' { |
| @@ -104,7 +172,7 @@ func (mathInlineParser) Parse(parent ast.Node, block text.Reader, pc parser.Cont |
| 104 | 172 | continue |
| 105 | 173 | } |
| 106 | 174 | if body[i] == '$' && body[i+1] == '$' { |
| 107 | | if i == 0 { |
| 175 | if i == 0 || !budgetFor(pc).take(i) { |
| 108 | 176 | break |
| 109 | 177 | } |
| 110 | 178 | block.Advance(i + 4) |
| @@ -115,33 +183,36 @@ func (mathInlineParser) Parse(parent ast.Node, block text.Reader, pc parser.Cont |
| 115 | 183 | block.Advance(2) |
| 116 | 184 | return ast.NewTextSegment(seg.WithStop(seg.Start + 2)) |
| 117 | 185 | } |
| 118 | | body := line[1:] |
| 119 | | if len(body) == 0 || isMathSpace(body[0]) { |
| 186 | if len(line) < 2 || isMathSpace(line[1]) { |
| 120 | 187 | return nil |
| 121 | 188 | } |
| 122 | | for i := 0; i < len(body); i++ { |
| 123 | | switch body[i] { |
| 124 | | case '\\': |
| 125 | | i++ |
| 126 | | case '$': |
| 127 | | if isMathSpace(body[i-1]) || i+1 < len(body) && body[i+1] >= '0' && body[i+1] <= '9' { |
| 128 | | continue |
| 129 | | } |
| 130 | | block.Advance(i + 2) |
| 131 | | return &mathInline{tex: string(body[:i])} |
| 132 | | } |
| 189 | start := seg.Start - seg.Padding |
| 190 | cache, _ := pc.Get(mathLineKey).(*mathLine) |
| 191 | if cache == nil || cache.stop != seg.Stop { |
| 192 | cache = &mathLine{stop: seg.Stop, closers: closers(line, start)} |
| 193 | pc.Set(mathLineKey, cache) |
| 194 | } |
| 195 | k := sort.SearchInts(cache.closers, start+2) |
| 196 | if k == len(cache.closers) { |
| 197 | return nil |
| 133 | 198 | } |
| 134 | | return nil |
| 199 | end := cache.closers[k] - start |
| 200 | if !budgetFor(pc).take(end - 1) { |
| 201 | return nil |
| 202 | } |
| 203 | block.Advance(end + 1) |
| 204 | return &mathInline{tex: string(line[1:end])} |
| 135 | 205 | } |
| 136 | 206 | |
| 137 | | // mathBlockParser opens on a line that is $$ alone, or $$…$$ whole, and |
| 138 | | // runs to the line that ends with $$. |
| 207 | // mathBlockParser opens on a line that is $$ alone, when a line ending in |
| 208 | // $$ follows before a blank line, or on a line that is $$…$$ whole. |
| 209 | // Anything else stays paragraph text. |
| 139 | 210 | type mathBlockParser struct{} |
| 140 | 211 | |
| 141 | 212 | func (mathBlockParser) Trigger() []byte { return []byte{'$'} } |
| 142 | 213 | |
| 143 | 214 | func (mathBlockParser) Open(parent ast.Node, reader text.Reader, pc parser.Context) (ast.Node, parser.State) { |
| 144 | | line, _ := reader.PeekLine() |
| 215 | line, seg := reader.PeekLine() |
| 145 | 216 | pos := pc.BlockOffset() |
| 146 | 217 | if pos < 0 || !bytes.HasPrefix(line[pos:], []byte("$$")) { |
| 147 | 218 | return nil, parser.NoChildren |
| @@ -150,8 +221,38 @@ func (mathBlockParser) Open(parent ast.Node, reader text.Reader, pc parser.Conte |
| 150 | 221 | n := &mathBlock{} |
| 151 | 222 | switch { |
| 152 | 223 | case len(rest) == 0: |
| 224 | // The lookahead stops at the first line ending in $$, which the |
| 225 | // block then consumes, so no line is looked at twice by it. |
| 226 | // It reads the source rather than moving the reader, whose |
| 227 | // SetPosition keeps the line it last peeked. |
| 228 | size, found := 0, false |
| 229 | src := reader.Source() |
| 230 | for i := seg.Stop; i < len(src); { |
| 231 | end := len(src) |
| 232 | if j := bytes.IndexByte(src[i:], '\n'); j >= 0 { |
| 233 | end = i + j + 1 |
| 234 | } |
| 235 | next := src[i:end] |
| 236 | if util.IsBlank(next) { |
| 237 | break |
| 238 | } |
| 239 | size += len(next) |
| 240 | if bytes.HasSuffix(util.TrimRightSpace(next), []byte("$$")) { |
| 241 | found = true |
| 242 | break |
| 243 | } |
| 244 | i = end |
| 245 | } |
| 246 | if !found || !budgetFor(pc).take(size) { |
| 247 | return nil, parser.NoChildren |
| 248 | } |
| 249 | n.source = append(n.source, "$$\n"...) |
| 153 | 250 | case len(rest) > 2 && bytes.HasSuffix(rest, []byte("$$")): |
| 251 | if !budgetFor(pc).take(len(rest) - 2) { |
| 252 | return nil, parser.NoChildren |
| 253 | } |
| 154 | 254 | n.tex, n.closed = rest[:len(rest)-2], true |
| 255 | n.source = append([]byte("$$"), rest...) |
| 155 | 256 | default: |
| 156 | 257 | return nil, parser.NoChildren |
| 157 | 258 | } |
| @@ -165,17 +266,19 @@ func (mathBlockParser) Continue(node ast.Node, reader text.Reader, pc parser.Con |
| 165 | 266 | return parser.Close |
| 166 | 267 | } |
| 167 | 268 | line, _ := reader.PeekLine() |
| 168 | | if line == nil { |
| 269 | if line == nil || util.IsBlank(line) { |
| 169 | 270 | return parser.Close |
| 170 | 271 | } |
| 171 | 272 | trimmed := util.TrimRightSpace(line) |
| 172 | 273 | if bytes.HasSuffix(trimmed, []byte("$$")) { |
| 173 | 274 | n.tex = append(n.tex, trimmed[:len(trimmed)-2]...) |
| 275 | n.source = append(n.source, trimmed...) |
| 174 | 276 | n.closed = true |
| 175 | 277 | reader.AdvanceToEOL() |
| 176 | 278 | return parser.Close |
| 177 | 279 | } |
| 178 | 280 | n.tex = append(n.tex, line...) |
| 281 | n.source = append(n.source, line...) |
| 179 | 282 | reader.AdvanceToEOL() |
| 180 | 283 | return parser.Continue | parser.NoChildren |
| 181 | 284 | } |
| @@ -204,12 +307,11 @@ func (mathRenderer) RegisterFuncs(reg renderer.NodeRendererFuncRegisterer) { |
| 204 | 307 | return ast.WalkContinue, nil |
| 205 | 308 | } |
| 206 | 309 | n := node.(*mathBlock) |
| 207 | | source := "$$" + string(n.tex) |
| 208 | | if n.closed { |
| 209 | | source += "$$" |
| 210 | | } |
| 310 | source := string(n.source) |
| 211 | 311 | out, ok := mathHTML(string(n.tex), source, true) |
| 212 | 312 | if !ok || !n.closed { |
| 313 | // A container that ended before the closing line leaves the |
| 314 | // block unclosed; show what it held. |
| 213 | 315 | out = "<pre>" + template.HTMLEscapeString(source) + "</pre>" |
| 214 | 316 | } |
| 215 | 317 | _, _ = w.WriteString(out + "\n") |
| @@ -229,6 +331,83 @@ func (mathExtension) Extend(m goldmark.Markdown) { |
| 229 | 331 | // Org: go-org already parses $…$, $$…$$, \(…\), \[…\] and \begin{…}…\end{…} |
| 230 | 332 | // as LaTeX fragments, and \begin{…} on its own lines as a LaTeX block; it |
| 231 | 333 | // writes them back out as text. These render them instead. |
| 334 | // |
| 335 | // The whole org document goes through ugcPolicy, which admits no MathML, so |
| 336 | // the writer leaves a placeholder for each expression and fill puts the |
| 337 | // mathPolicy-cleaned MathML back after sanitizing. The placeholder carries a |
| 338 | // random per-render prefix, so a document cannot spell one. |
| 339 | |
| 340 | type mathSlots struct { |
| 341 | prefix string |
| 342 | html []string |
| 343 | source []string |
| 344 | budget mathBudget |
| 345 | } |
| 346 | |
| 347 | func newMathSlots() *mathSlots { |
| 348 | var b [12]byte |
| 349 | _, _ = rand.Read(b[:]) |
| 350 | return &mathSlots{prefix: "gitbaymath" + hex.EncodeToString(b[:]) + "n"} |
| 351 | } |
| 352 | |
| 353 | func (m *mathSlots) put(html, source string) string { |
| 354 | m.html = append(m.html, html) |
| 355 | m.source = append(m.source, source) |
| 356 | return m.prefix + strconv.Itoa(len(m.html)-1) + "z" |
| 357 | } |
| 358 | |
| 359 | // fill replaces each placeholder in sanitized HTML with its MathML, or with |
| 360 | // its escaped source where the placeholder landed inside a tag (an |
| 361 | // attribute value). Sanitized output escapes < and > everywhere but in |
| 362 | // tags, so the last of them seen says whether the text is inside one. |
| 363 | func (m *mathSlots) fill(doc string) string { |
| 364 | if len(m.html) == 0 { |
| 365 | return doc |
| 366 | } |
| 367 | var b strings.Builder |
| 368 | inTag := false |
| 369 | for { |
| 370 | i := strings.Index(doc, m.prefix) |
| 371 | if i < 0 { |
| 372 | b.WriteString(doc) |
| 373 | return b.String() |
| 374 | } |
| 375 | before := doc[:i] |
| 376 | if j := strings.LastIndexAny(before, "<>"); j >= 0 { |
| 377 | inTag = before[j] == '<' |
| 378 | } |
| 379 | b.WriteString(before) |
| 380 | doc = doc[i+len(m.prefix):] |
| 381 | end := strings.IndexByte(doc, 'z') |
| 382 | if end < 0 { |
| 383 | b.WriteString(m.prefix) |
| 384 | continue |
| 385 | } |
| 386 | k, err := strconv.Atoi(doc[:end]) |
| 387 | if err != nil || k < 0 || k >= len(m.html) { |
| 388 | b.WriteString(m.prefix) |
| 389 | continue |
| 390 | } |
| 391 | doc = doc[end+1:] |
| 392 | if inTag { |
| 393 | b.WriteString(template.HTMLEscapeString(m.source[k])) |
| 394 | } else { |
| 395 | b.WriteString(m.html[k]) |
| 396 | } |
| 397 | } |
| 398 | } |
| 399 | |
| 400 | // writeMath writes an expression's placeholder, or its source as text when |
| 401 | // the converter refuses it or the document's budget is spent. |
| 402 | func (w *orgWriter) writeMath(tex, source string, display bool) { |
| 403 | if w.math.budget.take(len(tex)) { |
| 404 | if out, ok := mathHTML(tex, source, display); ok { |
| 405 | w.WriteString(w.math.put(out, source)) |
| 406 | return |
| 407 | } |
| 408 | } |
| 409 | w.WriteText(org.Text{Content: source, IsRaw: true}) |
| 410 | } |
| 232 | 411 | |
| 233 | 412 | func (w *orgWriter) WriteLatexFragment(l org.LatexFragment) { |
| 234 | 413 | tex := org.String(l.Content...) |
| @@ -242,19 +421,16 @@ func (w *orgWriter) WriteLatexFragment(l org.LatexFragment) { |
| 242 | 421 | if strings.HasPrefix(l.OpeningPair, `\begin{`) { |
| 243 | 422 | tex = source |
| 244 | 423 | } |
| 245 | | out, ok := mathHTML(tex, source, display) |
| 246 | | if !ok { |
| 247 | | w.WriteText(org.Text{Content: source, IsRaw: true}) |
| 248 | | return |
| 249 | | } |
| 250 | | w.WriteString(out) |
| 424 | w.writeMath(tex, source, display) |
| 251 | 425 | } |
| 252 | 426 | |
| 253 | 427 | func (w *orgWriter) WriteLatexBlock(b org.LatexBlock) { |
| 254 | 428 | tex := org.String(b.Content...) |
| 255 | | if out, ok := mathHTML(tex, tex, true); ok { |
| 256 | | w.WriteString(out + "\n") |
| 257 | | return |
| 429 | if w.math.budget.take(len(tex)) { |
| 430 | if out, ok := mathHTML(tex, tex, true); ok { |
| 431 | w.WriteString(w.math.put(out, tex) + "\n") |
| 432 | return |
| 433 | } |
| 258 | 434 | } |
| 259 | 435 | w.WriteString("<pre>" + template.HTMLEscapeString(tex) + "</pre>\n") |
| 260 | 436 | } |