Parse entities, sub/superscripts, LaTeX, radio targets as objects !96
12 files changed, +308 −27
Layout: unified · split
Sources/OrgCore/Commands/Footnotes.swift +2 −2
| @@ -127,7 +127,7 @@ extension EmacsBuffer { | ||
| 127 | 127 | return point >= title.range.lowerBound && tags.map { point < $0.range.lowerBound } ?? true |
| 128 | 128 | } |
| 129 | 129 | let objects: Set<SyntaxKind> = [.bold, .italic, .underline, .strikeThrough, .verbatim, .code, .link, .linkDescription, .timestamp, |
| 130 | .footnoteReference, .statisticsCookie, .target, .macro, .inlineSourceBlock, .latexFragment, .lineBreak, .superscript] | |
| 130 | .footnoteReference, .statisticsCookie, .target, .macro, .inlineSourceBlock, .latexFragment, .lineBreak, .superscript, .subscript, .entity, .radioTarget] | |
| 131 | 131 | let containers: Set<SyntaxKind> = [.document, .zerothSection, .section, .plainList, .item, .drawer] |
| 132 | 132 | let element = chain.last { !objects.contains($0.kind) && !containers.contains($0.kind) } |
| 133 | 133 | switch element?.kind { |
| @@ -160,7 +160,7 @@ extension EmacsBuffer { | ||
| 160 | 160 | if point == object.range.lowerBound { return true } |
| 161 | 161 | if chain.contains(where: { $0.kind == .link }) { return false } |
| 162 | 162 | switch object.kind { |
| 163 | case .bold, .italic, .underline, .strikeThrough, .superscript: | |
| 163 | case .bold, .italic, .underline, .strikeThrough, .superscript, .subscript, .radioTarget: | |
| 164 | 164 | return point > object.range.lowerBound && point <= object.range.upperBound - 1 |
| 165 | 165 | default: |
| 166 | 166 | return false |
Sources/OrgCore/Commands/Narrowing.swift +1 −1
| @@ -32,7 +32,7 @@ public enum Narrowing { | ||
| 32 | 32 | probe = buffer.lineStart(buffer.point) |
| 33 | 33 | } |
| 34 | 34 | let objects: Set<SyntaxKind> = [.bold, .italic, .underline, .strikeThrough, .verbatim, .code, .link, .linkDescription, .timestamp, |
| 35 | .footnoteReference, .statisticsCookie, .target, .macro, .inlineSourceBlock, .latexFragment, .lineBreak, .superscript] | |
| 35 | .footnoteReference, .statisticsCookie, .target, .macro, .inlineSourceBlock, .latexFragment, .lineBreak, .superscript, .subscript, .entity, .radioTarget] | |
| 36 | 36 | var chain: [SyntaxNode] = [] |
| 37 | 37 | var node = tree.root |
| 38 | 38 | while let child = node.child(containing: probe) { |
Sources/OrgCore/Export/HTMLExport.swift +16 −1
| @@ -235,6 +235,11 @@ final class HTMLRenderer { | ||
| 235 | 235 | return block(node) |
| 236 | 236 | case .horizontalRule: |
| 237 | 237 | return "<hr>\n" |
| 238 | case .latexEnvironment: | |
| 239 | // For MathJax, as ox-html leaves it with `tex:t`. | |
| 240 | guard options.tex else { return "" } | |
| 241 | hasMath = true | |
| 242 | return Self.escape(node.text.trimmingCharacters(in: .newlines)) + "\n" | |
| 238 | 243 | case .fixedWidth: |
| 239 | 244 | let lines = node.text.split(separator: "\n", omittingEmptySubsequences: false).map { line -> String in |
| 240 | 245 | let trimmed = line.drop { $0 == " " || $0 == "\t" } |
| @@ -637,7 +642,13 @@ final class HTMLRenderer { | ||
| 637 | 642 | case .linkDescription: return inline(node) |
| 638 | 643 | case .timestamp: return timestamp(node.text) |
| 639 | 644 | case .footnoteReference: return footnote(node.text) |
| 640 | case .superscript: | |
| 645 | case .entity: | |
| 646 | let name = node.text.dropFirst().replacingOccurrences(of: "{}", with: "") | |
| 647 | return Self.escape(OrgEntities.display[name] ?? node.text) | |
| 648 | case .radioTarget: | |
| 649 | let inner = String(node.text.dropFirst(3).dropLast(3)) | |
| 650 | return "<a id=\"\(Self.escape(slugOnly(inner)))\">\(Self.escape(inner))</a>" | |
| 651 | case .superscript, .subscript: | |
| 641 | 652 | let text = node.text |
| 642 | 653 | let body = text.dropFirst() |
| 643 | 654 | let braced = body.hasPrefix("{") && body.hasSuffix("}") |
| @@ -665,6 +676,10 @@ final class HTMLRenderer { | ||
| 665 | 676 | let path = node.tokens.first { $0.kind == .linkPath }?.text |
| 666 | 677 | let description = node.firstChild(.linkDescription) |
| 667 | 678 | guard let path else { |
| 679 | // A radio link goes to its target. | |
| 680 | if !node.text.hasPrefix("<"), !node.text.contains(":") { | |
| 681 | return "<a href=\"#\(Self.escape(slugOnly(node.text)))\">\(Self.escape(node.text))</a>" | |
| 682 | } | |
| 668 | 683 | // A plain or angle link. |
| 669 | 684 | var target = node.text |
| 670 | 685 | if target.hasPrefix("<"), target.hasSuffix(">") { target = String(target.dropFirst().dropLast()) } |
Sources/OrgCore/Export/MarkdownExport.swift +8 −1
| @@ -239,7 +239,12 @@ final class MarkdownRenderer { | ||
| 239 | 239 | } |
| 240 | 240 | return show(stamp.start) + (stamp.end.map { "–" + show($0) } ?? "") |
| 241 | 241 | case .footnoteReference: return footnote(node.text) |
| 242 | case .superscript: | |
| 242 | case .entity: | |
| 243 | let name = node.text.dropFirst().replacingOccurrences(of: "{}", with: "") | |
| 244 | return OrgEntities.display[name] ?? node.text | |
| 245 | case .radioTarget: | |
| 246 | return String(node.text.dropFirst(3).dropLast(3)) | |
| 247 | case .superscript, .subscript: | |
| 243 | 248 | let body = node.text.dropFirst() |
| 244 | 249 | let inner = body.hasPrefix("{") && body.hasSuffix("}") ? String(body.dropFirst().dropLast()) : String(body) |
| 245 | 250 | return node.text.hasPrefix("_") ? "<sub>\(inner)</sub>" : "<sup>\(inner)</sup>" |
| @@ -251,6 +256,8 @@ final class MarkdownRenderer { | ||
| 251 | 256 | |
| 252 | 257 | private func link(_ node: SyntaxNode) -> String { |
| 253 | 258 | guard let path = node.tokens.first(where: { $0.kind == .linkPath })?.text else { |
| 259 | // A radio link is its text. | |
| 260 | if !node.text.hasPrefix("<"), !node.text.contains(":") { return node.text } | |
| 254 | 261 | var target = node.text |
| 255 | 262 | if target.hasPrefix("<"), target.hasSuffix(">") { target = String(target.dropFirst().dropLast()) } |
| 256 | 263 | return "<\(target)>" |
Sources/OrgCore/Parser/Incremental.swift +6 −2
| @@ -72,12 +72,16 @@ private struct ReparseContext { | ||
| 72 | 72 | let start = lineStart(oldText, edit.range.lowerBound) |
| 73 | 73 | let oldLines = classes(slice(oldText, start, lineEnd(oldText, edit.range.upperBound))) |
| 74 | 74 | let newLines = classes(slice(newText, start, lineEnd(newText, edit.range.lowerBound + edit.replacement.utf16.count))) |
| 75 | // Radio targets make links all through the file. | |
| 76 | if slice(oldText, start, lineEnd(oldText, edit.range.upperBound)).contains("<<<") | |
| 77 | || slice(newText, start, lineEnd(newText, edit.range.lowerBound + edit.replacement.utf16.count)).contains("<<<") | |
| 78 | || slice(oldText, start, lineEnd(oldText, edit.range.upperBound)).contains(">>>") { return true } | |
| 75 | 79 | var movesBoundary = false |
| 76 | 80 | for line in oldLines + newLines { |
| 77 | 81 | switch line.cls { |
| 78 | 82 | case .keyword(let key) where Self.settingsKeys.contains(key): |
| 79 | 83 | return true |
| 80 | case .blockBegin, .blockEnd, .dynamicBegin, .dynamicEnd, .heading: | |
| 84 | case .blockBegin, .blockEnd, .dynamicBegin, .dynamicEnd, .latexBegin, .latexEnd, .heading: | |
| 81 | 85 | movesBoundary = true |
| 82 | 86 | default: |
| 83 | 87 | break |
| @@ -276,7 +280,7 @@ private struct ReparseContext { | ||
| 276 | 280 | var unmatched = false |
| 277 | 281 | for (index, line) in parser.info.enumerated() { |
| 278 | 282 | switch line.cls { |
| 279 | case .blockBegin, .dynamicBegin, .drawerBegin: | |
| 283 | case .blockBegin, .dynamicBegin, .drawerBegin, .latexBegin: | |
| 280 | 284 | if parser.blockEnds[index] == nil { unmatched = true } |
| 281 | 285 | default: |
| 282 | 286 | break |
Sources/OrgCore/Parser/Inline.swift +162 −8
| @@ -42,12 +42,15 @@ enum InlineMatch { | ||
| 42 | 42 | case emphasis(SyntaxKind, open: Int, close: Int) |
| 43 | 43 | case link(path: Range<Int>, description: Range<Int>?, whole: Range<Int>) |
| 44 | 44 | case object(SyntaxKind, Range<Int>) |
| 45 | /// An object whose contents hold objects, between markers. | |
| 46 | case container(SyntaxKind, contents: Range<Int>, whole: Range<Int>) | |
| 45 | 47 | |
| 46 | 48 | var end: Int { |
| 47 | 49 | switch self { |
| 48 | 50 | case .emphasis(_, _, let close): close + 1 |
| 49 | 51 | case .link(_, _, let whole): whole.upperBound |
| 50 | 52 | case .object(_, let range): range.upperBound |
| 53 | case .container(_, _, let whole): whole.upperBound | |
| 51 | 54 | } |
| 52 | 55 | } |
| 53 | 56 | } |
| @@ -57,12 +60,17 @@ enum InlineMatch { | ||
| 57 | 60 | /// grapheme clusters never need to be formed. |
| 58 | 61 | struct InlineScanner { |
| 59 | 62 | let chars: [Unicode.Scalar] |
| 63 | /// Radio targets, lowercased, longest first, and the scalars they start with. | |
| 64 | let radioTargets: [[Unicode.Scalar]] | |
| 65 | let radioStarts: Set<Unicode.Scalar> | |
| 60 | 66 | /// Where each scalar starts in `source`, plus its end, so token text is sliced from the |
| 61 | 67 | /// source instead of rebuilt scalar by scalar. |
| 62 | 68 | let source: Substring.UnicodeScalarView |
| 63 | 69 | let starts: [String.Index] |
| 64 | 70 | |
| 65 | init(_ text: Substring) { | |
| 71 | init(_ text: Substring, radioTargets: [String] = []) { | |
| 72 | self.radioTargets = radioTargets.map { Array($0.lowercased().unicodeScalars) } | |
| 73 | self.radioStarts = Set(self.radioTargets.compactMap(\.first).flatMap { [$0] + Array(String($0).uppercased().unicodeScalars) }) | |
| 66 | 74 | source = text.unicodeScalars |
| 67 | 75 | var chars: [Unicode.Scalar] = [] |
| 68 | 76 | var starts: [String.Index] = [] |
| @@ -83,7 +91,7 @@ struct InlineScanner { | ||
| 83 | 91 | var textStart = range.lowerBound |
| 84 | 92 | var i = range.lowerBound |
| 85 | 93 | while i < range.upperBound { |
| 86 | if Self.mayStartObject(chars[i]), let match = match(at: i, in: range, inLink: inLink) { | |
| 94 | if Self.mayStartObject(chars[i]) || radioStarts.contains(chars[i]), let match = match(at: i, in: range, inLink: inLink) { | |
| 87 | 95 | emitText(textStart..<i, into: &b) |
| 88 | 96 | emit(match, into: &b, inLink: inLink) |
| 89 | 97 | i = match.end |
| @@ -98,7 +106,7 @@ struct InlineScanner { | ||
| 98 | 106 | /// The first characters any recognizer accepts. |
| 99 | 107 | static func mayStartObject(_ c: Unicode.Scalar) -> Bool { |
| 100 | 108 | switch c { |
| 101 | case "[", "<", "{", "\\", "^", "s", "h", "m", "f", "*", "/", "_", "+", "=", "~": true | |
| 109 | case "[", "<", "{", "\\", "^", "s", "h", "m", "f", "*", "/", "_", "+", "=", "~", "$": true | |
| 102 | 110 | default: false |
| 103 | 111 | } |
| 104 | 112 | } |
| @@ -145,6 +153,12 @@ struct InlineScanner { | ||
| 145 | 153 | b.token(.marker, string(path.upperBound..<whole.upperBound)) |
| 146 | 154 | } |
| 147 | 155 | b.finish() |
| 156 | case .container(let kind, let contents, let whole): | |
| 157 | b.start(kind) | |
| 158 | b.token(.marker, string(whole.lowerBound..<contents.lowerBound)) | |
| 159 | scan(contents, into: &b, inLink: inLink) | |
| 160 | b.token(.marker, string(contents.upperBound..<whole.upperBound)) | |
| 161 | b.finish() | |
| 148 | 162 | case .object(let kind, let range): |
| 149 | 163 | b.start(kind) |
| 150 | 164 | emitText(range, into: &b) |
| @@ -172,6 +186,10 @@ struct InlineScanner { | ||
| 172 | 186 | let limit = range.upperBound |
| 173 | 187 | let previous: Unicode.Scalar? = i > range.lowerBound ? chars[i - 1] : nil |
| 174 | 188 | let afterWord = previous.map(isWordScalar) ?? false |
| 189 | let afterNonSpace = previous.map { !isSpace($0) } ?? false | |
| 190 | ||
| 191 | // Radio links: a radio target's words, case aside, between non-alphanumerics. | |
| 192 | if !inLink, !afterWord, radioStarts.contains(chars[i]), let end = radioLink(i, limit) { return .object(.link, i..<end) } | |
| 175 | 193 | |
| 176 | 194 | switch chars[i] { |
| 177 | 195 | case "[": |
| @@ -181,15 +199,22 @@ struct InlineScanner { | ||
| 181 | 199 | if let end = statisticsCookie(i, limit) { return .object(.statisticsCookie, i..<end) } |
| 182 | 200 | case "<": |
| 183 | 201 | if !inLink, let end = timestampEnd(chars, at: i, limit: limit) { return .object(.timestamp, i..<end) } |
| 202 | if let end = radioTarget(i, limit) { return .object(.radioTarget, i..<end) } | |
| 184 | 203 | if let end = target(i, limit) { return .object(.target, i..<end) } |
| 185 | 204 | if !inLink, let end = angleLink(i, limit) { return .object(.link, i..<end) } |
| 186 | 205 | case "{": |
| 187 | 206 | if let end = macro(i, limit) { return .object(.macro, i..<end) } |
| 188 | 207 | case "\\": |
| 189 | 208 | if let end = lineBreak(i, limit) { return .object(.lineBreak, i..<end) } |
| 209 | if let end = entity(i, limit) { return .object(.entity, i..<end) } | |
| 190 | 210 | if let end = latexFragment(i, limit) { return .object(.latexFragment, i..<end) } |
| 191 | case "^": | |
| 192 | if afterWord, let end = superscript(i, limit) { return .object(.superscript, i..<end) } | |
| 211 | case "$": | |
| 212 | if previous != "$", let end = dollarFragment(i, limit) { return .object(.latexFragment, i..<end) } | |
| 213 | case "^", "_": | |
| 214 | if afterNonSpace, let end = script(i, limit) { | |
| 215 | let kind: SyntaxKind = chars[i] == "^" ? .superscript : .subscript | |
| 216 | return chars[i + 1] == "{" ? .container(kind, contents: (i + 2)..<(end - 1), whole: i..<end) : .object(kind, i..<end) | |
| 217 | } | |
| 193 | 218 | case "s": |
| 194 | 219 | if !afterWord, let end = inlineSourceBlock(i, limit) { return .object(.inlineSourceBlock, i..<end) } |
| 195 | 220 | case "h", "m", "f": |
| @@ -359,14 +384,143 @@ struct InlineScanner { | ||
| 359 | 384 | return j |
| 360 | 385 | } |
| 361 | 386 | |
| 362 | /// `\(...\)` or `\[...\]`. | |
| 387 | /// `org-element-entity-parser`: `\name`, `\name{}` or `\_` and spaces, for a name in | |
| 388 | /// `org-entities`. | |
| 389 | func entity(_ i: Int, _ limit: Int) -> Int? { | |
| 390 | var j = i + 1 | |
| 391 | guard j < limit else { return nil } | |
| 392 | if chars[j] == "_" { | |
| 393 | j += 1 | |
| 394 | while j < limit, chars[j] == " " { j += 1 } | |
| 395 | guard j > i + 2, OrgEntities.display[string((i + 1)..<j)] != nil else { return nil } | |
| 396 | return j | |
| 397 | } | |
| 398 | while j < limit, chars[j].isASCII, chars[j].properties.isAlphabetic { j += 1 } | |
| 399 | // `there4`, `sup1`…, `frac12`… end in digits. | |
| 400 | if j < limit, chars[j].isASCII, chars[j].properties.numericType != nil { | |
| 401 | var k = j | |
| 402 | while k < limit, chars[k].isASCII, chars[k].properties.numericType != nil { k += 1 } | |
| 403 | if OrgEntities.display[string((i + 1)..<k)] != nil { j = k } | |
| 404 | } | |
| 405 | guard j > i + 1, OrgEntities.display[string((i + 1)..<j)] != nil else { return nil } | |
| 406 | if j < limit, chars[j].properties.isAlphabetic { return nil } | |
| 407 | if hasPrefix("{}", at: j, limit) { return j + 2 } | |
| 408 | return j | |
| 409 | } | |
| 410 | ||
| 411 | /// `$$...$$`, or `$...$` with no blank inside its ends and punctuation, a blank or the end | |
| 412 | /// after it, as `org-element-latex-fragment-parser`. | |
| 413 | func dollarFragment(_ i: Int, _ limit: Int) -> Int? { | |
| 414 | guard i + 1 < limit else { return nil } | |
| 415 | if chars[i + 1] == "$" { | |
| 416 | var j = i + 2 | |
| 417 | while j + 1 < limit { | |
| 418 | if chars[j] == "$", chars[j + 1] == "$" { return j + 2 } | |
| 419 | j += 1 | |
| 420 | } | |
| 421 | return nil | |
| 422 | } | |
| 423 | let next = chars[i + 1] | |
| 424 | guard ![" ", "\t", "\n", ",", ".", ";"].contains(next) else { return nil } | |
| 425 | var j = i + 1 | |
| 426 | while j < limit, chars[j] != "$" { j += 1 } | |
| 427 | guard j < limit, ![" ", "\t", "\n", ",", "."].contains(chars[j - 1]) else { return nil } | |
| 428 | if j + 1 < limit { | |
| 429 | let after = chars[j + 1] | |
| 430 | guard isSpace(after) || ".,;:!?'\"()[]{}<>-".unicodeScalars.contains(after) else { return nil } | |
| 431 | } | |
| 432 | return j + 1 | |
| 433 | } | |
| 434 | ||
| 435 | /// `org-match-substring-regexp` after its first character: `{…}` or `(…)` nested up to three | |
| 436 | /// deep, `*`, or `[+-]?[[:alnum:].,\\]*[[:alnum:]]`. | |
| 437 | func script(_ i: Int, _ limit: Int) -> Int? { | |
| 438 | let j = i + 1 | |
| 439 | guard j < limit else { return nil } | |
| 440 | if chars[j] == "{" || chars[j] == "(" { | |
| 441 | let open = chars[j], close: Unicode.Scalar = open == "{" ? "}" : ")" | |
| 442 | var depth = 0 | |
| 443 | var k = j | |
| 444 | while k < limit { | |
| 445 | if chars[k] == open { | |
| 446 | depth += 1 | |
| 447 | if depth > 4 { return nil } | |
| 448 | } else if chars[k] == close { | |
| 449 | depth -= 1 | |
| 450 | if depth == 0 { return k + 1 } | |
| 451 | } | |
| 452 | k += 1 | |
| 453 | } | |
| 454 | return nil | |
| 455 | } | |
| 456 | if chars[j] == "*" { return j + 1 } | |
| 457 | var k = j | |
| 458 | if k < limit, chars[k] == "+" || chars[k] == "-" { k += 1 } | |
| 459 | var lastAlnum: Int? | |
| 460 | while k < limit, isWordScalar(chars[k]) || chars[k] == "." || chars[k] == "," || chars[k] == "\\" { | |
| 461 | if isWordScalar(chars[k]) { lastAlnum = k } | |
| 462 | k += 1 | |
| 463 | } | |
| 464 | return lastAlnum.map { $0 + 1 } | |
| 465 | } | |
| 466 | ||
| 467 | /// `<<<target>>>`. | |
| 468 | func radioTarget(_ i: Int, _ limit: Int) -> Int? { | |
| 469 | guard hasPrefix("<<<", at: i, limit) else { return nil } | |
| 470 | let start = i + 3 | |
| 471 | var j = start | |
| 472 | while j < limit, chars[j] != ">", chars[j] != "<", !isNewline(chars[j]) { j += 1 } | |
| 473 | guard j > start, hasPrefix(">>>", at: j, limit), !isSpace(chars[start]), !isSpace(chars[j - 1]) else { return nil } | |
| 474 | return j + 3 | |
| 475 | } | |
| 476 | ||
| 477 | /// A radio target's text at `i`, its blanks matching any run of blanks, not followed by a | |
| 478 | /// letter or digit. | |
| 479 | func radioLink(_ i: Int, _ limit: Int) -> Int? { | |
| 480 | for target in radioTargets { | |
| 481 | var j = i | |
| 482 | var t = 0 | |
| 483 | var ok = true | |
| 484 | while t < target.count { | |
| 485 | guard j < limit else { ok = false; break } | |
| 486 | if target[t] == " " { | |
| 487 | guard isSpace(chars[j]) else { ok = false; break } | |
| 488 | while j < limit, isSpace(chars[j]) { j += 1 } | |
| 489 | while t < target.count, target[t] == " " { t += 1 } | |
| 490 | continue | |
| 491 | } | |
| 492 | guard String(chars[j]).lowercased().unicodeScalars.first == target[t] else { ok = false; break } | |
| 493 | j += 1 | |
| 494 | t += 1 | |
| 495 | } | |
| 496 | if ok, j == limit || !isWordScalar(chars[j]) { return j } | |
| 497 | } | |
| 498 | return nil | |
| 499 | } | |
| 500 | ||
| 501 | /// `\(...\)`, `\[...\]`, or a LaTeX command with its `[options]` and `{arguments}`. | |
| 363 | 502 | func latexFragment(_ i: Int, _ limit: Int) -> Int? { |
| 364 | 503 | guard i + 1 < limit else { return nil } |
| 365 | 504 | let closer: String |
| 366 | 505 | switch chars[i + 1] { |
| 367 | 506 | case "(": closer = "\\)" |
| 368 | 507 | case "[": closer = "\\]" |
| 369 | default: return nil | |
| 508 | default: | |
| 509 | var j = i + 1 | |
| 510 | while j < limit, chars[j].isASCII, chars[j].properties.isAlphabetic { j += 1 } | |
| 511 | guard j > i + 1 else { return nil } | |
| 512 | if j < limit, chars[j] == "*" { j += 1 } | |
| 513 | while j < limit, chars[j] == "[" || chars[j] == "{" { | |
| 514 | let close: Unicode.Scalar = chars[j] == "[" ? "]" : "}" | |
| 515 | var k = j + 1 | |
| 516 | while k < limit, chars[k] != close { | |
| 517 | if isNewline(chars[k]) || chars[k] == "{" || chars[k] == "}" || (close == "]" && chars[k] == "[") { return j } | |
| 518 | k += 1 | |
| 519 | } | |
| 520 | guard k < limit else { return j } | |
| 521 | j = k + 1 | |
| 522 | } | |
| 523 | return j | |
| 370 | 524 | } |
| 371 | 525 | let closing = Array(closer.unicodeScalars) |
| 372 | 526 | var j = i + 2 |
| @@ -427,7 +581,7 @@ struct InlineScanner { | ||
| 427 | 581 | |
| 428 | 582 | extension Parser { |
| 429 | 583 | mutating func inline(_ text: Substring) { |
| 430 | let scanner = InlineScanner(text) | |
| 584 | let scanner = InlineScanner(text, radioTargets: settings.radioTargets) | |
| 431 | 585 | scanner.scan(0..<scanner.chars.count, into: &builder) |
| 432 | 586 | } |
| 433 | 587 | } |
Sources/OrgCore/Parser/Lines.swift +15
| @@ -41,6 +41,9 @@ enum LineClass: Equatable { | ||
| 41 | 41 | case dynamicEnd |
| 42 | 42 | case drawerBegin(name: String) |
| 43 | 43 | case drawerEnd |
| 44 | /// `\begin{NAME}` and `\end{NAME}` of a LaTeX environment. | |
| 45 | case latexBegin(name: String) | |
| 46 | case latexEnd(name: String) | |
| 44 | 47 | case keyword(key: String) |
| 45 | 48 | case comment |
| 46 | 49 | case fixedWidth |
| @@ -102,6 +105,18 @@ private func lineClass(_ rest: Substring, columnZero: Bool) -> LineClass { | ||
| 102 | 105 | |
| 103 | 106 | if trimmed == "#" || rest.hasPrefix("# ") || rest.hasPrefix("#\t") { return .comment } |
| 104 | 107 | |
| 108 | if rest.hasPrefix("\\begin{") || rest.hasPrefix("\\end{") { | |
| 109 | let begin = rest.hasPrefix("\\begin{") | |
| 110 | let after = rest.dropFirst(begin ? 7 : 5) | |
| 111 | let name = after.prefix { $0.isLetter || $0.isNumber || $0 == "*" } | |
| 112 | if !name.isEmpty, after.dropFirst(name.count).first == "}" { | |
| 113 | // The end line has nothing after it. | |
| 114 | let tail = after.dropFirst(name.count + 1) | |
| 115 | if begin { return .latexBegin(name: String(name)) } | |
| 116 | if tail.trimmingCharacters(in: .whitespaces).isEmpty { return .latexEnd(name: String(name)) } | |
| 117 | } | |
| 118 | } | |
| 119 | ||
| 105 | 120 | if rest.first == ":" { |
| 106 | 121 | if trimmed == ":" || rest.hasPrefix(": ") || rest.hasPrefix(":\t") { return .fixedWidth } |
| 107 | 122 | if trimmed.uppercased() == ":END:" { return .drawerEnd } |
Sources/OrgCore/Parser/Parser.swift +24 −7
| @@ -21,7 +21,7 @@ struct Parser { | ||
| 21 | 21 | init(text: String, defaults: OrgSettings) { |
| 22 | 22 | let lines = splitRawLines(text) |
| 23 | 23 | let info = lines.map { classifyLine($0.content) } |
| 24 | let ends = Parser.matchEnds(info) | |
| 24 | let ends = Parser.matchEnds(info, lines) | |
| 25 | 25 | let settings = SettingsScanner.scan(lines: lines, info: info, blockEnds: ends, defaults: defaults) |
| 26 | 26 | self.init(lines: lines, info: info, blockEnds: ends, settings: settings) |
| 27 | 27 | } |
| @@ -30,7 +30,7 @@ struct Parser { | ||
| 30 | 30 | init(text: String, settings: OrgSettings) { |
| 31 | 31 | let lines = splitRawLines(text) |
| 32 | 32 | let info = lines.map { classifyLine($0.content) } |
| 33 | self.init(lines: lines, info: info, blockEnds: Parser.matchEnds(info), settings: settings) | |
| 33 | self.init(lines: lines, info: info, blockEnds: Parser.matchEnds(info, lines), settings: settings) | |
| 34 | 34 | } |
| 35 | 35 | |
| 36 | 36 | private init(lines: [RawLine], info: [ClassifiedLine], blockEnds: [Int: Int], settings: OrgSettings) { |
| @@ -40,13 +40,25 @@ struct Parser { | ||
| 40 | 40 | self.settings = settings |
| 41 | 41 | } |
| 42 | 42 | |
| 43 | static func matchEnds(_ info: [ClassifiedLine]) -> [Int: Int] { | |
| 43 | static func matchEnds(_ info: [ClassifiedLine], _ lines: [RawLine]) -> [Int: Int] { | |
| 44 | 44 | var ends: [Int: Int] = [:] |
| 45 | 45 | var k = 0 |
| 46 | 46 | while k < info.count { |
| 47 | 47 | let isEnd: ((LineClass) -> Bool)? |
| 48 | 48 | switch info[k].cls { |
| 49 | 49 | case .blockBegin(let name): isEnd = { $0 == .blockEnd(name: name) } |
| 50 | case .latexBegin(let name): | |
| 51 | // `\end{NAME}` ending a line, the first line's own included. | |
| 52 | let closer = "\\end{" + name + "}" | |
| 53 | var j = k | |
| 54 | while j < info.count { | |
| 55 | if j > k, case .heading = info[j].cls { break } | |
| 56 | let content = lines[j].content.trimmingTrailingWhitespace | |
| 57 | if content.hasSuffix(closer), j > k || content.count > closer.count + 7 + name.count { ends[k] = j; break } | |
| 58 | j += 1 | |
| 59 | } | |
| 60 | k += 1 | |
| 61 | continue | |
| 50 | 62 | case .dynamicBegin: isEnd = { $0 == .dynamicEnd } |
| 51 | 63 | case .drawerBegin: isEnd = { $0 == .drawerEnd } |
| 52 | 64 | default: isEnd = nil |
| @@ -143,9 +155,14 @@ struct Parser { | ||
| 143 | 155 | case .blank: |
| 144 | 156 | line(i) |
| 145 | 157 | i += 1 |
| 146 | case .blockBegin, .dynamicBegin: | |
| 158 | case .blockBegin, .dynamicBegin, .latexBegin: | |
| 147 | 159 | if let end = blockEnds[i], end < limit { |
| 148 | builder.start(info[i].cls == .dynamicBegin ? .dynamicBlock : .block) | |
| 160 | let kind: SyntaxKind = switch info[i].cls { | |
| 161 | case .dynamicBegin: .dynamicBlock | |
| 162 | case .latexBegin: .latexEnvironment | |
| 163 | default: .block | |
| 164 | } | |
| 165 | builder.start(kind) | |
| 149 | 166 | while i <= end { |
| 150 | 167 | line(i) |
| 151 | 168 | i += 1 |
| @@ -354,9 +371,9 @@ struct Parser { | ||
| 354 | 371 | /// Lines that don't start an element of their own. |
| 355 | 372 | func continuesParagraph(_ k: Int) -> Bool { |
| 356 | 373 | switch info[k].cls { |
| 357 | case .plain, .planning, .blockEnd, .dynamicEnd, .drawerEnd: | |
| 374 | case .plain, .planning, .blockEnd, .dynamicEnd, .drawerEnd, .latexEnd: | |
| 358 | 375 | return true |
| 359 | case .blockBegin, .dynamicBegin, .drawerBegin: | |
| 376 | case .blockBegin, .dynamicBegin, .drawerBegin, .latexBegin: | |
| 360 | 377 | return blockEnds[k] == nil |
| 361 | 378 | default: |
| 362 | 379 | return false |
Sources/OrgCore/Parser/Settings.swift +12 −2
| @@ -43,10 +43,13 @@ public struct Priorities: Sendable, Equatable { | ||
| 43 | 43 | public struct OrgSettings: Sendable, Equatable { |
| 44 | 44 | public var todoSequences: [TodoSequence] |
| 45 | 45 | public var priorities: Priorities |
| 46 | /// `<<<radio target>>>` texts in the file, longest first: their words elsewhere are links. | |
| 47 | public var radioTargets: [String] = [] | |
| 46 | 48 | |
| 47 | public init(todoSequences: [TodoSequence], priorities: Priorities) { | |
| 49 | public init(todoSequences: [TodoSequence], priorities: Priorities, radioTargets: [String] = []) { | |
| 48 | 50 | self.todoSequences = todoSequences |
| 49 | 51 | self.priorities = priorities |
| 52 | self.radioTargets = radioTargets | |
| 50 | 53 | } |
| 51 | 54 | |
| 52 | 55 | public static let `default` = OrgSettings( |
| @@ -104,7 +107,14 @@ enum SettingsScanner { | ||
| 104 | 107 | } |
| 105 | 108 | k += 1 |
| 106 | 109 | } |
| 107 | return OrgSettings(todoSequences: sequences.isEmpty ? defaults.todoSequences : sequences, priorities: priorities) | |
| 110 | var radio: [String] = [] | |
| 111 | for line in lines where line.content.contains("<<<") { | |
| 112 | for m in String(line.content).matches(of: /<<<([^<>\n\r \t](?:[^<>\n\r]*[^<>\n\r \t])?)>>>/) where !radio.contains(String(m.1)) { | |
| 113 | radio.append(String(m.1)) | |
| 114 | } | |
| 115 | } | |
| 116 | return OrgSettings(todoSequences: sequences.isEmpty ? defaults.todoSequences : sequences, priorities: priorities, | |
| 117 | radioTargets: radio.sorted { $0.count > $1.count }) | |
| 108 | 118 | } |
| 109 | 119 | |
| 110 | 120 | static func keywordValue(_ line: Substring) -> Substring { |
Sources/OrgCore/Syntax/SyntaxKind.swift +2 −1
| @@ -9,10 +9,11 @@ public enum SyntaxKind: String, Sendable { | ||
| 9 | 9 | case planning, propertyDrawer, nodeProperty, drawer, clock |
| 10 | 10 | case paragraph, plainList, item, table, tableRow, tableCell, tableFormula |
| 11 | 11 | case block, dynamicBlock, keyword, affiliatedKeyword |
| 12 | case comment, fixedWidth, horizontalRule, footnoteDefinition | |
| 12 | case comment, fixedWidth, horizontalRule, footnoteDefinition, latexEnvironment | |
| 13 | 13 | |
| 14 | 14 | // Objects |
| 15 | 15 | case bold, italic, underline, strikeThrough, verbatim, code |
| 16 | 16 | case link, linkDescription, timestamp, footnoteReference, statisticsCookie |
| 17 | 17 | case target, macro, inlineSourceBlock, latexFragment, lineBreak, superscript |
| 18 | case `subscript`, entity, radioTarget | |
| 18 | 19 | } |
Sources/OrgPresentation/Presentation.swift +2 −2
| @@ -152,9 +152,9 @@ public enum Presentation { | ||
| 152 | 152 | case .timestamp: .timestamp |
| 153 | 153 | case .footnoteReference, .footnoteDefinition: .footnote |
| 154 | 154 | case .statisticsCookie: .cookie |
| 155 | case .target: .target | |
| 155 | case .target, .radioTarget: .target | |
| 156 | 156 | case .macro: .macro |
| 157 | case .latexFragment: .latex | |
| 157 | case .latexFragment, .latexEnvironment: .latex | |
| 158 | 158 | case .inlineSourceBlock: .inlineSource |
| 159 | 159 | case .comment: .comment |
| 160 | 160 | case .keyword, .affiliatedKeyword: .keyword |
Tests/OrgCoreTests/ObjectParseTests.swift added +58
| @@ -0,0 +1,58 @@ | ||
| 1 | import Foundation | |
| 2 | import Testing | |
| 3 | @testable import OrgCore | |
| 4 | ||
| 5 | /// Where org-element finds objects of each kind, against our tree. | |
| 6 | struct ObjectParseTests { | |
| 7 | static let oracle = ProcessInfo.processInfo.environment["ORGSTAR_SKIP_ORACLE"] == nil | |
| 8 | ||
| 9 | static let text = """ | |
| 10 | * Objects | |
| 11 | Entities \\alpha, \\alpha{}beta, \\nbsp and \\notanentity. | |
| 12 | Scripts x_1 y^2 a_{b c} d^{e} f_(g) h^* snake_case_name n^-1 e_{x_{y}} 2_1.5a. | |
| 13 | Not after space _x ^y. | |
| 14 | Math $x$ and $x+y$, $$e=mc^2$$ and \\(a^2\\) and \\[b\\] and \\frac{1}{2} and \\textbf[opt]{bold}. | |
| 15 | Money $5 and $10, not math $ x$ nor $x $. | |
| 16 | A radio <<<Target Word>>> here; the target word links, Target Words don't. | |
| 17 | In *bold \\beta x_2* and [[link][\\gamma]]. | |
| 18 | \\begin{equation} | |
| 19 | x = y^2 | |
| 20 | \\end{equation} | |
| 21 | \\begin{align*} a \\end{align*} | |
| 22 | \\begin{x} | |
| 23 | never closed | |
| 24 | """ + "\n" | |
| 25 | ||
| 26 | static let kinds: [(lisp: String, ours: [SyntaxKind])] = [ | |
| 27 | ("entity", [.entity]), ("subscript", [.subscript]), ("superscript", [.superscript]), | |
| 28 | ("latex-fragment", [.latexFragment]), ("radio-target", [.radioTarget]), ("latex-environment", [.latexEnvironment]), | |
| 29 | ] | |
| 30 | ||
| 31 | func ours(_ kinds: [SyntaxKind]) -> [String] { | |
| 32 | OrgParser.parse(Self.text).root.descendants().filter { kinds.contains($0.kind) }.map { "\($0.range.lowerBound) \($0.range.upperBound)" } | |
| 33 | } | |
| 34 | ||
| 35 | @Test(.enabled(if: oracle)) | |
| 36 | func objectsMatchOrgElement() throws { | |
| 37 | let forms = Self.kinds.map { kind in | |
| 38 | """ | |
| 39 | (mapconcat (lambda (o) (format "%d %d" (1- (org-element-begin o)) | |
| 40 | (1- (save-excursion (goto-char (org-element-end o)) (skip-chars-backward " \\t") (point))))) | |
| 41 | (org-element-map (org-element-parse-buffer) '\(kind.lisp) #'identity) ",") | |
| 42 | """ | |
| 43 | } | |
| 44 | let emacs = try EmacsOracle.evaluate(Self.text, "(with-current-buffer (find-file-noselect \"oracle.org\") (list \(forms.joined(separator: " "))))") | |
| 45 | for (i, kind) in Self.kinds.enumerated() { | |
| 46 | let theirs = emacs[i].split(separator: ",").map(String.init) | |
| 47 | #expect(ours(kind.ours) == theirs, "\(kind.lisp)") | |
| 48 | } | |
| 49 | // Radio links: links whose text is a radio target's. | |
| 50 | let links = try EmacsOracle.evaluate(Self.text, """ | |
| 51 | (with-current-buffer (find-file-noselect "oracle.org") | |
| 52 | (list (mapconcat (lambda (o) (format "%d %d" (1- (org-element-begin o)) (1- (save-excursion (goto-char (org-element-end o)) (skip-chars-backward " \\t") (point))))) | |
| 53 | (org-element-map (org-element-parse-buffer) 'link (lambda (l) (and (equal (org-element-property :type l) "radio") l))) ","))) | |
| 54 | """) | |
| 55 | let radio = OrgParser.parse(Self.text).root.descendants().filter { $0.kind == .link && $0.tokens.first { $0.kind == .linkPath } == nil && !$0.text.hasPrefix("[") && !$0.text.hasPrefix("<") && !$0.text.contains(":") } | |
| 56 | #expect(radio.map { "\($0.range.lowerBound) \($0.range.upperBound)" } == links[0].split(separator: ",").map(String.init)) | |
| 57 | } | |
| 58 | } | |