Commit a1f5ce2f2c
Verified · cmc
Layout: unified · split
Sources/OrgCore/Parser/Inline.swift added +376
| @@ -0,0 +1,376 @@ | |||
| 1 | /// Characters allowed before an emphasis opener, besides whitespace and the start of the run. | ||
| 2 | private let emphasisPre: Set<Character> = ["-", "(", "{", "'", "\""] | ||
| 3 | |||
| 4 | /// Characters allowed after an emphasis closer, besides whitespace and the end of the run. | ||
| 5 | private let emphasisPost: Set<Character> = ["-", ".", ",", ":", "!", "?", ";", "'", "\"", ")", "}", "\\", "["] | ||
| 6 | |||
| 7 | private let emphasisKinds: [Character: SyntaxKind] = [ | ||
| 8 | "*": .bold, "/": .italic, "_": .underline, "+": .strikeThrough, "=": .verbatim, "~": .code, | ||
| 9 | ] | ||
| 10 | |||
| 11 | private let angleLinkSchemes: Set<String> = [ | ||
| 12 | "http", "https", "mailto", "file", "id", "doi", "ftp", "news", "shell", "elisp", "info", "help", "attachment", | ||
| 13 | ] | ||
| 14 | |||
| 15 | private let plainLinkPrefixes = ["https://", "http://", "mailto:", "file:"] | ||
| 16 | |||
| 17 | /// "\r\n" is a single Character, so both forms count. | ||
| 18 | func isNewline(_ c: Character) -> Bool { | ||
| 19 | c == "\n" || c == "\r\n" | ||
| 20 | } | ||
| 21 | |||
| 22 | enum InlineMatch { | ||
| 23 | case emphasis(SyntaxKind, open: Int, close: Int) | ||
| 24 | case link(path: Range<Int>, description: Range<Int>?, whole: Range<Int>) | ||
| 25 | case object(SyntaxKind, Range<Int>) | ||
| 26 | |||
| 27 | var end: Int { | ||
| 28 | switch self { | ||
| 29 | case .emphasis(_, _, let close): close + 1 | ||
| 30 | case .link(_, _, let whole): whole.upperBound | ||
| 31 | case .object(_, let range): range.upperBound | ||
| 32 | } | ||
| 33 | } | ||
| 34 | } | ||
| 35 | |||
| 36 | /// Turns a run of text into text, newline and object tokens. Every character ends up in | ||
| 37 | /// exactly one token. | ||
| 38 | struct InlineScanner { | ||
| 39 | let chars: [Character] | ||
| 40 | |||
| 41 | func scan(_ range: Range<Int>, into b: inout GreenBuilder, inLink: Bool = false) { | ||
| 42 | var textStart = range.lowerBound | ||
| 43 | var i = range.lowerBound | ||
| 44 | while i < range.upperBound { | ||
| 45 | if let match = match(at: i, in: range, inLink: inLink) { | ||
| 46 | emitText(textStart..<i, into: &b) | ||
| 47 | emit(match, into: &b, inLink: inLink) | ||
| 48 | i = match.end | ||
| 49 | textStart = i | ||
| 50 | } else { | ||
| 51 | i += 1 | ||
| 52 | } | ||
| 53 | } | ||
| 54 | emitText(textStart..<range.upperBound, into: &b) | ||
| 55 | } | ||
| 56 | |||
| 57 | func emitText(_ range: Range<Int>, into b: inout GreenBuilder) { | ||
| 58 | var start = range.lowerBound | ||
| 59 | for k in range where isNewline(chars[k]) { | ||
| 60 | if start < k { b.token(.text, string(start..<k)) } | ||
| 61 | b.token(.newline, String(chars[k])) | ||
| 62 | start = k + 1 | ||
| 63 | } | ||
| 64 | if start < range.upperBound { b.token(.text, string(start..<range.upperBound)) } | ||
| 65 | } | ||
| 66 | |||
| 67 | func emit(_ match: InlineMatch, into b: inout GreenBuilder, inLink: Bool) { | ||
| 68 | switch match { | ||
| 69 | case .emphasis(let kind, let open, let close): | ||
| 70 | b.start(kind) | ||
| 71 | b.token(.marker, string(open..<(open + 1))) | ||
| 72 | if kind == .verbatim || kind == .code { | ||
| 73 | emitText((open + 1)..<close, into: &b) | ||
| 74 | } else { | ||
| 75 | scan((open + 1)..<close, into: &b, inLink: inLink) | ||
| 76 | } | ||
| 77 | b.token(.marker, string(close..<(close + 1))) | ||
| 78 | b.finish() | ||
| 79 | case .link(let path, let description, let whole): | ||
| 80 | b.start(.link) | ||
| 81 | b.token(.marker, string(whole.lowerBound..<path.lowerBound)) | ||
| 82 | b.token(.linkPath, string(path)) | ||
| 83 | if let description { | ||
| 84 | b.token(.marker, string(path.upperBound..<description.lowerBound)) | ||
| 85 | b.start(.linkDescription) | ||
| 86 | scan(description, into: &b, inLink: true) | ||
| 87 | b.finish() | ||
| 88 | b.token(.marker, string(description.upperBound..<whole.upperBound)) | ||
| 89 | } else { | ||
| 90 | b.token(.marker, string(path.upperBound..<whole.upperBound)) | ||
| 91 | } | ||
| 92 | b.finish() | ||
| 93 | case .object(let kind, let range): | ||
| 94 | b.start(kind) | ||
| 95 | emitText(range, into: &b) | ||
| 96 | b.finish() | ||
| 97 | } | ||
| 98 | } | ||
| 99 | |||
| 100 | func string(_ range: Range<Int>) -> String { | ||
| 101 | String(chars[range]) | ||
| 102 | } | ||
| 103 | |||
| 104 | func hasPrefix(_ s: String, at i: Int, _ limit: Int) -> Bool { | ||
| 105 | var j = i | ||
| 106 | for c in s { | ||
| 107 | guard j < limit, chars[j] == c else { return false } | ||
| 108 | j += 1 | ||
| 109 | } | ||
| 110 | return true | ||
| 111 | } | ||
| 112 | |||
| 113 | // MARK: - Recognizers | ||
| 114 | |||
| 115 | func match(at i: Int, in range: Range<Int>, inLink: Bool) -> InlineMatch? { | ||
| 116 | let limit = range.upperBound | ||
| 117 | let previous: Character? = i > range.lowerBound ? chars[i - 1] : nil | ||
| 118 | let afterWord = previous.map { $0.isLetter || $0.isNumber } ?? false | ||
| 119 | |||
| 120 | switch chars[i] { | ||
| 121 | case "[": | ||
| 122 | if !inLink, let link = bracketLink(i, limit) { return link } | ||
| 123 | if let end = footnoteReference(i, limit) { return .object(.footnoteReference, i..<end) } | ||
| 124 | if let end = scanTimestamp(chars, at: i, limit: limit)?.end { return .object(.timestamp, i..<end) } | ||
| 125 | if let end = statisticsCookie(i, limit) { return .object(.statisticsCookie, i..<end) } | ||
| 126 | case "<": | ||
| 127 | if let end = scanTimestamp(chars, at: i, limit: limit)?.end { return .object(.timestamp, i..<end) } | ||
| 128 | if let end = target(i, limit) { return .object(.target, i..<end) } | ||
| 129 | if !inLink, let end = angleLink(i, limit) { return .object(.link, i..<end) } | ||
| 130 | case "{": | ||
| 131 | if let end = macro(i, limit) { return .object(.macro, i..<end) } | ||
| 132 | case "\\": | ||
| 133 | if let end = lineBreak(i, limit) { return .object(.lineBreak, i..<end) } | ||
| 134 | if let end = latexFragment(i, limit) { return .object(.latexFragment, i..<end) } | ||
| 135 | case "^": | ||
| 136 | if afterWord, let end = superscript(i, limit) { return .object(.superscript, i..<end) } | ||
| 137 | case "s": | ||
| 138 | if !afterWord, let end = inlineSourceBlock(i, limit) { return .object(.inlineSourceBlock, i..<end) } | ||
| 139 | default: | ||
| 140 | break | ||
| 141 | } | ||
| 142 | |||
| 143 | if !inLink, !afterWord, let end = plainLink(i, limit) { return .object(.link, i..<end) } | ||
| 144 | |||
| 145 | if let kind = emphasisKinds[chars[i]], | ||
| 146 | previous.map({ $0.isWhitespace || emphasisPre.contains($0) }) ?? true, | ||
| 147 | let close = emphasisClose(i, limit) { | ||
| 148 | return .emphasis(kind, open: i, close: close) | ||
| 149 | } | ||
| 150 | return nil | ||
| 151 | } | ||
| 152 | |||
| 153 | /// org's emphasis rules: the body neither starts nor ends with whitespace, spans at most | ||
| 154 | /// one line break, and the closer is followed by whitespace, punctuation or the end. | ||
| 155 | func emphasisClose(_ i: Int, _ limit: Int) -> Int? { | ||
| 156 | let marker = chars[i] | ||
| 157 | guard i + 1 < limit, !chars[i + 1].isWhitespace else { return nil } | ||
| 158 | var newlines = 0 | ||
| 159 | var j = i + 1 | ||
| 160 | while j < limit { | ||
| 161 | if isNewline(chars[j]) { | ||
| 162 | newlines += 1 | ||
| 163 | if newlines > 1 { return nil } | ||
| 164 | } else if chars[j] == marker, j > i + 1, !chars[j - 1].isWhitespace { | ||
| 165 | if j + 1 == limit || chars[j + 1].isWhitespace || emphasisPost.contains(chars[j + 1]) { return j } | ||
| 166 | } | ||
| 167 | j += 1 | ||
| 168 | } | ||
| 169 | return nil | ||
| 170 | } | ||
| 171 | |||
| 172 | /// `[[path]]` or `[[path][description]]`. | ||
| 173 | func bracketLink(_ i: Int, _ limit: Int) -> InlineMatch? { | ||
| 174 | guard i + 1 < limit, chars[i + 1] == "[" else { return nil } | ||
| 175 | var j = i + 2 | ||
| 176 | while j < limit, chars[j] != "]" { | ||
| 177 | if chars[j] == "[" || isNewline(chars[j]) { return nil } | ||
| 178 | if chars[j] == "\\", j + 1 < limit { j += 1 } | ||
| 179 | j += 1 | ||
| 180 | } | ||
| 181 | guard j > i + 2, j + 1 < limit else { return nil } | ||
| 182 | let path = (i + 2)..<j | ||
| 183 | if chars[j + 1] == "]" { return .link(path: path, description: nil, whole: i..<(j + 2)) } | ||
| 184 | guard chars[j + 1] == "[" else { return nil } | ||
| 185 | let descriptionStart = j + 2 | ||
| 186 | var depth = 0 | ||
| 187 | var k = descriptionStart | ||
| 188 | while k < limit { | ||
| 189 | if chars[k] == "[" { | ||
| 190 | depth += 1 | ||
| 191 | } else if chars[k] == "]" { | ||
| 192 | if depth == 0 { break } | ||
| 193 | depth -= 1 | ||
| 194 | } | ||
| 195 | k += 1 | ||
| 196 | } | ||
| 197 | guard k > descriptionStart, k + 1 < limit, chars[k + 1] == "]" else { return nil } | ||
| 198 | return .link(path: path, description: descriptionStart..<k, whole: i..<(k + 2)) | ||
| 199 | } | ||
| 200 | |||
| 201 | /// `[fn:label]`, `[fn:label:definition]` or `[fn::definition]`. | ||
| 202 | func footnoteReference(_ i: Int, _ limit: Int) -> Int? { | ||
| 203 | guard hasPrefix("[fn:", at: i, limit) else { return nil } | ||
| 204 | var j = i + 4 | ||
| 205 | while j < limit, chars[j].isLetter || chars[j].isNumber || chars[j] == "_" || chars[j] == "-" { j += 1 } | ||
| 206 | guard j < limit else { return nil } | ||
| 207 | if chars[j] == "]" { return j > i + 4 ? j + 1 : nil } | ||
| 208 | guard chars[j] == ":" else { return nil } | ||
| 209 | var depth = 0 | ||
| 210 | j += 1 | ||
| 211 | while j < limit { | ||
| 212 | if chars[j] == "[" { | ||
| 213 | depth += 1 | ||
| 214 | } else if chars[j] == "]" { | ||
| 215 | if depth == 0 { return j + 1 } | ||
| 216 | depth -= 1 | ||
| 217 | } | ||
| 218 | j += 1 | ||
| 219 | } | ||
| 220 | return nil | ||
| 221 | } | ||
| 222 | |||
| 223 | /// `[1/3]`, `[/]`, `[50%]` or `[%]`. | ||
| 224 | func statisticsCookie(_ i: Int, _ limit: Int) -> Int? { | ||
| 225 | var j = i + 1 | ||
| 226 | while j < limit, chars[j].isASCII, chars[j].isNumber { j += 1 } | ||
| 227 | guard j < limit else { return nil } | ||
| 228 | if chars[j] == "%" { | ||
| 229 | j += 1 | ||
| 230 | } else if chars[j] == "/" { | ||
| 231 | j += 1 | ||
| 232 | while j < limit, chars[j].isASCII, chars[j].isNumber { j += 1 } | ||
| 233 | } else { | ||
| 234 | return nil | ||
| 235 | } | ||
| 236 | guard j < limit, chars[j] == "]" else { return nil } | ||
| 237 | return j + 1 | ||
| 238 | } | ||
| 239 | |||
| 240 | /// `<<target>>`. | ||
| 241 | func target(_ i: Int, _ limit: Int) -> Int? { | ||
| 242 | guard hasPrefix("<<", at: i, limit), i + 2 < limit, chars[i + 2] != "<" else { return nil } | ||
| 243 | let start = i + 2 | ||
| 244 | var j = start | ||
| 245 | while j < limit, chars[j] != ">" { | ||
| 246 | if chars[j] == "<" || isNewline(chars[j]) { return nil } | ||
| 247 | j += 1 | ||
| 248 | } | ||
| 249 | guard j > start, j + 1 < limit, chars[j + 1] == ">", | ||
| 250 | !chars[start].isWhitespace, !chars[j - 1].isWhitespace else { return nil } | ||
| 251 | return j + 2 | ||
| 252 | } | ||
| 253 | |||
| 254 | /// `<scheme:path>` for a known scheme. | ||
| 255 | func angleLink(_ i: Int, _ limit: Int) -> Int? { | ||
| 256 | var j = i + 1 | ||
| 257 | while j < limit, chars[j].isLetter { j += 1 } | ||
| 258 | guard j < limit, chars[j] == ":", angleLinkSchemes.contains(string((i + 1)..<j).lowercased()) else { return nil } | ||
| 259 | j += 1 | ||
| 260 | let bodyStart = j | ||
| 261 | while j < limit, chars[j] != ">" { | ||
| 262 | if chars[j] == "<" || isNewline(chars[j]) { return nil } | ||
| 263 | j += 1 | ||
| 264 | } | ||
| 265 | guard j < limit, j > bodyStart else { return nil } | ||
| 266 | return j + 1 | ||
| 267 | } | ||
| 268 | |||
| 269 | /// A bare URL. Trailing sentence punctuation stays outside the link. | ||
| 270 | func plainLink(_ i: Int, _ limit: Int) -> Int? { | ||
| 271 | guard "hmf".contains(chars[i]), | ||
| 272 | let prefix = plainLinkPrefixes.first(where: { hasPrefix($0, at: i, limit) }) else { return nil } | ||
| 273 | let bodyStart = i + prefix.count | ||
| 274 | var j = bodyStart | ||
| 275 | while j < limit, !chars[j].isWhitespace, !"()<>[]\"".contains(chars[j]) { j += 1 } | ||
| 276 | while j > bodyStart, ".,;:!?'".contains(chars[j - 1]) { j -= 1 } | ||
| 277 | return j > bodyStart ? j : nil | ||
| 278 | } | ||
| 279 | |||
| 280 | /// `{{{name}}}` or `{{{name(arguments)}}}`. | ||
| 281 | func macro(_ i: Int, _ limit: Int) -> Int? { | ||
| 282 | guard hasPrefix("{{{", at: i, limit) else { return nil } | ||
| 283 | var j = i + 3 | ||
| 284 | guard j < limit, chars[j].isLetter else { return nil } | ||
| 285 | while j < limit, chars[j].isLetter || chars[j].isNumber || chars[j] == "-" || chars[j] == "_" { j += 1 } | ||
| 286 | if j < limit, chars[j] == "(" { | ||
| 287 | var k = j + 1 | ||
| 288 | while k < limit, !hasPrefix(")}}}", at: k, limit) { | ||
| 289 | if isNewline(chars[k]) { return nil } | ||
| 290 | k += 1 | ||
| 291 | } | ||
| 292 | return k < limit ? k + 4 : nil | ||
| 293 | } | ||
| 294 | return hasPrefix("}}}", at: j, limit) ? j + 3 : nil | ||
| 295 | } | ||
| 296 | |||
| 297 | /// `\\` at the end of a line, before optional trailing blanks. The newline stays outside. | ||
| 298 | func lineBreak(_ i: Int, _ limit: Int) -> Int? { | ||
| 299 | guard hasPrefix("\\\\", at: i, limit) else { return nil } | ||
| 300 | var j = i + 2 | ||
| 301 | while j < limit, chars[j] == " " || chars[j] == "\t" { j += 1 } | ||
| 302 | guard j == limit || isNewline(chars[j]) else { return nil } | ||
| 303 | return j | ||
| 304 | } | ||
| 305 | |||
| 306 | /// `\(...\)` or `\[...\]`. | ||
| 307 | func latexFragment(_ i: Int, _ limit: Int) -> Int? { | ||
| 308 | guard i + 1 < limit else { return nil } | ||
| 309 | let closer: String | ||
| 310 | switch chars[i + 1] { | ||
| 311 | case "(": closer = "\\)" | ||
| 312 | case "[": closer = "\\]" | ||
| 313 | default: return nil | ||
| 314 | } | ||
| 315 | var j = i + 2 | ||
| 316 | while j < limit { | ||
| 317 | if hasPrefix(closer, at: j, limit) { return j + 2 } | ||
| 318 | j += 1 | ||
| 319 | } | ||
| 320 | return nil | ||
| 321 | } | ||
| 322 | |||
| 323 | /// `^word` or `^{group}` after a letter or digit. | ||
| 324 | func superscript(_ i: Int, _ limit: Int) -> Int? { | ||
| 325 | var j = i + 1 | ||
| 326 | guard j < limit else { return nil } | ||
| 327 | if chars[j] == "{" { | ||
| 328 | j += 1 | ||
| 329 | while j < limit, chars[j] != "}" { | ||
| 330 | if isNewline(chars[j]) { return nil } | ||
| 331 | j += 1 | ||
| 332 | } | ||
| 333 | return j < limit ? j + 1 : nil | ||
| 334 | } | ||
| 335 | let start = j | ||
| 336 | while j < limit, chars[j].isLetter || chars[j].isNumber { j += 1 } | ||
| 337 | return j > start ? j : nil | ||
| 338 | } | ||
| 339 | |||
| 340 | /// `src_lang{body}` or `src_lang[headers]{body}`, on one line, with balanced braces. | ||
| 341 | func inlineSourceBlock(_ i: Int, _ limit: Int) -> Int? { | ||
| 342 | guard hasPrefix("src_", at: i, limit) else { return nil } | ||
| 343 | var j = i + 4 | ||
| 344 | let languageStart = j | ||
| 345 | while j < limit, !chars[j].isWhitespace, chars[j] != "[", chars[j] != "{" { j += 1 } | ||
| 346 | guard j > languageStart, j < limit else { return nil } | ||
| 347 | if chars[j] == "[" { | ||
| 348 | while j < limit, chars[j] != "]" { | ||
| 349 | if isNewline(chars[j]) { return nil } | ||
| 350 | j += 1 | ||
| 351 | } | ||
| 352 | guard j < limit else { return nil } | ||
| 353 | j += 1 | ||
| 354 | } | ||
| 355 | guard j < limit, chars[j] == "{" else { return nil } | ||
| 356 | var depth = 0 | ||
| 357 | while j < limit { | ||
| 358 | if isNewline(chars[j]) { return nil } | ||
| 359 | if chars[j] == "{" { | ||
| 360 | depth += 1 | ||
| 361 | } else if chars[j] == "}" { | ||
| 362 | depth -= 1 | ||
| 363 | if depth == 0 { return j + 1 } | ||
| 364 | } | ||
| 365 | j += 1 | ||
| 366 | } | ||
| 367 | return nil | ||
| 368 | } | ||
| 369 | } | ||
| 370 | |||
| 371 | extension Parser { | ||
| 372 | mutating func inline(_ text: Substring) { | ||
| 373 | let chars = Array(text) | ||
| 374 | InlineScanner(chars: chars).scan(0..<chars.count, into: &builder) | ||
| 375 | } | ||
| 376 | } | ||
Sources/OrgCore/Parser/Lines.swift +1 −1
| @@ -27,7 +27,7 @@ func splitRawLines(_ text: String) -> [RawLine] { | |||
| 27 | } | 27 | } |
| 28 | } | 28 | } |
| 29 | if lineStart != scalars.endIndex { | 29 | if lineStart != scalars.endIndex { |
| 30 | lines.append(RawLine(content: text[lineStart...], ending: "")) | 30 | lines.append(RawLine(content: text[lineStart...], ending: text[text.endIndex...])) |
| 31 | } | 31 | } |
| 32 | return lines | 32 | return lines |
| 33 | } | 33 | } |
Sources/OrgCore/Parser/Parser.swift +94 −21
| @@ -77,10 +77,7 @@ struct Parser { | |||
| 77 | headingLine(lines[i]) | 77 | headingLine(lines[i]) |
| 78 | i += 1 | 78 | i += 1 |
| 79 | if i < lines.count, info[i].cls == .planning { | 79 | if i < lines.count, info[i].cls == .planning { |
| 80 | builder.start(.planning) | 80 | single(.planning) |
| 81 | line(i) | ||
| 82 | i += 1 | ||
| 83 | builder.finish() | ||
| 84 | } | 81 | } |
| 85 | if i < lines.count, case .drawerBegin(let name) = info[i].cls, name.uppercased() == "PROPERTIES", | 82 | if i < lines.count, case .drawerBegin(let name) = info[i].cls, name.uppercased() == "PROPERTIES", |
| 86 | let end = blockEnds[i] { | 83 | let end = blockEnds[i] { |
| @@ -173,13 +170,28 @@ struct Parser { | |||
| 173 | } | 170 | } |
| 174 | } | 171 | } |
| 175 | 172 | ||
| 173 | /// Kinds whose single line holds inline objects (timestamps on planning and clock lines). | ||
| 174 | static let inlineLineKinds: Set<SyntaxKind> = [.planning, .clock] | ||
| 175 | |||
| 176 | mutating func single(_ kind: SyntaxKind) { | 176 | mutating func single(_ kind: SyntaxKind) { |
| 177 | builder.start(kind) | 177 | builder.start(kind) |
| 178 | line(i) | 178 | if Self.inlineLineKinds.contains(kind) { |
| 179 | let rest = whitespace(lines[i].content) | ||
| 180 | inline(rest) | ||
| 181 | builder.token(.newline, lines[i].ending) | ||
| 182 | } else { | ||
| 183 | line(i) | ||
| 184 | } | ||
| 179 | i += 1 | 185 | i += 1 |
| 180 | builder.finish() | 186 | builder.finish() |
| 181 | } | 187 | } |
| 182 | 188 | ||
| 189 | /// Source text from `start` through the end of line `last`, including its line ending. | ||
| 190 | func span(from start: Substring.Index, through last: Int) -> Substring { | ||
| 191 | let base = lines[last].ending.base | ||
| 192 | return base[start..<lines[last].ending.endIndex] | ||
| 193 | } | ||
| 194 | |||
| 183 | mutating func consecutive(_ kind: SyntaxKind, limit: Int, floor: Int?, matching: (LineClass) -> Bool) { | 195 | mutating func consecutive(_ kind: SyntaxKind, limit: Int, floor: Int?, matching: (LineClass) -> Bool) { |
| 184 | builder.start(kind) | 196 | builder.start(kind) |
| 185 | repeat { | 197 | repeat { |
| @@ -192,7 +204,7 @@ struct Parser { | |||
| 192 | mutating func table(limit: Int, floor: Int?) { | 204 | mutating func table(limit: Int, floor: Int?) { |
| 193 | builder.start(.table) | 205 | builder.start(.table) |
| 194 | while i < limit, info[i].cls == .tableRow, within(floor, i) { | 206 | while i < limit, info[i].cls == .tableRow, within(floor, i) { |
| 195 | single(.tableRow) | 207 | tableRow() |
| 196 | } | 208 | } |
| 197 | while i < limit, info[i].cls == .keyword(key: "TBLFM"), within(floor, i) { | 209 | while i < limit, info[i].cls == .keyword(key: "TBLFM"), within(floor, i) { |
| 198 | single(.tableFormula) | 210 | single(.tableFormula) |
| @@ -200,15 +212,47 @@ struct Parser { | |||
| 200 | builder.finish() | 212 | builder.finish() |
| 201 | } | 213 | } |
| 202 | 214 | ||
| 215 | /// A rule row (`|---+---|`) is one text token. Other rows alternate `|` markers and cells; | ||
| 216 | /// every pair of pipes gets a cell, even an empty one, so columns line up. | ||
| 217 | mutating func tableRow() { | ||
| 218 | builder.start(.tableRow) | ||
| 219 | let rest = whitespace(lines[i].content) | ||
| 220 | if rest.hasPrefix("|-") { | ||
| 221 | builder.token(.text, rest) | ||
| 222 | } else { | ||
| 223 | var cellStart = rest.startIndex | ||
| 224 | var index = rest.startIndex | ||
| 225 | while index < rest.endIndex { | ||
| 226 | if rest[index] == "|" { | ||
| 227 | if index > rest.startIndex { tableCell(rest[cellStart..<index]) } | ||
| 228 | builder.token(.marker, "|") | ||
| 229 | cellStart = rest.index(after: index) | ||
| 230 | } | ||
| 231 | index = rest.index(after: index) | ||
| 232 | } | ||
| 233 | if cellStart < rest.endIndex { tableCell(rest[cellStart...]) } | ||
| 234 | } | ||
| 235 | builder.token(.newline, lines[i].ending) | ||
| 236 | builder.finish() | ||
| 237 | i += 1 | ||
| 238 | } | ||
| 239 | |||
| 240 | mutating func tableCell(_ text: Substring) { | ||
| 241 | builder.start(.tableCell) | ||
| 242 | inline(text) | ||
| 243 | builder.finish() | ||
| 244 | } | ||
| 245 | |||
| 203 | mutating func footnoteDefinition(limit: Int) { | 246 | mutating func footnoteDefinition(limit: Int) { |
| 247 | var end = i + 1 | ||
| 248 | while end < limit, info[end].cls == .plain { end += 1 } | ||
| 249 | let content = lines[i].content | ||
| 250 | let close = content.firstIndex(of: "]")! | ||
| 204 | builder.start(.footnoteDefinition) | 251 | builder.start(.footnoteDefinition) |
| 205 | line(i) | 252 | builder.token(.marker, content[...close]) |
| 206 | i += 1 | 253 | inline(span(from: content.index(after: close), through: end - 1)) |
| 207 | while i < limit, info[i].cls == .plain { | ||
| 208 | line(i) | ||
| 209 | i += 1 | ||
| 210 | } | ||
| 211 | builder.finish() | 254 | builder.finish() |
| 255 | i = end | ||
| 212 | } | 256 | } |
| 213 | 257 | ||
| 214 | mutating func list(limit: Int, floor: Int?) { | 258 | mutating func list(limit: Int, floor: Int?) { |
| @@ -224,8 +268,25 @@ struct Parser { | |||
| 224 | /// inside the item when the item or list continues after it; two end the list. | 268 | /// inside the item when the item or list continues after it; two end the list. |
| 225 | mutating func item(base: Int, limit: Int) { | 269 | mutating func item(base: Int, limit: Int) { |
| 226 | builder.start(.item) | 270 | builder.start(.item) |
| 227 | line(i) | 271 | var rest = whitespace(lines[i].content) |
| 228 | i += 1 | 272 | let bullet = rest.prefix { $0 != " " && $0 != "\t" } |
| 273 | builder.token(.bullet, bullet) | ||
| 274 | rest = whitespace(rest.dropFirst(bullet.count)) | ||
| 275 | if let box = checkbox(rest) { | ||
| 276 | builder.token(.checkbox, box) | ||
| 277 | rest = whitespace(rest.dropFirst(box.count)) | ||
| 278 | } | ||
| 279 | // The rest of the bullet line and its continuation lines are the item's first paragraph. | ||
| 280 | var end = i + 1 | ||
| 281 | while end < limit, within(base, end), continuesParagraph(end) { end += 1 } | ||
| 282 | if rest.isEmpty, end == i + 1 { | ||
| 283 | builder.token(.newline, lines[i].ending) | ||
| 284 | } else { | ||
| 285 | builder.start(.paragraph) | ||
| 286 | inline(span(from: rest.startIndex, through: end - 1)) | ||
| 287 | builder.finish() | ||
| 288 | } | ||
| 289 | i = end | ||
| 229 | while i < limit, !isHeading(i) { | 290 | while i < limit, !isHeading(i) { |
| 230 | if info[i].cls == .blank { | 291 | if info[i].cls == .blank { |
| 231 | var j = i | 292 | var j = i |
| @@ -245,15 +306,23 @@ struct Parser { | |||
| 245 | builder.finish() | 306 | builder.finish() |
| 246 | } | 307 | } |
| 247 | 308 | ||
| 309 | /// `[ ]`, `[X]`, `[x]` or `[-]`, followed by whitespace or end of line. | ||
| 310 | func checkbox(_ s: Substring) -> Substring? { | ||
| 311 | guard s.count >= 3, s.first == "[", "Xx -".contains(s.dropFirst().first!), | ||
| 312 | s.dropFirst(2).first == "]" else { return nil } | ||
| 313 | let after = s.dropFirst(3) | ||
| 314 | guard after.isEmpty || after.first == " " || after.first == "\t" else { return nil } | ||
| 315 | return s.prefix(3) | ||
| 316 | } | ||
| 317 | |||
| 318 | /// The paragraph's lines are one inline run, so emphasis and links can cross a line break. | ||
| 248 | mutating func paragraph(limit: Int, floor: Int?) { | 319 | mutating func paragraph(limit: Int, floor: Int?) { |
| 320 | var end = i + 1 | ||
| 321 | while end < limit, within(floor, end), continuesParagraph(end) { end += 1 } | ||
| 249 | builder.start(.paragraph) | 322 | builder.start(.paragraph) |
| 250 | line(i) | 323 | inline(span(from: lines[i].content.startIndex, through: end - 1)) |
| 251 | i += 1 | ||
| 252 | while i < limit, within(floor, i), continuesParagraph(i) { | ||
| 253 | line(i) | ||
| 254 | i += 1 | ||
| 255 | } | ||
| 256 | builder.finish() | 324 | builder.finish() |
| 325 | i = end | ||
| 257 | } | 326 | } |
| 258 | 327 | ||
| 259 | /// Lines that don't start an element of their own. | 328 | /// Lines that don't start an element of their own. |
| @@ -307,7 +376,11 @@ struct Parser { | |||
| 307 | } | 376 | } |
| 308 | 377 | ||
| 309 | let parts = splitTags(rest) | 378 | let parts = splitTags(rest) |
| 310 | builder.token(.title, parts.title) | 379 | if !parts.title.isEmpty { |
| 380 | builder.start(.title) | ||
| 381 | inline(parts.title) | ||
| 382 | builder.finish() | ||
| 383 | } | ||
| 311 | builder.token(.whitespace, parts.gap) | 384 | builder.token(.whitespace, parts.gap) |
| 312 | builder.token(.tags, parts.tags) | 385 | builder.token(.tags, parts.tags) |
| 313 | builder.token(.whitespace, parts.trailing) | 386 | builder.token(.whitespace, parts.trailing) |
Sources/OrgCore/Syntax/SyntaxKind.swift +10 −4
| @@ -1,12 +1,18 @@ | |||
| 1 | public enum SyntaxKind: String, Sendable { | 1 | public enum SyntaxKind: String, Sendable { |
| 2 | // Tokens | 2 | // Tokens |
| 3 | case text, newline, whitespace | 3 | case text, newline, whitespace |
| 4 | case stars, todoKeyword, priority, title, tags | 4 | case stars, todoKeyword, priority, tags |
| 5 | case marker, linkPath, bullet, checkbox | ||
| 5 | 6 | ||
| 6 | // Nodes | 7 | // Elements |
| 7 | case document, zerothSection, section, heading | 8 | case document, zerothSection, section, heading, title |
| 8 | case planning, propertyDrawer, nodeProperty, drawer, clock | 9 | case planning, propertyDrawer, nodeProperty, drawer, clock |
| 9 | case paragraph, plainList, item, table, tableRow, tableFormula | 10 | case paragraph, plainList, item, table, tableRow, tableCell, tableFormula |
| 10 | case block, dynamicBlock, keyword, affiliatedKeyword | 11 | case block, dynamicBlock, keyword, affiliatedKeyword |
| 11 | case comment, fixedWidth, horizontalRule, footnoteDefinition | 12 | case comment, fixedWidth, horizontalRule, footnoteDefinition |
| 13 | |||
| 14 | // Objects | ||
| 15 | case bold, italic, underline, strikeThrough, verbatim, code | ||
| 16 | case link, linkDescription, timestamp, footnoteReference, statisticsCookie | ||
| 17 | case target, macro, inlineSourceBlock, latexFragment, lineBreak, superscript | ||
| 12 | } | 18 | } |
Tests/OrgCoreTests/InlineTests.swift added +89
| @@ -0,0 +1,89 @@ | |||
| 1 | import Testing | ||
| 2 | @testable import OrgCore | ||
| 3 | |||
| 4 | /// `kind:text` for each object directly inside the first paragraph. | ||
| 5 | func objects(_ text: String) -> [String] { | ||
| 6 | let paragraph = OrgParser.parse(text).root.descendants().first { $0.kind == .paragraph }! | ||
| 7 | return paragraph.children.map { "\($0.kind.rawValue):\($0.text)" } | ||
| 8 | } | ||
| 9 | |||
| 10 | struct InlineTests { | ||
| 11 | @Test(arguments: [ | ||
| 12 | ("*b* /i/ _u_ +s+ =v= ~c~", ["bold:*b*", "italic:/i/", "underline:_u_", "strikeThrough:+s+", "verbatim:=v=", "code:~c~"]), | ||
| 13 | ("a*b* (*c*) \"*d*\"", ["bold:*c*", "bold:*d*"]), | ||
| 14 | ("x *y * z", []), | ||
| 15 | ("*y*z", []), | ||
| 16 | ("*a\nb*", ["bold:*a\nb*"]), | ||
| 17 | ("*a\nb\nc*", []), | ||
| 18 | ("=*not bold*=", ["verbatim:=*not bold*="]), | ||
| 19 | ("[[https://a.b][the *site*]]", ["link:[[https://a.b][the *site*]]"]), | ||
| 20 | ("[[file:x.org]]", ["link:[[file:x.org]]"]), | ||
| 21 | ("<https://a.b/c>", ["link:<https://a.b/c>"]), | ||
| 22 | ("see https://a.b/c.", ["link:https://a.b/c"]), | ||
| 23 | ("<2026-10-04 Sun 10:00-11:30 +1w -2d>", ["timestamp:<2026-10-04 Sun 10:00-11:30 +1w -2d>"]), | ||
| 24 | ("[2026-10-04 Sun]--[2026-10-06 Tue]", ["timestamp:[2026-10-04 Sun]--[2026-10-06 Tue]"]), | ||
| 25 | ("a [fn:1] and [fn::inline [x] note]", ["footnoteReference:[fn:1]", "footnoteReference:[fn::inline [x] note]"]), | ||
| 26 | ("a [1/3] [50%]", ["statisticsCookie:[1/3]", "statisticsCookie:[50%]"]), | ||
| 27 | ("a <<target>> {{{m(x, y)}}}", ["target:<<target>>", "macro:{{{m(x, y)}}}"]), | ||
| 28 | ("\\(x^2\\) and src_sh[:results raw]{echo {a}}", ["latexFragment:\\(x^2\\)", "inlineSourceBlock:src_sh[:results raw]{echo {a}}"]), | ||
| 29 | ("N^2 and e^{i}", ["superscript:^2", "superscript:^{i}"]), | ||
| 30 | ("end\\\\\nnext", ["lineBreak:\\\\"]), | ||
| 31 | ]) | ||
| 32 | func recognizes(text: String, expected: [String]) { | ||
| 33 | #expect(objects(text) == expected) | ||
| 34 | } | ||
| 35 | |||
| 36 | @Test func emphasisNests() { | ||
| 37 | let bold = OrgParser.parse("*bold /italic/ x*\n").root.descendants().first { $0.kind == .bold }! | ||
| 38 | #expect(bold.children.map(\.kind) == [.italic]) | ||
| 39 | #expect(bold.tokens.first?.kind == .marker) | ||
| 40 | } | ||
| 41 | |||
| 42 | @Test func linkParts() { | ||
| 43 | let link = OrgParser.parse("[[id:abc][desc]]\n").root.descendants().first { $0.kind == .link }! | ||
| 44 | #expect(link.tokens.map(\.kind) == [.marker, .linkPath, .marker, .marker]) | ||
| 45 | #expect(link.tokens.first { $0.kind == .linkPath }?.text == "id:abc") | ||
| 46 | #expect(link.children.map(\.kind) == [.linkDescription]) | ||
| 47 | } | ||
| 48 | |||
| 49 | @Test func noLinksInsideLinkDescriptions() { | ||
| 50 | let link = OrgParser.parse("[[a][see https://b.c]]\n").root.descendants().first { $0.kind == .link }! | ||
| 51 | #expect(link.descendants().filter { $0.kind == .link }.count == 1) | ||
| 52 | } | ||
| 53 | |||
| 54 | @Test func headingTitlesHoldObjects() { | ||
| 55 | let title = OrgParser.parse("* TODO Read [[https://a.b][it]] [1/2]\n").root.descendants().first { $0.kind == .title }! | ||
| 56 | #expect(title.children.map(\.kind) == [.link, .statisticsCookie]) | ||
| 57 | } | ||
| 58 | |||
| 59 | @Test func planningHoldsTimestamps() { | ||
| 60 | let planning = OrgParser.parse("* a\nDEADLINE: <2026-10-04 Sun -2d> SCHEDULED: <2026-10-01 Thu>\n").root | ||
| 61 | .descendants().first { $0.kind == .planning }! | ||
| 62 | #expect(planning.children.map(\.kind) == [.timestamp, .timestamp]) | ||
| 63 | } | ||
| 64 | |||
| 65 | @Test func itemsSplitBulletAndCheckbox() { | ||
| 66 | let item = OrgParser.parse(" - [X] done *now*\n more\n").root.descendants().first { $0.kind == .item }! | ||
| 67 | #expect(item.tokens.map(\.kind) == [.whitespace, .bullet, .whitespace, .checkbox, .whitespace]) | ||
| 68 | #expect(item.children.map(\.kind) == [.paragraph]) | ||
| 69 | #expect(item.children[0].text == "done *now*\n more\n") | ||
| 70 | } | ||
| 71 | |||
| 72 | @Test func emptyItem() { | ||
| 73 | let item = OrgParser.parse("-\n").root.descendants().first { $0.kind == .item }! | ||
| 74 | #expect(item.tokens.map(\.kind) == [.bullet, .newline]) | ||
| 75 | } | ||
| 76 | |||
| 77 | @Test func tableCells() { | ||
| 78 | let rows = OrgParser.parse("| *a* || b\n|---+---|\n").root.descendants().filter { $0.kind == .tableRow } | ||
| 79 | #expect(rows[0].children.map(\.text) == [" *a* ", "", " b"]) | ||
| 80 | #expect(rows[0].children[0].children.map(\.kind) == [.bold]) | ||
| 81 | #expect(rows[1].children.isEmpty) | ||
| 82 | } | ||
| 83 | |||
| 84 | @Test func footnoteDefinitionLabelIsNotAReference() { | ||
| 85 | let definition = OrgParser.parse("[fn:1] see [fn:2]\n").root.descendants().first { $0.kind == .footnoteDefinition }! | ||
| 86 | #expect(definition.tokens.first?.text == "[fn:1]") | ||
| 87 | #expect(definition.children.map(\.kind) == [.footnoteReference]) | ||
| 88 | } | ||
| 89 | } | ||
Tests/OrgCoreTests/ParserSectionTests.swift +5 −4
| @@ -17,7 +17,7 @@ struct ParserSectionTests { | |||
| 17 | } | 17 | } |
| 18 | 18 | ||
| 19 | @Test func zerothSectionHoldsPreamble() { | 19 | @Test func zerothSectionHoldsPreamble() { |
| 20 | #expect(nodeKinds("text\n* a\n") == [.document, .zerothSection, .paragraph, .section, .heading]) | 20 | #expect(nodeKinds("text\n* a\n") == [.document, .zerothSection, .paragraph, .section, .heading, .title]) |
| 21 | } | 21 | } |
| 22 | 22 | ||
| 23 | @Test func sectionsNestByLevel() { | 23 | @Test func sectionsNestByLevel() { |
| @@ -30,10 +30,11 @@ struct ParserSectionTests { | |||
| 30 | } | 30 | } |
| 31 | 31 | ||
| 32 | @Test func headingTokens() { | 32 | @Test func headingTokens() { |
| 33 | let parts = tokens(of: .heading, in: "** TODO [#A] Write the plan :work:urgent: \n") | 33 | let text = "** TODO [#A] Write the plan :work:urgent: \n" |
| 34 | #expect(parts.map(\.kind) == [.stars, .whitespace, .todoKeyword, .whitespace, .priority, .whitespace, .title, .whitespace, .tags, .whitespace, .newline]) | 34 | let parts = tokens(of: .heading, in: text) |
| 35 | #expect(parts.map(\.kind) == [.stars, .whitespace, .todoKeyword, .whitespace, .priority, .whitespace, .whitespace, .tags, .whitespace, .newline]) | ||
| 35 | #expect(parts.first { $0.kind == .tags }?.text == ":work:urgent:") | 36 | #expect(parts.first { $0.kind == .tags }?.text == ":work:urgent:") |
| 36 | #expect(parts.first { $0.kind == .title }?.text == "Write the plan") | 37 | #expect(OrgParser.parse(text).root.descendants().first { $0.kind == .title }?.text == "Write the plan") |
| 37 | } | 38 | } |
| 38 | 39 | ||
| 39 | @Test func todoKeywordsComeFromSettings() { | 40 | @Test func todoKeywordsComeFromSettings() { |