Commit 6fb6b30dba
Verified · cmc
Layout: unified · split
docs/plans/2026-10-04-inline-objects.md added +1085
| @@ -0,0 +1,1085 @@ | ||
| 1 | # Inline Objects Implementation Plan | |
| 2 | ||
| 3 | > **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. | |
| 4 | ||
| 5 | **Goal:** Parse org's inline objects into the lossless tree: emphasis, links, timestamps, footnote references, statistics cookies, targets, macros, inline source blocks, LaTeX fragments, line breaks and superscripts. | |
| 6 | ||
| 7 | **Architecture:** A character-level `InlineScanner` turns a run of text into text, newline and object nodes through the existing `GreenBuilder`. The parser feeds it whole paragraphs (so emphasis and links can cross one line break), heading titles, table cells, item paragraphs, footnote definitions, and planning and clock lines. `Timestamp` is both the recognizer and a public value type with repeaters and warnings. | |
| 8 | ||
| 9 | **Tech Stack:** Swift 6.2 tools, Swift Testing, Foundation only. | |
| 10 | ||
| 11 | **Spec:** `docs/design.md` ("OrgCore data model"); OrgSwift's `OrgInlineParser.swift` for the emphasis border rules. | |
| 12 | ||
| 13 | ## Global Constraints | |
| 14 | ||
| 15 | - Every character of the input lands in exactly one token; tree text equals source text. | |
| 16 | - Emphasis follows org's defaults: opener preceded by start, whitespace or `-({'"`; closer followed by end, whitespace or `-.,:!?;'")}\[`; body neither starts nor ends with whitespace; at most one line break. | |
| 17 | - `OrgCore` imports Foundation only. | |
| 18 | ||
| 19 | ## Out of scope | |
| 20 | ||
| 21 | Entities (`\alpha`), subscripts, `$...$` LaTeX, export snippets, citations, radio targets, inline footnote definitions as parsed content, and description-list terms. Each is recorded for a later plan. | |
| 22 | ||
| 23 | ## File structure | |
| 24 | ||
| 25 | | File | Responsibility | | |
| 26 | | --- | --- | | |
| 27 | | `Sources/OrgCore/Timestamp.swift` | `Timestamp` value type and `scanTimestamp` recognizer | | |
| 28 | | `Sources/OrgCore/Parser/Inline.swift` | `InlineScanner` and `Parser.inline(_:)` | | |
| 29 | | `Sources/OrgCore/Syntax/SyntaxKind.swift` | New token and object kinds; `title` becomes a node | | |
| 30 | | `Sources/OrgCore/Parser/Parser.swift` | Paragraph spans, item bullets and checkboxes, table cells, footnote labels, inline planning and clock lines | | |
| 31 | | `Sources/OrgCore/Parser/Lines.swift` | Empty final line ending sliced from the source, so spans can end there | | |
| 32 | ||
| 33 | --- | |
| 34 | ||
| 35 | ### Task 1: Timestamps | |
| 36 | ||
| 37 | **Files:** | |
| 38 | - Create: `Sources/OrgCore/Timestamp.swift` | |
| 39 | - Test: `Tests/OrgCoreTests/TimestampTests.swift` | |
| 40 | ||
| 41 | **Interfaces:** | |
| 42 | - Produces: `Timestamp` (`active`, `start: Point`, `end: Point?`, `repeater: Repeater?`, `warning: Warning?`), `Timestamp.parse(_:) -> Timestamp?`; internal `scanTimestamp(_ chars: [Character], at: Int, limit: Int) -> (stamp: Timestamp, end: Int)?`. | |
| 43 | ||
| 44 | - [ ] **Step 1: Write the failing tests** | |
| 45 | ||
| 46 | ```swift | |
| 47 | import Testing | |
| 48 | @testable import OrgCore | |
| 49 | ||
| 50 | struct TimestampTests { | |
| 51 | @Test func fullTimestamp() throws { | |
| 52 | let stamp = try #require(Timestamp.parse("<2026-10-04 Sun 10:00-11:30 .+1w/2w --2d>")) | |
| 53 | #expect(stamp.active) | |
| 54 | #expect(stamp.start == Timestamp.Point(year: 2026, month: 10, day: 4, hour: 10, minute: 0)) | |
| 55 | #expect(stamp.end == Timestamp.Point(year: 2026, month: 10, day: 4, hour: 11, minute: 30)) | |
| 56 | #expect(stamp.repeater == Timestamp.Repeater( | |
| 57 | kind: .restart, | |
| 58 | interval: .init(value: 1, unit: .week), | |
| 59 | habitDeadline: .init(value: 2, unit: .week) | |
| 60 | )) | |
| 61 | #expect(stamp.warning == Timestamp.Warning(firstOccurrenceOnly: true, interval: .init(value: 2, unit: .day))) | |
| 62 | } | |
| 63 | ||
| 64 | @Test func dateOnlyInactive() throws { | |
| 65 | let stamp = try #require(Timestamp.parse("[2026-10-04]")) | |
| 66 | #expect(!stamp.active) | |
| 67 | #expect(stamp.start.hour == nil) | |
| 68 | } | |
| 69 | ||
| 70 | @Test func dateRange() throws { | |
| 71 | let stamp = try #require(Timestamp.parse("[2026-10-04 Sun]--[2026-10-06 Tue]")) | |
| 72 | #expect(stamp.end?.day == 6) | |
| 73 | } | |
| 74 | ||
| 75 | @Test func repeaterKinds() { | |
| 76 | #expect(Timestamp.parse("<2026-10-04 +1d>")?.repeater?.kind == .cumulate) | |
| 77 | #expect(Timestamp.parse("<2026-10-04 ++1m>")?.repeater?.kind == .catchUp) | |
| 78 | #expect(Timestamp.parse("<2026-10-04 -3d +1y>")?.repeater?.interval == .init(value: 1, unit: .year)) | |
| 79 | } | |
| 80 | ||
| 81 | @Test func oneDigitHourAndOtherLanguages() { | |
| 82 | #expect(Timestamp.parse("<2026-10-04 9:05>")?.start.hour == 9) | |
| 83 | #expect(Timestamp.parse("<2026-10-04 dim.>") != nil) | |
| 84 | } | |
| 85 | ||
| 86 | @Test(arguments: ["<2026-10-4>", "<2026-10-04 Sun", "[2026-10-04 Sun]x", "<2026-10-04 Sun +1x>", "<2026-10-04]", "2026-10-04"]) | |
| 87 | func rejects(text: String) { | |
| 88 | #expect(Timestamp.parse(text) == nil) | |
| 89 | } | |
| 90 | } | |
| 91 | ``` | |
| 92 | ||
| 93 | - [ ] **Step 2: Run to verify failure** | |
| 94 | ||
| 95 | Run: `swift test --filter TimestampTests` | |
| 96 | Expected: build failure, `cannot find 'Timestamp' in scope`. | |
| 97 | ||
| 98 | - [ ] **Step 3: Implement** | |
| 99 | ||
| 100 | ```swift | |
| 101 | public struct Timestamp: Sendable, Equatable { | |
| 102 | public enum Unit: Character, Sendable { | |
| 103 | case hour = "h", day = "d", week = "w", month = "m", year = "y" | |
| 104 | } | |
| 105 | ||
| 106 | public struct Interval: Sendable, Equatable { | |
| 107 | public var value: Int | |
| 108 | public var unit: Unit | |
| 109 | } | |
| 110 | ||
| 111 | public enum RepeaterKind: String, Sendable { | |
| 112 | case cumulate = "+", catchUp = "++", restart = ".+" | |
| 113 | } | |
| 114 | ||
| 115 | public struct Repeater: Sendable, Equatable { | |
| 116 | public var kind: RepeaterKind | |
| 117 | public var interval: Interval | |
| 118 | /// The habit deadline from `.+2d/3d`. | |
| 119 | public var habitDeadline: Interval? | |
| 120 | } | |
| 121 | ||
| 122 | public struct Warning: Sendable, Equatable { | |
| 123 | /// `--` warns only for the first occurrence of a repeated timestamp. | |
| 124 | public var firstOccurrenceOnly: Bool | |
| 125 | public var interval: Interval | |
| 126 | } | |
| 127 | ||
| 128 | public struct Point: Sendable, Equatable { | |
| 129 | public var year: Int | |
| 130 | public var month: Int | |
| 131 | public var day: Int | |
| 132 | public var hour: Int? | |
| 133 | public var minute: Int? | |
| 134 | } | |
| 135 | ||
| 136 | public var active: Bool | |
| 137 | public var start: Point | |
| 138 | /// End of a same-day time range or of a `--` date range. | |
| 139 | public var end: Point? | |
| 140 | public var repeater: Repeater? | |
| 141 | public var warning: Warning? | |
| 142 | ||
| 143 | /// Parses exactly one timestamp or range, with nothing before or after it. | |
| 144 | public static func parse(_ text: some StringProtocol) -> Timestamp? { | |
| 145 | let chars = Array(text) | |
| 146 | guard let result = scanTimestamp(chars, at: 0, limit: chars.count), result.end == chars.count else { return nil } | |
| 147 | return result.stamp | |
| 148 | } | |
| 149 | } | |
| 150 | ||
| 151 | /// A timestamp or `--` range starting at `start`, and the index after it. | |
| 152 | func scanTimestamp(_ chars: [Character], at start: Int, limit: Int) -> (stamp: Timestamp, end: Int)? { | |
| 153 | guard let first = scanSingleTimestamp(chars, at: start, limit: limit) else { return nil } | |
| 154 | if first.stamp.end == nil, first.end + 2 < limit, chars[first.end] == "-", chars[first.end + 1] == "-", | |
| 155 | let second = scanSingleTimestamp(chars, at: first.end + 2, limit: limit), | |
| 156 | second.stamp.active == first.stamp.active, second.stamp.end == nil { | |
| 157 | var range = first.stamp | |
| 158 | range.end = second.stamp.start | |
| 159 | return (range, second.end) | |
| 160 | } | |
| 161 | return first | |
| 162 | } | |
| 163 | ||
| 164 | /// `<YYYY-MM-DD DAY HH:MM-HH:MM REPEATER WARNING>`, or the same in `[...]` for inactive. | |
| 165 | private func scanSingleTimestamp(_ chars: [Character], at start: Int, limit: Int) -> (stamp: Timestamp, end: Int)? { | |
| 166 | guard start < limit, chars[start] == "<" || chars[start] == "[" else { return nil } | |
| 167 | let active = chars[start] == "<" | |
| 168 | let close: Character = active ? ">" : "]" | |
| 169 | var j = start + 1 | |
| 170 | ||
| 171 | func number(_ minDigits: Int, _ maxDigits: Int) -> Int? { | |
| 172 | var k = j | |
| 173 | while k < limit, k - j < maxDigits, chars[k].isASCII, chars[k].isNumber { k += 1 } | |
| 174 | guard k - j >= minDigits else { return nil } | |
| 175 | let value = Int(String(chars[j..<k]))! | |
| 176 | j = k | |
| 177 | return value | |
| 178 | } | |
| 179 | ||
| 180 | func take(_ c: Character) -> Bool { | |
| 181 | guard j < limit, chars[j] == c else { return false } | |
| 182 | j += 1 | |
| 183 | return true | |
| 184 | } | |
| 185 | ||
| 186 | func interval() -> Timestamp.Interval? { | |
| 187 | let before = j | |
| 188 | guard let value = number(1, 9), j < limit, let unit = Timestamp.Unit(rawValue: chars[j]) else { | |
| 189 | j = before | |
| 190 | return nil | |
| 191 | } | |
| 192 | j += 1 | |
| 193 | return Timestamp.Interval(value: value, unit: unit) | |
| 194 | } | |
| 195 | ||
| 196 | func repeater() -> Timestamp.Repeater? { | |
| 197 | let before = j | |
| 198 | let kind: Timestamp.RepeaterKind | |
| 199 | if j + 1 < limit, chars[j] == ".", chars[j + 1] == "+" { | |
| 200 | kind = .restart | |
| 201 | j += 2 | |
| 202 | } else if j + 1 < limit, chars[j] == "+", chars[j + 1] == "+" { | |
| 203 | kind = .catchUp | |
| 204 | j += 2 | |
| 205 | } else if take("+") { | |
| 206 | kind = .cumulate | |
| 207 | } else { | |
| 208 | return nil | |
| 209 | } | |
| 210 | guard let value = interval() else { | |
| 211 | j = before | |
| 212 | return nil | |
| 213 | } | |
| 214 | var deadline: Timestamp.Interval? | |
| 215 | if take("/") { | |
| 216 | deadline = interval() | |
| 217 | if deadline == nil { | |
| 218 | j = before | |
| 219 | return nil | |
| 220 | } | |
| 221 | } | |
| 222 | return Timestamp.Repeater(kind: kind, interval: value, habitDeadline: deadline) | |
| 223 | } | |
| 224 | ||
| 225 | func warning() -> Timestamp.Warning? { | |
| 226 | let before = j | |
| 227 | guard take("-") else { return nil } | |
| 228 | let firstOnly = take("-") | |
| 229 | guard let value = interval() else { | |
| 230 | j = before | |
| 231 | return nil | |
| 232 | } | |
| 233 | return Timestamp.Warning(firstOccurrenceOnly: firstOnly, interval: value) | |
| 234 | } | |
| 235 | ||
| 236 | guard let year = number(4, 4), take("-"), let month = number(2, 2), take("-"), let day = number(2, 2) else { | |
| 237 | return nil | |
| 238 | } | |
| 239 | var stamp = Timestamp( | |
| 240 | active: active, | |
| 241 | start: Timestamp.Point(year: year, month: month, day: day, hour: nil, minute: nil), | |
| 242 | end: nil, repeater: nil, warning: nil | |
| 243 | ) | |
| 244 | ||
| 245 | // Day name: anything but digits, whitespace, `+`, `-`, `]` and `>`, in any language. | |
| 246 | if j < limit, chars[j] == " " { | |
| 247 | var k = j + 1 | |
| 248 | while k < limit, !(chars[k].isNumber || chars[k].isWhitespace || "+-]>".contains(chars[k])) { k += 1 } | |
| 249 | if k > j + 1 { j = k } | |
| 250 | } | |
| 251 | ||
| 252 | let beforeTime = j | |
| 253 | if take(" "), let hour = number(1, 2), take(":"), let minute = number(2, 2) { | |
| 254 | stamp.start.hour = hour | |
| 255 | stamp.start.minute = minute | |
| 256 | let beforeEnd = j | |
| 257 | if take("-"), let endHour = number(1, 2), take(":"), let endMinute = number(2, 2) { | |
| 258 | stamp.end = Timestamp.Point(year: year, month: month, day: day, hour: endHour, minute: endMinute) | |
| 259 | } else { | |
| 260 | j = beforeEnd | |
| 261 | } | |
| 262 | } else { | |
| 263 | j = beforeTime | |
| 264 | } | |
| 265 | ||
| 266 | while j < limit, chars[j] == " " { | |
| 267 | let beforeModifier = j | |
| 268 | j += 1 | |
| 269 | if let value = repeater() { | |
| 270 | stamp.repeater = value | |
| 271 | } else if let value = warning() { | |
| 272 | stamp.warning = value | |
| 273 | } else { | |
| 274 | j = beforeModifier | |
| 275 | break | |
| 276 | } | |
| 277 | } | |
| 278 | ||
| 279 | guard take(close) else { return nil } | |
| 280 | return (stamp, j) | |
| 281 | } | |
| 282 | ``` | |
| 283 | ||
| 284 | - [ ] **Step 4: Run to verify pass** | |
| 285 | ||
| 286 | Run: `swift test --filter TimestampTests` | |
| 287 | Expected: all pass. | |
| 288 | ||
| 289 | - [ ] **Step 5: Commit** | |
| 290 | ||
| 291 | ```bash | |
| 292 | git add Sources/OrgCore/Timestamp.swift Tests/OrgCoreTests/TimestampTests.swift | |
| 293 | git commit -m "Add timestamp recognizer and value type" | |
| 294 | ``` | |
| 295 | ||
| 296 | --- | |
| 297 | ||
| 298 | ### Task 2: Inline scanner and parser integration | |
| 299 | ||
| 300 | **Files:** | |
| 301 | - Create: `Sources/OrgCore/Parser/Inline.swift` | |
| 302 | - Modify: `Sources/OrgCore/Syntax/SyntaxKind.swift`, `Sources/OrgCore/Parser/Parser.swift`, `Sources/OrgCore/Parser/Lines.swift` | |
| 303 | - Test: `Tests/OrgCoreTests/InlineTests.swift`; update `Tests/OrgCoreTests/ParserSectionTests.swift` (title is a node) | |
| 304 | ||
| 305 | **Interfaces:** | |
| 306 | - Consumes: `scanTimestamp` (Task 1), `GreenBuilder`, `Parser`. | |
| 307 | - Produces: `InlineScanner(chars:)` with `scan(_:into:inLink:)`; `Parser.inline(_ text: Substring)`; `Parser.span(from:through:)`; node kinds `title`, `tableCell`, `bold`, `italic`, `underline`, `strikeThrough`, `verbatim`, `code`, `link`, `linkDescription`, `timestamp`, `footnoteReference`, `statisticsCookie`, `target`, `macro`, `inlineSourceBlock`, `latexFragment`, `lineBreak`, `superscript`; token kinds `marker`, `linkPath`, `bullet`, `checkbox`. | |
| 308 | ||
| 309 | - [ ] **Step 1: Write the failing tests** | |
| 310 | ||
| 311 | ```swift | |
| 312 | import Testing | |
| 313 | @testable import OrgCore | |
| 314 | ||
| 315 | /// `kind:text` for each object directly inside the first paragraph. | |
| 316 | func objects(_ text: String) -> [String] { | |
| 317 | let paragraph = OrgParser.parse(text).root.descendants().first { $0.kind == .paragraph }! | |
| 318 | return paragraph.children.map { "\($0.kind.rawValue):\($0.text)" } | |
| 319 | } | |
| 320 | ||
| 321 | struct InlineTests { | |
| 322 | @Test(arguments: [ | |
| 323 | ("*b* /i/ _u_ +s+ =v= ~c~", ["bold:*b*", "italic:/i/", "underline:_u_", "strikeThrough:+s+", "verbatim:=v=", "code:~c~"]), | |
| 324 | ("a*b* (*c*) \"*d*\"", ["bold:*c*", "bold:*d*"]), | |
| 325 | ("x *y * z", []), | |
| 326 | ("*y*z", []), | |
| 327 | ("*a\nb*", ["bold:*a\nb*"]), | |
| 328 | ("*a\nb\nc*", []), | |
| 329 | ("=*not bold*=", ["verbatim:=*not bold*="]), | |
| 330 | ("[[https://a.b][the *site*]]", ["link:[[https://a.b][the *site*]]"]), | |
| 331 | ("[[file:x.org]]", ["link:[[file:x.org]]"]), | |
| 332 | ("<https://a.b/c>", ["link:<https://a.b/c>"]), | |
| 333 | ("see https://a.b/c.", ["link:https://a.b/c"]), | |
| 334 | ("<2026-10-04 Sun 10:00-11:30 +1w -2d>", ["timestamp:<2026-10-04 Sun 10:00-11:30 +1w -2d>"]), | |
| 335 | ("[2026-10-04 Sun]--[2026-10-06 Tue]", ["timestamp:[2026-10-04 Sun]--[2026-10-06 Tue]"]), | |
| 336 | ("a [fn:1] and [fn::inline [x] note]", ["footnoteReference:[fn:1]", "footnoteReference:[fn::inline [x] note]"]), | |
| 337 | ("a [1/3] [50%]", ["statisticsCookie:[1/3]", "statisticsCookie:[50%]"]), | |
| 338 | ("a <<target>> {{{m(x, y)}}}", ["target:<<target>>", "macro:{{{m(x, y)}}}"]), | |
| 339 | ("\\(x^2\\) and src_sh[:results raw]{echo {a}}", ["latexFragment:\\(x^2\\)", "inlineSourceBlock:src_sh[:results raw]{echo {a}}"]), | |
| 340 | ("N^2 and e^{i}", ["superscript:^2", "superscript:^{i}"]), | |
| 341 | ("end\\\\\nnext", ["lineBreak:\\\\"]), | |
| 342 | ]) | |
| 343 | func recognizes(text: String, expected: [String]) { | |
| 344 | #expect(objects(text) == expected) | |
| 345 | } | |
| 346 | ||
| 347 | @Test func emphasisNests() { | |
| 348 | let bold = OrgParser.parse("*bold /italic/ x*\n").root.descendants().first { $0.kind == .bold }! | |
| 349 | #expect(bold.children.map(\.kind) == [.italic]) | |
| 350 | #expect(bold.tokens.first?.kind == .marker) | |
| 351 | } | |
| 352 | ||
| 353 | @Test func linkParts() { | |
| 354 | let link = OrgParser.parse("[[id:abc][desc]]\n").root.descendants().first { $0.kind == .link }! | |
| 355 | #expect(link.tokens.map(\.kind) == [.marker, .linkPath, .marker, .marker]) | |
| 356 | #expect(link.tokens.first { $0.kind == .linkPath }?.text == "id:abc") | |
| 357 | #expect(link.children.map(\.kind) == [.linkDescription]) | |
| 358 | } | |
| 359 | ||
| 360 | @Test func noLinksInsideLinkDescriptions() { | |
| 361 | let link = OrgParser.parse("[[a][see https://b.c]]\n").root.descendants().first { $0.kind == .link }! | |
| 362 | #expect(link.descendants().filter { $0.kind == .link }.count == 1) | |
| 363 | } | |
| 364 | ||
| 365 | @Test func headingTitlesHoldObjects() { | |
| 366 | let title = OrgParser.parse("* TODO Read [[https://a.b][it]] [1/2]\n").root.descendants().first { $0.kind == .title }! | |
| 367 | #expect(title.children.map(\.kind) == [.link, .statisticsCookie]) | |
| 368 | } | |
| 369 | ||
| 370 | @Test func planningHoldsTimestamps() { | |
| 371 | let planning = OrgParser.parse("* a\nDEADLINE: <2026-10-04 Sun -2d> SCHEDULED: <2026-10-01 Thu>\n").root | |
| 372 | .descendants().first { $0.kind == .planning }! | |
| 373 | #expect(planning.children.map(\.kind) == [.timestamp, .timestamp]) | |
| 374 | } | |
| 375 | ||
| 376 | @Test func itemsSplitBulletAndCheckbox() { | |
| 377 | let item = OrgParser.parse(" - [X] done *now*\n more\n").root.descendants().first { $0.kind == .item }! | |
| 378 | #expect(item.tokens.map(\.kind) == [.whitespace, .bullet, .whitespace, .checkbox, .whitespace]) | |
| 379 | #expect(item.children.map(\.kind) == [.paragraph]) | |
| 380 | #expect(item.children[0].text == "done *now*\n more\n") | |
| 381 | } | |
| 382 | ||
| 383 | @Test func emptyItem() { | |
| 384 | let item = OrgParser.parse("-\n").root.descendants().first { $0.kind == .item }! | |
| 385 | #expect(item.tokens.map(\.kind) == [.bullet, .newline]) | |
| 386 | } | |
| 387 | ||
| 388 | @Test func tableCells() { | |
| 389 | let rows = OrgParser.parse("| *a* || b\n|---+---|\n").root.descendants().filter { $0.kind == .tableRow } | |
| 390 | #expect(rows[0].children.map(\.text) == [" *a* ", "", " b"]) | |
| 391 | #expect(rows[0].children[0].children.map(\.kind) == [.bold]) | |
| 392 | #expect(rows[1].children.isEmpty) | |
| 393 | } | |
| 394 | ||
| 395 | @Test func footnoteDefinitionLabelIsNotAReference() { | |
| 396 | let definition = OrgParser.parse("[fn:1] see [fn:2]\n").root.descendants().first { $0.kind == .footnoteDefinition }! | |
| 397 | #expect(definition.tokens.first?.text == "[fn:1]") | |
| 398 | #expect(definition.children.map(\.kind) == [.footnoteReference]) | |
| 399 | } | |
| 400 | } | |
| 401 | ``` | |
| 402 | ||
| 403 | Update the two `ParserSectionTests` expectations that assumed `title` was a token: | |
| 404 | ||
| 405 | ```diff | |
| 406 | @@ -17,7 +17,7 @@ struct ParserSectionTests { | |
| 407 | } | |
| 408 | ||
| 409 | @Test func zerothSectionHoldsPreamble() { | |
| 410 | - #expect(nodeKinds("text\n* a\n") == [.document, .zerothSection, .paragraph, .section, .heading]) | |
| 411 | + #expect(nodeKinds("text\n* a\n") == [.document, .zerothSection, .paragraph, .section, .heading, .title]) | |
| 412 | } | |
| 413 | ||
| 414 | @Test func sectionsNestByLevel() { | |
| 415 | @@ -30,10 +30,11 @@ struct ParserSectionTests { | |
| 416 | } | |
| 417 | ||
| 418 | @Test func headingTokens() { | |
| 419 | - let parts = tokens(of: .heading, in: "** TODO [#A] Write the plan :work:urgent: \n") | |
| 420 | - #expect(parts.map(\.kind) == [.stars, .whitespace, .todoKeyword, .whitespace, .priority, .whitespace, .title, .whitespace, .tags, .whitespace, .newline]) | |
| 421 | + let text = "** TODO [#A] Write the plan :work:urgent: \n" | |
| 422 | + let parts = tokens(of: .heading, in: text) | |
| 423 | + #expect(parts.map(\.kind) == [.stars, .whitespace, .todoKeyword, .whitespace, .priority, .whitespace, .whitespace, .tags, .whitespace, .newline]) | |
| 424 | #expect(parts.first { $0.kind == .tags }?.text == ":work:urgent:") | |
| 425 | - #expect(parts.first { $0.kind == .title }?.text == "Write the plan") | |
| 426 | + #expect(OrgParser.parse(text).root.descendants().first { $0.kind == .title }?.text == "Write the plan") | |
| 427 | } | |
| 428 | ||
| 429 | @Test func todoKeywordsComeFromSettings() { | |
| 430 | ``` | |
| 431 | ||
| 432 | - [ ] **Step 2: Run to verify failure** | |
| 433 | ||
| 434 | Run: `swift test --filter InlineTests` | |
| 435 | Expected: build failure on the new `SyntaxKind` cases. | |
| 436 | ||
| 437 | - [ ] **Step 3: Replace `SyntaxKind.swift`** | |
| 438 | ||
| 439 | ```swift | |
| 440 | public enum SyntaxKind: String, Sendable { | |
| 441 | // Tokens | |
| 442 | case text, newline, whitespace | |
| 443 | case stars, todoKeyword, priority, tags | |
| 444 | case marker, linkPath, bullet, checkbox | |
| 445 | ||
| 446 | // Elements | |
| 447 | case document, zerothSection, section, heading, title | |
| 448 | case planning, propertyDrawer, nodeProperty, drawer, clock | |
| 449 | case paragraph, plainList, item, table, tableRow, tableCell, tableFormula | |
| 450 | case block, dynamicBlock, keyword, affiliatedKeyword | |
| 451 | case comment, fixedWidth, horizontalRule, footnoteDefinition | |
| 452 | ||
| 453 | // Objects | |
| 454 | case bold, italic, underline, strikeThrough, verbatim, code | |
| 455 | case link, linkDescription, timestamp, footnoteReference, statisticsCookie | |
| 456 | case target, macro, inlineSourceBlock, latexFragment, lineBreak, superscript | |
| 457 | } | |
| 458 | ``` | |
| 459 | ||
| 460 | - [ ] **Step 4: Create `Inline.swift`** | |
| 461 | ||
| 462 | ```swift | |
| 463 | /// Characters allowed before an emphasis opener, besides whitespace and the start of the run. | |
| 464 | private let emphasisPre: Set<Character> = ["-", "(", "{", "'", "\""] | |
| 465 | ||
| 466 | /// Characters allowed after an emphasis closer, besides whitespace and the end of the run. | |
| 467 | private let emphasisPost: Set<Character> = ["-", ".", ",", ":", "!", "?", ";", "'", "\"", ")", "}", "\\", "["] | |
| 468 | ||
| 469 | private let emphasisKinds: [Character: SyntaxKind] = [ | |
| 470 | "*": .bold, "/": .italic, "_": .underline, "+": .strikeThrough, "=": .verbatim, "~": .code, | |
| 471 | ] | |
| 472 | ||
| 473 | private let angleLinkSchemes: Set<String> = [ | |
| 474 | "http", "https", "mailto", "file", "id", "doi", "ftp", "news", "shell", "elisp", "info", "help", "attachment", | |
| 475 | ] | |
| 476 | ||
| 477 | private let plainLinkPrefixes = ["https://", "http://", "mailto:", "file:"] | |
| 478 | ||
| 479 | /// "\r\n" is a single Character, so both forms count. | |
| 480 | func isNewline(_ c: Character) -> Bool { | |
| 481 | c == "\n" || c == "\r\n" | |
| 482 | } | |
| 483 | ||
| 484 | enum InlineMatch { | |
| 485 | case emphasis(SyntaxKind, open: Int, close: Int) | |
| 486 | case link(path: Range<Int>, description: Range<Int>?, whole: Range<Int>) | |
| 487 | case object(SyntaxKind, Range<Int>) | |
| 488 | ||
| 489 | var end: Int { | |
| 490 | switch self { | |
| 491 | case .emphasis(_, _, let close): close + 1 | |
| 492 | case .link(_, _, let whole): whole.upperBound | |
| 493 | case .object(_, let range): range.upperBound | |
| 494 | } | |
| 495 | } | |
| 496 | } | |
| 497 | ||
| 498 | /// Turns a run of text into text, newline and object tokens. Every character ends up in | |
| 499 | /// exactly one token. | |
| 500 | struct InlineScanner { | |
| 501 | let chars: [Character] | |
| 502 | ||
| 503 | func scan(_ range: Range<Int>, into b: inout GreenBuilder, inLink: Bool = false) { | |
| 504 | var textStart = range.lowerBound | |
| 505 | var i = range.lowerBound | |
| 506 | while i < range.upperBound { | |
| 507 | if let match = match(at: i, in: range, inLink: inLink) { | |
| 508 | emitText(textStart..<i, into: &b) | |
| 509 | emit(match, into: &b, inLink: inLink) | |
| 510 | i = match.end | |
| 511 | textStart = i | |
| 512 | } else { | |
| 513 | i += 1 | |
| 514 | } | |
| 515 | } | |
| 516 | emitText(textStart..<range.upperBound, into: &b) | |
| 517 | } | |
| 518 | ||
| 519 | func emitText(_ range: Range<Int>, into b: inout GreenBuilder) { | |
| 520 | var start = range.lowerBound | |
| 521 | for k in range where isNewline(chars[k]) { | |
| 522 | if start < k { b.token(.text, string(start..<k)) } | |
| 523 | b.token(.newline, String(chars[k])) | |
| 524 | start = k + 1 | |
| 525 | } | |
| 526 | if start < range.upperBound { b.token(.text, string(start..<range.upperBound)) } | |
| 527 | } | |
| 528 | ||
| 529 | func emit(_ match: InlineMatch, into b: inout GreenBuilder, inLink: Bool) { | |
| 530 | switch match { | |
| 531 | case .emphasis(let kind, let open, let close): | |
| 532 | b.start(kind) | |
| 533 | b.token(.marker, string(open..<(open + 1))) | |
| 534 | if kind == .verbatim || kind == .code { | |
| 535 | emitText((open + 1)..<close, into: &b) | |
| 536 | } else { | |
| 537 | scan((open + 1)..<close, into: &b, inLink: inLink) | |
| 538 | } | |
| 539 | b.token(.marker, string(close..<(close + 1))) | |
| 540 | b.finish() | |
| 541 | case .link(let path, let description, let whole): | |
| 542 | b.start(.link) | |
| 543 | b.token(.marker, string(whole.lowerBound..<path.lowerBound)) | |
| 544 | b.token(.linkPath, string(path)) | |
| 545 | if let description { | |
| 546 | b.token(.marker, string(path.upperBound..<description.lowerBound)) | |
| 547 | b.start(.linkDescription) | |
| 548 | scan(description, into: &b, inLink: true) | |
| 549 | b.finish() | |
| 550 | b.token(.marker, string(description.upperBound..<whole.upperBound)) | |
| 551 | } else { | |
| 552 | b.token(.marker, string(path.upperBound..<whole.upperBound)) | |
| 553 | } | |
| 554 | b.finish() | |
| 555 | case .object(let kind, let range): | |
| 556 | b.start(kind) | |
| 557 | emitText(range, into: &b) | |
| 558 | b.finish() | |
| 559 | } | |
| 560 | } | |
| 561 | ||
| 562 | func string(_ range: Range<Int>) -> String { | |
| 563 | String(chars[range]) | |
| 564 | } | |
| 565 | ||
| 566 | func hasPrefix(_ s: String, at i: Int, _ limit: Int) -> Bool { | |
| 567 | var j = i | |
| 568 | for c in s { | |
| 569 | guard j < limit, chars[j] == c else { return false } | |
| 570 | j += 1 | |
| 571 | } | |
| 572 | return true | |
| 573 | } | |
| 574 | ||
| 575 | // MARK: - Recognizers | |
| 576 | ||
| 577 | func match(at i: Int, in range: Range<Int>, inLink: Bool) -> InlineMatch? { | |
| 578 | let limit = range.upperBound | |
| 579 | let previous: Character? = i > range.lowerBound ? chars[i - 1] : nil | |
| 580 | let afterWord = previous.map { $0.isLetter || $0.isNumber } ?? false | |
| 581 | ||
| 582 | switch chars[i] { | |
| 583 | case "[": | |
| 584 | if !inLink, let link = bracketLink(i, limit) { return link } | |
| 585 | if let end = footnoteReference(i, limit) { return .object(.footnoteReference, i..<end) } | |
| 586 | if let end = scanTimestamp(chars, at: i, limit: limit)?.end { return .object(.timestamp, i..<end) } | |
| 587 | if let end = statisticsCookie(i, limit) { return .object(.statisticsCookie, i..<end) } | |
| 588 | case "<": | |
| 589 | if let end = scanTimestamp(chars, at: i, limit: limit)?.end { return .object(.timestamp, i..<end) } | |
| 590 | if let end = target(i, limit) { return .object(.target, i..<end) } | |
| 591 | if !inLink, let end = angleLink(i, limit) { return .object(.link, i..<end) } | |
| 592 | case "{": | |
| 593 | if let end = macro(i, limit) { return .object(.macro, i..<end) } | |
| 594 | case "\\": | |
| 595 | if let end = lineBreak(i, limit) { return .object(.lineBreak, i..<end) } | |
| 596 | if let end = latexFragment(i, limit) { return .object(.latexFragment, i..<end) } | |
| 597 | case "^": | |
| 598 | if afterWord, let end = superscript(i, limit) { return .object(.superscript, i..<end) } | |
| 599 | case "s": | |
| 600 | if !afterWord, let end = inlineSourceBlock(i, limit) { return .object(.inlineSourceBlock, i..<end) } | |
| 601 | default: | |
| 602 | break | |
| 603 | } | |
| 604 | ||
| 605 | if !inLink, !afterWord, let end = plainLink(i, limit) { return .object(.link, i..<end) } | |
| 606 | ||
| 607 | if let kind = emphasisKinds[chars[i]], | |
| 608 | previous.map({ $0.isWhitespace || emphasisPre.contains($0) }) ?? true, | |
| 609 | let close = emphasisClose(i, limit) { | |
| 610 | return .emphasis(kind, open: i, close: close) | |
| 611 | } | |
| 612 | return nil | |
| 613 | } | |
| 614 | ||
| 615 | /// org's emphasis rules: the body neither starts nor ends with whitespace, spans at most | |
| 616 | /// one line break, and the closer is followed by whitespace, punctuation or the end. | |
| 617 | func emphasisClose(_ i: Int, _ limit: Int) -> Int? { | |
| 618 | let marker = chars[i] | |
| 619 | guard i + 1 < limit, !chars[i + 1].isWhitespace else { return nil } | |
| 620 | var newlines = 0 | |
| 621 | var j = i + 1 | |
| 622 | while j < limit { | |
| 623 | if isNewline(chars[j]) { | |
| 624 | newlines += 1 | |
| 625 | if newlines > 1 { return nil } | |
| 626 | } else if chars[j] == marker, j > i + 1, !chars[j - 1].isWhitespace { | |
| 627 | if j + 1 == limit || chars[j + 1].isWhitespace || emphasisPost.contains(chars[j + 1]) { return j } | |
| 628 | } | |
| 629 | j += 1 | |
| 630 | } | |
| 631 | return nil | |
| 632 | } | |
| 633 | ||
| 634 | /// `[[path]]` or `[[path][description]]`. | |
| 635 | func bracketLink(_ i: Int, _ limit: Int) -> InlineMatch? { | |
| 636 | guard i + 1 < limit, chars[i + 1] == "[" else { return nil } | |
| 637 | var j = i + 2 | |
| 638 | while j < limit, chars[j] != "]" { | |
| 639 | if chars[j] == "[" || isNewline(chars[j]) { return nil } | |
| 640 | if chars[j] == "\\", j + 1 < limit { j += 1 } | |
| 641 | j += 1 | |
| 642 | } | |
| 643 | guard j > i + 2, j + 1 < limit else { return nil } | |
| 644 | let path = (i + 2)..<j | |
| 645 | if chars[j + 1] == "]" { return .link(path: path, description: nil, whole: i..<(j + 2)) } | |
| 646 | guard chars[j + 1] == "[" else { return nil } | |
| 647 | let descriptionStart = j + 2 | |
| 648 | var depth = 0 | |
| 649 | var k = descriptionStart | |
| 650 | while k < limit { | |
| 651 | if chars[k] == "[" { | |
| 652 | depth += 1 | |
| 653 | } else if chars[k] == "]" { | |
| 654 | if depth == 0 { break } | |
| 655 | depth -= 1 | |
| 656 | } | |
| 657 | k += 1 | |
| 658 | } | |
| 659 | guard k > descriptionStart, k + 1 < limit, chars[k + 1] == "]" else { return nil } | |
| 660 | return .link(path: path, description: descriptionStart..<k, whole: i..<(k + 2)) | |
| 661 | } | |
| 662 | ||
| 663 | /// `[fn:label]`, `[fn:label:definition]` or `[fn::definition]`. | |
| 664 | func footnoteReference(_ i: Int, _ limit: Int) -> Int? { | |
| 665 | guard hasPrefix("[fn:", at: i, limit) else { return nil } | |
| 666 | var j = i + 4 | |
| 667 | while j < limit, chars[j].isLetter || chars[j].isNumber || chars[j] == "_" || chars[j] == "-" { j += 1 } | |
| 668 | guard j < limit else { return nil } | |
| 669 | if chars[j] == "]" { return j > i + 4 ? j + 1 : nil } | |
| 670 | guard chars[j] == ":" else { return nil } | |
| 671 | var depth = 0 | |
| 672 | j += 1 | |
| 673 | while j < limit { | |
| 674 | if chars[j] == "[" { | |
| 675 | depth += 1 | |
| 676 | } else if chars[j] == "]" { | |
| 677 | if depth == 0 { return j + 1 } | |
| 678 | depth -= 1 | |
| 679 | } | |
| 680 | j += 1 | |
| 681 | } | |
| 682 | return nil | |
| 683 | } | |
| 684 | ||
| 685 | /// `[1/3]`, `[/]`, `[50%]` or `[%]`. | |
| 686 | func statisticsCookie(_ i: Int, _ limit: Int) -> Int? { | |
| 687 | var j = i + 1 | |
| 688 | while j < limit, chars[j].isASCII, chars[j].isNumber { j += 1 } | |
| 689 | guard j < limit else { return nil } | |
| 690 | if chars[j] == "%" { | |
| 691 | j += 1 | |
| 692 | } else if chars[j] == "/" { | |
| 693 | j += 1 | |
| 694 | while j < limit, chars[j].isASCII, chars[j].isNumber { j += 1 } | |
| 695 | } else { | |
| 696 | return nil | |
| 697 | } | |
| 698 | guard j < limit, chars[j] == "]" else { return nil } | |
| 699 | return j + 1 | |
| 700 | } | |
| 701 | ||
| 702 | /// `<<target>>`. | |
| 703 | func target(_ i: Int, _ limit: Int) -> Int? { | |
| 704 | guard hasPrefix("<<", at: i, limit), i + 2 < limit, chars[i + 2] != "<" else { return nil } | |
| 705 | let start = i + 2 | |
| 706 | var j = start | |
| 707 | while j < limit, chars[j] != ">" { | |
| 708 | if chars[j] == "<" || isNewline(chars[j]) { return nil } | |
| 709 | j += 1 | |
| 710 | } | |
| 711 | guard j > start, j + 1 < limit, chars[j + 1] == ">", | |
| 712 | !chars[start].isWhitespace, !chars[j - 1].isWhitespace else { return nil } | |
| 713 | return j + 2 | |
| 714 | } | |
| 715 | ||
| 716 | /// `<scheme:path>` for a known scheme. | |
| 717 | func angleLink(_ i: Int, _ limit: Int) -> Int? { | |
| 718 | var j = i + 1 | |
| 719 | while j < limit, chars[j].isLetter { j += 1 } | |
| 720 | guard j < limit, chars[j] == ":", angleLinkSchemes.contains(string((i + 1)..<j).lowercased()) else { return nil } | |
| 721 | j += 1 | |
| 722 | let bodyStart = j | |
| 723 | while j < limit, chars[j] != ">" { | |
| 724 | if chars[j] == "<" || isNewline(chars[j]) { return nil } | |
| 725 | j += 1 | |
| 726 | } | |
| 727 | guard j < limit, j > bodyStart else { return nil } | |
| 728 | return j + 1 | |
| 729 | } | |
| 730 | ||
| 731 | /// A bare URL. Trailing sentence punctuation stays outside the link. | |
| 732 | func plainLink(_ i: Int, _ limit: Int) -> Int? { | |
| 733 | guard "hmf".contains(chars[i]), | |
| 734 | let prefix = plainLinkPrefixes.first(where: { hasPrefix($0, at: i, limit) }) else { return nil } | |
| 735 | let bodyStart = i + prefix.count | |
| 736 | var j = bodyStart | |
| 737 | while j < limit, !chars[j].isWhitespace, !"()<>[]\"".contains(chars[j]) { j += 1 } | |
| 738 | while j > bodyStart, ".,;:!?'".contains(chars[j - 1]) { j -= 1 } | |
| 739 | return j > bodyStart ? j : nil | |
| 740 | } | |
| 741 | ||
| 742 | /// `{{{name}}}` or `{{{name(arguments)}}}`. | |
| 743 | func macro(_ i: Int, _ limit: Int) -> Int? { | |
| 744 | guard hasPrefix("{{{", at: i, limit) else { return nil } | |
| 745 | var j = i + 3 | |
| 746 | guard j < limit, chars[j].isLetter else { return nil } | |
| 747 | while j < limit, chars[j].isLetter || chars[j].isNumber || chars[j] == "-" || chars[j] == "_" { j += 1 } | |
| 748 | if j < limit, chars[j] == "(" { | |
| 749 | var k = j + 1 | |
| 750 | while k < limit, !hasPrefix(")}}}", at: k, limit) { | |
| 751 | if isNewline(chars[k]) { return nil } | |
| 752 | k += 1 | |
| 753 | } | |
| 754 | return k < limit ? k + 4 : nil | |
| 755 | } | |
| 756 | return hasPrefix("}}}", at: j, limit) ? j + 3 : nil | |
| 757 | } | |
| 758 | ||
| 759 | /// `\\` at the end of a line, before optional trailing blanks. The newline stays outside. | |
| 760 | func lineBreak(_ i: Int, _ limit: Int) -> Int? { | |
| 761 | guard hasPrefix("\\\\", at: i, limit) else { return nil } | |
| 762 | var j = i + 2 | |
| 763 | while j < limit, chars[j] == " " || chars[j] == "\t" { j += 1 } | |
| 764 | guard j == limit || isNewline(chars[j]) else { return nil } | |
| 765 | return j | |
| 766 | } | |
| 767 | ||
| 768 | /// `\(...\)` or `\[...\]`. | |
| 769 | func latexFragment(_ i: Int, _ limit: Int) -> Int? { | |
| 770 | guard i + 1 < limit else { return nil } | |
| 771 | let closer: String | |
| 772 | switch chars[i + 1] { | |
| 773 | case "(": closer = "\\)" | |
| 774 | case "[": closer = "\\]" | |
| 775 | default: return nil | |
| 776 | } | |
| 777 | var j = i + 2 | |
| 778 | while j < limit { | |
| 779 | if hasPrefix(closer, at: j, limit) { return j + 2 } | |
| 780 | j += 1 | |
| 781 | } | |
| 782 | return nil | |
| 783 | } | |
| 784 | ||
| 785 | /// `^word` or `^{group}` after a letter or digit. | |
| 786 | func superscript(_ i: Int, _ limit: Int) -> Int? { | |
| 787 | var j = i + 1 | |
| 788 | guard j < limit else { return nil } | |
| 789 | if chars[j] == "{" { | |
| 790 | j += 1 | |
| 791 | while j < limit, chars[j] != "}" { | |
| 792 | if isNewline(chars[j]) { return nil } | |
| 793 | j += 1 | |
| 794 | } | |
| 795 | return j < limit ? j + 1 : nil | |
| 796 | } | |
| 797 | let start = j | |
| 798 | while j < limit, chars[j].isLetter || chars[j].isNumber { j += 1 } | |
| 799 | return j > start ? j : nil | |
| 800 | } | |
| 801 | ||
| 802 | /// `src_lang{body}` or `src_lang[headers]{body}`, on one line, with balanced braces. | |
| 803 | func inlineSourceBlock(_ i: Int, _ limit: Int) -> Int? { | |
| 804 | guard hasPrefix("src_", at: i, limit) else { return nil } | |
| 805 | var j = i + 4 | |
| 806 | let languageStart = j | |
| 807 | while j < limit, !chars[j].isWhitespace, chars[j] != "[", chars[j] != "{" { j += 1 } | |
| 808 | guard j > languageStart, j < limit else { return nil } | |
| 809 | if chars[j] == "[" { | |
| 810 | while j < limit, chars[j] != "]" { | |
| 811 | if isNewline(chars[j]) { return nil } | |
| 812 | j += 1 | |
| 813 | } | |
| 814 | guard j < limit else { return nil } | |
| 815 | j += 1 | |
| 816 | } | |
| 817 | guard j < limit, chars[j] == "{" else { return nil } | |
| 818 | var depth = 0 | |
| 819 | while j < limit { | |
| 820 | if isNewline(chars[j]) { return nil } | |
| 821 | if chars[j] == "{" { | |
| 822 | depth += 1 | |
| 823 | } else if chars[j] == "}" { | |
| 824 | depth -= 1 | |
| 825 | if depth == 0 { return j + 1 } | |
| 826 | } | |
| 827 | j += 1 | |
| 828 | } | |
| 829 | return nil | |
| 830 | } | |
| 831 | } | |
| 832 | ||
| 833 | extension Parser { | |
| 834 | mutating func inline(_ text: Substring) { | |
| 835 | let chars = Array(text) | |
| 836 | InlineScanner(chars: chars).scan(0..<chars.count, into: &builder) | |
| 837 | } | |
| 838 | } | |
| 839 | ``` | |
| 840 | ||
| 841 | - [ ] **Step 5: Wire the scanner into the parser** | |
| 842 | ||
| 843 | ```diff | |
| 844 | diff --git a/Sources/OrgCore/Parser/Lines.swift b/Sources/OrgCore/Parser/Lines.swift | |
| 845 | index 32a778a..5e9a077 100644 | |
| 846 | --- a/Sources/OrgCore/Parser/Lines.swift | |
| 847 | +++ b/Sources/OrgCore/Parser/Lines.swift | |
| 848 | @@ -27,7 +27,7 @@ func splitRawLines(_ text: String) -> [RawLine] { | |
| 849 | } | |
| 850 | } | |
| 851 | if lineStart != scalars.endIndex { | |
| 852 | - lines.append(RawLine(content: text[lineStart...], ending: "")) | |
| 853 | + lines.append(RawLine(content: text[lineStart...], ending: text[text.endIndex...])) | |
| 854 | } | |
| 855 | return lines | |
| 856 | } | |
| 857 | diff --git a/Sources/OrgCore/Parser/Parser.swift b/Sources/OrgCore/Parser/Parser.swift | |
| 858 | index ac210ff..6460486 100644 | |
| 859 | --- a/Sources/OrgCore/Parser/Parser.swift | |
| 860 | +++ b/Sources/OrgCore/Parser/Parser.swift | |
| 861 | @@ -77,10 +77,7 @@ struct Parser { | |
| 862 | headingLine(lines[i]) | |
| 863 | i += 1 | |
| 864 | if i < lines.count, info[i].cls == .planning { | |
| 865 | - builder.start(.planning) | |
| 866 | - line(i) | |
| 867 | - i += 1 | |
| 868 | - builder.finish() | |
| 869 | + single(.planning) | |
| 870 | } | |
| 871 | if i < lines.count, case .drawerBegin(let name) = info[i].cls, name.uppercased() == "PROPERTIES", | |
| 872 | let end = blockEnds[i] { | |
| 873 | @@ -173,13 +170,28 @@ struct Parser { | |
| 874 | } | |
| 875 | } | |
| 876 | ||
| 877 | + /// Kinds whose single line holds inline objects (timestamps on planning and clock lines). | |
| 878 | + static let inlineLineKinds: Set<SyntaxKind> = [.planning, .clock] | |
| 879 | + | |
| 880 | mutating func single(_ kind: SyntaxKind) { | |
| 881 | builder.start(kind) | |
| 882 | - line(i) | |
| 883 | + if Self.inlineLineKinds.contains(kind) { | |
| 884 | + let rest = whitespace(lines[i].content) | |
| 885 | + inline(rest) | |
| 886 | + builder.token(.newline, lines[i].ending) | |
| 887 | + } else { | |
| 888 | + line(i) | |
| 889 | + } | |
| 890 | i += 1 | |
| 891 | builder.finish() | |
| 892 | } | |
| 893 | ||
| 894 | + /// Source text from `start` through the end of line `last`, including its line ending. | |
| 895 | + func span(from start: Substring.Index, through last: Int) -> Substring { | |
| 896 | + let base = lines[last].ending.base | |
| 897 | + return base[start..<lines[last].ending.endIndex] | |
| 898 | + } | |
| 899 | + | |
| 900 | mutating func consecutive(_ kind: SyntaxKind, limit: Int, floor: Int?, matching: (LineClass) -> Bool) { | |
| 901 | builder.start(kind) | |
| 902 | repeat { | |
| 903 | @@ -192,7 +204,7 @@ struct Parser { | |
| 904 | mutating func table(limit: Int, floor: Int?) { | |
| 905 | builder.start(.table) | |
| 906 | while i < limit, info[i].cls == .tableRow, within(floor, i) { | |
| 907 | - single(.tableRow) | |
| 908 | + tableRow() | |
| 909 | } | |
| 910 | while i < limit, info[i].cls == .keyword(key: "TBLFM"), within(floor, i) { | |
| 911 | single(.tableFormula) | |
| 912 | @@ -200,15 +212,47 @@ struct Parser { | |
| 913 | builder.finish() | |
| 914 | } | |
| 915 | ||
| 916 | - mutating func footnoteDefinition(limit: Int) { | |
| 917 | - builder.start(.footnoteDefinition) | |
| 918 | - line(i) | |
| 919 | - i += 1 | |
| 920 | - while i < limit, info[i].cls == .plain { | |
| 921 | - line(i) | |
| 922 | - i += 1 | |
| 923 | + /// A rule row (`|---+---|`) is one text token. Other rows alternate `|` markers and cells; | |
| 924 | + /// every pair of pipes gets a cell, even an empty one, so columns line up. | |
| 925 | + mutating func tableRow() { | |
| 926 | + builder.start(.tableRow) | |
| 927 | + let rest = whitespace(lines[i].content) | |
| 928 | + if rest.hasPrefix("|-") { | |
| 929 | + builder.token(.text, rest) | |
| 930 | + } else { | |
| 931 | + var cellStart = rest.startIndex | |
| 932 | + var index = rest.startIndex | |
| 933 | + while index < rest.endIndex { | |
| 934 | + if rest[index] == "|" { | |
| 935 | + if index > rest.startIndex { tableCell(rest[cellStart..<index]) } | |
| 936 | + builder.token(.marker, "|") | |
| 937 | + cellStart = rest.index(after: index) | |
| 938 | + } | |
| 939 | + index = rest.index(after: index) | |
| 940 | + } | |
| 941 | + if cellStart < rest.endIndex { tableCell(rest[cellStart...]) } | |
| 942 | } | |
| 943 | + builder.token(.newline, lines[i].ending) | |
| 944 | builder.finish() | |
| 945 | + i += 1 | |
| 946 | + } | |
| 947 | + | |
| 948 | + mutating func tableCell(_ text: Substring) { | |
| 949 | + builder.start(.tableCell) | |
| 950 | + inline(text) | |
| 951 | + builder.finish() | |
| 952 | + } | |
| 953 | + | |
| 954 | + mutating func footnoteDefinition(limit: Int) { | |
| 955 | + var end = i + 1 | |
| 956 | + while end < limit, info[end].cls == .plain { end += 1 } | |
| 957 | + let content = lines[i].content | |
| 958 | + let close = content.firstIndex(of: "]")! | |
| 959 | + builder.start(.footnoteDefinition) | |
| 960 | + builder.token(.marker, content[...close]) | |
| 961 | + inline(span(from: content.index(after: close), through: end - 1)) | |
| 962 | + builder.finish() | |
| 963 | + i = end | |
| 964 | } | |
| 965 | ||
| 966 | mutating func list(limit: Int, floor: Int?) { | |
| 967 | @@ -224,8 +268,25 @@ struct Parser { | |
| 968 | /// inside the item when the item or list continues after it; two end the list. | |
| 969 | mutating func item(base: Int, limit: Int) { | |
| 970 | builder.start(.item) | |
| 971 | - line(i) | |
| 972 | - i += 1 | |
| 973 | + var rest = whitespace(lines[i].content) | |
| 974 | + let bullet = rest.prefix { $0 != " " && $0 != "\t" } | |
| 975 | + builder.token(.bullet, bullet) | |
| 976 | + rest = whitespace(rest.dropFirst(bullet.count)) | |
| 977 | + if let box = checkbox(rest) { | |
| 978 | + builder.token(.checkbox, box) | |
| 979 | + rest = whitespace(rest.dropFirst(box.count)) | |
| 980 | + } | |
| 981 | + // The rest of the bullet line and its continuation lines are the item's first paragraph. | |
| 982 | + var end = i + 1 | |
| 983 | + while end < limit, within(base, end), continuesParagraph(end) { end += 1 } | |
| 984 | + if rest.isEmpty, end == i + 1 { | |
| 985 | + builder.token(.newline, lines[i].ending) | |
| 986 | + } else { | |
| 987 | + builder.start(.paragraph) | |
| 988 | + inline(span(from: rest.startIndex, through: end - 1)) | |
| 989 | + builder.finish() | |
| 990 | + } | |
| 991 | + i = end | |
| 992 | while i < limit, !isHeading(i) { | |
| 993 | if info[i].cls == .blank { | |
| 994 | var j = i | |
| 995 | @@ -245,15 +306,23 @@ struct Parser { | |
| 996 | builder.finish() | |
| 997 | } | |
| 998 | ||
| 999 | + /// `[ ]`, `[X]`, `[x]` or `[-]`, followed by whitespace or end of line. | |
| 1000 | + func checkbox(_ s: Substring) -> Substring? { | |
| 1001 | + guard s.count >= 3, s.first == "[", "Xx -".contains(s.dropFirst().first!), | |
| 1002 | + s.dropFirst(2).first == "]" else { return nil } | |
| 1003 | + let after = s.dropFirst(3) | |
| 1004 | + guard after.isEmpty || after.first == " " || after.first == "\t" else { return nil } | |
| 1005 | + return s.prefix(3) | |
| 1006 | + } | |
| 1007 | + | |
| 1008 | + /// The paragraph's lines are one inline run, so emphasis and links can cross a line break. | |
| 1009 | mutating func paragraph(limit: Int, floor: Int?) { | |
| 1010 | + var end = i + 1 | |
| 1011 | + while end < limit, within(floor, end), continuesParagraph(end) { end += 1 } | |
| 1012 | builder.start(.paragraph) | |
| 1013 | - line(i) | |
| 1014 | - i += 1 | |
| 1015 | - while i < limit, within(floor, i), continuesParagraph(i) { | |
| 1016 | - line(i) | |
| 1017 | - i += 1 | |
| 1018 | - } | |
| 1019 | + inline(span(from: lines[i].content.startIndex, through: end - 1)) | |
| 1020 | builder.finish() | |
| 1021 | + i = end | |
| 1022 | } | |
| 1023 | ||
| 1024 | /// Lines that don't start an element of their own. | |
| 1025 | @@ -307,7 +376,11 @@ struct Parser { | |
| 1026 | } | |
| 1027 | ||
| 1028 | let parts = splitTags(rest) | |
| 1029 | - builder.token(.title, parts.title) | |
| 1030 | + if !parts.title.isEmpty { | |
| 1031 | + builder.start(.title) | |
| 1032 | + inline(parts.title) | |
| 1033 | + builder.finish() | |
| 1034 | + } | |
| 1035 | builder.token(.whitespace, parts.gap) | |
| 1036 | builder.token(.tags, parts.tags) | |
| 1037 | builder.token(.whitespace, parts.trailing) | |
| 1038 | ``` | |
| 1039 | ||
| 1040 | - [ ] **Step 6: Run all tests** | |
| 1041 | ||
| 1042 | Run: `swift test` | |
| 1043 | Expected: all pass. | |
| 1044 | ||
| 1045 | - [ ] **Step 7: Commit** | |
| 1046 | ||
| 1047 | ```bash | |
| 1048 | git add Sources Tests | |
| 1049 | git commit -m "Parse inline objects" | |
| 1050 | ``` | |
| 1051 | ||
| 1052 | --- | |
| 1053 | ||
| 1054 | ### Task 3: Fuzz the inline layer | |
| 1055 | ||
| 1056 | **Files:** | |
| 1057 | - Modify: `Tests/OrgCoreTests/RoundTripTests.swift` (fragment list) | |
| 1058 | ||
| 1059 | - [ ] **Step 1: Add inline fragments** | |
| 1060 | ||
| 1061 | ```diff | |
| 1062 | @@ -20,6 +20,10 @@ let fragments = [ | |
| 1063 | ":LOGBOOK:", "CLOCK: [2026-10-04 Sun 10:00]", "SCHEDULED: <2026-10-04 Sun>", "- item", " - nested", | |
| 1064 | "\t+ tab", "1. one", "| a | b |", "|---+---|", "#+TBLFM: $2=$1", "# comment", ": fixed", "-----", | |
| 1065 | "[fn:1] note", "#+NAME: x", "plain text", "é", "😀", "e\u{301}", " ", "\t", "\n", "\n", "\r\n", "\r", "", | |
| 1066 | + "*bold*", "/it/ ", "=v=", "~c~", "_u_", "+s+", "*", "/", "=", "[[https://a.b][d *b*]]", "[[x]]", "[[", "]]", | |
| 1067 | + "<2026-10-04 Sun 10:00 +1w -2d>", "[2026-10-04]--[2026-10-05]", "<", ">", "[fn:2]", "[fn::in [x]]", "[1/3]", | |
| 1068 | + "[50%]", "<<t>>", "{{{m(a)}}}", "\\(x\\)", "a\\\\", "x^2", "^{y}", "src_sh{echo}", "https://e.com.", | |
| 1069 | + "<https://e.com>", "| *a* | b |", "||", "- [X] done", "+ [ ] todo", "-", | |
| 1070 | ] | |
| 1071 | ||
| 1072 | func randomDocument(_ rng: inout SeededGenerator) -> String { | |
| 1073 | ``` | |
| 1074 | ||
| 1075 | - [ ] **Step 2: Run the fuzz test and the private corpus** | |
| 1076 | ||
| 1077 | Run: `swift test` then `ORGSTAR_CORPUS=~/Documents/notes swift test --filter corpusRoundTrips` | |
| 1078 | Expected: all pass. | |
| 1079 | ||
| 1080 | - [ ] **Step 3: Commit** | |
| 1081 | ||
| 1082 | ```bash | |
| 1083 | git add Tests | |
| 1084 | git commit -m "Fuzz inline objects" | |
| 1085 | ``` | |