Commit 6fb6b30dba
Verified · cmc
Layout: unified · split
docs/plans/2026-10-04-inline-objects.md added +1085
| @@ -0,0 +1,1085 @@ | |||
| 1 | # Inline Objects Implementation Plan | ||
| 2 | |||
| 3 | > **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. | ||
| 4 | |||
| 5 | **Goal:** Parse org's inline objects into the lossless tree: emphasis, links, timestamps, footnote references, statistics cookies, targets, macros, inline source blocks, LaTeX fragments, line breaks and superscripts. | ||
| 6 | |||
| 7 | **Architecture:** A character-level `InlineScanner` turns a run of text into text, newline and object nodes through the existing `GreenBuilder`. The parser feeds it whole paragraphs (so emphasis and links can cross one line break), heading titles, table cells, item paragraphs, footnote definitions, and planning and clock lines. `Timestamp` is both the recognizer and a public value type with repeaters and warnings. | ||
| 8 | |||
| 9 | **Tech Stack:** Swift 6.2 tools, Swift Testing, Foundation only. | ||
| 10 | |||
| 11 | **Spec:** `docs/design.md` ("OrgCore data model"); OrgSwift's `OrgInlineParser.swift` for the emphasis border rules. | ||
| 12 | |||
| 13 | ## Global Constraints | ||
| 14 | |||
| 15 | - Every character of the input lands in exactly one token; tree text equals source text. | ||
| 16 | - Emphasis follows org's defaults: opener preceded by start, whitespace or `-({'"`; closer followed by end, whitespace or `-.,:!?;'")}\[`; body neither starts nor ends with whitespace; at most one line break. | ||
| 17 | - `OrgCore` imports Foundation only. | ||
| 18 | |||
| 19 | ## Out of scope | ||
| 20 | |||
| 21 | Entities (`\alpha`), subscripts, `$...$` LaTeX, export snippets, citations, radio targets, inline footnote definitions as parsed content, and description-list terms. Each is recorded for a later plan. | ||
| 22 | |||
| 23 | ## File structure | ||
| 24 | |||
| 25 | | File | Responsibility | | ||
| 26 | | --- | --- | | ||
| 27 | | `Sources/OrgCore/Timestamp.swift` | `Timestamp` value type and `scanTimestamp` recognizer | | ||
| 28 | | `Sources/OrgCore/Parser/Inline.swift` | `InlineScanner` and `Parser.inline(_:)` | | ||
| 29 | | `Sources/OrgCore/Syntax/SyntaxKind.swift` | New token and object kinds; `title` becomes a node | | ||
| 30 | | `Sources/OrgCore/Parser/Parser.swift` | Paragraph spans, item bullets and checkboxes, table cells, footnote labels, inline planning and clock lines | | ||
| 31 | | `Sources/OrgCore/Parser/Lines.swift` | Empty final line ending sliced from the source, so spans can end there | | ||
| 32 | |||
| 33 | --- | ||
| 34 | |||
| 35 | ### Task 1: Timestamps | ||
| 36 | |||
| 37 | **Files:** | ||
| 38 | - Create: `Sources/OrgCore/Timestamp.swift` | ||
| 39 | - Test: `Tests/OrgCoreTests/TimestampTests.swift` | ||
| 40 | |||
| 41 | **Interfaces:** | ||
| 42 | - Produces: `Timestamp` (`active`, `start: Point`, `end: Point?`, `repeater: Repeater?`, `warning: Warning?`), `Timestamp.parse(_:) -> Timestamp?`; internal `scanTimestamp(_ chars: [Character], at: Int, limit: Int) -> (stamp: Timestamp, end: Int)?`. | ||
| 43 | |||
| 44 | - [ ] **Step 1: Write the failing tests** | ||
| 45 | |||
| 46 | ```swift | ||
| 47 | import Testing | ||
| 48 | @testable import OrgCore | ||
| 49 | |||
| 50 | struct TimestampTests { | ||
| 51 | @Test func fullTimestamp() throws { | ||
| 52 | let stamp = try #require(Timestamp.parse("<2026-10-04 Sun 10:00-11:30 .+1w/2w --2d>")) | ||
| 53 | #expect(stamp.active) | ||
| 54 | #expect(stamp.start == Timestamp.Point(year: 2026, month: 10, day: 4, hour: 10, minute: 0)) | ||
| 55 | #expect(stamp.end == Timestamp.Point(year: 2026, month: 10, day: 4, hour: 11, minute: 30)) | ||
| 56 | #expect(stamp.repeater == Timestamp.Repeater( | ||
| 57 | kind: .restart, | ||
| 58 | interval: .init(value: 1, unit: .week), | ||
| 59 | habitDeadline: .init(value: 2, unit: .week) | ||
| 60 | )) | ||
| 61 | #expect(stamp.warning == Timestamp.Warning(firstOccurrenceOnly: true, interval: .init(value: 2, unit: .day))) | ||
| 62 | } | ||
| 63 | |||
| 64 | @Test func dateOnlyInactive() throws { | ||
| 65 | let stamp = try #require(Timestamp.parse("[2026-10-04]")) | ||
| 66 | #expect(!stamp.active) | ||
| 67 | #expect(stamp.start.hour == nil) | ||
| 68 | } | ||
| 69 | |||
| 70 | @Test func dateRange() throws { | ||
| 71 | let stamp = try #require(Timestamp.parse("[2026-10-04 Sun]--[2026-10-06 Tue]")) | ||
| 72 | #expect(stamp.end?.day == 6) | ||
| 73 | } | ||
| 74 | |||
| 75 | @Test func repeaterKinds() { | ||
| 76 | #expect(Timestamp.parse("<2026-10-04 +1d>")?.repeater?.kind == .cumulate) | ||
| 77 | #expect(Timestamp.parse("<2026-10-04 ++1m>")?.repeater?.kind == .catchUp) | ||
| 78 | #expect(Timestamp.parse("<2026-10-04 -3d +1y>")?.repeater?.interval == .init(value: 1, unit: .year)) | ||
| 79 | } | ||
| 80 | |||
| 81 | @Test func oneDigitHourAndOtherLanguages() { | ||
| 82 | #expect(Timestamp.parse("<2026-10-04 9:05>")?.start.hour == 9) | ||
| 83 | #expect(Timestamp.parse("<2026-10-04 dim.>") != nil) | ||
| 84 | } | ||
| 85 | |||
| 86 | @Test(arguments: ["<2026-10-4>", "<2026-10-04 Sun", "[2026-10-04 Sun]x", "<2026-10-04 Sun +1x>", "<2026-10-04]", "2026-10-04"]) | ||
| 87 | func rejects(text: String) { | ||
| 88 | #expect(Timestamp.parse(text) == nil) | ||
| 89 | } | ||
| 90 | } | ||
| 91 | ``` | ||
| 92 | |||
| 93 | - [ ] **Step 2: Run to verify failure** | ||
| 94 | |||
| 95 | Run: `swift test --filter TimestampTests` | ||
| 96 | Expected: build failure, `cannot find 'Timestamp' in scope`. | ||
| 97 | |||
| 98 | - [ ] **Step 3: Implement** | ||
| 99 | |||
| 100 | ```swift | ||
| 101 | public struct Timestamp: Sendable, Equatable { | ||
| 102 | public enum Unit: Character, Sendable { | ||
| 103 | case hour = "h", day = "d", week = "w", month = "m", year = "y" | ||
| 104 | } | ||
| 105 | |||
| 106 | public struct Interval: Sendable, Equatable { | ||
| 107 | public var value: Int | ||
| 108 | public var unit: Unit | ||
| 109 | } | ||
| 110 | |||
| 111 | public enum RepeaterKind: String, Sendable { | ||
| 112 | case cumulate = "+", catchUp = "++", restart = ".+" | ||
| 113 | } | ||
| 114 | |||
| 115 | public struct Repeater: Sendable, Equatable { | ||
| 116 | public var kind: RepeaterKind | ||
| 117 | public var interval: Interval | ||
| 118 | /// The habit deadline from `.+2d/3d`. | ||
| 119 | public var habitDeadline: Interval? | ||
| 120 | } | ||
| 121 | |||
| 122 | public struct Warning: Sendable, Equatable { | ||
| 123 | /// `--` warns only for the first occurrence of a repeated timestamp. | ||
| 124 | public var firstOccurrenceOnly: Bool | ||
| 125 | public var interval: Interval | ||
| 126 | } | ||
| 127 | |||
| 128 | public struct Point: Sendable, Equatable { | ||
| 129 | public var year: Int | ||
| 130 | public var month: Int | ||
| 131 | public var day: Int | ||
| 132 | public var hour: Int? | ||
| 133 | public var minute: Int? | ||
| 134 | } | ||
| 135 | |||
| 136 | public var active: Bool | ||
| 137 | public var start: Point | ||
| 138 | /// End of a same-day time range or of a `--` date range. | ||
| 139 | public var end: Point? | ||
| 140 | public var repeater: Repeater? | ||
| 141 | public var warning: Warning? | ||
| 142 | |||
| 143 | /// Parses exactly one timestamp or range, with nothing before or after it. | ||
| 144 | public static func parse(_ text: some StringProtocol) -> Timestamp? { | ||
| 145 | let chars = Array(text) | ||
| 146 | guard let result = scanTimestamp(chars, at: 0, limit: chars.count), result.end == chars.count else { return nil } | ||
| 147 | return result.stamp | ||
| 148 | } | ||
| 149 | } | ||
| 150 | |||
| 151 | /// A timestamp or `--` range starting at `start`, and the index after it. | ||
| 152 | func scanTimestamp(_ chars: [Character], at start: Int, limit: Int) -> (stamp: Timestamp, end: Int)? { | ||
| 153 | guard let first = scanSingleTimestamp(chars, at: start, limit: limit) else { return nil } | ||
| 154 | if first.stamp.end == nil, first.end + 2 < limit, chars[first.end] == "-", chars[first.end + 1] == "-", | ||
| 155 | let second = scanSingleTimestamp(chars, at: first.end + 2, limit: limit), | ||
| 156 | second.stamp.active == first.stamp.active, second.stamp.end == nil { | ||
| 157 | var range = first.stamp | ||
| 158 | range.end = second.stamp.start | ||
| 159 | return (range, second.end) | ||
| 160 | } | ||
| 161 | return first | ||
| 162 | } | ||
| 163 | |||
| 164 | /// `<YYYY-MM-DD DAY HH:MM-HH:MM REPEATER WARNING>`, or the same in `[...]` for inactive. | ||
| 165 | private func scanSingleTimestamp(_ chars: [Character], at start: Int, limit: Int) -> (stamp: Timestamp, end: Int)? { | ||
| 166 | guard start < limit, chars[start] == "<" || chars[start] == "[" else { return nil } | ||
| 167 | let active = chars[start] == "<" | ||
| 168 | let close: Character = active ? ">" : "]" | ||
| 169 | var j = start + 1 | ||
| 170 | |||
| 171 | func number(_ minDigits: Int, _ maxDigits: Int) -> Int? { | ||
| 172 | var k = j | ||
| 173 | while k < limit, k - j < maxDigits, chars[k].isASCII, chars[k].isNumber { k += 1 } | ||
| 174 | guard k - j >= minDigits else { return nil } | ||
| 175 | let value = Int(String(chars[j..<k]))! | ||
| 176 | j = k | ||
| 177 | return value | ||
| 178 | } | ||
| 179 | |||
| 180 | func take(_ c: Character) -> Bool { | ||
| 181 | guard j < limit, chars[j] == c else { return false } | ||
| 182 | j += 1 | ||
| 183 | return true | ||
| 184 | } | ||
| 185 | |||
| 186 | func interval() -> Timestamp.Interval? { | ||
| 187 | let before = j | ||
| 188 | guard let value = number(1, 9), j < limit, let unit = Timestamp.Unit(rawValue: chars[j]) else { | ||
| 189 | j = before | ||
| 190 | return nil | ||
| 191 | } | ||
| 192 | j += 1 | ||
| 193 | return Timestamp.Interval(value: value, unit: unit) | ||
| 194 | } | ||
| 195 | |||
| 196 | func repeater() -> Timestamp.Repeater? { | ||
| 197 | let before = j | ||
| 198 | let kind: Timestamp.RepeaterKind | ||
| 199 | if j + 1 < limit, chars[j] == ".", chars[j + 1] == "+" { | ||
| 200 | kind = .restart | ||
| 201 | j += 2 | ||
| 202 | } else if j + 1 < limit, chars[j] == "+", chars[j + 1] == "+" { | ||
| 203 | kind = .catchUp | ||
| 204 | j += 2 | ||
| 205 | } else if take("+") { | ||
| 206 | kind = .cumulate | ||
| 207 | } else { | ||
| 208 | return nil | ||
| 209 | } | ||
| 210 | guard let value = interval() else { | ||
| 211 | j = before | ||
| 212 | return nil | ||
| 213 | } | ||
| 214 | var deadline: Timestamp.Interval? | ||
| 215 | if take("/") { | ||
| 216 | deadline = interval() | ||
| 217 | if deadline == nil { | ||
| 218 | j = before | ||
| 219 | return nil | ||
| 220 | } | ||
| 221 | } | ||
| 222 | return Timestamp.Repeater(kind: kind, interval: value, habitDeadline: deadline) | ||
| 223 | } | ||
| 224 | |||
| 225 | func warning() -> Timestamp.Warning? { | ||
| 226 | let before = j | ||
| 227 | guard take("-") else { return nil } | ||
| 228 | let firstOnly = take("-") | ||
| 229 | guard let value = interval() else { | ||
| 230 | j = before | ||
| 231 | return nil | ||
| 232 | } | ||
| 233 | return Timestamp.Warning(firstOccurrenceOnly: firstOnly, interval: value) | ||
| 234 | } | ||
| 235 | |||
| 236 | guard let year = number(4, 4), take("-"), let month = number(2, 2), take("-"), let day = number(2, 2) else { | ||
| 237 | return nil | ||
| 238 | } | ||
| 239 | var stamp = Timestamp( | ||
| 240 | active: active, | ||
| 241 | start: Timestamp.Point(year: year, month: month, day: day, hour: nil, minute: nil), | ||
| 242 | end: nil, repeater: nil, warning: nil | ||
| 243 | ) | ||
| 244 | |||
| 245 | // Day name: anything but digits, whitespace, `+`, `-`, `]` and `>`, in any language. | ||
| 246 | if j < limit, chars[j] == " " { | ||
| 247 | var k = j + 1 | ||
| 248 | while k < limit, !(chars[k].isNumber || chars[k].isWhitespace || "+-]>".contains(chars[k])) { k += 1 } | ||
| 249 | if k > j + 1 { j = k } | ||
| 250 | } | ||
| 251 | |||
| 252 | let beforeTime = j | ||
| 253 | if take(" "), let hour = number(1, 2), take(":"), let minute = number(2, 2) { | ||
| 254 | stamp.start.hour = hour | ||
| 255 | stamp.start.minute = minute | ||
| 256 | let beforeEnd = j | ||
| 257 | if take("-"), let endHour = number(1, 2), take(":"), let endMinute = number(2, 2) { | ||
| 258 | stamp.end = Timestamp.Point(year: year, month: month, day: day, hour: endHour, minute: endMinute) | ||
| 259 | } else { | ||
| 260 | j = beforeEnd | ||
| 261 | } | ||
| 262 | } else { | ||
| 263 | j = beforeTime | ||
| 264 | } | ||
| 265 | |||
| 266 | while j < limit, chars[j] == " " { | ||
| 267 | let beforeModifier = j | ||
| 268 | j += 1 | ||
| 269 | if let value = repeater() { | ||
| 270 | stamp.repeater = value | ||
| 271 | } else if let value = warning() { | ||
| 272 | stamp.warning = value | ||
| 273 | } else { | ||
| 274 | j = beforeModifier | ||
| 275 | break | ||
| 276 | } | ||
| 277 | } | ||
| 278 | |||
| 279 | guard take(close) else { return nil } | ||
| 280 | return (stamp, j) | ||
| 281 | } | ||
| 282 | ``` | ||
| 283 | |||
| 284 | - [ ] **Step 4: Run to verify pass** | ||
| 285 | |||
| 286 | Run: `swift test --filter TimestampTests` | ||
| 287 | Expected: all pass. | ||
| 288 | |||
| 289 | - [ ] **Step 5: Commit** | ||
| 290 | |||
| 291 | ```bash | ||
| 292 | git add Sources/OrgCore/Timestamp.swift Tests/OrgCoreTests/TimestampTests.swift | ||
| 293 | git commit -m "Add timestamp recognizer and value type" | ||
| 294 | ``` | ||
| 295 | |||
| 296 | --- | ||
| 297 | |||
| 298 | ### Task 2: Inline scanner and parser integration | ||
| 299 | |||
| 300 | **Files:** | ||
| 301 | - Create: `Sources/OrgCore/Parser/Inline.swift` | ||
| 302 | - Modify: `Sources/OrgCore/Syntax/SyntaxKind.swift`, `Sources/OrgCore/Parser/Parser.swift`, `Sources/OrgCore/Parser/Lines.swift` | ||
| 303 | - Test: `Tests/OrgCoreTests/InlineTests.swift`; update `Tests/OrgCoreTests/ParserSectionTests.swift` (title is a node) | ||
| 304 | |||
| 305 | **Interfaces:** | ||
| 306 | - Consumes: `scanTimestamp` (Task 1), `GreenBuilder`, `Parser`. | ||
| 307 | - Produces: `InlineScanner(chars:)` with `scan(_:into:inLink:)`; `Parser.inline(_ text: Substring)`; `Parser.span(from:through:)`; node kinds `title`, `tableCell`, `bold`, `italic`, `underline`, `strikeThrough`, `verbatim`, `code`, `link`, `linkDescription`, `timestamp`, `footnoteReference`, `statisticsCookie`, `target`, `macro`, `inlineSourceBlock`, `latexFragment`, `lineBreak`, `superscript`; token kinds `marker`, `linkPath`, `bullet`, `checkbox`. | ||
| 308 | |||
| 309 | - [ ] **Step 1: Write the failing tests** | ||
| 310 | |||
| 311 | ```swift | ||
| 312 | import Testing | ||
| 313 | @testable import OrgCore | ||
| 314 | |||
| 315 | /// `kind:text` for each object directly inside the first paragraph. | ||
| 316 | func objects(_ text: String) -> [String] { | ||
| 317 | let paragraph = OrgParser.parse(text).root.descendants().first { $0.kind == .paragraph }! | ||
| 318 | return paragraph.children.map { "\($0.kind.rawValue):\($0.text)" } | ||
| 319 | } | ||
| 320 | |||
| 321 | struct InlineTests { | ||
| 322 | @Test(arguments: [ | ||
| 323 | ("*b* /i/ _u_ +s+ =v= ~c~", ["bold:*b*", "italic:/i/", "underline:_u_", "strikeThrough:+s+", "verbatim:=v=", "code:~c~"]), | ||
| 324 | ("a*b* (*c*) \"*d*\"", ["bold:*c*", "bold:*d*"]), | ||
| 325 | ("x *y * z", []), | ||
| 326 | ("*y*z", []), | ||
| 327 | ("*a\nb*", ["bold:*a\nb*"]), | ||
| 328 | ("*a\nb\nc*", []), | ||
| 329 | ("=*not bold*=", ["verbatim:=*not bold*="]), | ||
| 330 | ("[[https://a.b][the *site*]]", ["link:[[https://a.b][the *site*]]"]), | ||
| 331 | ("[[file:x.org]]", ["link:[[file:x.org]]"]), | ||
| 332 | ("<https://a.b/c>", ["link:<https://a.b/c>"]), | ||
| 333 | ("see https://a.b/c.", ["link:https://a.b/c"]), | ||
| 334 | ("<2026-10-04 Sun 10:00-11:30 +1w -2d>", ["timestamp:<2026-10-04 Sun 10:00-11:30 +1w -2d>"]), | ||
| 335 | ("[2026-10-04 Sun]--[2026-10-06 Tue]", ["timestamp:[2026-10-04 Sun]--[2026-10-06 Tue]"]), | ||
| 336 | ("a [fn:1] and [fn::inline [x] note]", ["footnoteReference:[fn:1]", "footnoteReference:[fn::inline [x] note]"]), | ||
| 337 | ("a [1/3] [50%]", ["statisticsCookie:[1/3]", "statisticsCookie:[50%]"]), | ||
| 338 | ("a <<target>> {{{m(x, y)}}}", ["target:<<target>>", "macro:{{{m(x, y)}}}"]), | ||
| 339 | ("\\(x^2\\) and src_sh[:results raw]{echo {a}}", ["latexFragment:\\(x^2\\)", "inlineSourceBlock:src_sh[:results raw]{echo {a}}"]), | ||
| 340 | ("N^2 and e^{i}", ["superscript:^2", "superscript:^{i}"]), | ||
| 341 | ("end\\\\\nnext", ["lineBreak:\\\\"]), | ||
| 342 | ]) | ||
| 343 | func recognizes(text: String, expected: [String]) { | ||
| 344 | #expect(objects(text) == expected) | ||
| 345 | } | ||
| 346 | |||
| 347 | @Test func emphasisNests() { | ||
| 348 | let bold = OrgParser.parse("*bold /italic/ x*\n").root.descendants().first { $0.kind == .bold }! | ||
| 349 | #expect(bold.children.map(\.kind) == [.italic]) | ||
| 350 | #expect(bold.tokens.first?.kind == .marker) | ||
| 351 | } | ||
| 352 | |||
| 353 | @Test func linkParts() { | ||
| 354 | let link = OrgParser.parse("[[id:abc][desc]]\n").root.descendants().first { $0.kind == .link }! | ||
| 355 | #expect(link.tokens.map(\.kind) == [.marker, .linkPath, .marker, .marker]) | ||
| 356 | #expect(link.tokens.first { $0.kind == .linkPath }?.text == "id:abc") | ||
| 357 | #expect(link.children.map(\.kind) == [.linkDescription]) | ||
| 358 | } | ||
| 359 | |||
| 360 | @Test func noLinksInsideLinkDescriptions() { | ||
| 361 | let link = OrgParser.parse("[[a][see https://b.c]]\n").root.descendants().first { $0.kind == .link }! | ||
| 362 | #expect(link.descendants().filter { $0.kind == .link }.count == 1) | ||
| 363 | } | ||
| 364 | |||
| 365 | @Test func headingTitlesHoldObjects() { | ||
| 366 | let title = OrgParser.parse("* TODO Read [[https://a.b][it]] [1/2]\n").root.descendants().first { $0.kind == .title }! | ||
| 367 | #expect(title.children.map(\.kind) == [.link, .statisticsCookie]) | ||
| 368 | } | ||
| 369 | |||
| 370 | @Test func planningHoldsTimestamps() { | ||
| 371 | let planning = OrgParser.parse("* a\nDEADLINE: <2026-10-04 Sun -2d> SCHEDULED: <2026-10-01 Thu>\n").root | ||
| 372 | .descendants().first { $0.kind == .planning }! | ||
| 373 | #expect(planning.children.map(\.kind) == [.timestamp, .timestamp]) | ||
| 374 | } | ||
| 375 | |||
| 376 | @Test func itemsSplitBulletAndCheckbox() { | ||
| 377 | let item = OrgParser.parse(" - [X] done *now*\n more\n").root.descendants().first { $0.kind == .item }! | ||
| 378 | #expect(item.tokens.map(\.kind) == [.whitespace, .bullet, .whitespace, .checkbox, .whitespace]) | ||
| 379 | #expect(item.children.map(\.kind) == [.paragraph]) | ||
| 380 | #expect(item.children[0].text == "done *now*\n more\n") | ||
| 381 | } | ||
| 382 | |||
| 383 | @Test func emptyItem() { | ||
| 384 | let item = OrgParser.parse("-\n").root.descendants().first { $0.kind == .item }! | ||
| 385 | #expect(item.tokens.map(\.kind) == [.bullet, .newline]) | ||
| 386 | } | ||
| 387 | |||
| 388 | @Test func tableCells() { | ||
| 389 | let rows = OrgParser.parse("| *a* || b\n|---+---|\n").root.descendants().filter { $0.kind == .tableRow } | ||
| 390 | #expect(rows[0].children.map(\.text) == [" *a* ", "", " b"]) | ||
| 391 | #expect(rows[0].children[0].children.map(\.kind) == [.bold]) | ||
| 392 | #expect(rows[1].children.isEmpty) | ||
| 393 | } | ||
| 394 | |||
| 395 | @Test func footnoteDefinitionLabelIsNotAReference() { | ||
| 396 | let definition = OrgParser.parse("[fn:1] see [fn:2]\n").root.descendants().first { $0.kind == .footnoteDefinition }! | ||
| 397 | #expect(definition.tokens.first?.text == "[fn:1]") | ||
| 398 | #expect(definition.children.map(\.kind) == [.footnoteReference]) | ||
| 399 | } | ||
| 400 | } | ||
| 401 | ``` | ||
| 402 | |||
| 403 | Update the two `ParserSectionTests` expectations that assumed `title` was a token: | ||
| 404 | |||
| 405 | ```diff | ||
| 406 | @@ -17,7 +17,7 @@ struct ParserSectionTests { | ||
| 407 | } | ||
| 408 | |||
| 409 | @Test func zerothSectionHoldsPreamble() { | ||
| 410 | - #expect(nodeKinds("text\n* a\n") == [.document, .zerothSection, .paragraph, .section, .heading]) | ||
| 411 | + #expect(nodeKinds("text\n* a\n") == [.document, .zerothSection, .paragraph, .section, .heading, .title]) | ||
| 412 | } | ||
| 413 | |||
| 414 | @Test func sectionsNestByLevel() { | ||
| 415 | @@ -30,10 +30,11 @@ struct ParserSectionTests { | ||
| 416 | } | ||
| 417 | |||
| 418 | @Test func headingTokens() { | ||
| 419 | - let parts = tokens(of: .heading, in: "** TODO [#A] Write the plan :work:urgent: \n") | ||
| 420 | - #expect(parts.map(\.kind) == [.stars, .whitespace, .todoKeyword, .whitespace, .priority, .whitespace, .title, .whitespace, .tags, .whitespace, .newline]) | ||
| 421 | + let text = "** TODO [#A] Write the plan :work:urgent: \n" | ||
| 422 | + let parts = tokens(of: .heading, in: text) | ||
| 423 | + #expect(parts.map(\.kind) == [.stars, .whitespace, .todoKeyword, .whitespace, .priority, .whitespace, .whitespace, .tags, .whitespace, .newline]) | ||
| 424 | #expect(parts.first { $0.kind == .tags }?.text == ":work:urgent:") | ||
| 425 | - #expect(parts.first { $0.kind == .title }?.text == "Write the plan") | ||
| 426 | + #expect(OrgParser.parse(text).root.descendants().first { $0.kind == .title }?.text == "Write the plan") | ||
| 427 | } | ||
| 428 | |||
| 429 | @Test func todoKeywordsComeFromSettings() { | ||
| 430 | ``` | ||
| 431 | |||
| 432 | - [ ] **Step 2: Run to verify failure** | ||
| 433 | |||
| 434 | Run: `swift test --filter InlineTests` | ||
| 435 | Expected: build failure on the new `SyntaxKind` cases. | ||
| 436 | |||
| 437 | - [ ] **Step 3: Replace `SyntaxKind.swift`** | ||
| 438 | |||
| 439 | ```swift | ||
| 440 | public enum SyntaxKind: String, Sendable { | ||
| 441 | // Tokens | ||
| 442 | case text, newline, whitespace | ||
| 443 | case stars, todoKeyword, priority, tags | ||
| 444 | case marker, linkPath, bullet, checkbox | ||
| 445 | |||
| 446 | // Elements | ||
| 447 | case document, zerothSection, section, heading, title | ||
| 448 | case planning, propertyDrawer, nodeProperty, drawer, clock | ||
| 449 | case paragraph, plainList, item, table, tableRow, tableCell, tableFormula | ||
| 450 | case block, dynamicBlock, keyword, affiliatedKeyword | ||
| 451 | case comment, fixedWidth, horizontalRule, footnoteDefinition | ||
| 452 | |||
| 453 | // Objects | ||
| 454 | case bold, italic, underline, strikeThrough, verbatim, code | ||
| 455 | case link, linkDescription, timestamp, footnoteReference, statisticsCookie | ||
| 456 | case target, macro, inlineSourceBlock, latexFragment, lineBreak, superscript | ||
| 457 | } | ||
| 458 | ``` | ||
| 459 | |||
| 460 | - [ ] **Step 4: Create `Inline.swift`** | ||
| 461 | |||
| 462 | ```swift | ||
| 463 | /// Characters allowed before an emphasis opener, besides whitespace and the start of the run. | ||
| 464 | private let emphasisPre: Set<Character> = ["-", "(", "{", "'", "\""] | ||
| 465 | |||
| 466 | /// Characters allowed after an emphasis closer, besides whitespace and the end of the run. | ||
| 467 | private let emphasisPost: Set<Character> = ["-", ".", ",", ":", "!", "?", ";", "'", "\"", ")", "}", "\\", "["] | ||
| 468 | |||
| 469 | private let emphasisKinds: [Character: SyntaxKind] = [ | ||
| 470 | "*": .bold, "/": .italic, "_": .underline, "+": .strikeThrough, "=": .verbatim, "~": .code, | ||
| 471 | ] | ||
| 472 | |||
| 473 | private let angleLinkSchemes: Set<String> = [ | ||
| 474 | "http", "https", "mailto", "file", "id", "doi", "ftp", "news", "shell", "elisp", "info", "help", "attachment", | ||
| 475 | ] | ||
| 476 | |||
| 477 | private let plainLinkPrefixes = ["https://", "http://", "mailto:", "file:"] | ||
| 478 | |||
| 479 | /// "\r\n" is a single Character, so both forms count. | ||
| 480 | func isNewline(_ c: Character) -> Bool { | ||
| 481 | c == "\n" || c == "\r\n" | ||
| 482 | } | ||
| 483 | |||
| 484 | enum InlineMatch { | ||
| 485 | case emphasis(SyntaxKind, open: Int, close: Int) | ||
| 486 | case link(path: Range<Int>, description: Range<Int>?, whole: Range<Int>) | ||
| 487 | case object(SyntaxKind, Range<Int>) | ||
| 488 | |||
| 489 | var end: Int { | ||
| 490 | switch self { | ||
| 491 | case .emphasis(_, _, let close): close + 1 | ||
| 492 | case .link(_, _, let whole): whole.upperBound | ||
| 493 | case .object(_, let range): range.upperBound | ||
| 494 | } | ||
| 495 | } | ||
| 496 | } | ||
| 497 | |||
| 498 | /// Turns a run of text into text, newline and object tokens. Every character ends up in | ||
| 499 | /// exactly one token. | ||
| 500 | struct InlineScanner { | ||
| 501 | let chars: [Character] | ||
| 502 | |||
| 503 | func scan(_ range: Range<Int>, into b: inout GreenBuilder, inLink: Bool = false) { | ||
| 504 | var textStart = range.lowerBound | ||
| 505 | var i = range.lowerBound | ||
| 506 | while i < range.upperBound { | ||
| 507 | if let match = match(at: i, in: range, inLink: inLink) { | ||
| 508 | emitText(textStart..<i, into: &b) | ||
| 509 | emit(match, into: &b, inLink: inLink) | ||
| 510 | i = match.end | ||
| 511 | textStart = i | ||
| 512 | } else { | ||
| 513 | i += 1 | ||
| 514 | } | ||
| 515 | } | ||
| 516 | emitText(textStart..<range.upperBound, into: &b) | ||
| 517 | } | ||
| 518 | |||
| 519 | func emitText(_ range: Range<Int>, into b: inout GreenBuilder) { | ||
| 520 | var start = range.lowerBound | ||
| 521 | for k in range where isNewline(chars[k]) { | ||
| 522 | if start < k { b.token(.text, string(start..<k)) } | ||
| 523 | b.token(.newline, String(chars[k])) | ||
| 524 | start = k + 1 | ||
| 525 | } | ||
| 526 | if start < range.upperBound { b.token(.text, string(start..<range.upperBound)) } | ||
| 527 | } | ||
| 528 | |||
| 529 | func emit(_ match: InlineMatch, into b: inout GreenBuilder, inLink: Bool) { | ||
| 530 | switch match { | ||
| 531 | case .emphasis(let kind, let open, let close): | ||
| 532 | b.start(kind) | ||
| 533 | b.token(.marker, string(open..<(open + 1))) | ||
| 534 | if kind == .verbatim || kind == .code { | ||
| 535 | emitText((open + 1)..<close, into: &b) | ||
| 536 | } else { | ||
| 537 | scan((open + 1)..<close, into: &b, inLink: inLink) | ||
| 538 | } | ||
| 539 | b.token(.marker, string(close..<(close + 1))) | ||
| 540 | b.finish() | ||
| 541 | case .link(let path, let description, let whole): | ||
| 542 | b.start(.link) | ||
| 543 | b.token(.marker, string(whole.lowerBound..<path.lowerBound)) | ||
| 544 | b.token(.linkPath, string(path)) | ||
| 545 | if let description { | ||
| 546 | b.token(.marker, string(path.upperBound..<description.lowerBound)) | ||
| 547 | b.start(.linkDescription) | ||
| 548 | scan(description, into: &b, inLink: true) | ||
| 549 | b.finish() | ||
| 550 | b.token(.marker, string(description.upperBound..<whole.upperBound)) | ||
| 551 | } else { | ||
| 552 | b.token(.marker, string(path.upperBound..<whole.upperBound)) | ||
| 553 | } | ||
| 554 | b.finish() | ||
| 555 | case .object(let kind, let range): | ||
| 556 | b.start(kind) | ||
| 557 | emitText(range, into: &b) | ||
| 558 | b.finish() | ||
| 559 | } | ||
| 560 | } | ||
| 561 | |||
| 562 | func string(_ range: Range<Int>) -> String { | ||
| 563 | String(chars[range]) | ||
| 564 | } | ||
| 565 | |||
| 566 | func hasPrefix(_ s: String, at i: Int, _ limit: Int) -> Bool { | ||
| 567 | var j = i | ||
| 568 | for c in s { | ||
| 569 | guard j < limit, chars[j] == c else { return false } | ||
| 570 | j += 1 | ||
| 571 | } | ||
| 572 | return true | ||
| 573 | } | ||
| 574 | |||
| 575 | // MARK: - Recognizers | ||
| 576 | |||
| 577 | func match(at i: Int, in range: Range<Int>, inLink: Bool) -> InlineMatch? { | ||
| 578 | let limit = range.upperBound | ||
| 579 | let previous: Character? = i > range.lowerBound ? chars[i - 1] : nil | ||
| 580 | let afterWord = previous.map { $0.isLetter || $0.isNumber } ?? false | ||
| 581 | |||
| 582 | switch chars[i] { | ||
| 583 | case "[": | ||
| 584 | if !inLink, let link = bracketLink(i, limit) { return link } | ||
| 585 | if let end = footnoteReference(i, limit) { return .object(.footnoteReference, i..<end) } | ||
| 586 | if let end = scanTimestamp(chars, at: i, limit: limit)?.end { return .object(.timestamp, i..<end) } | ||
| 587 | if let end = statisticsCookie(i, limit) { return .object(.statisticsCookie, i..<end) } | ||
| 588 | case "<": | ||
| 589 | if let end = scanTimestamp(chars, at: i, limit: limit)?.end { return .object(.timestamp, i..<end) } | ||
| 590 | if let end = target(i, limit) { return .object(.target, i..<end) } | ||
| 591 | if !inLink, let end = angleLink(i, limit) { return .object(.link, i..<end) } | ||
| 592 | case "{": | ||
| 593 | if let end = macro(i, limit) { return .object(.macro, i..<end) } | ||
| 594 | case "\\": | ||
| 595 | if let end = lineBreak(i, limit) { return .object(.lineBreak, i..<end) } | ||
| 596 | if let end = latexFragment(i, limit) { return .object(.latexFragment, i..<end) } | ||
| 597 | case "^": | ||
| 598 | if afterWord, let end = superscript(i, limit) { return .object(.superscript, i..<end) } | ||
| 599 | case "s": | ||
| 600 | if !afterWord, let end = inlineSourceBlock(i, limit) { return .object(.inlineSourceBlock, i..<end) } | ||
| 601 | default: | ||
| 602 | break | ||
| 603 | } | ||
| 604 | |||
| 605 | if !inLink, !afterWord, let end = plainLink(i, limit) { return .object(.link, i..<end) } | ||
| 606 | |||
| 607 | if let kind = emphasisKinds[chars[i]], | ||
| 608 | previous.map({ $0.isWhitespace || emphasisPre.contains($0) }) ?? true, | ||
| 609 | let close = emphasisClose(i, limit) { | ||
| 610 | return .emphasis(kind, open: i, close: close) | ||
| 611 | } | ||
| 612 | return nil | ||
| 613 | } | ||
| 614 | |||
| 615 | /// org's emphasis rules: the body neither starts nor ends with whitespace, spans at most | ||
| 616 | /// one line break, and the closer is followed by whitespace, punctuation or the end. | ||
| 617 | func emphasisClose(_ i: Int, _ limit: Int) -> Int? { | ||
| 618 | let marker = chars[i] | ||
| 619 | guard i + 1 < limit, !chars[i + 1].isWhitespace else { return nil } | ||
| 620 | var newlines = 0 | ||
| 621 | var j = i + 1 | ||
| 622 | while j < limit { | ||
| 623 | if isNewline(chars[j]) { | ||
| 624 | newlines += 1 | ||
| 625 | if newlines > 1 { return nil } | ||
| 626 | } else if chars[j] == marker, j > i + 1, !chars[j - 1].isWhitespace { | ||
| 627 | if j + 1 == limit || chars[j + 1].isWhitespace || emphasisPost.contains(chars[j + 1]) { return j } | ||
| 628 | } | ||
| 629 | j += 1 | ||
| 630 | } | ||
| 631 | return nil | ||
| 632 | } | ||
| 633 | |||
| 634 | /// `[[path]]` or `[[path][description]]`. | ||
| 635 | func bracketLink(_ i: Int, _ limit: Int) -> InlineMatch? { | ||
| 636 | guard i + 1 < limit, chars[i + 1] == "[" else { return nil } | ||
| 637 | var j = i + 2 | ||
| 638 | while j < limit, chars[j] != "]" { | ||
| 639 | if chars[j] == "[" || isNewline(chars[j]) { return nil } | ||
| 640 | if chars[j] == "\\", j + 1 < limit { j += 1 } | ||
| 641 | j += 1 | ||
| 642 | } | ||
| 643 | guard j > i + 2, j + 1 < limit else { return nil } | ||
| 644 | let path = (i + 2)..<j | ||
| 645 | if chars[j + 1] == "]" { return .link(path: path, description: nil, whole: i..<(j + 2)) } | ||
| 646 | guard chars[j + 1] == "[" else { return nil } | ||
| 647 | let descriptionStart = j + 2 | ||
| 648 | var depth = 0 | ||
| 649 | var k = descriptionStart | ||
| 650 | while k < limit { | ||
| 651 | if chars[k] == "[" { | ||
| 652 | depth += 1 | ||
| 653 | } else if chars[k] == "]" { | ||
| 654 | if depth == 0 { break } | ||
| 655 | depth -= 1 | ||
| 656 | } | ||
| 657 | k += 1 | ||
| 658 | } | ||
| 659 | guard k > descriptionStart, k + 1 < limit, chars[k + 1] == "]" else { return nil } | ||
| 660 | return .link(path: path, description: descriptionStart..<k, whole: i..<(k + 2)) | ||
| 661 | } | ||
| 662 | |||
| 663 | /// `[fn:label]`, `[fn:label:definition]` or `[fn::definition]`. | ||
| 664 | func footnoteReference(_ i: Int, _ limit: Int) -> Int? { | ||
| 665 | guard hasPrefix("[fn:", at: i, limit) else { return nil } | ||
| 666 | var j = i + 4 | ||
| 667 | while j < limit, chars[j].isLetter || chars[j].isNumber || chars[j] == "_" || chars[j] == "-" { j += 1 } | ||
| 668 | guard j < limit else { return nil } | ||
| 669 | if chars[j] == "]" { return j > i + 4 ? j + 1 : nil } | ||
| 670 | guard chars[j] == ":" else { return nil } | ||
| 671 | var depth = 0 | ||
| 672 | j += 1 | ||
| 673 | while j < limit { | ||
| 674 | if chars[j] == "[" { | ||
| 675 | depth += 1 | ||
| 676 | } else if chars[j] == "]" { | ||
| 677 | if depth == 0 { return j + 1 } | ||
| 678 | depth -= 1 | ||
| 679 | } | ||
| 680 | j += 1 | ||
| 681 | } | ||
| 682 | return nil | ||
| 683 | } | ||
| 684 | |||
| 685 | /// `[1/3]`, `[/]`, `[50%]` or `[%]`. | ||
| 686 | func statisticsCookie(_ i: Int, _ limit: Int) -> Int? { | ||
| 687 | var j = i + 1 | ||
| 688 | while j < limit, chars[j].isASCII, chars[j].isNumber { j += 1 } | ||
| 689 | guard j < limit else { return nil } | ||
| 690 | if chars[j] == "%" { | ||
| 691 | j += 1 | ||
| 692 | } else if chars[j] == "/" { | ||
| 693 | j += 1 | ||
| 694 | while j < limit, chars[j].isASCII, chars[j].isNumber { j += 1 } | ||
| 695 | } else { | ||
| 696 | return nil | ||
| 697 | } | ||
| 698 | guard j < limit, chars[j] == "]" else { return nil } | ||
| 699 | return j + 1 | ||
| 700 | } | ||
| 701 | |||
| 702 | /// `<<target>>`. | ||
| 703 | func target(_ i: Int, _ limit: Int) -> Int? { | ||
| 704 | guard hasPrefix("<<", at: i, limit), i + 2 < limit, chars[i + 2] != "<" else { return nil } | ||
| 705 | let start = i + 2 | ||
| 706 | var j = start | ||
| 707 | while j < limit, chars[j] != ">" { | ||
| 708 | if chars[j] == "<" || isNewline(chars[j]) { return nil } | ||
| 709 | j += 1 | ||
| 710 | } | ||
| 711 | guard j > start, j + 1 < limit, chars[j + 1] == ">", | ||
| 712 | !chars[start].isWhitespace, !chars[j - 1].isWhitespace else { return nil } | ||
| 713 | return j + 2 | ||
| 714 | } | ||
| 715 | |||
| 716 | /// `<scheme:path>` for a known scheme. | ||
| 717 | func angleLink(_ i: Int, _ limit: Int) -> Int? { | ||
| 718 | var j = i + 1 | ||
| 719 | while j < limit, chars[j].isLetter { j += 1 } | ||
| 720 | guard j < limit, chars[j] == ":", angleLinkSchemes.contains(string((i + 1)..<j).lowercased()) else { return nil } | ||
| 721 | j += 1 | ||
| 722 | let bodyStart = j | ||
| 723 | while j < limit, chars[j] != ">" { | ||
| 724 | if chars[j] == "<" || isNewline(chars[j]) { return nil } | ||
| 725 | j += 1 | ||
| 726 | } | ||
| 727 | guard j < limit, j > bodyStart else { return nil } | ||
| 728 | return j + 1 | ||
| 729 | } | ||
| 730 | |||
| 731 | /// A bare URL. Trailing sentence punctuation stays outside the link. | ||
| 732 | func plainLink(_ i: Int, _ limit: Int) -> Int? { | ||
| 733 | guard "hmf".contains(chars[i]), | ||
| 734 | let prefix = plainLinkPrefixes.first(where: { hasPrefix($0, at: i, limit) }) else { return nil } | ||
| 735 | let bodyStart = i + prefix.count | ||
| 736 | var j = bodyStart | ||
| 737 | while j < limit, !chars[j].isWhitespace, !"()<>[]\"".contains(chars[j]) { j += 1 } | ||
| 738 | while j > bodyStart, ".,;:!?'".contains(chars[j - 1]) { j -= 1 } | ||
| 739 | return j > bodyStart ? j : nil | ||
| 740 | } | ||
| 741 | |||
| 742 | /// `{{{name}}}` or `{{{name(arguments)}}}`. | ||
| 743 | func macro(_ i: Int, _ limit: Int) -> Int? { | ||
| 744 | guard hasPrefix("{{{", at: i, limit) else { return nil } | ||
| 745 | var j = i + 3 | ||
| 746 | guard j < limit, chars[j].isLetter else { return nil } | ||
| 747 | while j < limit, chars[j].isLetter || chars[j].isNumber || chars[j] == "-" || chars[j] == "_" { j += 1 } | ||
| 748 | if j < limit, chars[j] == "(" { | ||
| 749 | var k = j + 1 | ||
| 750 | while k < limit, !hasPrefix(")}}}", at: k, limit) { | ||
| 751 | if isNewline(chars[k]) { return nil } | ||
| 752 | k += 1 | ||
| 753 | } | ||
| 754 | return k < limit ? k + 4 : nil | ||
| 755 | } | ||
| 756 | return hasPrefix("}}}", at: j, limit) ? j + 3 : nil | ||
| 757 | } | ||
| 758 | |||
| 759 | /// `\\` at the end of a line, before optional trailing blanks. The newline stays outside. | ||
| 760 | func lineBreak(_ i: Int, _ limit: Int) -> Int? { | ||
| 761 | guard hasPrefix("\\\\", at: i, limit) else { return nil } | ||
| 762 | var j = i + 2 | ||
| 763 | while j < limit, chars[j] == " " || chars[j] == "\t" { j += 1 } | ||
| 764 | guard j == limit || isNewline(chars[j]) else { return nil } | ||
| 765 | return j | ||
| 766 | } | ||
| 767 | |||
| 768 | /// `\(...\)` or `\[...\]`. | ||
| 769 | func latexFragment(_ i: Int, _ limit: Int) -> Int? { | ||
| 770 | guard i + 1 < limit else { return nil } | ||
| 771 | let closer: String | ||
| 772 | switch chars[i + 1] { | ||
| 773 | case "(": closer = "\\)" | ||
| 774 | case "[": closer = "\\]" | ||
| 775 | default: return nil | ||
| 776 | } | ||
| 777 | var j = i + 2 | ||
| 778 | while j < limit { | ||
| 779 | if hasPrefix(closer, at: j, limit) { return j + 2 } | ||
| 780 | j += 1 | ||
| 781 | } | ||
| 782 | return nil | ||
| 783 | } | ||
| 784 | |||
| 785 | /// `^word` or `^{group}` after a letter or digit. | ||
| 786 | func superscript(_ i: Int, _ limit: Int) -> Int? { | ||
| 787 | var j = i + 1 | ||
| 788 | guard j < limit else { return nil } | ||
| 789 | if chars[j] == "{" { | ||
| 790 | j += 1 | ||
| 791 | while j < limit, chars[j] != "}" { | ||
| 792 | if isNewline(chars[j]) { return nil } | ||
| 793 | j += 1 | ||
| 794 | } | ||
| 795 | return j < limit ? j + 1 : nil | ||
| 796 | } | ||
| 797 | let start = j | ||
| 798 | while j < limit, chars[j].isLetter || chars[j].isNumber { j += 1 } | ||
| 799 | return j > start ? j : nil | ||
| 800 | } | ||
| 801 | |||
| 802 | /// `src_lang{body}` or `src_lang[headers]{body}`, on one line, with balanced braces. | ||
| 803 | func inlineSourceBlock(_ i: Int, _ limit: Int) -> Int? { | ||
| 804 | guard hasPrefix("src_", at: i, limit) else { return nil } | ||
| 805 | var j = i + 4 | ||
| 806 | let languageStart = j | ||
| 807 | while j < limit, !chars[j].isWhitespace, chars[j] != "[", chars[j] != "{" { j += 1 } | ||
| 808 | guard j > languageStart, j < limit else { return nil } | ||
| 809 | if chars[j] == "[" { | ||
| 810 | while j < limit, chars[j] != "]" { | ||
| 811 | if isNewline(chars[j]) { return nil } | ||
| 812 | j += 1 | ||
| 813 | } | ||
| 814 | guard j < limit else { return nil } | ||
| 815 | j += 1 | ||
| 816 | } | ||
| 817 | guard j < limit, chars[j] == "{" else { return nil } | ||
| 818 | var depth = 0 | ||
| 819 | while j < limit { | ||
| 820 | if isNewline(chars[j]) { return nil } | ||
| 821 | if chars[j] == "{" { | ||
| 822 | depth += 1 | ||
| 823 | } else if chars[j] == "}" { | ||
| 824 | depth -= 1 | ||
| 825 | if depth == 0 { return j + 1 } | ||
| 826 | } | ||
| 827 | j += 1 | ||
| 828 | } | ||
| 829 | return nil | ||
| 830 | } | ||
| 831 | } | ||
| 832 | |||
| 833 | extension Parser { | ||
| 834 | mutating func inline(_ text: Substring) { | ||
| 835 | let chars = Array(text) | ||
| 836 | InlineScanner(chars: chars).scan(0..<chars.count, into: &builder) | ||
| 837 | } | ||
| 838 | } | ||
| 839 | ``` | ||
| 840 | |||
| 841 | - [ ] **Step 5: Wire the scanner into the parser** | ||
| 842 | |||
| 843 | ```diff | ||
| 844 | diff --git a/Sources/OrgCore/Parser/Lines.swift b/Sources/OrgCore/Parser/Lines.swift | ||
| 845 | index 32a778a..5e9a077 100644 | ||
| 846 | --- a/Sources/OrgCore/Parser/Lines.swift | ||
| 847 | +++ b/Sources/OrgCore/Parser/Lines.swift | ||
| 848 | @@ -27,7 +27,7 @@ func splitRawLines(_ text: String) -> [RawLine] { | ||
| 849 | } | ||
| 850 | } | ||
| 851 | if lineStart != scalars.endIndex { | ||
| 852 | - lines.append(RawLine(content: text[lineStart...], ending: "")) | ||
| 853 | + lines.append(RawLine(content: text[lineStart...], ending: text[text.endIndex...])) | ||
| 854 | } | ||
| 855 | return lines | ||
| 856 | } | ||
| 857 | diff --git a/Sources/OrgCore/Parser/Parser.swift b/Sources/OrgCore/Parser/Parser.swift | ||
| 858 | index ac210ff..6460486 100644 | ||
| 859 | --- a/Sources/OrgCore/Parser/Parser.swift | ||
| 860 | +++ b/Sources/OrgCore/Parser/Parser.swift | ||
| 861 | @@ -77,10 +77,7 @@ struct Parser { | ||
| 862 | headingLine(lines[i]) | ||
| 863 | i += 1 | ||
| 864 | if i < lines.count, info[i].cls == .planning { | ||
| 865 | - builder.start(.planning) | ||
| 866 | - line(i) | ||
| 867 | - i += 1 | ||
| 868 | - builder.finish() | ||
| 869 | + single(.planning) | ||
| 870 | } | ||
| 871 | if i < lines.count, case .drawerBegin(let name) = info[i].cls, name.uppercased() == "PROPERTIES", | ||
| 872 | let end = blockEnds[i] { | ||
| 873 | @@ -173,13 +170,28 @@ struct Parser { | ||
| 874 | } | ||
| 875 | } | ||
| 876 | |||
| 877 | + /// Kinds whose single line holds inline objects (timestamps on planning and clock lines). | ||
| 878 | + static let inlineLineKinds: Set<SyntaxKind> = [.planning, .clock] | ||
| 879 | + | ||
| 880 | mutating func single(_ kind: SyntaxKind) { | ||
| 881 | builder.start(kind) | ||
| 882 | - line(i) | ||
| 883 | + if Self.inlineLineKinds.contains(kind) { | ||
| 884 | + let rest = whitespace(lines[i].content) | ||
| 885 | + inline(rest) | ||
| 886 | + builder.token(.newline, lines[i].ending) | ||
| 887 | + } else { | ||
| 888 | + line(i) | ||
| 889 | + } | ||
| 890 | i += 1 | ||
| 891 | builder.finish() | ||
| 892 | } | ||
| 893 | |||
| 894 | + /// Source text from `start` through the end of line `last`, including its line ending. | ||
| 895 | + func span(from start: Substring.Index, through last: Int) -> Substring { | ||
| 896 | + let base = lines[last].ending.base | ||
| 897 | + return base[start..<lines[last].ending.endIndex] | ||
| 898 | + } | ||
| 899 | + | ||
| 900 | mutating func consecutive(_ kind: SyntaxKind, limit: Int, floor: Int?, matching: (LineClass) -> Bool) { | ||
| 901 | builder.start(kind) | ||
| 902 | repeat { | ||
| 903 | @@ -192,7 +204,7 @@ struct Parser { | ||
| 904 | mutating func table(limit: Int, floor: Int?) { | ||
| 905 | builder.start(.table) | ||
| 906 | while i < limit, info[i].cls == .tableRow, within(floor, i) { | ||
| 907 | - single(.tableRow) | ||
| 908 | + tableRow() | ||
| 909 | } | ||
| 910 | while i < limit, info[i].cls == .keyword(key: "TBLFM"), within(floor, i) { | ||
| 911 | single(.tableFormula) | ||
| 912 | @@ -200,15 +212,47 @@ struct Parser { | ||
| 913 | builder.finish() | ||
| 914 | } | ||
| 915 | |||
| 916 | - mutating func footnoteDefinition(limit: Int) { | ||
| 917 | - builder.start(.footnoteDefinition) | ||
| 918 | - line(i) | ||
| 919 | - i += 1 | ||
| 920 | - while i < limit, info[i].cls == .plain { | ||
| 921 | - line(i) | ||
| 922 | - i += 1 | ||
| 923 | + /// A rule row (`|---+---|`) is one text token. Other rows alternate `|` markers and cells; | ||
| 924 | + /// every pair of pipes gets a cell, even an empty one, so columns line up. | ||
| 925 | + mutating func tableRow() { | ||
| 926 | + builder.start(.tableRow) | ||
| 927 | + let rest = whitespace(lines[i].content) | ||
| 928 | + if rest.hasPrefix("|-") { | ||
| 929 | + builder.token(.text, rest) | ||
| 930 | + } else { | ||
| 931 | + var cellStart = rest.startIndex | ||
| 932 | + var index = rest.startIndex | ||
| 933 | + while index < rest.endIndex { | ||
| 934 | + if rest[index] == "|" { | ||
| 935 | + if index > rest.startIndex { tableCell(rest[cellStart..<index]) } | ||
| 936 | + builder.token(.marker, "|") | ||
| 937 | + cellStart = rest.index(after: index) | ||
| 938 | + } | ||
| 939 | + index = rest.index(after: index) | ||
| 940 | + } | ||
| 941 | + if cellStart < rest.endIndex { tableCell(rest[cellStart...]) } | ||
| 942 | } | ||
| 943 | + builder.token(.newline, lines[i].ending) | ||
| 944 | builder.finish() | ||
| 945 | + i += 1 | ||
| 946 | + } | ||
| 947 | + | ||
| 948 | + mutating func tableCell(_ text: Substring) { | ||
| 949 | + builder.start(.tableCell) | ||
| 950 | + inline(text) | ||
| 951 | + builder.finish() | ||
| 952 | + } | ||
| 953 | + | ||
| 954 | + mutating func footnoteDefinition(limit: Int) { | ||
| 955 | + var end = i + 1 | ||
| 956 | + while end < limit, info[end].cls == .plain { end += 1 } | ||
| 957 | + let content = lines[i].content | ||
| 958 | + let close = content.firstIndex(of: "]")! | ||
| 959 | + builder.start(.footnoteDefinition) | ||
| 960 | + builder.token(.marker, content[...close]) | ||
| 961 | + inline(span(from: content.index(after: close), through: end - 1)) | ||
| 962 | + builder.finish() | ||
| 963 | + i = end | ||
| 964 | } | ||
| 965 | |||
| 966 | mutating func list(limit: Int, floor: Int?) { | ||
| 967 | @@ -224,8 +268,25 @@ struct Parser { | ||
| 968 | /// inside the item when the item or list continues after it; two end the list. | ||
| 969 | mutating func item(base: Int, limit: Int) { | ||
| 970 | builder.start(.item) | ||
| 971 | - line(i) | ||
| 972 | - i += 1 | ||
| 973 | + var rest = whitespace(lines[i].content) | ||
| 974 | + let bullet = rest.prefix { $0 != " " && $0 != "\t" } | ||
| 975 | + builder.token(.bullet, bullet) | ||
| 976 | + rest = whitespace(rest.dropFirst(bullet.count)) | ||
| 977 | + if let box = checkbox(rest) { | ||
| 978 | + builder.token(.checkbox, box) | ||
| 979 | + rest = whitespace(rest.dropFirst(box.count)) | ||
| 980 | + } | ||
| 981 | + // The rest of the bullet line and its continuation lines are the item's first paragraph. | ||
| 982 | + var end = i + 1 | ||
| 983 | + while end < limit, within(base, end), continuesParagraph(end) { end += 1 } | ||
| 984 | + if rest.isEmpty, end == i + 1 { | ||
| 985 | + builder.token(.newline, lines[i].ending) | ||
| 986 | + } else { | ||
| 987 | + builder.start(.paragraph) | ||
| 988 | + inline(span(from: rest.startIndex, through: end - 1)) | ||
| 989 | + builder.finish() | ||
| 990 | + } | ||
| 991 | + i = end | ||
| 992 | while i < limit, !isHeading(i) { | ||
| 993 | if info[i].cls == .blank { | ||
| 994 | var j = i | ||
| 995 | @@ -245,15 +306,23 @@ struct Parser { | ||
| 996 | builder.finish() | ||
| 997 | } | ||
| 998 | |||
| 999 | + /// `[ ]`, `[X]`, `[x]` or `[-]`, followed by whitespace or end of line. | ||
| 1000 | + func checkbox(_ s: Substring) -> Substring? { | ||
| 1001 | + guard s.count >= 3, s.first == "[", "Xx -".contains(s.dropFirst().first!), | ||
| 1002 | + s.dropFirst(2).first == "]" else { return nil } | ||
| 1003 | + let after = s.dropFirst(3) | ||
| 1004 | + guard after.isEmpty || after.first == " " || after.first == "\t" else { return nil } | ||
| 1005 | + return s.prefix(3) | ||
| 1006 | + } | ||
| 1007 | + | ||
| 1008 | + /// The paragraph's lines are one inline run, so emphasis and links can cross a line break. | ||
| 1009 | mutating func paragraph(limit: Int, floor: Int?) { | ||
| 1010 | + var end = i + 1 | ||
| 1011 | + while end < limit, within(floor, end), continuesParagraph(end) { end += 1 } | ||
| 1012 | builder.start(.paragraph) | ||
| 1013 | - line(i) | ||
| 1014 | - i += 1 | ||
| 1015 | - while i < limit, within(floor, i), continuesParagraph(i) { | ||
| 1016 | - line(i) | ||
| 1017 | - i += 1 | ||
| 1018 | - } | ||
| 1019 | + inline(span(from: lines[i].content.startIndex, through: end - 1)) | ||
| 1020 | builder.finish() | ||
| 1021 | + i = end | ||
| 1022 | } | ||
| 1023 | |||
| 1024 | /// Lines that don't start an element of their own. | ||
| 1025 | @@ -307,7 +376,11 @@ struct Parser { | ||
| 1026 | } | ||
| 1027 | |||
| 1028 | let parts = splitTags(rest) | ||
| 1029 | - builder.token(.title, parts.title) | ||
| 1030 | + if !parts.title.isEmpty { | ||
| 1031 | + builder.start(.title) | ||
| 1032 | + inline(parts.title) | ||
| 1033 | + builder.finish() | ||
| 1034 | + } | ||
| 1035 | builder.token(.whitespace, parts.gap) | ||
| 1036 | builder.token(.tags, parts.tags) | ||
| 1037 | builder.token(.whitespace, parts.trailing) | ||
| 1038 | ``` | ||
| 1039 | |||
| 1040 | - [ ] **Step 6: Run all tests** | ||
| 1041 | |||
| 1042 | Run: `swift test` | ||
| 1043 | Expected: all pass. | ||
| 1044 | |||
| 1045 | - [ ] **Step 7: Commit** | ||
| 1046 | |||
| 1047 | ```bash | ||
| 1048 | git add Sources Tests | ||
| 1049 | git commit -m "Parse inline objects" | ||
| 1050 | ``` | ||
| 1051 | |||
| 1052 | --- | ||
| 1053 | |||
| 1054 | ### Task 3: Fuzz the inline layer | ||
| 1055 | |||
| 1056 | **Files:** | ||
| 1057 | - Modify: `Tests/OrgCoreTests/RoundTripTests.swift` (fragment list) | ||
| 1058 | |||
| 1059 | - [ ] **Step 1: Add inline fragments** | ||
| 1060 | |||
| 1061 | ```diff | ||
| 1062 | @@ -20,6 +20,10 @@ let fragments = [ | ||
| 1063 | ":LOGBOOK:", "CLOCK: [2026-10-04 Sun 10:00]", "SCHEDULED: <2026-10-04 Sun>", "- item", " - nested", | ||
| 1064 | "\t+ tab", "1. one", "| a | b |", "|---+---|", "#+TBLFM: $2=$1", "# comment", ": fixed", "-----", | ||
| 1065 | "[fn:1] note", "#+NAME: x", "plain text", "é", "😀", "e\u{301}", " ", "\t", "\n", "\n", "\r\n", "\r", "", | ||
| 1066 | + "*bold*", "/it/ ", "=v=", "~c~", "_u_", "+s+", "*", "/", "=", "[[https://a.b][d *b*]]", "[[x]]", "[[", "]]", | ||
| 1067 | + "<2026-10-04 Sun 10:00 +1w -2d>", "[2026-10-04]--[2026-10-05]", "<", ">", "[fn:2]", "[fn::in [x]]", "[1/3]", | ||
| 1068 | + "[50%]", "<<t>>", "{{{m(a)}}}", "\\(x\\)", "a\\\\", "x^2", "^{y}", "src_sh{echo}", "https://e.com.", | ||
| 1069 | + "<https://e.com>", "| *a* | b |", "||", "- [X] done", "+ [ ] todo", "-", | ||
| 1070 | ] | ||
| 1071 | |||
| 1072 | func randomDocument(_ rng: inout SeededGenerator) -> String { | ||
| 1073 | ``` | ||
| 1074 | |||
| 1075 | - [ ] **Step 2: Run the fuzz test and the private corpus** | ||
| 1076 | |||
| 1077 | Run: `swift test` then `ORGSTAR_CORPUS=~/Documents/notes swift test --filter corpusRoundTrips` | ||
| 1078 | Expected: all pass. | ||
| 1079 | |||
| 1080 | - [ ] **Step 3: Commit** | ||
| 1081 | |||
| 1082 | ```bash | ||
| 1083 | git add Tests | ||
| 1084 | git commit -m "Fuzz inline objects" | ||
| 1085 | ``` | ||