| @@ -0,0 +1,1085 @@ |
| |
1 | # Inline Objects Implementation Plan |
| |
2 | |
| |
3 | > **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. |
| |
4 | |
| |
5 | **Goal:** Parse org's inline objects into the lossless tree: emphasis, links, timestamps, footnote references, statistics cookies, targets, macros, inline source blocks, LaTeX fragments, line breaks and superscripts. |
| |
6 | |
| |
7 | **Architecture:** A character-level `InlineScanner` turns a run of text into text, newline and object nodes through the existing `GreenBuilder`. The parser feeds it whole paragraphs (so emphasis and links can cross one line break), heading titles, table cells, item paragraphs, footnote definitions, and planning and clock lines. `Timestamp` is both the recognizer and a public value type with repeaters and warnings. |
| |
8 | |
| |
9 | **Tech Stack:** Swift 6.2 tools, Swift Testing, Foundation only. |
| |
10 | |
| |
11 | **Spec:** `docs/design.md` ("OrgCore data model"); OrgSwift's `OrgInlineParser.swift` for the emphasis border rules. |
| |
12 | |
| |
13 | ## Global Constraints |
| |
14 | |
| |
15 | - Every character of the input lands in exactly one token; tree text equals source text. |
| |
16 | - Emphasis follows org's defaults: opener preceded by start, whitespace or `-({'"`; closer followed by end, whitespace or `-.,:!?;'")}\[`; body neither starts nor ends with whitespace; at most one line break. |
| |
17 | - `OrgCore` imports Foundation only. |
| |
18 | |
| |
19 | ## Out of scope |
| |
20 | |
| |
21 | Entities (`\alpha`), subscripts, `$...$` LaTeX, export snippets, citations, radio targets, inline footnote definitions as parsed content, and description-list terms. Each is recorded for a later plan. |
| |
22 | |
| |
23 | ## File structure |
| |
24 | |
| |
25 | | File | Responsibility | |
| |
26 | | --- | --- | |
| |
27 | | `Sources/OrgCore/Timestamp.swift` | `Timestamp` value type and `scanTimestamp` recognizer | |
| |
28 | | `Sources/OrgCore/Parser/Inline.swift` | `InlineScanner` and `Parser.inline(_:)` | |
| |
29 | | `Sources/OrgCore/Syntax/SyntaxKind.swift` | New token and object kinds; `title` becomes a node | |
| |
30 | | `Sources/OrgCore/Parser/Parser.swift` | Paragraph spans, item bullets and checkboxes, table cells, footnote labels, inline planning and clock lines | |
| |
31 | | `Sources/OrgCore/Parser/Lines.swift` | Empty final line ending sliced from the source, so spans can end there | |
| |
32 | |
| |
33 | --- |
| |
34 | |
| |
35 | ### Task 1: Timestamps |
| |
36 | |
| |
37 | **Files:** |
| |
38 | - Create: `Sources/OrgCore/Timestamp.swift` |
| |
39 | - Test: `Tests/OrgCoreTests/TimestampTests.swift` |
| |
40 | |
| |
41 | **Interfaces:** |
| |
42 | - Produces: `Timestamp` (`active`, `start: Point`, `end: Point?`, `repeater: Repeater?`, `warning: Warning?`), `Timestamp.parse(_:) -> Timestamp?`; internal `scanTimestamp(_ chars: [Character], at: Int, limit: Int) -> (stamp: Timestamp, end: Int)?`. |
| |
43 | |
| |
44 | - [ ] **Step 1: Write the failing tests** |
| |
45 | |
| |
46 | ```swift |
| |
47 | import Testing |
| |
48 | @testable import OrgCore |
| |
49 | |
| |
50 | struct TimestampTests { |
| |
51 | @Test func fullTimestamp() throws { |
| |
52 | let stamp = try #require(Timestamp.parse("<2026-10-04 Sun 10:00-11:30 .+1w/2w --2d>")) |
| |
53 | #expect(stamp.active) |
| |
54 | #expect(stamp.start == Timestamp.Point(year: 2026, month: 10, day: 4, hour: 10, minute: 0)) |
| |
55 | #expect(stamp.end == Timestamp.Point(year: 2026, month: 10, day: 4, hour: 11, minute: 30)) |
| |
56 | #expect(stamp.repeater == Timestamp.Repeater( |
| |
57 | kind: .restart, |
| |
58 | interval: .init(value: 1, unit: .week), |
| |
59 | habitDeadline: .init(value: 2, unit: .week) |
| |
60 | )) |
| |
61 | #expect(stamp.warning == Timestamp.Warning(firstOccurrenceOnly: true, interval: .init(value: 2, unit: .day))) |
| |
62 | } |
| |
63 | |
| |
64 | @Test func dateOnlyInactive() throws { |
| |
65 | let stamp = try #require(Timestamp.parse("[2026-10-04]")) |
| |
66 | #expect(!stamp.active) |
| |
67 | #expect(stamp.start.hour == nil) |
| |
68 | } |
| |
69 | |
| |
70 | @Test func dateRange() throws { |
| |
71 | let stamp = try #require(Timestamp.parse("[2026-10-04 Sun]--[2026-10-06 Tue]")) |
| |
72 | #expect(stamp.end?.day == 6) |
| |
73 | } |
| |
74 | |
| |
75 | @Test func repeaterKinds() { |
| |
76 | #expect(Timestamp.parse("<2026-10-04 +1d>")?.repeater?.kind == .cumulate) |
| |
77 | #expect(Timestamp.parse("<2026-10-04 ++1m>")?.repeater?.kind == .catchUp) |
| |
78 | #expect(Timestamp.parse("<2026-10-04 -3d +1y>")?.repeater?.interval == .init(value: 1, unit: .year)) |
| |
79 | } |
| |
80 | |
| |
81 | @Test func oneDigitHourAndOtherLanguages() { |
| |
82 | #expect(Timestamp.parse("<2026-10-04 9:05>")?.start.hour == 9) |
| |
83 | #expect(Timestamp.parse("<2026-10-04 dim.>") != nil) |
| |
84 | } |
| |
85 | |
| |
86 | @Test(arguments: ["<2026-10-4>", "<2026-10-04 Sun", "[2026-10-04 Sun]x", "<2026-10-04 Sun +1x>", "<2026-10-04]", "2026-10-04"]) |
| |
87 | func rejects(text: String) { |
| |
88 | #expect(Timestamp.parse(text) == nil) |
| |
89 | } |
| |
90 | } |
| |
91 | ``` |
| |
92 | |
| |
93 | - [ ] **Step 2: Run to verify failure** |
| |
94 | |
| |
95 | Run: `swift test --filter TimestampTests` |
| |
96 | Expected: build failure, `cannot find 'Timestamp' in scope`. |
| |
97 | |
| |
98 | - [ ] **Step 3: Implement** |
| |
99 | |
| |
100 | ```swift |
| |
101 | public struct Timestamp: Sendable, Equatable { |
| |
102 | public enum Unit: Character, Sendable { |
| |
103 | case hour = "h", day = "d", week = "w", month = "m", year = "y" |
| |
104 | } |
| |
105 | |
| |
106 | public struct Interval: Sendable, Equatable { |
| |
107 | public var value: Int |
| |
108 | public var unit: Unit |
| |
109 | } |
| |
110 | |
| |
111 | public enum RepeaterKind: String, Sendable { |
| |
112 | case cumulate = "+", catchUp = "++", restart = ".+" |
| |
113 | } |
| |
114 | |
| |
115 | public struct Repeater: Sendable, Equatable { |
| |
116 | public var kind: RepeaterKind |
| |
117 | public var interval: Interval |
| |
118 | /// The habit deadline from `.+2d/3d`. |
| |
119 | public var habitDeadline: Interval? |
| |
120 | } |
| |
121 | |
| |
122 | public struct Warning: Sendable, Equatable { |
| |
123 | /// `--` warns only for the first occurrence of a repeated timestamp. |
| |
124 | public var firstOccurrenceOnly: Bool |
| |
125 | public var interval: Interval |
| |
126 | } |
| |
127 | |
| |
128 | public struct Point: Sendable, Equatable { |
| |
129 | public var year: Int |
| |
130 | public var month: Int |
| |
131 | public var day: Int |
| |
132 | public var hour: Int? |
| |
133 | public var minute: Int? |
| |
134 | } |
| |
135 | |
| |
136 | public var active: Bool |
| |
137 | public var start: Point |
| |
138 | /// End of a same-day time range or of a `--` date range. |
| |
139 | public var end: Point? |
| |
140 | public var repeater: Repeater? |
| |
141 | public var warning: Warning? |
| |
142 | |
| |
143 | /// Parses exactly one timestamp or range, with nothing before or after it. |
| |
144 | public static func parse(_ text: some StringProtocol) -> Timestamp? { |
| |
145 | let chars = Array(text) |
| |
146 | guard let result = scanTimestamp(chars, at: 0, limit: chars.count), result.end == chars.count else { return nil } |
| |
147 | return result.stamp |
| |
148 | } |
| |
149 | } |
| |
150 | |
| |
151 | /// A timestamp or `--` range starting at `start`, and the index after it. |
| |
152 | func scanTimestamp(_ chars: [Character], at start: Int, limit: Int) -> (stamp: Timestamp, end: Int)? { |
| |
153 | guard let first = scanSingleTimestamp(chars, at: start, limit: limit) else { return nil } |
| |
154 | if first.stamp.end == nil, first.end + 2 < limit, chars[first.end] == "-", chars[first.end + 1] == "-", |
| |
155 | let second = scanSingleTimestamp(chars, at: first.end + 2, limit: limit), |
| |
156 | second.stamp.active == first.stamp.active, second.stamp.end == nil { |
| |
157 | var range = first.stamp |
| |
158 | range.end = second.stamp.start |
| |
159 | return (range, second.end) |
| |
160 | } |
| |
161 | return first |
| |
162 | } |
| |
163 | |
| |
164 | /// `<YYYY-MM-DD DAY HH:MM-HH:MM REPEATER WARNING>`, or the same in `[...]` for inactive. |
| |
165 | private func scanSingleTimestamp(_ chars: [Character], at start: Int, limit: Int) -> (stamp: Timestamp, end: Int)? { |
| |
166 | guard start < limit, chars[start] == "<" || chars[start] == "[" else { return nil } |
| |
167 | let active = chars[start] == "<" |
| |
168 | let close: Character = active ? ">" : "]" |
| |
169 | var j = start + 1 |
| |
170 | |
| |
171 | func number(_ minDigits: Int, _ maxDigits: Int) -> Int? { |
| |
172 | var k = j |
| |
173 | while k < limit, k - j < maxDigits, chars[k].isASCII, chars[k].isNumber { k += 1 } |
| |
174 | guard k - j >= minDigits else { return nil } |
| |
175 | let value = Int(String(chars[j..<k]))! |
| |
176 | j = k |
| |
177 | return value |
| |
178 | } |
| |
179 | |
| |
180 | func take(_ c: Character) -> Bool { |
| |
181 | guard j < limit, chars[j] == c else { return false } |
| |
182 | j += 1 |
| |
183 | return true |
| |
184 | } |
| |
185 | |
| |
186 | func interval() -> Timestamp.Interval? { |
| |
187 | let before = j |
| |
188 | guard let value = number(1, 9), j < limit, let unit = Timestamp.Unit(rawValue: chars[j]) else { |
| |
189 | j = before |
| |
190 | return nil |
| |
191 | } |
| |
192 | j += 1 |
| |
193 | return Timestamp.Interval(value: value, unit: unit) |
| |
194 | } |
| |
195 | |
| |
196 | func repeater() -> Timestamp.Repeater? { |
| |
197 | let before = j |
| |
198 | let kind: Timestamp.RepeaterKind |
| |
199 | if j + 1 < limit, chars[j] == ".", chars[j + 1] == "+" { |
| |
200 | kind = .restart |
| |
201 | j += 2 |
| |
202 | } else if j + 1 < limit, chars[j] == "+", chars[j + 1] == "+" { |
| |
203 | kind = .catchUp |
| |
204 | j += 2 |
| |
205 | } else if take("+") { |
| |
206 | kind = .cumulate |
| |
207 | } else { |
| |
208 | return nil |
| |
209 | } |
| |
210 | guard let value = interval() else { |
| |
211 | j = before |
| |
212 | return nil |
| |
213 | } |
| |
214 | var deadline: Timestamp.Interval? |
| |
215 | if take("/") { |
| |
216 | deadline = interval() |
| |
217 | if deadline == nil { |
| |
218 | j = before |
| |
219 | return nil |
| |
220 | } |
| |
221 | } |
| |
222 | return Timestamp.Repeater(kind: kind, interval: value, habitDeadline: deadline) |
| |
223 | } |
| |
224 | |
| |
225 | func warning() -> Timestamp.Warning? { |
| |
226 | let before = j |
| |
227 | guard take("-") else { return nil } |
| |
228 | let firstOnly = take("-") |
| |
229 | guard let value = interval() else { |
| |
230 | j = before |
| |
231 | return nil |
| |
232 | } |
| |
233 | return Timestamp.Warning(firstOccurrenceOnly: firstOnly, interval: value) |
| |
234 | } |
| |
235 | |
| |
236 | guard let year = number(4, 4), take("-"), let month = number(2, 2), take("-"), let day = number(2, 2) else { |
| |
237 | return nil |
| |
238 | } |
| |
239 | var stamp = Timestamp( |
| |
240 | active: active, |
| |
241 | start: Timestamp.Point(year: year, month: month, day: day, hour: nil, minute: nil), |
| |
242 | end: nil, repeater: nil, warning: nil |
| |
243 | ) |
| |
244 | |
| |
245 | // Day name: anything but digits, whitespace, `+`, `-`, `]` and `>`, in any language. |
| |
246 | if j < limit, chars[j] == " " { |
| |
247 | var k = j + 1 |
| |
248 | while k < limit, !(chars[k].isNumber || chars[k].isWhitespace || "+-]>".contains(chars[k])) { k += 1 } |
| |
249 | if k > j + 1 { j = k } |
| |
250 | } |
| |
251 | |
| |
252 | let beforeTime = j |
| |
253 | if take(" "), let hour = number(1, 2), take(":"), let minute = number(2, 2) { |
| |
254 | stamp.start.hour = hour |
| |
255 | stamp.start.minute = minute |
| |
256 | let beforeEnd = j |
| |
257 | if take("-"), let endHour = number(1, 2), take(":"), let endMinute = number(2, 2) { |
| |
258 | stamp.end = Timestamp.Point(year: year, month: month, day: day, hour: endHour, minute: endMinute) |
| |
259 | } else { |
| |
260 | j = beforeEnd |
| |
261 | } |
| |
262 | } else { |
| |
263 | j = beforeTime |
| |
264 | } |
| |
265 | |
| |
266 | while j < limit, chars[j] == " " { |
| |
267 | let beforeModifier = j |
| |
268 | j += 1 |
| |
269 | if let value = repeater() { |
| |
270 | stamp.repeater = value |
| |
271 | } else if let value = warning() { |
| |
272 | stamp.warning = value |
| |
273 | } else { |
| |
274 | j = beforeModifier |
| |
275 | break |
| |
276 | } |
| |
277 | } |
| |
278 | |
| |
279 | guard take(close) else { return nil } |
| |
280 | return (stamp, j) |
| |
281 | } |
| |
282 | ``` |
| |
283 | |
| |
284 | - [ ] **Step 4: Run to verify pass** |
| |
285 | |
| |
286 | Run: `swift test --filter TimestampTests` |
| |
287 | Expected: all pass. |
| |
288 | |
| |
289 | - [ ] **Step 5: Commit** |
| |
290 | |
| |
291 | ```bash |
| |
292 | git add Sources/OrgCore/Timestamp.swift Tests/OrgCoreTests/TimestampTests.swift |
| |
293 | git commit -m "Add timestamp recognizer and value type" |
| |
294 | ``` |
| |
295 | |
| |
296 | --- |
| |
297 | |
| |
298 | ### Task 2: Inline scanner and parser integration |
| |
299 | |
| |
300 | **Files:** |
| |
301 | - Create: `Sources/OrgCore/Parser/Inline.swift` |
| |
302 | - Modify: `Sources/OrgCore/Syntax/SyntaxKind.swift`, `Sources/OrgCore/Parser/Parser.swift`, `Sources/OrgCore/Parser/Lines.swift` |
| |
303 | - Test: `Tests/OrgCoreTests/InlineTests.swift`; update `Tests/OrgCoreTests/ParserSectionTests.swift` (title is a node) |
| |
304 | |
| |
305 | **Interfaces:** |
| |
306 | - Consumes: `scanTimestamp` (Task 1), `GreenBuilder`, `Parser`. |
| |
307 | - Produces: `InlineScanner(chars:)` with `scan(_:into:inLink:)`; `Parser.inline(_ text: Substring)`; `Parser.span(from:through:)`; node kinds `title`, `tableCell`, `bold`, `italic`, `underline`, `strikeThrough`, `verbatim`, `code`, `link`, `linkDescription`, `timestamp`, `footnoteReference`, `statisticsCookie`, `target`, `macro`, `inlineSourceBlock`, `latexFragment`, `lineBreak`, `superscript`; token kinds `marker`, `linkPath`, `bullet`, `checkbox`. |
| |
308 | |
| |
309 | - [ ] **Step 1: Write the failing tests** |
| |
310 | |
| |
311 | ```swift |
| |
312 | import Testing |
| |
313 | @testable import OrgCore |
| |
314 | |
| |
315 | /// `kind:text` for each object directly inside the first paragraph. |
| |
316 | func objects(_ text: String) -> [String] { |
| |
317 | let paragraph = OrgParser.parse(text).root.descendants().first { $0.kind == .paragraph }! |
| |
318 | return paragraph.children.map { "\($0.kind.rawValue):\($0.text)" } |
| |
319 | } |
| |
320 | |
| |
321 | struct InlineTests { |
| |
322 | @Test(arguments: [ |
| |
323 | ("*b* /i/ _u_ +s+ =v= ~c~", ["bold:*b*", "italic:/i/", "underline:_u_", "strikeThrough:+s+", "verbatim:=v=", "code:~c~"]), |
| |
324 | ("a*b* (*c*) \"*d*\"", ["bold:*c*", "bold:*d*"]), |
| |
325 | ("x *y * z", []), |
| |
326 | ("*y*z", []), |
| |
327 | ("*a\nb*", ["bold:*a\nb*"]), |
| |
328 | ("*a\nb\nc*", []), |
| |
329 | ("=*not bold*=", ["verbatim:=*not bold*="]), |
| |
330 | ("[[https://a.b][the *site*]]", ["link:[[https://a.b][the *site*]]"]), |
| |
331 | ("[[file:x.org]]", ["link:[[file:x.org]]"]), |
| |
332 | ("<https://a.b/c>", ["link:<https://a.b/c>"]), |
| |
333 | ("see https://a.b/c.", ["link:https://a.b/c"]), |
| |
334 | ("<2026-10-04 Sun 10:00-11:30 +1w -2d>", ["timestamp:<2026-10-04 Sun 10:00-11:30 +1w -2d>"]), |
| |
335 | ("[2026-10-04 Sun]--[2026-10-06 Tue]", ["timestamp:[2026-10-04 Sun]--[2026-10-06 Tue]"]), |
| |
336 | ("a [fn:1] and [fn::inline [x] note]", ["footnoteReference:[fn:1]", "footnoteReference:[fn::inline [x] note]"]), |
| |
337 | ("a [1/3] [50%]", ["statisticsCookie:[1/3]", "statisticsCookie:[50%]"]), |
| |
338 | ("a <<target>> {{{m(x, y)}}}", ["target:<<target>>", "macro:{{{m(x, y)}}}"]), |
| |
339 | ("\\(x^2\\) and src_sh[:results raw]{echo {a}}", ["latexFragment:\\(x^2\\)", "inlineSourceBlock:src_sh[:results raw]{echo {a}}"]), |
| |
340 | ("N^2 and e^{i}", ["superscript:^2", "superscript:^{i}"]), |
| |
341 | ("end\\\\\nnext", ["lineBreak:\\\\"]), |
| |
342 | ]) |
| |
343 | func recognizes(text: String, expected: [String]) { |
| |
344 | #expect(objects(text) == expected) |
| |
345 | } |
| |
346 | |
| |
347 | @Test func emphasisNests() { |
| |
348 | let bold = OrgParser.parse("*bold /italic/ x*\n").root.descendants().first { $0.kind == .bold }! |
| |
349 | #expect(bold.children.map(\.kind) == [.italic]) |
| |
350 | #expect(bold.tokens.first?.kind == .marker) |
| |
351 | } |
| |
352 | |
| |
353 | @Test func linkParts() { |
| |
354 | let link = OrgParser.parse("[[id:abc][desc]]\n").root.descendants().first { $0.kind == .link }! |
| |
355 | #expect(link.tokens.map(\.kind) == [.marker, .linkPath, .marker, .marker]) |
| |
356 | #expect(link.tokens.first { $0.kind == .linkPath }?.text == "id:abc") |
| |
357 | #expect(link.children.map(\.kind) == [.linkDescription]) |
| |
358 | } |
| |
359 | |
| |
360 | @Test func noLinksInsideLinkDescriptions() { |
| |
361 | let link = OrgParser.parse("[[a][see https://b.c]]\n").root.descendants().first { $0.kind == .link }! |
| |
362 | #expect(link.descendants().filter { $0.kind == .link }.count == 1) |
| |
363 | } |
| |
364 | |
| |
365 | @Test func headingTitlesHoldObjects() { |
| |
366 | let title = OrgParser.parse("* TODO Read [[https://a.b][it]] [1/2]\n").root.descendants().first { $0.kind == .title }! |
| |
367 | #expect(title.children.map(\.kind) == [.link, .statisticsCookie]) |
| |
368 | } |
| |
369 | |
| |
370 | @Test func planningHoldsTimestamps() { |
| |
371 | let planning = OrgParser.parse("* a\nDEADLINE: <2026-10-04 Sun -2d> SCHEDULED: <2026-10-01 Thu>\n").root |
| |
372 | .descendants().first { $0.kind == .planning }! |
| |
373 | #expect(planning.children.map(\.kind) == [.timestamp, .timestamp]) |
| |
374 | } |
| |
375 | |
| |
376 | @Test func itemsSplitBulletAndCheckbox() { |
| |
377 | let item = OrgParser.parse(" - [X] done *now*\n more\n").root.descendants().first { $0.kind == .item }! |
| |
378 | #expect(item.tokens.map(\.kind) == [.whitespace, .bullet, .whitespace, .checkbox, .whitespace]) |
| |
379 | #expect(item.children.map(\.kind) == [.paragraph]) |
| |
380 | #expect(item.children[0].text == "done *now*\n more\n") |
| |
381 | } |
| |
382 | |
| |
383 | @Test func emptyItem() { |
| |
384 | let item = OrgParser.parse("-\n").root.descendants().first { $0.kind == .item }! |
| |
385 | #expect(item.tokens.map(\.kind) == [.bullet, .newline]) |
| |
386 | } |
| |
387 | |
| |
388 | @Test func tableCells() { |
| |
389 | let rows = OrgParser.parse("| *a* || b\n|---+---|\n").root.descendants().filter { $0.kind == .tableRow } |
| |
390 | #expect(rows[0].children.map(\.text) == [" *a* ", "", " b"]) |
| |
391 | #expect(rows[0].children[0].children.map(\.kind) == [.bold]) |
| |
392 | #expect(rows[1].children.isEmpty) |
| |
393 | } |
| |
394 | |
| |
395 | @Test func footnoteDefinitionLabelIsNotAReference() { |
| |
396 | let definition = OrgParser.parse("[fn:1] see [fn:2]\n").root.descendants().first { $0.kind == .footnoteDefinition }! |
| |
397 | #expect(definition.tokens.first?.text == "[fn:1]") |
| |
398 | #expect(definition.children.map(\.kind) == [.footnoteReference]) |
| |
399 | } |
| |
400 | } |
| |
401 | ``` |
| |
402 | |
| |
403 | Update the two `ParserSectionTests` expectations that assumed `title` was a token: |
| |
404 | |
| |
405 | ```diff |
| |
406 | @@ -17,7 +17,7 @@ struct ParserSectionTests { |
| |
407 | } |
| |
408 | |
| |
409 | @Test func zerothSectionHoldsPreamble() { |
| |
410 | - #expect(nodeKinds("text\n* a\n") == [.document, .zerothSection, .paragraph, .section, .heading]) |
| |
411 | + #expect(nodeKinds("text\n* a\n") == [.document, .zerothSection, .paragraph, .section, .heading, .title]) |
| |
412 | } |
| |
413 | |
| |
414 | @Test func sectionsNestByLevel() { |
| |
415 | @@ -30,10 +30,11 @@ struct ParserSectionTests { |
| |
416 | } |
| |
417 | |
| |
418 | @Test func headingTokens() { |
| |
419 | - let parts = tokens(of: .heading, in: "** TODO [#A] Write the plan :work:urgent: \n") |
| |
420 | - #expect(parts.map(\.kind) == [.stars, .whitespace, .todoKeyword, .whitespace, .priority, .whitespace, .title, .whitespace, .tags, .whitespace, .newline]) |
| |
421 | + let text = "** TODO [#A] Write the plan :work:urgent: \n" |
| |
422 | + let parts = tokens(of: .heading, in: text) |
| |
423 | + #expect(parts.map(\.kind) == [.stars, .whitespace, .todoKeyword, .whitespace, .priority, .whitespace, .whitespace, .tags, .whitespace, .newline]) |
| |
424 | #expect(parts.first { $0.kind == .tags }?.text == ":work:urgent:") |
| |
425 | - #expect(parts.first { $0.kind == .title }?.text == "Write the plan") |
| |
426 | + #expect(OrgParser.parse(text).root.descendants().first { $0.kind == .title }?.text == "Write the plan") |
| |
427 | } |
| |
428 | |
| |
429 | @Test func todoKeywordsComeFromSettings() { |
| |
430 | ``` |
| |
431 | |
| |
432 | - [ ] **Step 2: Run to verify failure** |
| |
433 | |
| |
434 | Run: `swift test --filter InlineTests` |
| |
435 | Expected: build failure on the new `SyntaxKind` cases. |
| |
436 | |
| |
437 | - [ ] **Step 3: Replace `SyntaxKind.swift`** |
| |
438 | |
| |
439 | ```swift |
| |
440 | public enum SyntaxKind: String, Sendable { |
| |
441 | // Tokens |
| |
442 | case text, newline, whitespace |
| |
443 | case stars, todoKeyword, priority, tags |
| |
444 | case marker, linkPath, bullet, checkbox |
| |
445 | |
| |
446 | // Elements |
| |
447 | case document, zerothSection, section, heading, title |
| |
448 | case planning, propertyDrawer, nodeProperty, drawer, clock |
| |
449 | case paragraph, plainList, item, table, tableRow, tableCell, tableFormula |
| |
450 | case block, dynamicBlock, keyword, affiliatedKeyword |
| |
451 | case comment, fixedWidth, horizontalRule, footnoteDefinition |
| |
452 | |
| |
453 | // Objects |
| |
454 | case bold, italic, underline, strikeThrough, verbatim, code |
| |
455 | case link, linkDescription, timestamp, footnoteReference, statisticsCookie |
| |
456 | case target, macro, inlineSourceBlock, latexFragment, lineBreak, superscript |
| |
457 | } |
| |
458 | ``` |
| |
459 | |
| |
460 | - [ ] **Step 4: Create `Inline.swift`** |
| |
461 | |
| |
462 | ```swift |
| |
463 | /// Characters allowed before an emphasis opener, besides whitespace and the start of the run. |
| |
464 | private let emphasisPre: Set<Character> = ["-", "(", "{", "'", "\""] |
| |
465 | |
| |
466 | /// Characters allowed after an emphasis closer, besides whitespace and the end of the run. |
| |
467 | private let emphasisPost: Set<Character> = ["-", ".", ",", ":", "!", "?", ";", "'", "\"", ")", "}", "\\", "["] |
| |
468 | |
| |
469 | private let emphasisKinds: [Character: SyntaxKind] = [ |
| |
470 | "*": .bold, "/": .italic, "_": .underline, "+": .strikeThrough, "=": .verbatim, "~": .code, |
| |
471 | ] |
| |
472 | |
| |
473 | private let angleLinkSchemes: Set<String> = [ |
| |
474 | "http", "https", "mailto", "file", "id", "doi", "ftp", "news", "shell", "elisp", "info", "help", "attachment", |
| |
475 | ] |
| |
476 | |
| |
477 | private let plainLinkPrefixes = ["https://", "http://", "mailto:", "file:"] |
| |
478 | |
| |
479 | /// "\r\n" is a single Character, so both forms count. |
| |
480 | func isNewline(_ c: Character) -> Bool { |
| |
481 | c == "\n" || c == "\r\n" |
| |
482 | } |
| |
483 | |
| |
484 | enum InlineMatch { |
| |
485 | case emphasis(SyntaxKind, open: Int, close: Int) |
| |
486 | case link(path: Range<Int>, description: Range<Int>?, whole: Range<Int>) |
| |
487 | case object(SyntaxKind, Range<Int>) |
| |
488 | |
| |
489 | var end: Int { |
| |
490 | switch self { |
| |
491 | case .emphasis(_, _, let close): close + 1 |
| |
492 | case .link(_, _, let whole): whole.upperBound |
| |
493 | case .object(_, let range): range.upperBound |
| |
494 | } |
| |
495 | } |
| |
496 | } |
| |
497 | |
| |
498 | /// Turns a run of text into text, newline and object tokens. Every character ends up in |
| |
499 | /// exactly one token. |
| |
500 | struct InlineScanner { |
| |
501 | let chars: [Character] |
| |
502 | |
| |
503 | func scan(_ range: Range<Int>, into b: inout GreenBuilder, inLink: Bool = false) { |
| |
504 | var textStart = range.lowerBound |
| |
505 | var i = range.lowerBound |
| |
506 | while i < range.upperBound { |
| |
507 | if let match = match(at: i, in: range, inLink: inLink) { |
| |
508 | emitText(textStart..<i, into: &b) |
| |
509 | emit(match, into: &b, inLink: inLink) |
| |
510 | i = match.end |
| |
511 | textStart = i |
| |
512 | } else { |
| |
513 | i += 1 |
| |
514 | } |
| |
515 | } |
| |
516 | emitText(textStart..<range.upperBound, into: &b) |
| |
517 | } |
| |
518 | |
| |
519 | func emitText(_ range: Range<Int>, into b: inout GreenBuilder) { |
| |
520 | var start = range.lowerBound |
| |
521 | for k in range where isNewline(chars[k]) { |
| |
522 | if start < k { b.token(.text, string(start..<k)) } |
| |
523 | b.token(.newline, String(chars[k])) |
| |
524 | start = k + 1 |
| |
525 | } |
| |
526 | if start < range.upperBound { b.token(.text, string(start..<range.upperBound)) } |
| |
527 | } |
| |
528 | |
| |
529 | func emit(_ match: InlineMatch, into b: inout GreenBuilder, inLink: Bool) { |
| |
530 | switch match { |
| |
531 | case .emphasis(let kind, let open, let close): |
| |
532 | b.start(kind) |
| |
533 | b.token(.marker, string(open..<(open + 1))) |
| |
534 | if kind == .verbatim || kind == .code { |
| |
535 | emitText((open + 1)..<close, into: &b) |
| |
536 | } else { |
| |
537 | scan((open + 1)..<close, into: &b, inLink: inLink) |
| |
538 | } |
| |
539 | b.token(.marker, string(close..<(close + 1))) |
| |
540 | b.finish() |
| |
541 | case .link(let path, let description, let whole): |
| |
542 | b.start(.link) |
| |
543 | b.token(.marker, string(whole.lowerBound..<path.lowerBound)) |
| |
544 | b.token(.linkPath, string(path)) |
| |
545 | if let description { |
| |
546 | b.token(.marker, string(path.upperBound..<description.lowerBound)) |
| |
547 | b.start(.linkDescription) |
| |
548 | scan(description, into: &b, inLink: true) |
| |
549 | b.finish() |
| |
550 | b.token(.marker, string(description.upperBound..<whole.upperBound)) |
| |
551 | } else { |
| |
552 | b.token(.marker, string(path.upperBound..<whole.upperBound)) |
| |
553 | } |
| |
554 | b.finish() |
| |
555 | case .object(let kind, let range): |
| |
556 | b.start(kind) |
| |
557 | emitText(range, into: &b) |
| |
558 | b.finish() |
| |
559 | } |
| |
560 | } |
| |
561 | |
| |
562 | func string(_ range: Range<Int>) -> String { |
| |
563 | String(chars[range]) |
| |
564 | } |
| |
565 | |
| |
566 | func hasPrefix(_ s: String, at i: Int, _ limit: Int) -> Bool { |
| |
567 | var j = i |
| |
568 | for c in s { |
| |
569 | guard j < limit, chars[j] == c else { return false } |
| |
570 | j += 1 |
| |
571 | } |
| |
572 | return true |
| |
573 | } |
| |
574 | |
| |
575 | // MARK: - Recognizers |
| |
576 | |
| |
577 | func match(at i: Int, in range: Range<Int>, inLink: Bool) -> InlineMatch? { |
| |
578 | let limit = range.upperBound |
| |
579 | let previous: Character? = i > range.lowerBound ? chars[i - 1] : nil |
| |
580 | let afterWord = previous.map { $0.isLetter || $0.isNumber } ?? false |
| |
581 | |
| |
582 | switch chars[i] { |
| |
583 | case "[": |
| |
584 | if !inLink, let link = bracketLink(i, limit) { return link } |
| |
585 | if let end = footnoteReference(i, limit) { return .object(.footnoteReference, i..<end) } |
| |
586 | if let end = scanTimestamp(chars, at: i, limit: limit)?.end { return .object(.timestamp, i..<end) } |
| |
587 | if let end = statisticsCookie(i, limit) { return .object(.statisticsCookie, i..<end) } |
| |
588 | case "<": |
| |
589 | if let end = scanTimestamp(chars, at: i, limit: limit)?.end { return .object(.timestamp, i..<end) } |
| |
590 | if let end = target(i, limit) { return .object(.target, i..<end) } |
| |
591 | if !inLink, let end = angleLink(i, limit) { return .object(.link, i..<end) } |
| |
592 | case "{": |
| |
593 | if let end = macro(i, limit) { return .object(.macro, i..<end) } |
| |
594 | case "\\": |
| |
595 | if let end = lineBreak(i, limit) { return .object(.lineBreak, i..<end) } |
| |
596 | if let end = latexFragment(i, limit) { return .object(.latexFragment, i..<end) } |
| |
597 | case "^": |
| |
598 | if afterWord, let end = superscript(i, limit) { return .object(.superscript, i..<end) } |
| |
599 | case "s": |
| |
600 | if !afterWord, let end = inlineSourceBlock(i, limit) { return .object(.inlineSourceBlock, i..<end) } |
| |
601 | default: |
| |
602 | break |
| |
603 | } |
| |
604 | |
| |
605 | if !inLink, !afterWord, let end = plainLink(i, limit) { return .object(.link, i..<end) } |
| |
606 | |
| |
607 | if let kind = emphasisKinds[chars[i]], |
| |
608 | previous.map({ $0.isWhitespace || emphasisPre.contains($0) }) ?? true, |
| |
609 | let close = emphasisClose(i, limit) { |
| |
610 | return .emphasis(kind, open: i, close: close) |
| |
611 | } |
| |
612 | return nil |
| |
613 | } |
| |
614 | |
| |
615 | /// org's emphasis rules: the body neither starts nor ends with whitespace, spans at most |
| |
616 | /// one line break, and the closer is followed by whitespace, punctuation or the end. |
| |
617 | func emphasisClose(_ i: Int, _ limit: Int) -> Int? { |
| |
618 | let marker = chars[i] |
| |
619 | guard i + 1 < limit, !chars[i + 1].isWhitespace else { return nil } |
| |
620 | var newlines = 0 |
| |
621 | var j = i + 1 |
| |
622 | while j < limit { |
| |
623 | if isNewline(chars[j]) { |
| |
624 | newlines += 1 |
| |
625 | if newlines > 1 { return nil } |
| |
626 | } else if chars[j] == marker, j > i + 1, !chars[j - 1].isWhitespace { |
| |
627 | if j + 1 == limit || chars[j + 1].isWhitespace || emphasisPost.contains(chars[j + 1]) { return j } |
| |
628 | } |
| |
629 | j += 1 |
| |
630 | } |
| |
631 | return nil |
| |
632 | } |
| |
633 | |
| |
634 | /// `[[path]]` or `[[path][description]]`. |
| |
635 | func bracketLink(_ i: Int, _ limit: Int) -> InlineMatch? { |
| |
636 | guard i + 1 < limit, chars[i + 1] == "[" else { return nil } |
| |
637 | var j = i + 2 |
| |
638 | while j < limit, chars[j] != "]" { |
| |
639 | if chars[j] == "[" || isNewline(chars[j]) { return nil } |
| |
640 | if chars[j] == "\\", j + 1 < limit { j += 1 } |
| |
641 | j += 1 |
| |
642 | } |
| |
643 | guard j > i + 2, j + 1 < limit else { return nil } |
| |
644 | let path = (i + 2)..<j |
| |
645 | if chars[j + 1] == "]" { return .link(path: path, description: nil, whole: i..<(j + 2)) } |
| |
646 | guard chars[j + 1] == "[" else { return nil } |
| |
647 | let descriptionStart = j + 2 |
| |
648 | var depth = 0 |
| |
649 | var k = descriptionStart |
| |
650 | while k < limit { |
| |
651 | if chars[k] == "[" { |
| |
652 | depth += 1 |
| |
653 | } else if chars[k] == "]" { |
| |
654 | if depth == 0 { break } |
| |
655 | depth -= 1 |
| |
656 | } |
| |
657 | k += 1 |
| |
658 | } |
| |
659 | guard k > descriptionStart, k + 1 < limit, chars[k + 1] == "]" else { return nil } |
| |
660 | return .link(path: path, description: descriptionStart..<k, whole: i..<(k + 2)) |
| |
661 | } |
| |
662 | |
| |
663 | /// `[fn:label]`, `[fn:label:definition]` or `[fn::definition]`. |
| |
664 | func footnoteReference(_ i: Int, _ limit: Int) -> Int? { |
| |
665 | guard hasPrefix("[fn:", at: i, limit) else { return nil } |
| |
666 | var j = i + 4 |
| |
667 | while j < limit, chars[j].isLetter || chars[j].isNumber || chars[j] == "_" || chars[j] == "-" { j += 1 } |
| |
668 | guard j < limit else { return nil } |
| |
669 | if chars[j] == "]" { return j > i + 4 ? j + 1 : nil } |
| |
670 | guard chars[j] == ":" else { return nil } |
| |
671 | var depth = 0 |
| |
672 | j += 1 |
| |
673 | while j < limit { |
| |
674 | if chars[j] == "[" { |
| |
675 | depth += 1 |
| |
676 | } else if chars[j] == "]" { |
| |
677 | if depth == 0 { return j + 1 } |
| |
678 | depth -= 1 |
| |
679 | } |
| |
680 | j += 1 |
| |
681 | } |
| |
682 | return nil |
| |
683 | } |
| |
684 | |
| |
685 | /// `[1/3]`, `[/]`, `[50%]` or `[%]`. |
| |
686 | func statisticsCookie(_ i: Int, _ limit: Int) -> Int? { |
| |
687 | var j = i + 1 |
| |
688 | while j < limit, chars[j].isASCII, chars[j].isNumber { j += 1 } |
| |
689 | guard j < limit else { return nil } |
| |
690 | if chars[j] == "%" { |
| |
691 | j += 1 |
| |
692 | } else if chars[j] == "/" { |
| |
693 | j += 1 |
| |
694 | while j < limit, chars[j].isASCII, chars[j].isNumber { j += 1 } |
| |
695 | } else { |
| |
696 | return nil |
| |
697 | } |
| |
698 | guard j < limit, chars[j] == "]" else { return nil } |
| |
699 | return j + 1 |
| |
700 | } |
| |
701 | |
| |
702 | /// `<<target>>`. |
| |
703 | func target(_ i: Int, _ limit: Int) -> Int? { |
| |
704 | guard hasPrefix("<<", at: i, limit), i + 2 < limit, chars[i + 2] != "<" else { return nil } |
| |
705 | let start = i + 2 |
| |
706 | var j = start |
| |
707 | while j < limit, chars[j] != ">" { |
| |
708 | if chars[j] == "<" || isNewline(chars[j]) { return nil } |
| |
709 | j += 1 |
| |
710 | } |
| |
711 | guard j > start, j + 1 < limit, chars[j + 1] == ">", |
| |
712 | !chars[start].isWhitespace, !chars[j - 1].isWhitespace else { return nil } |
| |
713 | return j + 2 |
| |
714 | } |
| |
715 | |
| |
716 | /// `<scheme:path>` for a known scheme. |
| |
717 | func angleLink(_ i: Int, _ limit: Int) -> Int? { |
| |
718 | var j = i + 1 |
| |
719 | while j < limit, chars[j].isLetter { j += 1 } |
| |
720 | guard j < limit, chars[j] == ":", angleLinkSchemes.contains(string((i + 1)..<j).lowercased()) else { return nil } |
| |
721 | j += 1 |
| |
722 | let bodyStart = j |
| |
723 | while j < limit, chars[j] != ">" { |
| |
724 | if chars[j] == "<" || isNewline(chars[j]) { return nil } |
| |
725 | j += 1 |
| |
726 | } |
| |
727 | guard j < limit, j > bodyStart else { return nil } |
| |
728 | return j + 1 |
| |
729 | } |
| |
730 | |
| |
731 | /// A bare URL. Trailing sentence punctuation stays outside the link. |
| |
732 | func plainLink(_ i: Int, _ limit: Int) -> Int? { |
| |
733 | guard "hmf".contains(chars[i]), |
| |
734 | let prefix = plainLinkPrefixes.first(where: { hasPrefix($0, at: i, limit) }) else { return nil } |
| |
735 | let bodyStart = i + prefix.count |
| |
736 | var j = bodyStart |
| |
737 | while j < limit, !chars[j].isWhitespace, !"()<>[]\"".contains(chars[j]) { j += 1 } |
| |
738 | while j > bodyStart, ".,;:!?'".contains(chars[j - 1]) { j -= 1 } |
| |
739 | return j > bodyStart ? j : nil |
| |
740 | } |
| |
741 | |
| |
742 | /// `{{{name}}}` or `{{{name(arguments)}}}`. |
| |
743 | func macro(_ i: Int, _ limit: Int) -> Int? { |
| |
744 | guard hasPrefix("{{{", at: i, limit) else { return nil } |
| |
745 | var j = i + 3 |
| |
746 | guard j < limit, chars[j].isLetter else { return nil } |
| |
747 | while j < limit, chars[j].isLetter || chars[j].isNumber || chars[j] == "-" || chars[j] == "_" { j += 1 } |
| |
748 | if j < limit, chars[j] == "(" { |
| |
749 | var k = j + 1 |
| |
750 | while k < limit, !hasPrefix(")}}}", at: k, limit) { |
| |
751 | if isNewline(chars[k]) { return nil } |
| |
752 | k += 1 |
| |
753 | } |
| |
754 | return k < limit ? k + 4 : nil |
| |
755 | } |
| |
756 | return hasPrefix("}}}", at: j, limit) ? j + 3 : nil |
| |
757 | } |
| |
758 | |
| |
759 | /// `\\` at the end of a line, before optional trailing blanks. The newline stays outside. |
| |
760 | func lineBreak(_ i: Int, _ limit: Int) -> Int? { |
| |
761 | guard hasPrefix("\\\\", at: i, limit) else { return nil } |
| |
762 | var j = i + 2 |
| |
763 | while j < limit, chars[j] == " " || chars[j] == "\t" { j += 1 } |
| |
764 | guard j == limit || isNewline(chars[j]) else { return nil } |
| |
765 | return j |
| |
766 | } |
| |
767 | |
| |
768 | /// `\(...\)` or `\[...\]`. |
| |
769 | func latexFragment(_ i: Int, _ limit: Int) -> Int? { |
| |
770 | guard i + 1 < limit else { return nil } |
| |
771 | let closer: String |
| |
772 | switch chars[i + 1] { |
| |
773 | case "(": closer = "\\)" |
| |
774 | case "[": closer = "\\]" |
| |
775 | default: return nil |
| |
776 | } |
| |
777 | var j = i + 2 |
| |
778 | while j < limit { |
| |
779 | if hasPrefix(closer, at: j, limit) { return j + 2 } |
| |
780 | j += 1 |
| |
781 | } |
| |
782 | return nil |
| |
783 | } |
| |
784 | |
| |
785 | /// `^word` or `^{group}` after a letter or digit. |
| |
786 | func superscript(_ i: Int, _ limit: Int) -> Int? { |
| |
787 | var j = i + 1 |
| |
788 | guard j < limit else { return nil } |
| |
789 | if chars[j] == "{" { |
| |
790 | j += 1 |
| |
791 | while j < limit, chars[j] != "}" { |
| |
792 | if isNewline(chars[j]) { return nil } |
| |
793 | j += 1 |
| |
794 | } |
| |
795 | return j < limit ? j + 1 : nil |
| |
796 | } |
| |
797 | let start = j |
| |
798 | while j < limit, chars[j].isLetter || chars[j].isNumber { j += 1 } |
| |
799 | return j > start ? j : nil |
| |
800 | } |
| |
801 | |
| |
802 | /// `src_lang{body}` or `src_lang[headers]{body}`, on one line, with balanced braces. |
| |
803 | func inlineSourceBlock(_ i: Int, _ limit: Int) -> Int? { |
| |
804 | guard hasPrefix("src_", at: i, limit) else { return nil } |
| |
805 | var j = i + 4 |
| |
806 | let languageStart = j |
| |
807 | while j < limit, !chars[j].isWhitespace, chars[j] != "[", chars[j] != "{" { j += 1 } |
| |
808 | guard j > languageStart, j < limit else { return nil } |
| |
809 | if chars[j] == "[" { |
| |
810 | while j < limit, chars[j] != "]" { |
| |
811 | if isNewline(chars[j]) { return nil } |
| |
812 | j += 1 |
| |
813 | } |
| |
814 | guard j < limit else { return nil } |
| |
815 | j += 1 |
| |
816 | } |
| |
817 | guard j < limit, chars[j] == "{" else { return nil } |
| |
818 | var depth = 0 |
| |
819 | while j < limit { |
| |
820 | if isNewline(chars[j]) { return nil } |
| |
821 | if chars[j] == "{" { |
| |
822 | depth += 1 |
| |
823 | } else if chars[j] == "}" { |
| |
824 | depth -= 1 |
| |
825 | if depth == 0 { return j + 1 } |
| |
826 | } |
| |
827 | j += 1 |
| |
828 | } |
| |
829 | return nil |
| |
830 | } |
| |
831 | } |
| |
832 | |
| |
833 | extension Parser { |
| |
834 | mutating func inline(_ text: Substring) { |
| |
835 | let chars = Array(text) |
| |
836 | InlineScanner(chars: chars).scan(0..<chars.count, into: &builder) |
| |
837 | } |
| |
838 | } |
| |
839 | ``` |
| |
840 | |
| |
841 | - [ ] **Step 5: Wire the scanner into the parser** |
| |
842 | |
| |
843 | ```diff |
| |
844 | diff --git a/Sources/OrgCore/Parser/Lines.swift b/Sources/OrgCore/Parser/Lines.swift |
| |
845 | index 32a778a..5e9a077 100644 |
| |
846 | --- a/Sources/OrgCore/Parser/Lines.swift |
| |
847 | +++ b/Sources/OrgCore/Parser/Lines.swift |
| |
848 | @@ -27,7 +27,7 @@ func splitRawLines(_ text: String) -> [RawLine] { |
| |
849 | } |
| |
850 | } |
| |
851 | if lineStart != scalars.endIndex { |
| |
852 | - lines.append(RawLine(content: text[lineStart...], ending: "")) |
| |
853 | + lines.append(RawLine(content: text[lineStart...], ending: text[text.endIndex...])) |
| |
854 | } |
| |
855 | return lines |
| |
856 | } |
| |
857 | diff --git a/Sources/OrgCore/Parser/Parser.swift b/Sources/OrgCore/Parser/Parser.swift |
| |
858 | index ac210ff..6460486 100644 |
| |
859 | --- a/Sources/OrgCore/Parser/Parser.swift |
| |
860 | +++ b/Sources/OrgCore/Parser/Parser.swift |
| |
861 | @@ -77,10 +77,7 @@ struct Parser { |
| |
862 | headingLine(lines[i]) |
| |
863 | i += 1 |
| |
864 | if i < lines.count, info[i].cls == .planning { |
| |
865 | - builder.start(.planning) |
| |
866 | - line(i) |
| |
867 | - i += 1 |
| |
868 | - builder.finish() |
| |
869 | + single(.planning) |
| |
870 | } |
| |
871 | if i < lines.count, case .drawerBegin(let name) = info[i].cls, name.uppercased() == "PROPERTIES", |
| |
872 | let end = blockEnds[i] { |
| |
873 | @@ -173,13 +170,28 @@ struct Parser { |
| |
874 | } |
| |
875 | } |
| |
876 | |
| |
877 | + /// Kinds whose single line holds inline objects (timestamps on planning and clock lines). |
| |
878 | + static let inlineLineKinds: Set<SyntaxKind> = [.planning, .clock] |
| |
879 | + |
| |
880 | mutating func single(_ kind: SyntaxKind) { |
| |
881 | builder.start(kind) |
| |
882 | - line(i) |
| |
883 | + if Self.inlineLineKinds.contains(kind) { |
| |
884 | + let rest = whitespace(lines[i].content) |
| |
885 | + inline(rest) |
| |
886 | + builder.token(.newline, lines[i].ending) |
| |
887 | + } else { |
| |
888 | + line(i) |
| |
889 | + } |
| |
890 | i += 1 |
| |
891 | builder.finish() |
| |
892 | } |
| |
893 | |
| |
894 | + /// Source text from `start` through the end of line `last`, including its line ending. |
| |
895 | + func span(from start: Substring.Index, through last: Int) -> Substring { |
| |
896 | + let base = lines[last].ending.base |
| |
897 | + return base[start..<lines[last].ending.endIndex] |
| |
898 | + } |
| |
899 | + |
| |
900 | mutating func consecutive(_ kind: SyntaxKind, limit: Int, floor: Int?, matching: (LineClass) -> Bool) { |
| |
901 | builder.start(kind) |
| |
902 | repeat { |
| |
903 | @@ -192,7 +204,7 @@ struct Parser { |
| |
904 | mutating func table(limit: Int, floor: Int?) { |
| |
905 | builder.start(.table) |
| |
906 | while i < limit, info[i].cls == .tableRow, within(floor, i) { |
| |
907 | - single(.tableRow) |
| |
908 | + tableRow() |
| |
909 | } |
| |
910 | while i < limit, info[i].cls == .keyword(key: "TBLFM"), within(floor, i) { |
| |
911 | single(.tableFormula) |
| |
912 | @@ -200,15 +212,47 @@ struct Parser { |
| |
913 | builder.finish() |
| |
914 | } |
| |
915 | |
| |
916 | - mutating func footnoteDefinition(limit: Int) { |
| |
917 | - builder.start(.footnoteDefinition) |
| |
918 | - line(i) |
| |
919 | - i += 1 |
| |
920 | - while i < limit, info[i].cls == .plain { |
| |
921 | - line(i) |
| |
922 | - i += 1 |
| |
923 | + /// A rule row (`|---+---|`) is one text token. Other rows alternate `|` markers and cells; |
| |
924 | + /// every pair of pipes gets a cell, even an empty one, so columns line up. |
| |
925 | + mutating func tableRow() { |
| |
926 | + builder.start(.tableRow) |
| |
927 | + let rest = whitespace(lines[i].content) |
| |
928 | + if rest.hasPrefix("|-") { |
| |
929 | + builder.token(.text, rest) |
| |
930 | + } else { |
| |
931 | + var cellStart = rest.startIndex |
| |
932 | + var index = rest.startIndex |
| |
933 | + while index < rest.endIndex { |
| |
934 | + if rest[index] == "|" { |
| |
935 | + if index > rest.startIndex { tableCell(rest[cellStart..<index]) } |
| |
936 | + builder.token(.marker, "|") |
| |
937 | + cellStart = rest.index(after: index) |
| |
938 | + } |
| |
939 | + index = rest.index(after: index) |
| |
940 | + } |
| |
941 | + if cellStart < rest.endIndex { tableCell(rest[cellStart...]) } |
| |
942 | } |
| |
943 | + builder.token(.newline, lines[i].ending) |
| |
944 | builder.finish() |
| |
945 | + i += 1 |
| |
946 | + } |
| |
947 | + |
| |
948 | + mutating func tableCell(_ text: Substring) { |
| |
949 | + builder.start(.tableCell) |
| |
950 | + inline(text) |
| |
951 | + builder.finish() |
| |
952 | + } |
| |
953 | + |
| |
954 | + mutating func footnoteDefinition(limit: Int) { |
| |
955 | + var end = i + 1 |
| |
956 | + while end < limit, info[end].cls == .plain { end += 1 } |
| |
957 | + let content = lines[i].content |
| |
958 | + let close = content.firstIndex(of: "]")! |
| |
959 | + builder.start(.footnoteDefinition) |
| |
960 | + builder.token(.marker, content[...close]) |
| |
961 | + inline(span(from: content.index(after: close), through: end - 1)) |
| |
962 | + builder.finish() |
| |
963 | + i = end |
| |
964 | } |
| |
965 | |
| |
966 | mutating func list(limit: Int, floor: Int?) { |
| |
967 | @@ -224,8 +268,25 @@ struct Parser { |
| |
968 | /// inside the item when the item or list continues after it; two end the list. |
| |
969 | mutating func item(base: Int, limit: Int) { |
| |
970 | builder.start(.item) |
| |
971 | - line(i) |
| |
972 | - i += 1 |
| |
973 | + var rest = whitespace(lines[i].content) |
| |
974 | + let bullet = rest.prefix { $0 != " " && $0 != "\t" } |
| |
975 | + builder.token(.bullet, bullet) |
| |
976 | + rest = whitespace(rest.dropFirst(bullet.count)) |
| |
977 | + if let box = checkbox(rest) { |
| |
978 | + builder.token(.checkbox, box) |
| |
979 | + rest = whitespace(rest.dropFirst(box.count)) |
| |
980 | + } |
| |
981 | + // The rest of the bullet line and its continuation lines are the item's first paragraph. |
| |
982 | + var end = i + 1 |
| |
983 | + while end < limit, within(base, end), continuesParagraph(end) { end += 1 } |
| |
984 | + if rest.isEmpty, end == i + 1 { |
| |
985 | + builder.token(.newline, lines[i].ending) |
| |
986 | + } else { |
| |
987 | + builder.start(.paragraph) |
| |
988 | + inline(span(from: rest.startIndex, through: end - 1)) |
| |
989 | + builder.finish() |
| |
990 | + } |
| |
991 | + i = end |
| |
992 | while i < limit, !isHeading(i) { |
| |
993 | if info[i].cls == .blank { |
| |
994 | var j = i |
| |
995 | @@ -245,15 +306,23 @@ struct Parser { |
| |
996 | builder.finish() |
| |
997 | } |
| |
998 | |
| |
999 | + /// `[ ]`, `[X]`, `[x]` or `[-]`, followed by whitespace or end of line. |
| |
1000 | + func checkbox(_ s: Substring) -> Substring? { |
| |
1001 | + guard s.count >= 3, s.first == "[", "Xx -".contains(s.dropFirst().first!), |
| |
1002 | + s.dropFirst(2).first == "]" else { return nil } |
| |
1003 | + let after = s.dropFirst(3) |
| |
1004 | + guard after.isEmpty || after.first == " " || after.first == "\t" else { return nil } |
| |
1005 | + return s.prefix(3) |
| |
1006 | + } |
| |
1007 | + |
| |
1008 | + /// The paragraph's lines are one inline run, so emphasis and links can cross a line break. |
| |
1009 | mutating func paragraph(limit: Int, floor: Int?) { |
| |
1010 | + var end = i + 1 |
| |
1011 | + while end < limit, within(floor, end), continuesParagraph(end) { end += 1 } |
| |
1012 | builder.start(.paragraph) |
| |
1013 | - line(i) |
| |
1014 | - i += 1 |
| |
1015 | - while i < limit, within(floor, i), continuesParagraph(i) { |
| |
1016 | - line(i) |
| |
1017 | - i += 1 |
| |
1018 | - } |
| |
1019 | + inline(span(from: lines[i].content.startIndex, through: end - 1)) |
| |
1020 | builder.finish() |
| |
1021 | + i = end |
| |
1022 | } |
| |
1023 | |
| |
1024 | /// Lines that don't start an element of their own. |
| |
1025 | @@ -307,7 +376,11 @@ struct Parser { |
| |
1026 | } |
| |
1027 | |
| |
1028 | let parts = splitTags(rest) |
| |
1029 | - builder.token(.title, parts.title) |
| |
1030 | + if !parts.title.isEmpty { |
| |
1031 | + builder.start(.title) |
| |
1032 | + inline(parts.title) |
| |
1033 | + builder.finish() |
| |
1034 | + } |
| |
1035 | builder.token(.whitespace, parts.gap) |
| |
1036 | builder.token(.tags, parts.tags) |
| |
1037 | builder.token(.whitespace, parts.trailing) |
| |
1038 | ``` |
| |
1039 | |
| |
1040 | - [ ] **Step 6: Run all tests** |
| |
1041 | |
| |
1042 | Run: `swift test` |
| |
1043 | Expected: all pass. |
| |
1044 | |
| |
1045 | - [ ] **Step 7: Commit** |
| |
1046 | |
| |
1047 | ```bash |
| |
1048 | git add Sources Tests |
| |
1049 | git commit -m "Parse inline objects" |
| |
1050 | ``` |
| |
1051 | |
| |
1052 | --- |
| |
1053 | |
| |
1054 | ### Task 3: Fuzz the inline layer |
| |
1055 | |
| |
1056 | **Files:** |
| |
1057 | - Modify: `Tests/OrgCoreTests/RoundTripTests.swift` (fragment list) |
| |
1058 | |
| |
1059 | - [ ] **Step 1: Add inline fragments** |
| |
1060 | |
| |
1061 | ```diff |
| |
1062 | @@ -20,6 +20,10 @@ let fragments = [ |
| |
1063 | ":LOGBOOK:", "CLOCK: [2026-10-04 Sun 10:00]", "SCHEDULED: <2026-10-04 Sun>", "- item", " - nested", |
| |
1064 | "\t+ tab", "1. one", "| a | b |", "|---+---|", "#+TBLFM: $2=$1", "# comment", ": fixed", "-----", |
| |
1065 | "[fn:1] note", "#+NAME: x", "plain text", "é", "😀", "e\u{301}", " ", "\t", "\n", "\n", "\r\n", "\r", "", |
| |
1066 | + "*bold*", "/it/ ", "=v=", "~c~", "_u_", "+s+", "*", "/", "=", "[[https://a.b][d *b*]]", "[[x]]", "[[", "]]", |
| |
1067 | + "<2026-10-04 Sun 10:00 +1w -2d>", "[2026-10-04]--[2026-10-05]", "<", ">", "[fn:2]", "[fn::in [x]]", "[1/3]", |
| |
1068 | + "[50%]", "<<t>>", "{{{m(a)}}}", "\\(x\\)", "a\\\\", "x^2", "^{y}", "src_sh{echo}", "https://e.com.", |
| |
1069 | + "<https://e.com>", "| *a* | b |", "||", "- [X] done", "+ [ ] todo", "-", |
| |
1070 | ] |
| |
1071 | |
| |
1072 | func randomDocument(_ rng: inout SeededGenerator) -> String { |
| |
1073 | ``` |
| |
1074 | |
| |
1075 | - [ ] **Step 2: Run the fuzz test and the private corpus** |
| |
1076 | |
| |
1077 | Run: `swift test` then `ORGSTAR_CORPUS=~/Documents/notes swift test --filter corpusRoundTrips` |
| |
1078 | Expected: all pass. |
| |
1079 | |
| |
1080 | - [ ] **Step 3: Commit** |
| |
1081 | |
| |
1082 | ```bash |
| |
1083 | git add Tests |
| |
1084 | git commit -m "Fuzz inline objects" |
| |
1085 | ``` |