OrgCore parser foundation !1

merged merged by cmc on 2026-10-04 23:22 UTC · krz/orgstar:phase1-orgcore-parser into main

17 files changed, +3252 −0

Layout: unified · split

Package.swift added +14
@@ -0,0 +1,14 @@
1// swift-tools-version: 6.2
2import PackageDescription
3
4let package = Package(
5 name: "Orgstar",
6 platforms: [.macOS(.v26), .iOS(.v26)],
7 products: [
8 .library(name: "OrgCore", targets: ["OrgCore"])
9 ],
10 targets: [
11 .target(name: "OrgCore"),
12 .testTarget(name: "OrgCoreTests", dependencies: ["OrgCore"])
13 ]
14)
Sources/OrgCore/Parser/Lines.swift added +153
@@ -0,0 +1,153 @@
1/// One line of source: its content and its terminator ("", "\n" or "\r\n"), both as slices of
2/// the original text.
3struct RawLine {
4 let content: Substring
5 let ending: Substring
6}
7
8/// Splits on "\n" without normalizing anything. Works on unicode scalars, because String
9/// treats "\r\n" as a single Character.
10func splitRawLines(_ text: String) -> [RawLine] {
11 var lines: [RawLine] = []
12 let scalars = text.unicodeScalars
13 var lineStart = scalars.startIndex
14 var i = lineStart
15 while i != scalars.endIndex {
16 if scalars[i] == "\n" {
17 var contentEnd = i
18 if contentEnd > lineStart, scalars[scalars.index(before: i)] == "\r" {
19 contentEnd = scalars.index(before: i)
20 }
21 let next = scalars.index(after: i)
22 lines.append(RawLine(content: text[lineStart..<contentEnd], ending: text[contentEnd..<next]))
23 lineStart = next
24 i = next
25 } else {
26 i = scalars.index(after: i)
27 }
28 }
29 if lineStart != scalars.endIndex {
30 lines.append(RawLine(content: text[lineStart...], ending: ""))
31 }
32 return lines
33}
34
35enum LineClass: Equatable {
36 case blank
37 case heading(level: Int)
38 case blockBegin(name: String)
39 case blockEnd(name: String)
40 case dynamicBegin
41 case dynamicEnd
42 case drawerBegin(name: String)
43 case drawerEnd
44 case keyword(key: String)
45 case comment
46 case fixedWidth
47 case horizontalRule
48 case tableRow
49 case footnoteDefinition
50 case clock
51 case planning
52 case listItem
53 case plain
54}
55
56struct ClassifiedLine {
57 let cls: LineClass
58 /// Column of the first non-blank character, with tabs advancing to the next multiple of 8.
59 let indent: Int
60}
61
62func classifyLine(_ line: Substring) -> ClassifiedLine {
63 var column = 0
64 var rest = line
65 while let c = rest.first, c == " " || c == "\t" {
66 column = c == "\t" ? (column / 8 + 1) * 8 : column + 1
67 rest = rest.dropFirst()
68 }
69 if rest.isEmpty { return ClassifiedLine(cls: .blank, indent: column) }
70 return ClassifiedLine(cls: lineClass(rest, columnZero: column == 0), indent: column)
71}
72
73private func lineClass(_ rest: Substring, columnZero: Bool) -> LineClass {
74 let trimmed = rest.trimmingTrailingWhitespace
75
76 if columnZero, rest.first == "*" {
77 let stars = rest.prefix { $0 == "*" }
78 let after = rest.dropFirst(stars.count)
79 if after.isEmpty || after.first == " " || after.first == "\t" {
80 return .heading(level: stars.count)
81 }
82 }
83
84 if rest.hasPrefix("#+") {
85 let lower = trimmed.lowercased()
86 if lower.hasPrefix("#+begin_") {
87 let name = lower.dropFirst(8).prefix { !$0.isWhitespace }
88 if !name.isEmpty { return .blockBegin(name: String(name)) }
89 }
90 if lower.hasPrefix("#+end_") {
91 let name = lower.dropFirst(6)
92 if !name.isEmpty, !name.contains(where: \.isWhitespace) { return .blockEnd(name: String(name)) }
93 }
94 if lower.hasPrefix("#+begin:") { return .dynamicBegin }
95 if lower == "#+end:" { return .dynamicEnd }
96 if let colon = rest.firstIndex(of: ":") {
97 let key = rest[rest.index(rest.startIndex, offsetBy: 2)..<colon]
98 if !key.isEmpty, !key.contains(where: \.isWhitespace) { return .keyword(key: key.uppercased()) }
99 }
100 }
101
102 if trimmed == "#" || rest.hasPrefix("# ") || rest.hasPrefix("#\t") { return .comment }
103
104 if rest.first == ":" {
105 if trimmed == ":" || rest.hasPrefix(": ") || rest.hasPrefix(":\t") { return .fixedWidth }
106 if trimmed.uppercased() == ":END:" { return .drawerEnd }
107 if trimmed.count >= 3, trimmed.last == ":" {
108 let name = trimmed.dropFirst().dropLast()
109 if name.allSatisfy({ $0.isLetter || $0.isNumber || $0 == "_" || $0 == "-" }) {
110 return .drawerBegin(name: String(name))
111 }
112 }
113 }
114
115 if rest.first == "|" { return .tableRow }
116 if trimmed.count >= 5, trimmed.allSatisfy({ $0 == "-" }) { return .horizontalRule }
117
118 if columnZero, rest.hasPrefix("[fn:"), let close = rest.firstIndex(of: "]"),
119 close > rest.index(rest.startIndex, offsetBy: 4) {
120 return .footnoteDefinition
121 }
122
123 if rest.hasPrefix("CLOCK:") { return .clock }
124 if rest.hasPrefix("SCHEDULED:") || rest.hasPrefix("DEADLINE:") || rest.hasPrefix("CLOSED:") { return .planning }
125 if isListBullet(rest, indented: !columnZero) { return .listItem }
126 return .plain
127}
128
129/// `-`, `+`, `*` (indented only), `1.` or `1)`, followed by whitespace or end of line.
130/// Alphabetical bullets are off, as in org's default.
131private func isListBullet(_ rest: Substring, indented: Bool) -> Bool {
132 guard let first = rest.first else { return false }
133 let afterBullet: Substring
134 if first == "-" || first == "+" || (first == "*" && indented) {
135 afterBullet = rest.dropFirst()
136 } else if first.isASCII, first.isNumber {
137 let digits = rest.prefix { $0.isASCII && $0.isNumber }
138 let tail = rest.dropFirst(digits.count)
139 guard let separator = tail.first, separator == "." || separator == ")" else { return false }
140 afterBullet = tail.dropFirst()
141 } else {
142 return false
143 }
144 return afterBullet.isEmpty || afterBullet.first == " " || afterBullet.first == "\t"
145}
146
147extension Substring {
148 var trimmingTrailingWhitespace: Substring {
149 var s = self
150 while let last = s.last, last == " " || last == "\t" { s = s.dropLast() }
151 return s
152 }
153}
Sources/OrgCore/Parser/Parser.swift added +348
@@ -0,0 +1,348 @@
1public enum OrgParser {
2 public static func parse(_ text: String, defaults: OrgSettings = .default) -> OrgTree {
3 var parser = Parser(text: text, defaults: defaults)
4 return parser.run()
5 }
6}
7
8struct Parser {
9 let lines: [RawLine]
10 let info: [ClassifiedLine]
11 /// Begin line → end line, for blocks, dynamic blocks and drawers that are closed before the
12 /// next heading.
13 let blockEnds: [Int: Int]
14 let settings: OrgSettings
15 var builder = GreenBuilder()
16 var i = 0
17
18 init(text: String, defaults: OrgSettings) {
19 lines = splitRawLines(text)
20 info = lines.map { classifyLine($0.content) }
21 blockEnds = Parser.matchEnds(info)
22 settings = SettingsScanner.scan(lines: lines, info: info, blockEnds: blockEnds, defaults: defaults)
23 }
24
25 static func matchEnds(_ info: [ClassifiedLine]) -> [Int: Int] {
26 var ends: [Int: Int] = [:]
27 var k = 0
28 while k < info.count {
29 let isEnd: ((LineClass) -> Bool)?
30 switch info[k].cls {
31 case .blockBegin(let name): isEnd = { $0 == .blockEnd(name: name) }
32 case .dynamicBegin: isEnd = { $0 == .dynamicEnd }
33 case .drawerBegin: isEnd = { $0 == .drawerEnd }
34 default: isEnd = nil
35 }
36 if let isEnd {
37 var j = k + 1
38 while j < info.count {
39 if case .heading = info[j].cls { break }
40 if isEnd(info[j].cls) { ends[k] = j; break }
41 j += 1
42 }
43 // Block contents are verbatim, so nothing inside starts another element.
44 if let end = ends[k], !isDrawer(info[k].cls) { k = end }
45 }
46 k += 1
47 }
48 return ends
49 }
50
51 static func isDrawer(_ cls: LineClass) -> Bool {
52 if case .drawerBegin = cls { return true }
53 return false
54 }
55
56 mutating func run() -> OrgTree {
57 builder.start(.document)
58 if !lines.isEmpty, !isHeading(0) {
59 builder.start(.zerothSection)
60 parseContent(limit: lines.count)
61 builder.finish()
62 }
63 while i < lines.count, case .heading(let level) = info[i].cls {
64 parseSection(level: level)
65 }
66 builder.finish()
67 return OrgTree(green: builder.build(), settings: settings)
68 }
69
70 func isHeading(_ k: Int) -> Bool {
71 if case .heading = info[k].cls { return true }
72 return false
73 }
74
75 mutating func parseSection(level: Int) {
76 builder.start(.section)
77 headingLine(lines[i])
78 i += 1
79 if i < lines.count, info[i].cls == .planning {
80 builder.start(.planning)
81 line(i)
82 i += 1
83 builder.finish()
84 }
85 if i < lines.count, case .drawerBegin(let name) = info[i].cls, name.uppercased() == "PROPERTIES",
86 let end = blockEnds[i] {
87 propertyDrawer(end: end)
88 }
89 parseContent(limit: lines.count)
90 while i < lines.count, case .heading(let child) = info[i].cls, child > level {
91 parseSection(level: child)
92 }
93 builder.finish()
94 }
95
96 mutating func propertyDrawer(end: Int) {
97 builder.start(.propertyDrawer)
98 line(i)
99 i += 1
100 while i < end {
101 if info[i].cls == .blank {
102 line(i)
103 } else {
104 builder.start(.nodeProperty)
105 line(i)
106 builder.finish()
107 }
108 i += 1
109 }
110 line(i)
111 i += 1
112 builder.finish()
113 }
114
115 /// Elements until `limit` or the next heading.
116 mutating func parseContent(limit: Int) {
117 while i < limit, !isHeading(i) {
118 element(limit: limit, floor: nil)
119 }
120 }
121
122 static let affiliatedKeys: Set<String> = ["NAME", "CAPTION", "RESULTS", "HEADER", "PLOT"]
123
124 /// One element starting at `i`. `floor` is the indent of the enclosing list item, if any:
125 /// non-blank lines at or left of it end the element.
126 mutating func element(limit: Int, floor: Int?) {
127 switch info[i].cls {
128 case .blank:
129 line(i)
130 i += 1
131 case .blockBegin, .dynamicBegin:
132 if let end = blockEnds[i], end < limit {
133 builder.start(info[i].cls == .dynamicBegin ? .dynamicBlock : .block)
134 while i <= end {
135 line(i)
136 i += 1
137 }
138 builder.finish()
139 } else {
140 paragraph(limit: limit, floor: floor)
141 }
142 case .drawerBegin:
143 if let end = blockEnds[i], end < limit {
144 builder.start(.drawer)
145 line(i)
146 i += 1
147 parseContent(limit: end)
148 line(i)
149 i += 1
150 builder.finish()
151 } else {
152 paragraph(limit: limit, floor: floor)
153 }
154 case .keyword(let key):
155 let affiliated = Self.affiliatedKeys.contains(key) || key.hasPrefix("ATTR_")
156 single(affiliated ? .affiliatedKeyword : .keyword)
157 case .comment:
158 consecutive(.comment, limit: limit, floor: floor) { $0 == .comment }
159 case .fixedWidth:
160 consecutive(.fixedWidth, limit: limit, floor: floor) { $0 == .fixedWidth }
161 case .horizontalRule:
162 single(.horizontalRule)
163 case .clock:
164 single(.clock)
165 case .tableRow:
166 table(limit: limit, floor: floor)
167 case .footnoteDefinition:
168 footnoteDefinition(limit: limit)
169 case .listItem:
170 list(limit: limit, floor: floor)
171 default:
172 paragraph(limit: limit, floor: floor)
173 }
174 }
175
176 mutating func single(_ kind: SyntaxKind) {
177 builder.start(kind)
178 line(i)
179 i += 1
180 builder.finish()
181 }
182
183 mutating func consecutive(_ kind: SyntaxKind, limit: Int, floor: Int?, matching: (LineClass) -> Bool) {
184 builder.start(kind)
185 repeat {
186 line(i)
187 i += 1
188 } while i < limit && matching(info[i].cls) && within(floor, i)
189 builder.finish()
190 }
191
192 mutating func table(limit: Int, floor: Int?) {
193 builder.start(.table)
194 while i < limit, info[i].cls == .tableRow, within(floor, i) {
195 single(.tableRow)
196 }
197 while i < limit, info[i].cls == .keyword(key: "TBLFM"), within(floor, i) {
198 single(.tableFormula)
199 }
200 builder.finish()
201 }
202
203 mutating func footnoteDefinition(limit: Int) {
204 builder.start(.footnoteDefinition)
205 line(i)
206 i += 1
207 while i < limit, info[i].cls == .plain {
208 line(i)
209 i += 1
210 }
211 builder.finish()
212 }
213
214 mutating func list(limit: Int, floor: Int?) {
215 let base = info[i].indent
216 builder.start(.plainList)
217 while i < limit, info[i].cls == .listItem, info[i].indent == base, within(floor, i) {
218 item(base: base, limit: limit)
219 }
220 builder.finish()
221 }
222
223 /// An item's first line, then everything indented past its bullet. One blank line stays
224 /// inside the item when the item or list continues after it; two end the list.
225 mutating func item(base: Int, limit: Int) {
226 builder.start(.item)
227 line(i)
228 i += 1
229 while i < limit, !isHeading(i) {
230 if info[i].cls == .blank {
231 var j = i
232 while j < limit, info[j].cls == .blank { j += 1 }
233 guard j - i < 2, j < limit else { break }
234 let continuesItem = info[j].indent > base
235 let nextSibling = info[j].cls == .listItem && info[j].indent == base
236 guard continuesItem || nextSibling else { break }
237 line(i)
238 i += 1
239 if nextSibling { break }
240 continue
241 }
242 guard info[i].indent > base else { break }
243 element(limit: limit, floor: base)
244 }
245 builder.finish()
246 }
247
248 mutating func paragraph(limit: Int, floor: Int?) {
249 builder.start(.paragraph)
250 line(i)
251 i += 1
252 while i < limit, within(floor, i), continuesParagraph(i) {
253 line(i)
254 i += 1
255 }
256 builder.finish()
257 }
258
259 /// Lines that don't start an element of their own.
260 func continuesParagraph(_ k: Int) -> Bool {
261 switch info[k].cls {
262 case .plain, .planning, .blockEnd, .dynamicEnd, .drawerEnd:
263 return true
264 case .blockBegin, .dynamicBegin, .drawerBegin:
265 return blockEnds[k] == nil
266 default:
267 return false
268 }
269 }
270
271 func within(_ floor: Int?, _ k: Int) -> Bool {
272 guard let floor else { return true }
273 return info[k].indent > floor
274 }
275
276 // MARK: - Tokens
277
278 /// A whole line as leading whitespace, content and line ending.
279 mutating func line(_ k: Int) {
280 let rest = whitespace(lines[k].content)
281 builder.token(.text, rest)
282 builder.token(.newline, lines[k].ending)
283 }
284
285 mutating func whitespace(_ s: Substring) -> Substring {
286 let ws = s.prefix { $0 == " " || $0 == "\t" }
287 builder.token(.whitespace, ws)
288 return s.dropFirst(ws.count)
289 }
290
291 mutating func headingLine(_ raw: RawLine) {
292 builder.start(.heading)
293 var rest = raw.content
294 let stars = rest.prefix { $0 == "*" }
295 builder.token(.stars, stars)
296 rest = whitespace(rest.dropFirst(stars.count))
297
298 let word = rest.prefix { $0 != " " && $0 != "\t" }
299 if !word.isEmpty, settings.todoKeywordNames.contains(String(word)) {
300 builder.token(.todoKeyword, word)
301 rest = whitespace(rest.dropFirst(word.count))
302 }
303
304 if let cookie = priorityCookie(rest) {
305 builder.token(.priority, cookie)
306 rest = whitespace(rest.dropFirst(cookie.count))
307 }
308
309 let parts = splitTags(rest)
310 builder.token(.title, parts.title)
311 builder.token(.whitespace, parts.gap)
312 builder.token(.tags, parts.tags)
313 builder.token(.whitespace, parts.trailing)
314 builder.token(.newline, raw.ending)
315 builder.finish()
316 }
317
318 /// `[#A]` or `[#10]`, followed by whitespace or end of line.
319 func priorityCookie(_ s: Substring) -> Substring? {
320 guard s.hasPrefix("[#"), let close = s.firstIndex(of: "]") else { return nil }
321 let value = s[s.index(s.startIndex, offsetBy: 2)..<close]
322 let valid = (value.count == 1 && value.first!.isLetter && value.first!.isUppercase)
323 || (!value.isEmpty && value.allSatisfy { $0.isASCII && $0.isNumber })
324 guard valid else { return nil }
325 let after = s[s.index(after: close)...]
326 guard after.isEmpty || after.first == " " || after.first == "\t" else { return nil }
327 return s[...close]
328 }
329
330 func splitTags(_ s: Substring) -> (title: Substring, gap: Substring, tags: Substring, trailing: Substring) {
331 let trimmed = s.trimmingTrailingWhitespace
332 let trailing = s[trimmed.endIndex...]
333 let none = (title: trimmed, gap: Substring(), tags: Substring(), trailing: trailing)
334 guard trimmed.last == ":" else { return none }
335 let tagStart = trimmed.lastIndex { $0 == " " || $0 == "\t" }.map { trimmed.index(after: $0) } ?? trimmed.startIndex
336 let tags = trimmed[tagStart...]
337 guard tags.count >= 3, tags.first == ":", isTagString(tags) else { return none }
338 let before = trimmed[..<tagStart]
339 let title = before.trimmingTrailingWhitespace
340 return (title, before[title.endIndex...], tags, trailing)
341 }
342
343 func isTagString(_ tags: Substring) -> Bool {
344 tags.dropFirst().dropLast().split(separator: ":", omittingEmptySubsequences: false).allSatisfy { tag in
345 !tag.isEmpty && tag.allSatisfy { $0.isLetter || $0.isNumber || "_@#%".contains($0) }
346 }
347 }
348}
Sources/OrgCore/Parser/Settings.swift added +129
@@ -0,0 +1,129 @@
1public struct TodoKeyword: Sendable, Hashable {
2 public var name: String
3 public var fastKey: Character?
4 /// Logging flag when entering the state (`!` or `@`), from `NAME(k!/@)`.
5 public var logOnEnter: String?
6 /// Logging flag when leaving the state.
7 public var logOnLeave: String?
8
9 public init(name: String, fastKey: Character? = nil, logOnEnter: String? = nil, logOnLeave: String? = nil) {
10 self.name = name
11 self.fastKey = fastKey
12 self.logOnEnter = logOnEnter
13 self.logOnLeave = logOnLeave
14 }
15}
16
17public struct TodoSequence: Sendable, Equatable {
18 public enum Kind: Sendable, Equatable { case sequence, type }
19
20 public var kind: Kind
21 public var active: [TodoKeyword]
22 public var done: [TodoKeyword]
23
24 public init(kind: Kind, active: [TodoKeyword], done: [TodoKeyword]) {
25 self.kind = kind
26 self.active = active
27 self.done = done
28 }
29}
30
31public struct Priorities: Sendable, Equatable {
32 public var highest: String
33 public var lowest: String
34 public var `default`: String
35
36 public init(highest: String, lowest: String, default: String) {
37 self.highest = highest
38 self.lowest = lowest
39 self.default = `default`
40 }
41}
42
43public struct OrgSettings: Sendable, Equatable {
44 public var todoSequences: [TodoSequence]
45 public var priorities: Priorities
46
47 public init(todoSequences: [TodoSequence], priorities: Priorities) {
48 self.todoSequences = todoSequences
49 self.priorities = priorities
50 }
51
52 public static let `default` = OrgSettings(
53 todoSequences: [TodoSequence(kind: .sequence, active: [TodoKeyword(name: "TODO")], done: [TodoKeyword(name: "DONE")])],
54 priorities: Priorities(highest: "A", lowest: "C", default: "B")
55 )
56
57 public var todoKeywordNames: Set<String> {
58 Set(todoSequences.flatMap { ($0.active + $0.done).map(\.name) })
59 }
60
61 public func isDone(_ name: String) -> Bool {
62 todoSequences.contains { $0.done.contains { $0.name == name } }
63 }
64}
65
66enum SettingsScanner {
67 /// Reads `#+TODO`, `#+SEQ_TODO`, `#+TYP_TODO` and `#+PRIORITIES` outside blocks. Any TODO
68 /// line replaces the default sequences, as in org.
69 static func scan(lines: [RawLine], info: [ClassifiedLine], blockEnds: [Int: Int], defaults: OrgSettings) -> OrgSettings {
70 var sequences: [TodoSequence] = []
71 var priorities = defaults.priorities
72 var k = 0
73 while k < lines.count {
74 switch info[k].cls {
75 case .blockBegin, .dynamicBegin:
76 if let end = blockEnds[k] { k = end }
77 case .keyword(let key):
78 let value = keywordValue(lines[k].content)
79 switch key {
80 case "TODO", "SEQ_TODO":
81 if let s = todoSequence(value, kind: .sequence) { sequences.append(s) }
82 case "TYP_TODO":
83 if let s = todoSequence(value, kind: .type) { sequences.append(s) }
84 case "PRIORITIES":
85 let words = value.split(whereSeparator: \.isWhitespace)
86 if words.count == 3 {
87 priorities = Priorities(highest: String(words[0]), lowest: String(words[1]), default: String(words[2]))
88 }
89 default:
90 break
91 }
92 default:
93 break
94 }
95 k += 1
96 }
97 return OrgSettings(todoSequences: sequences.isEmpty ? defaults.todoSequences : sequences, priorities: priorities)
98 }
99
100 static func keywordValue(_ line: Substring) -> Substring {
101 guard let colon = line.firstIndex(of: ":") else { return "" }
102 return line[line.index(after: colon)...]
103 }
104
105 static func todoSequence(_ value: Substring, kind: TodoSequence.Kind) -> TodoSequence? {
106 let words = value.split(whereSeparator: \.isWhitespace)
107 guard !words.isEmpty else { return nil }
108 if let bar = words.firstIndex(of: "|") {
109 return TodoSequence(kind: kind, active: words[..<bar].map(todoKeyword), done: words[(bar + 1)...].map(todoKeyword))
110 }
111 return TodoSequence(kind: kind, active: words.dropLast().map(todoKeyword), done: [todoKeyword(words.last!)])
112 }
113
114 /// `NAME`, or `NAME(spec)` where spec is an optional fast key followed by `enter/leave`
115 /// logging flags.
116 static func todoKeyword(_ word: Substring) -> TodoKeyword {
117 guard let open = word.firstIndex(of: "("), word.last == ")" else { return TodoKeyword(name: String(word)) }
118 var spec = word[word.index(after: open)..<word.index(before: word.endIndex)]
119 var fastKey: Character?
120 if let first = spec.first, first != "!", first != "@", first != "/" {
121 fastKey = first
122 spec = spec.dropFirst()
123 }
124 let parts = spec.split(separator: "/", omittingEmptySubsequences: false)
125 let enter = parts.first.flatMap { $0.isEmpty ? nil : String($0) }
126 let leave = parts.count > 1 && !parts[1].isEmpty ? String(parts[1]) : nil
127 return TodoKeyword(name: String(word[..<open]), fastKey: fastKey, logOnEnter: enter, logOnLeave: leave)
128 }
129}
Sources/OrgCore/SourceText.swift added +40
@@ -0,0 +1,40 @@
1import Foundation
2
3/// A file's bytes and their decoded text. Only UTF-8 (with or without a BOM) is editable;
4/// anything else decodes with replacement characters for display and is never written back.
5public struct SourceText: Sendable {
6 public let originalBytes: [UInt8]
7 public let hasBOM: Bool
8 public let isValidUTF8: Bool
9 /// Decoded text without the BOM.
10 public let text: String
11
12 private static let bom: [UInt8] = [0xEF, 0xBB, 0xBF]
13
14 public init(bytes: [UInt8]) {
15 originalBytes = bytes
16 hasBOM = bytes.starts(with: Self.bom)
17 let body = hasBOM ? Array(bytes.dropFirst(3)) : bytes
18 if let decoded = String(validating: body, as: UTF8.self) {
19 text = decoded
20 isValidUTF8 = true
21 } else {
22 text = String(decoding: body, as: UTF8.self)
23 isValidUTF8 = false
24 }
25 }
26
27 public init(_ text: String) {
28 self.init(bytes: Array(text.utf8))
29 }
30
31 public var isEditable: Bool { isValidUTF8 }
32
33 /// Bytes to write for `newText`. Unchanged text returns the original bytes. Valid UTF-8
34 /// round-trips through `String` unchanged, so untouched spans keep their exact bytes.
35 public func encode(_ newText: String) -> [UInt8] {
36 if newText == text { return originalBytes }
37 precondition(isEditable, "a read-only document cannot be re-encoded")
38 return (hasBOM ? Self.bom : []) + Array(newText.utf8)
39 }
40}
Sources/OrgCore/Syntax/GreenTree.swift added +87
@@ -0,0 +1,87 @@
1public struct GreenToken: Sendable, Equatable {
2 public let kind: SyntaxKind
3 public let text: String
4 /// Length in UTF-16 code units.
5 public let length: Int
6
7 public init(kind: SyntaxKind, text: String) {
8 self.kind = kind
9 self.text = text
10 self.length = text.utf16.count
11 }
12}
13
14/// An immutable node. Stores only kind, children and length, so unchanged subtrees can be
15/// shared between versions of a document.
16public final class GreenNode: Sendable, Equatable {
17 public let kind: SyntaxKind
18 public let children: [GreenElement]
19 /// Length in UTF-16 code units.
20 public let length: Int
21
22 public init(kind: SyntaxKind, children: [GreenElement]) {
23 self.kind = kind
24 self.children = children
25 self.length = children.reduce(0) { $0 + $1.length }
26 }
27
28 public static func == (lhs: GreenNode, rhs: GreenNode) -> Bool {
29 lhs === rhs || (lhs.kind == rhs.kind && lhs.children == rhs.children)
30 }
31
32 public var text: String {
33 var out = ""
34 write(to: &out)
35 return out
36 }
37
38 func write(to out: inout String) {
39 for child in children {
40 switch child {
41 case .node(let node): node.write(to: &out)
42 case .token(let token): out += token.text
43 }
44 }
45 }
46}
47
48public enum GreenElement: Sendable, Equatable {
49 case node(GreenNode)
50 case token(GreenToken)
51
52 public var length: Int {
53 switch self {
54 case .node(let node): node.length
55 case .token(let token): token.length
56 }
57 }
58}
59
60struct GreenBuilder {
61 private var stack: [(kind: SyntaxKind, children: [GreenElement])] = []
62 private var root: GreenNode?
63
64 mutating func start(_ kind: SyntaxKind) {
65 stack.append((kind, []))
66 }
67
68 mutating func token(_ kind: SyntaxKind, _ text: some StringProtocol) {
69 guard !text.isEmpty else { return }
70 stack[stack.count - 1].children.append(.token(GreenToken(kind: kind, text: String(text))))
71 }
72
73 mutating func finish() {
74 let (kind, children) = stack.removeLast()
75 let node = GreenNode(kind: kind, children: children)
76 if stack.isEmpty {
77 root = node
78 } else {
79 stack[stack.count - 1].children.append(.node(node))
80 }
81 }
82
83 func build() -> GreenNode {
84 precondition(stack.isEmpty, "unfinished nodes")
85 return root!
86 }
87}
Sources/OrgCore/Syntax/SyntaxKind.swift added +12
@@ -0,0 +1,12 @@
1public enum SyntaxKind: String, Sendable {
2 // Tokens
3 case text, newline, whitespace
4 case stars, todoKeyword, priority, title, tags
5
6 // Nodes
7 case document, zerothSection, section, heading
8 case planning, propertyDrawer, nodeProperty, drawer, clock
9 case paragraph, plainList, item, table, tableRow, tableFormula
10 case block, dynamicBlock, keyword, affiliatedKeyword
11 case comment, fixedWidth, horizontalRule, footnoteDefinition
12}
Sources/OrgCore/Syntax/SyntaxNode.swift added +59
@@ -0,0 +1,59 @@
1/// A view of a green node at an absolute offset, with a parent link. Created on demand.
2public final class SyntaxNode: Sendable {
3 public let green: GreenNode
4 public let offset: Int
5 public let parent: SyntaxNode?
6
7 init(green: GreenNode, offset: Int, parent: SyntaxNode?) {
8 self.green = green
9 self.offset = offset
10 self.parent = parent
11 }
12
13 public var kind: SyntaxKind { green.kind }
14 public var range: Range<Int> { offset..<(offset + green.length) }
15 public var text: String { green.text }
16
17 public var children: [SyntaxNode] {
18 var result: [SyntaxNode] = []
19 var at = offset
20 for child in green.children {
21 if case .node(let node) = child {
22 result.append(SyntaxNode(green: node, offset: at, parent: self))
23 }
24 at += child.length
25 }
26 return result
27 }
28
29 public var tokens: [SyntaxToken] {
30 var result: [SyntaxToken] = []
31 var at = offset
32 for child in green.children {
33 if case .token(let token) = child {
34 result.append(SyntaxToken(kind: token.kind, text: token.text, range: at..<(at + token.length)))
35 }
36 at += child.length
37 }
38 return result
39 }
40
41 /// This node and every node below it, in document order.
42 public func descendants() -> [SyntaxNode] {
43 [self] + children.flatMap { $0.descendants() }
44 }
45}
46
47public struct SyntaxToken: Sendable, Equatable {
48 public let kind: SyntaxKind
49 public let text: String
50 public let range: Range<Int>
51}
52
53public struct OrgTree: Sendable {
54 public let green: GreenNode
55 public let settings: OrgSettings
56
57 public var root: SyntaxNode { SyntaxNode(green: green, offset: 0, parent: nil) }
58 public var text: String { green.text }
59}
Tests/OrgCoreTests/LinesTests.swift added +64
@@ -0,0 +1,64 @@
1import Testing
2@testable import OrgCore
3
4struct LinesTests {
5 @Test func splitKeepsEveryEnding() {
6 let lines = splitRawLines("a\r\nb\n\nc")
7 #expect(lines.map { String($0.content) } == ["a", "b", "", "c"])
8 #expect(lines.map { String($0.ending) } == ["\r\n", "\n", "\n", ""])
9 #expect(splitRawLines("").isEmpty)
10 #expect(splitRawLines("x\n").count == 1)
11 }
12
13 @Test func splitIsLossless() {
14 let text = "\r\n\n a\r b\r\n😀\n"
15 #expect(splitRawLines(text).map { String($0.content) + String($0.ending) }.joined() == text)
16 }
17
18 @Test(arguments: [
19 ("", LineClass.blank),
20 (" \t", .blank),
21 ("* a", .heading(level: 1)),
22 ("*** ", .heading(level: 3)),
23 ("*", .heading(level: 1)),
24 ("*bold* text", .plain),
25 (" * a", .listItem),
26 ("#+BEGIN_SRC sh :results output", .blockBegin(name: "src")),
27 ("#+end_src", .blockEnd(name: "src")),
28 ("#+BEGIN: clocktable :scope file", .dynamicBegin),
29 ("#+END:", .dynamicEnd),
30 ("#+TITLE: x", .keyword(key: "TITLE")),
31 ("#+tblfm: $2=$1", .keyword(key: "TBLFM")),
32 ("# comment", .comment),
33 ("#", .comment),
34 ("#hashtag", .plain),
35 (": fixed", .fixedWidth),
36 (":", .fixedWidth),
37 (":PROPERTIES:", .drawerBegin(name: "PROPERTIES")),
38 (" :LOGBOOK:", .drawerBegin(name: "LOGBOOK")),
39 (":END:", .drawerEnd),
40 ("| a | b |", .tableRow),
41 ("-----", .horizontalRule),
42 ("----", .plain),
43 ("[fn:1] note", .footnoteDefinition),
44 ("CLOCK: [2026-10-04 Sun 10:00]", .clock),
45 ("SCHEDULED: <2026-10-04 Sun>", .planning),
46 ("- item", .listItem),
47 ("+ item", .listItem),
48 ("1. item", .listItem),
49 ("2) item", .listItem),
50 ("-", .listItem),
51 ("-x", .plain),
52 ("1.5 apples", .plain),
53 ("plain text", .plain),
54 ])
55 func classify(line: String, expected: LineClass) {
56 #expect(classifyLine(line[...]).cls == expected)
57 }
58
59 @Test func indentCountsTabsToEight() {
60 #expect(classifyLine("\t- a").indent == 8)
61 #expect(classifyLine(" \t- a").indent == 8)
62 #expect(classifyLine(" - a").indent == 3)
63 }
64}
Tests/OrgCoreTests/ParserElementTests.swift added +79
@@ -0,0 +1,79 @@
1import Testing
2@testable import OrgCore
3
4func childKinds(_ text: String) -> [SyntaxKind] {
5 let root = OrgParser.parse(text).root
6 let container = root.children.first { $0.kind == .zerothSection || $0.kind == .section }!
7 return container.children.map(\.kind)
8}
9
10struct ParserElementTests {
11 @Test func planningAndPropertiesFollowHeading() {
12 let text = "* a\nSCHEDULED: <2026-10-04 Sun>\n:PROPERTIES:\n:ID: x\n:END:\nbody\n"
13 #expect(childKinds(text) == [.heading, .planning, .propertyDrawer, .paragraph])
14 }
15
16 @Test func planningElsewhereIsText() {
17 #expect(childKinds("SCHEDULED: <2026-10-04 Sun>\n") == [.paragraph])
18 }
19
20 @Test func blocks() {
21 #expect(childKinds("#+begin_src sh\n,* escaped\n:END:\n#+end_src\nafter\n") == [.block, .paragraph])
22 #expect(childKinds("#+BEGIN: clocktable\n#+END:\n") == [.dynamicBlock])
23 }
24
25 @Test func headingsEndBlocks() {
26 #expect(childKinds("#+begin_src sh\n* heading\n#+end_src\n") == [.paragraph])
27 }
28
29 @Test func unclosedBlockIsParagraph() {
30 #expect(childKinds("#+begin_src sh\necho\n") == [.paragraph])
31 }
32
33 @Test func drawersHoldElements() {
34 let root = OrgParser.parse(":LOGBOOK:\nCLOCK: [2026-10-04 Sun 10:00]\n:END:\n").root
35 let drawer = root.children[0].children[0]
36 #expect(drawer.kind == .drawer)
37 #expect(drawer.children.map(\.kind) == [.clock])
38 }
39
40 @Test func keywords() {
41 #expect(childKinds("#+TITLE: x\n#+NAME: t\n#+ATTR_HTML: :width 10\n") == [.keyword, .affiliatedKeyword, .affiliatedKeyword])
42 }
43
44 @Test func commentsAndFixedWidthGroup() {
45 #expect(childKinds("# a\n# b\n: c\n: d\n-----\n") == [.comment, .fixedWidth, .horizontalRule])
46 }
47
48 @Test func tableWithFormulas() {
49 let root = OrgParser.parse("| a |\n|---|\n| 1 |\n#+TBLFM: $1=2\n#+TBLFM: $1=3\n").root
50 let table = root.children[0].children[0]
51 #expect(table.kind == .table)
52 #expect(table.children.map(\.kind) == [.tableRow, .tableRow, .tableRow, .tableFormula, .tableFormula])
53 }
54
55 @Test func footnoteDefinition() {
56 #expect(childKinds("[fn:1] note\ncontinued\n\nafter\n") == [.footnoteDefinition, .paragraph])
57 }
58
59 @Test func listsNestByIndent() {
60 let text = "- a\n more\n - b\n- c\n\nafter\n"
61 let root = OrgParser.parse(text).root
62 let list = root.children[0].children[0]
63 #expect(list.kind == .plainList)
64 #expect(list.children.map(\.kind) == [.item, .item])
65 #expect(list.children[0].children.map(\.kind) == [.paragraph, .plainList])
66 #expect(childKinds(text) == [.plainList, .paragraph])
67 }
68
69 @Test func twoBlankLinesEndAList() {
70 #expect(childKinds("- a\n\n\n- b\n") == [.plainList, .plainList])
71 #expect(childKinds("- a\n\n- b\n") == [.plainList])
72 }
73
74 @Test func paragraphStopsAtElementStart() {
75 #expect(childKinds("text\n| a |\n") == [.paragraph, .table])
76 #expect(childKinds("text\n#+begin_quote\nq\n#+end_quote\n") == [.paragraph, .block])
77 #expect(childKinds("text\n#+begin_quote\nq\n") == [.paragraph])
78 }
79}
Tests/OrgCoreTests/ParserSectionTests.swift added +66
@@ -0,0 +1,66 @@
1import Testing
2@testable import OrgCore
3
4func nodeKinds(_ text: String) -> [SyntaxKind] {
5 OrgParser.parse(text).root.descendants().map(\.kind)
6}
7
8func tokens(of kind: SyntaxKind, in text: String) -> [SyntaxToken] {
9 OrgParser.parse(text).root.descendants().filter { $0.kind == kind }.flatMap(\.tokens)
10}
11
12struct ParserSectionTests {
13 @Test func emptyDocument() {
14 let tree = OrgParser.parse("")
15 #expect(tree.text == "")
16 #expect(nodeKinds("") == [.document])
17 }
18
19 @Test func zerothSectionHoldsPreamble() {
20 #expect(nodeKinds("text\n* a\n") == [.document, .zerothSection, .paragraph, .section, .heading])
21 }
22
23 @Test func sectionsNestByLevel() {
24 let text = "* a\n** b\n*** c\n** d\n* e\n"
25 let root = OrgParser.parse(text).root
26 let top = root.children
27 #expect(top.map(\.kind) == [.section, .section])
28 #expect(top[0].children.map(\.kind) == [.heading, .section, .section])
29 #expect(top[0].children[1].children.map(\.kind) == [.heading, .section])
30 }
31
32 @Test func headingTokens() {
33 let parts = tokens(of: .heading, in: "** TODO [#A] Write the plan :work:urgent: \n")
34 #expect(parts.map(\.kind) == [.stars, .whitespace, .todoKeyword, .whitespace, .priority, .whitespace, .title, .whitespace, .tags, .whitespace, .newline])
35 #expect(parts.first { $0.kind == .tags }?.text == ":work:urgent:")
36 #expect(parts.first { $0.kind == .title }?.text == "Write the plan")
37 }
38
39 @Test func todoKeywordsComeFromSettings() {
40 let text = "#+TODO: NEXT | DONE\n* NEXT a\n* TODO b\n"
41 let todo = tokens(of: .heading, in: text).filter { $0.kind == .todoKeyword }.map(\.text)
42 #expect(todo == ["NEXT"])
43 }
44
45 @Test func priorityNeedsValidValueAndSpace() {
46 #expect(tokens(of: .heading, in: "* [#B] x\n").contains { $0.kind == .priority })
47 #expect(tokens(of: .heading, in: "* [#10] x\n").contains { $0.kind == .priority })
48 #expect(!tokens(of: .heading, in: "* [#AB] x\n").contains { $0.kind == .priority })
49 #expect(!tokens(of: .heading, in: "* [#A]x\n").contains { $0.kind == .priority })
50 }
51
52 @Test func tagsNeedValidCharacters() {
53 #expect(tokens(of: .heading, in: "* a :b@c_1:\n").contains { $0.kind == .tags })
54 #expect(!tokens(of: .heading, in: "* a :b c:\n").contains { $0.kind == .tags })
55 #expect(!tokens(of: .heading, in: "* a :b:c\n").contains { $0.kind == .tags })
56 }
57
58 @Test func headingWithoutNewlineAtEnd() {
59 #expect(OrgParser.parse("* a").text == "* a")
60 }
61
62 @Test(arguments: ["* a\n", "*\n", "* TODO\n", "text\r\n* a\r\n** b\r\n", "\n\n* a\n\n"])
63 func roundTrip(text: String) {
64 #expect(OrgParser.parse(text).text == text)
65 }
66}
Tests/OrgCoreTests/RoundTripTests.swift added +75
@@ -0,0 +1,75 @@
1import Foundation
2import Testing
3@testable import OrgCore
4
5/// SplitMix64, so failures reproduce from the seed.
6struct SeededGenerator: RandomNumberGenerator {
7 var state: UInt64
8 mutating func next() -> UInt64 {
9 state &+= 0x9E37_79B9_7F4A_7C15
10 var z = state
11 z = (z ^ (z >> 30)) &* 0xBF58_476D_1CE4_E5B9
12 z = (z ^ (z >> 27)) &* 0x94D0_49BB_1331_11EB
13 return z ^ (z >> 31)
14 }
15}
16
17let fragments = [
18 "* ", "** TODO [#A] title :a:b:", "*** DONE", "#+TODO: NEXT | DONE", "#+begin_src sh", "#+end_src",
19 "#+BEGIN_QUOTE", "#+end_quote", "#+BEGIN: clocktable", "#+END:", ":PROPERTIES:", ":ID: x", ":END:",
20 ":LOGBOOK:", "CLOCK: [2026-10-04 Sun 10:00]", "SCHEDULED: <2026-10-04 Sun>", "- item", " - nested",
21 "\t+ tab", "1. one", "| a | b |", "|---+---|", "#+TBLFM: $2=$1", "# comment", ": fixed", "-----",
22 "[fn:1] note", "#+NAME: x", "plain text", "é", "😀", "e\u{301}", " ", "\t", "\n", "\n", "\r\n", "\r", "",
23]
24
25func randomDocument(_ rng: inout SeededGenerator) -> String {
26 (0..<Int.random(in: 0...40, using: &rng)).map { _ in
27 fragments.randomElement(using: &rng)! + (Bool.random(using: &rng) ? "\n" : "")
28 }.joined()
29}
30
31func checkLengths(_ node: SyntaxNode) -> Bool {
32 let sum = node.green.children.reduce(0) { $0 + $1.length }
33 return sum == node.green.length && node.children.allSatisfy(checkLengths)
34}
35
36struct RoundTripTests {
37 @Test func fuzzedDocumentsRoundTrip() {
38 var rng = SeededGenerator(state: 20261004)
39 for n in 0..<2_000 {
40 let text = randomDocument(&rng)
41 let tree = OrgParser.parse(text)
42 #expect(tree.text == text, "document \(n)")
43 #expect(checkLengths(tree.root), "document \(n)")
44 }
45 }
46
47 @Test(arguments: [
48 [0xEF, 0xBB, 0xBF] + Array("* a\r\n".utf8),
49 Array("* a\r\n- b\n\tc".utf8),
50 Array("* a".utf8),
51 Array("😀 e\u{301}\n".utf8),
52 [0x2A, 0x20, 0xFF, 0x0A] as [UInt8],
53 ])
54 func encodingFixturesRoundTrip(bytes: [UInt8]) {
55 let source = SourceText(bytes: bytes)
56 let tree = OrgParser.parse(source.text)
57 #expect(source.encode(tree.text) == bytes)
58 }
59
60 /// Private corpus: `ORGSTAR_CORPUS=/path/to/org swift test --filter corpus`.
61 @Test(.enabled(if: ProcessInfo.processInfo.environment["ORGSTAR_CORPUS"] != nil))
62 func corpusRoundTrips() throws {
63 let root = URL(fileURLWithPath: ProcessInfo.processInfo.environment["ORGSTAR_CORPUS"]!)
64 let files = FileManager.default.enumerator(at: root, includingPropertiesForKeys: nil)!
65 .compactMap { $0 as? URL }
66 .filter { ["org", "org_archive"].contains($0.pathExtension) }
67 #expect(!files.isEmpty)
68 for file in files {
69 let bytes = try [UInt8](Data(contentsOf: file))
70 let source = SourceText(bytes: bytes)
71 let tree = OrgParser.parse(source.text)
72 #expect(source.encode(tree.text) == bytes, "\(file.path)")
73 }
74 }
75}
Tests/OrgCoreTests/SettingsTests.swift added +48
@@ -0,0 +1,48 @@
1import Testing
2@testable import OrgCore
3
4struct SettingsTests {
5 func scan(_ text: String, blockEnds: [Int: Int] = [:]) -> OrgSettings {
6 let lines = splitRawLines(text)
7 let info = lines.map { classifyLine($0.content) }
8 return SettingsScanner.scan(lines: lines, info: info, blockEnds: blockEnds, defaults: .default)
9 }
10
11 @Test func defaultsWithoutKeywords() {
12 let settings = scan("* TODO a\n")
13 #expect(settings.todoKeywordNames == ["TODO", "DONE"])
14 #expect(settings.isDone("DONE"))
15 }
16
17 @Test func fileKeywordsReplaceDefaults() {
18 let settings = scan("#+TODO: NEXT(n) WAIT(w@/!) | DONE(d!) CANCELED(c@)\n")
19 #expect(settings.todoKeywordNames == ["NEXT", "WAIT", "DONE", "CANCELED"])
20 let sequence = settings.todoSequences[0]
21 #expect(sequence.active.map(\.name) == ["NEXT", "WAIT"])
22 #expect(sequence.done.map(\.name) == ["DONE", "CANCELED"])
23 #expect(sequence.active[1] == TodoKeyword(name: "WAIT", fastKey: "w", logOnEnter: "@", logOnLeave: "!"))
24 #expect(sequence.done[0] == TodoKeyword(name: "DONE", fastKey: "d", logOnEnter: "!", logOnLeave: nil))
25 }
26
27 @Test func lastWordIsDoneWithoutSeparator() {
28 let settings = scan("#+SEQ_TODO: A B C\n")
29 #expect(settings.todoSequences[0].active.map(\.name) == ["A", "B"])
30 #expect(settings.todoSequences[0].done.map(\.name) == ["C"])
31 }
32
33 @Test func severalLinesMakeSeveralSequences() {
34 let settings = scan("#+TODO: A | B\n#+TYP_TODO: X | Y\n")
35 #expect(settings.todoSequences.count == 2)
36 #expect(settings.todoSequences[1].kind == .type)
37 }
38
39 @Test func keywordsInsideBlocksAreIgnored() {
40 let text = "#+begin_example\n#+TODO: X | Y\n#+end_example\n"
41 #expect(scan(text, blockEnds: [0: 2]).todoKeywordNames == ["TODO", "DONE"])
42 }
43
44 @Test func priorities() {
45 #expect(scan("#+PRIORITIES: 1 10 5\n").priorities == Priorities(highest: "1", lowest: "10", default: "5"))
46 #expect(scan("").priorities == Priorities(highest: "A", lowest: "C", default: "B"))
47 }
48}
Tests/OrgCoreTests/SourceTextTests.swift added +40
@@ -0,0 +1,40 @@
1import Testing
2@testable import OrgCore
3
4struct SourceTextTests {
5 @Test func plainUTF8() {
6 let source = SourceText(bytes: Array("* a\n".utf8))
7 #expect(source.text == "* a\n")
8 #expect(!source.hasBOM)
9 #expect(source.isEditable)
10 }
11
12 @Test func bomIsStrippedAndRestored() {
13 let bytes: [UInt8] = [0xEF, 0xBB, 0xBF] + Array("x\n".utf8)
14 let source = SourceText(bytes: bytes)
15 #expect(source.hasBOM)
16 #expect(source.text == "x\n")
17 #expect(source.encode(source.text) == bytes)
18 #expect(source.encode("y\n") == [0xEF, 0xBB, 0xBF] + Array("y\n".utf8))
19 }
20
21 @Test func crlfAndMixedEndingsSurvive() {
22 let bytes = Array("a\r\nb\nc\r\n".utf8)
23 let source = SourceText(bytes: bytes)
24 #expect(source.encode(source.text) == bytes)
25 }
26
27 @Test func invalidUTF8IsReadOnlyAndUnchanged() {
28 let bytes: [UInt8] = [0x61, 0xFF, 0x0A]
29 let source = SourceText(bytes: bytes)
30 #expect(!source.isValidUTF8)
31 #expect(!source.isEditable)
32 #expect(source.encode(source.text) == bytes)
33 }
34
35 @Test func nonBMPAndCombiningSurvive() {
36 let bytes = Array("😀 e\u{301}\n".utf8)
37 let source = SourceText(bytes: bytes)
38 #expect(source.encode(source.text) == bytes)
39 }
40}
Tests/OrgCoreTests/SyntaxTreeTests.swift added +45
@@ -0,0 +1,45 @@
1import Testing
2@testable import OrgCore
3
4struct SyntaxTreeTests {
5 func sample() -> GreenNode {
6 var b = GreenBuilder()
7 b.start(.document)
8 b.start(.paragraph)
9 b.token(.text, "hé😀")
10 b.token(.newline, "\n")
11 b.finish()
12 b.token(.newline, "\r\n")
13 b.finish()
14 return b.build()
15 }
16
17 @Test func lengthsAreUTF16() {
18 let green = sample()
19 #expect(green.length == 4 + 1 + 2)
20 #expect(green.text == "hé😀\n\r\n")
21 }
22
23 @Test func redNodesCarryOffsets() {
24 let root = SyntaxNode(green: sample(), offset: 0, parent: nil)
25 let paragraph = root.children[0]
26 #expect(paragraph.kind == .paragraph)
27 #expect(paragraph.range == 0..<5)
28 #expect(paragraph.parent === root)
29 #expect(root.tokens.map(\.range) == [5..<7])
30 #expect(paragraph.tokens.map(\.kind) == [.text, .newline])
31 }
32
33 @Test func builderSkipsEmptyTokens() {
34 var b = GreenBuilder()
35 b.start(.document)
36 b.token(.whitespace, "")
37 b.finish()
38 #expect(b.build().children.isEmpty)
39 }
40
41 @Test func descendantsArePreorder() {
42 let root = SyntaxNode(green: sample(), offset: 0, parent: nil)
43 #expect(root.descendants().map(\.kind) == [.document, .paragraph])
44 }
45}
docs/design.md added +393
@@ -0,0 +1,393 @@
1# Orgstar Design Doc
2
3Oct 4, 2026
4
5## Summary
6
7A native macOS app, with iOS to follow, that edits a folder of org files directly and can replace Emacs for someone who only uses org-mode. The buffer is the file: the app is a text editor with an org-aware syntax tree, rendering drawn over the text, and views (agenda, table recalculation, export) that write back minimal edits.
8
9**Goals**
10
11- Cover the org features used daily: cycling, TODO/priority/tags, planning and timestamps, tables with formulas, source blocks with execution, `M-q` fill, agenda, export.
12- Byte-for-byte fidelity: text the user did not edit never changes, so Emacs, beorg and Syncthing can work on the same files.
13- macOS first. Every core decision keeps an iOS port possible without a rewrite.
14
15**Non-goals for v1**
16
17- Emacs features unrelated to org (an elisp runtime, packages, other major modes).
18- Block-based storage, or a database as the source of truth.
19- Mac App Store distribution.
20- Babel execution on iOS.
21- Native Calc symbolic math in table formulas.
22- Sync providers other than Syncthing and iCloud Drive.
23
24## Decisions
25
26| Area | Decision | Reason |
27| --- | --- | --- |
28| Editing model | Text editor; the buffer is the file. Rendering and views sit on top. | Fill, table alignment and fidelity need plain text. Block editors can't round-trip. |
29| Platforms | macOS v1. iOS after the Mac app is mature. | One platform at a time; core stays portable. |
30| Distribution | Direct download (notarized, Sparkle) and Homebrew cask. No sandbox. | Babel needs to spawn interpreters. |
31| Language | Swift throughout. Core in a separate package with no AppKit/UIKit imports. | TextKit uses UTF-16 offsets natively; no FFI layer; direct path to iOS. |
32| Parser | New lossless parser (`OrgCore`) with source ranges. Port inline rules from OrgSwift. | OrgSwift and orgo are export parsers: no positions, drop drawers/comments, flat headings. |
33| Key bindings | Command registry; keymaps as data. Presets: Emacs (default), Mac, Doom. | One registry serves keys, palette, menus and iOS. Doom needs a modal engine, so it ships last. |
34| Babel | Execute `sh`/`bash`/`shell` and `python` with real header args; generic interpreter runner; `emacs-lisp` via `emacs --batch`. Tree-sitter highlighting for ~20 languages. iOS execution: maybe. | Covers the most-used languages; generic runner adds ruby, node, R, sqlite3, awk cheaply. |
35| Table formulas | Native evaluator for the common subset; `emacs --batch` fallback for the rest. | Runs on iOS; Calc can't be reproduced exactly. |
36| Storage | Plain folders: Syncthing-synced and iCloud Drive. SQLite index is a rebuildable cache. | Files stay the source of truth. |
37
38## Architecture
39
40Three layers: a thin app shell per platform, a platform layer for system services, and `OrgCore`, which holds all org logic and is shared unchanged by both apps.
41
42Diagram (described in text):
43
44- macOS app (v1, AppKit shell, menus) and iOS app (later, UIKit shell, touch input) both sit on the platform layer.
45- Platform layer (AppKit or UIKit): Editor view (TextKit 2, rendering over text), Notifications (scheduled from agenda queries), Processes (Babel runs and `emacs --batch`, talks to external interpreters: sh, python, emacs), Files (bookmarks, watching, saving; reads/writes the org folders on Syncthing and iCloud, which are the source of truth).
46- The platform layer calls into OrgCore (commands, tree queries).
47- OrgCore (Swift package, no UI imports): Parser (lossless tree, UTF-16 ranges), Commands (pure functions, minimal edits), Agenda (index queries, saved views), Compute (TBLFM evaluator, Babel headers). OrgCore writes to OrgIndex (SQLite + FTS5, rebuildable cache).
48- Keymaps (TOML; Emacs, Mac, Doom) resolve keys to OrgCore commands.
49
50**Packages**
51
52- `OrgCore`: parser, syntax tree, semantic layer, commands, agenda queries, table formula evaluator, Babel header parsing and results writing. Foundation only.
53- `OrgIndex`: GRDB schema, indexing and queries. Foundation only.
54- `OrgPlatform`: TextKit 2 editor view, file access and watching, process runner, notifications. Separate macOS and iOS implementations behind shared protocols.
55- `OrgSwift` (existing): HTML and SwiftUI rendering, used for export and preview.
56- App target: windows, sidebar, menus, settings, keymap loading.
57
58## OrgCore data model
59
60`OrgCore` parses a file into a lossless syntax tree: concatenating its tokens reproduces the file exactly, and every node knows its range in UTF-16 code units, the unit `NSTextStorage` uses.
61
62**Tree structure**
63
64- Two layers, as in rust-analyzer's rowan: immutable *green* nodes store kind and length and can be shared between versions; *red* nodes are created on demand with absolute offsets and parent links.
65- Whitespace, blank lines and newlines are tokens in the tree, not discarded.
66- Anything the parser doesn't recognize becomes a `Raw` node that keeps its text, so unknown syntax survives every edit.
67
68**Bytes and encoding**
69
70The tree holds text, so byte fidelity needs its own contract.
71
72- Supported: UTF-8, with or without a BOM. Anything else (invalid UTF-8, UTF-16, legacy encodings) opens read-only with a notice; it is never converted on save.
73- Each document keeps its original bytes plus metadata: BOM present, line-ending style per line (LF, CRLF, mixed), trailing newline present.
74- Line endings stay as tokens in the tree; nothing normalizes CRLF.
75- Saving writes untouched byte spans from the original buffer and encodes only edited spans. A UTF-16 edit maps to a byte range through a per-line offset table.
76- Fixtures: BOM, CRLF, mixed endings, no trailing newline, non-BMP characters, combining sequences, tabs, invalid UTF-8.
77
78**Node kinds**
79
80- Containers: `Document`, `Section` (heading line, its content, child sections).
81- Heading parts: stars, TODO keyword, priority cookie, title objects, tags.
82- Heading metadata: `Planning` (SCHEDULED/DEADLINE/CLOSED), `PropertyDrawer`, `Drawer` (incl. LOGBOOK), `Clock`.
83- Elements: `Paragraph`, `PlainList`/`Item` (with checkbox and description term), `Table`/`TableRow`/`TableFormula`, blocks (src, example, quote, center, verse, export, special, dynamic), `Keyword`, `AffiliatedKeyword` (`#+NAME`, `#+CAPTION`, `#+RESULTS`), `Comment`, `FixedWidth`, `HorizontalRule`, `FootnoteDefinition`.
84- Objects: emphasis, link, timestamp (with repeater and warning cookies), footnote reference, entity, inline src block, macro, statistics cookie, LaTeX fragment, target.
85
86**Parse settings**
87
88Some in-buffer settings change how text parses.
89
90- Syntax-affecting: `#+TODO`/`#+SEQ_TODO`/`#+TYP_TODO`, `#+PRIORITIES`. Semantic only: `#+FILETAGS`, `#+STARTUP`, `#+PROPERTY`.
91- The settings pass is element-aware: a `#+TODO` line inside a src, example or export block is not a setting.
92- TODO precedence follows org: any file-level TODO line replaces the app default sequences for that file; several lines define several sequences; `|` separates active from done states; fast-select keys and logging suffixes (`TODO(t!)`) are parsed and kept.
93- An edit that adds, removes or changes a syntax-affecting line triggers a full reparse. A semantic-only change invalidates the semantic layer and the file's index rows, not the tree.
94
95**Incremental reparse**
96
971. Compute the damaged range from both the old and the new text: the edited lines plus one line on each side.
982. Classify the damaged lines in both versions (heading, block or drawer delimiter, list item, footnote definition, keyword, table row, plain). If any class differs between old and new, the edit changed structure.
993. No structural change: reparse the enclosing element and reuse everything else.
1004. Structural change: reparse forward from the start of the enclosing section until the new tree and the old tree agree on a section boundary at the same shifted offset (a synchronization point), then reuse the old tree from there.
1015. Content before the first heading is a zeroth section, so every offset has an enclosing section.
1026. When the classifier is unsure, or no synchronization point is found, reparse the whole file.
103
104A differential test asserts that the incremental result always equals a full parse, with edits that create and delete delimiters, join lines, remove indentation, add list tabs and change keyword lines. Latency targets are in Phases.
105
106**Semantic layer**
107
108Typed views over the tree, computed and cached per tree version: `HeadingInfo` (todo, priority, tags with inheritance, properties with inheritance, planning, clocks, ID), `TableModel` (cells, column widths, formulas), `SrcBlockInfo` (language, header args merged from `#+PROPERTY`, property drawers and the block line). Commands and the index read these, never raw text.
109
110Inheritance is not one rule, so the semantic layer stores local values and resolved values separately, and each consumer picks a policy:
111
112- Tags: inherited by default, minus an exclusion list (`org-tags-exclude-from-inheritance`); `#+FILETAGS` apply to the whole file.
113- Properties: not inherited for search by default; opt-in per property (`org-use-property-inheritance`). Special properties (`ID`, `CATEGORY`, `ARCHIVE`, `COLUMNS`) have their own rules; `ID` is never inherited.
114- Babel header args: resolved from `#+PROPERTY: header-args`, language-specific `header-args:lang`, additive `+` forms, heading properties, `#+HEADER:` lines and the block line, in org's order.
115- Every resolved value records where it came from, so the UI can show it and tests can check it.
116
117**From OrgSwift**
118
119Port the inline rules: emphasis border characters (matched to orgo), link forms, timestamp parsing and ranges. OrgSwift itself stays the HTML/SwiftUI renderer; later it can render from the `OrgCore` tree so there is one parser.
120
121## Commands and keymaps
122
123Every feature is a named command, implemented as a pure function in `OrgCore`. Keys, menus, the command palette and (later) iOS touch controls all invoke the same commands.
124
125**Command shape**
126
127```swift
128protocol OrgCommand {
129 static var id: String { get } // "org.todo.cycle"
130 static var title: String { get } // shown in the palette
131 func applies(in ctx: EditContext) -> Bool
132 func run(in ctx: EditContext) throws -> CommandStep
133}
134
135struct EditContext {
136 let document: DocumentID
137 let revision: Int // the revision the command read
138 let text: String
139 let tree: OrgTree
140 let selection: [Range<Int>]
141 let settings: OrgSettings
142 let now: Date // injected; commands never read the clock
143 let calendar: Calendar // includes time zone
144 let answers: [String: String] // replies to earlier prompts
145}
146
147enum CommandStep {
148 case commit(EditResult)
149 case prompt(Prompt) // ask, then rerun with the answer in `answers`
150}
151
152struct EditResult { var baseRevision: Int; var edits: [TextEdit]; var selection: [Range<Int>]?; var effects: [Effect] }
153struct TextEdit { let range: Range<Int>; let replacement: String } // UTF-16, against baseRevision, non-overlapping
154```
155
156- `edits` are minimal replacements, applied as one undo group.
157- `effects` cover anything that isn't a text change: run a src block, open the agenda, show a message, fold a subtree. The platform layer performs them.
158- Context dispatch works like org's TAB: one key can bind several commands, and the first whose `applies` returns true runs (cycle on a heading, next cell in a table, indent in a list).
159
160**Revisions, time and prompts**
161
162- Every edit result names the revision it was computed against. The document session applies it only if that is still the current revision; otherwise the command reruns on the new state.
163- Prompts (a note when logging a state change, a date for SCHEDULED) use the prepare/prompt/commit loop above. Cancelling at any prompt produces no edit.
164- Repeaters, `LOGBOOK` notes and `CLOSED` stamps use the injected `now` and `calendar`, so tests are deterministic.
165- Asynchronous effects (a Babel run) carry the revision and the target element's identity. On completion they re-find the target in the current tree (by `#+NAME`, or by the block's position relative to its neighbours) and are rejected with a message if it is gone or ambiguous. Result insertion is its own undo step.
166
167**Document session**
168
169One session per open file owns the text, original bytes, tree, revision counter, merge base and undo history. Each window or split has its own selection, fold state and narrowing, mapped through every edit. On an external reload or merge, view state is mapped through the diff; folds on headings that no longer exist are dropped.
170
171**Keymap format**
172
173Keymaps are TOML files, user-editable, loaded in layers: preset, then user overrides.
174
175```toml
176[[bind]]
177keys = "C-c C-t"
178command = "org.todo.cycle"
179
180[[bind]]
181keys = "TAB"
182command = "org.cycle"
183when = "heading"
184
185[[bind]]
186keys = "SPC m t"
187command = "org.todo.cycle"
188mode = "normal" # Doom preset only
189```
190
191- Key sequences of any length (`C-c C-x C-i`), with an echo area for the pending prefix and a which-key popup after a short delay.
192- `when` names a context predicate; `mode` names a modal state. Both exist in the format from day one even though only Doom uses `mode`.
193- An "Option as Meta" setting, per side (left/right), like Terminal and iTerm2.
194- `⌘` shortcuts (save, undo, find) stay active in every preset.
195
196**Presets**
197
198| Preset | Default | Needs | Ships |
199| --- | --- | --- | --- |
200| Emacs | Yes | Prefix-key engine | Phase 2 |
201| Mac | No | Menu wiring | Phase 2 |
202| Doom | No | Modal engine: normal/insert/visual, operators + motions + text objects, counts, `.` repeat, registers; `SPC` leader; evil-org bindings | After phase 3 |
203
204## Workspace, storage and index
205
206A workspace is one or more root folders. The files are the only source of truth; the SQLite index can be deleted and rebuilt at any time.
207
208**Folders and access**
209
210- The user adds folders through an open panel. Store security-scoped bookmarks even though the Mac app is unsandboxed, because iOS will require them.
211- Three scopes, configured separately:
212 - Discovery: every `.org` and `.org_archive` file under the roots. Indexed for search and link resolution.
213 - Agenda: folders or globs, the equivalent of `org-agenda-files`. Archive files and subtrees tagged `ARCHIVE` are excluded by default.
214 - Link resolution: all discovered files, including archives, so `id:` links into archived subtrees still resolve. Duplicate IDs are reported. A link to a file outside the roots opens it but doesn't index it.
215- Syncthing conflict copies are discovered but excluded from agenda, search and ID resolution; they appear only in the conflict UI.
216
217**Watching for external changes**
218
219- FSEvents (macOS) and `NSFilePresenter` are hints that something changed, not a complete log.
220- On launch, and whenever FSEvents reports dropped events, a root change, or must-scan-subdirs, rescan the affected tree and reconcile the index: compare path, size, mtime and hash; detect renames by hash; delete rows for missing files. Reconciliation runs in one transaction per root.
221- Roots are identified by bookmark, not path, so a moved root is followed or reported.
222- iOS (later): `NSFilePresenter` plus `NSMetadataQuery` for iCloud.
223- Ignore Syncthing temp files (`.syncthing.*.tmp`). Show Syncthing conflict files (`*.sync-conflict-*`) with a diff against the original.
224- iCloud placeholders: a file not yet downloaded is requested with `startDownloadingUbiquitousItem` and shown as loading.
225
226**Saving**
227
228- Save mode is a user setting: autosave after an idle delay, or explicit save only.
229- Save sequence, inside a coordinated write:
230 1. Read the current disk bytes and hash.
231 2. If the hash equals the merge base, write. Otherwise three-way merge (base = merge base, ours = buffer, theirs = disk). On conflict, stop and show the conflict; write nothing.
232 3. Write to a temp file in the same folder and replace atomically.
233 4. Read back the hash; it becomes the new merge base.
234- Emacs and Syncthing don't use `NSFileCoordinator`, so another writer can still replace the file between steps 1 and 3. We can't fully prevent that; we narrow the window and detect it: after the write, if FSEvents reports a change whose hash is neither ours nor the base, treat it as a new external edit and merge again.
235- Before any write that replaces a version we didn't produce, keep a copy of both versions in a recovery folder (last 20 per file) so nothing is lost.
236- An unedited open file reloads silently on external change, keeping cursor and folds. An edited one merges automatically when the merge is clean, and shows the conflict otherwise.
237- Fault-injection tests replace the file at each step of the sequence.
238
239**Index schema (GRDB, SQLite + FTS5)**
240
241| Table | Columns |
242| --- | --- |
243| `files` | id, root, path, size, mtime, hash, parsed_at |
244| `headings` | id, file_id, parent_id, start, end, level, todo, priority, title, outline_path, org_id |
245| `tags` | heading_id, tag, inherited |
246| `properties` | heading_id, key, value, inherited |
247| `timestamps` | heading_id, kind (scheduled, deadline, closed, active, inactive), start, end, repeater, warning |
248| `clocks` | heading_id, start, end, minutes |
249| `links` | heading_id, type, target |
250| `headings_fts` | FTS5 over title and body text |
251
252- A file is reindexed when its disk hash changes, and when an app setting that changes semantics (TODO defaults, inheritance policies) changes. Index rows record the settings version they were built with.
253- Open documents with unsaved edits overlay the index: their rows are computed in memory from the session's tree and replace that file's disk rows in every query. Agenda and search therefore reflect unsaved TODO, tag and planning edits in explicit-save mode.
254- Index rows store positions with the revision they came from. Jumping to a heading checks the target session's revision and re-finds the heading by ID or outline path if the positions are stale.
255- Every query the UI runs (agenda, tag search, TODO list, saved views) is a query over the index plus the overlay.
256
257## Babel and table formulas
258
259Both live in `OrgCore` as parsing and planning logic; only the process launch is platform code. Results are written back as a minimal edit to `#+RESULTS:`.
260
261**Babel execution (macOS)**
262
263- Execution plan first: resolve all header args, then check them before launching anything. `:eval never`/`no` blocks execution; `:eval query` always asks. Any header that changes what runs and isn't supported yet (`:session`, `:noweb yes`, `:prologue`, `:epilogue`, `:file`) stops execution with a message naming it, rather than running different code.
264- Runner: write the expanded body to a temp file, launch the interpreter with `Process` in `:dir` (default: the file's folder), capture stdout and stderr, support timeout and cancel.
265- Each language has an adapter for `:var` serialization (scalars, lists, tables) and for `:results value` (python wraps the body in a function, as org does; shells take the last output). Interpreter invocation alone doesn't provide org semantics.
266- Never execute on open or on export without confirmation. Confirm per block, with a per-file trust setting, like `org-confirm-babel-evaluate`. Trust is keyed to the block's content hash, so an external edit that changes the code asks again.
267
268| Language | Runner | Header args in v1 |
269| --- | --- | --- |
270| `sh`, `bash`, `shell` | Native | `:results`, `:var`, `:dir`, `:cmd`, `:exports` |
271| `python` | Native; `:results value` wraps the body in a function, as org does | Same |
272| `ruby`, `js`, `R`, `sqlite`, `awk` | Generic: configured command per language (`ruby`, `node`, `Rscript`, `sqlite3`, `awk`) | `:results output`, `:dir`, `:cmd` |
273| `emacs-lisp`, `elisp` | `emacs --batch` if Emacs is installed | `:results` |
274
275- `:results` handling in v1: `output`/`value`, `verbatim`/`table`/`list`/`raw`/`drawer`, `replace`/`append`/`silent`.
276- Result ownership, matched to org: the results element is the `#+RESULTS:` keyword plus exactly one element after it (fixed-width lines, a table, a list, an example block, or a `:RESULTS:` drawer). `replace` swaps that element; `append` adds after it; `silent` writes nothing. `raw` output has no boundary, so `raw` without `drawer` refuses to replace existing results and asks first. Changing the format replaces the old element by its old format's boundary.
277- Results are re-found on completion as described in Commands; affiliated keywords (`#+NAME`, `#+CAPTION`) on the results stay.
278- Tests run the same block repeatedly and switch formats between runs.
279- Later: `:session`, `:file` with inline images (dot, plantuml, mermaid), `:noweb`, `:tangle`.
280- Highlighting: tree-sitter (SwiftTreeSitter) grammars for about 20 languages: shells, emacs-lisp, python, C, R, js, java, lisp, scheme, clojure, haskell, rust, go, sql, ruby, org, latex, dot, plantuml, mermaid.
281
282**Table formula evaluator**
283
284The native evaluator handles a defined numeric domain. A table that uses anything outside it is recalculated through Emacs.
285
286- **Native domain:** integers within ±2^53 and decimals with at most 15 significant digits, using Decimal. Results within the domain must match Calc's output byte for byte; an input or intermediate outside it (overflow, cancellation flagged by precision loss, unsupported mode flags) routes the whole table to Emacs.
287- **Emacs fallback:** run `emacs -Q --batch` on a snapshot of the whole file (so `#+CONSTANTS`, properties, named tables for `remote()` and file-local settings are present) in the file's own folder. Put point in the table, select the chosen `#+TBLFM` line, and call `(org-table-recalculate 'all)`; without that argument it recalculates only the current row. Splice back only the table's text.
288- **Execution authorization:** Lisp formulas (`'(...)`) and anything that evaluates Lisp run arbitrary code, so they need the same confirmation and content-hash trust as Babel. The fallback disables file-local variable evaluation (`enable-local-variables` set to `:safe`).
289
290| Area | Native | Notes |
291| --- | --- | --- |
292| Arithmetic, `^`, parentheses | Yes | |
293| References: `$N`, `@N`, `@N$M`, `@<`, `@>`, `@I`/`@II`, `$#`, `@#`, relative `@-1` | Yes | |
294| Ranges: `$1..$3`, `@2$1..@>$1` | Yes | |
295| `vsum`, `vmean`, `vmax`, `vmin`, `vcount`, `vmedian` | Yes | |
296| `sqrt`, `exp`, `ln`, `log10`, trig | Yes | |
297| Column vs field formulas, field takes precedence | Yes | |
298| Several `#+TBLFM:` lines, chosen by cursor line | Yes | |
299| Format flags `;%.Nf`, `;N`, `;E` | Yes | |
300| Calc display format (e.g. `1.4142136`) | Yes, matched | Must be tested against Emacs output byte for byte. |
301| Precision | Bounded | Native within the defined domain only; outside it, Emacs. |
302| Dates and durations (`;T`, `;t`, `HH:MM`, timestamp subtraction) | Later | Separate piece of work. |
303| `remote(name, ref)` | Later | |
304| Calc symbolic math: `taylor`, `deriv`, `integ`, `solve` | No | Emacs fallback. |
305| Lisp formulas `'(...)` | No | Emacs fallback. |
306| Calc units, vectors/matrices beyond `v*` | No | Emacs fallback. |
307| Iterate to convergence (`C-u C-u C-c C-c`) | No | Emacs fallback. |
308
309On iOS, tables that need Emacs keep their last computed values and show a "recalculate on Mac" marker.
310
311## Testing
312
313Fidelity is checked against Emacs itself, using the same oracle pattern as orgo's `tests/oracle.rs`.
314
315| Layer | Test | Corpus |
316| --- | --- | --- |
317| Parser | Round trip: tree text equals file bytes | Your org files, org-conformance cases, org manual examples |
318| Parser | Incremental equals full: random edits, then compare incremental tree with a fresh parse | Same, plus fuzzed edits |
319| Parser | Conformance: HTML skeleton from an `OrgCore`-based renderer matches the goldens | org-conformance |
320| Commands | Oracle: same file, cursor and command in `emacs --batch`; diff resulting bytes | Generated cases per command |
321| Table formulas | Oracle: `org-table-recalculate` output vs native, byte for byte | Tables from your files plus a generated set |
322| Babel | Results block text vs Emacs for the same block and header args | sh and python cases |
323| Performance | Parse, reparse and restyle times on the largest real files | Your largest files, plus synthetic 10x copies |
324
325- Command oracle mapping examples: `org.todo.cycle` → `org-todo`, `org.heading.demote` → `org-metaright`, `org.fill` → `org-fill-paragraph`, `org.table.align` → `org-table-align`.
326- The command oracle is stateful. Each case fixes: file, point and mark (converted from UTF-16 offsets to Emacs character positions), prefix argument, active region, the org settings that matter (`org-todo-keywords`, `org-log-done`, `fill-column`, `tab-width`, `indent-tabs-mode`), frozen time via a stubbed `current-time`, locale and time zone, and scripted answers to prompts. Emacs runs with `-Q` plus that explicit configuration.
327- Oracle comparisons cover resulting bytes, resulting point and mark, and the text after one undo.
328- Pin Emacs 31.1 and Org 9.8.7, recorded in the repo. CI requires the pinned Emacs; a missing oracle fails the job instead of skipping.
329- Encoding fixtures (BOM, CRLF, mixed endings, invalid UTF-8, non-BMP, combining characters) are public, in the repo.
330- Save fault-injection tests replace the file on disk at each step of the save sequence.
331- Personal org files are a second, private corpus run locally; phase gates use the public corpus.
332
333## Phases
334
335Seven phases, each usable on its own. Phase 1 is read-only, so it can run next to Emacs from the first build.
336
3371. **Core and viewer:** `OrgCore` parser, workspace, index, read-only rendering, folding, search.
3382. **Editor:** text editing, structure and heading commands, timestamps, `M-q`, table alignment, command palette, Emacs and Mac keymaps, remaining tree-sitter grammars.
3393. **Agenda:** day/week views, tag and TODO search, saved custom views, notifications from SCHEDULED/DEADLINE.
3404. **Computation:** native table formulas with Emacs fallback, Babel execution and results.
3415. **Export and capture:** native HTML and Markdown export, pandoc or Emacs for the rest, capture templates, global capture hotkey.
3426. **Beyond Emacs:** clock reports, habit charts, table and kanban views over properties, Doom keymap.
3437. **iOS:** same core; touch input, capture and agenda first.
344
345**Phase 1 scope**
346
347- [ ] `OrgCore` package: lossless parser with UTF-16 ranges and the bytes-and-encoding contract; element-aware settings pass; inline rules ported from OrgSwift
348- [ ] Incremental reparse with old/new classification and synchronization points; differential tests
349- [ ] Document session: original bytes, revisions, merge base, per-view state
350- [ ] Save path with merge, recovery copies and fault-injection tests (used by the editing spike, not exposed in the viewer)
351- [ ] Editable TextKit 2 spike: folding and org-indent display with correct caret movement, selection, IME composition, VoiceOver and copy/paste across folded text
352- [ ] Workspace: folders, bookmarks, discovery/agenda/link scopes, FSEvents with reconciliation scans, Syncthing temp/conflict handling, iCloud placeholders
353- [ ] Index: schema, settings version, dirty-buffer overlay, FTS5 search
354- [ ] Mac app (read-only): sidebar of folders and files, outline of the current file, quick open (`⌘P`), full-text search
355- [ ] Rendering over the text: heading styles, TODO/priority/tag styling, links, emphasis, tables, src blocks highlighted for org, emacs-lisp, sh and python only (the rest move to phase 2)
356- [ ] Folding: local and global cycling, `#+STARTUP` visibility
357- [ ] Benchmarks recorded for the files listed in the exit gates
358
359**Phase 1 exit gates** (measured on an M1 MacBook Air, 8 GB, the slowest supported reference machine):
360
361| Gate | Target |
362| --- | --- |
363| Round trip | 100% byte-identical on the public corpus and the private corpus |
364| Incremental equals full | 0 mismatches over 100,000 fuzzed edits |
365| Open to first render | p95 under 100 ms for a 1 MB file; under 500 ms for a 10 MB file |
366| Keystroke to restyled frame (spike) | p95 under 16 ms for a 1 MB file, including one file whose content is a single top-level section and one with a 5,000-row table |
367| Peak memory | Under 10x file size for an open document |
368| Index | Full rebuild of 10,000 files under 30 s; reconciliation after restart under 2 s with no changes |
369| Save safety | All fault-injection cases end with both versions recoverable |
370| Editing spike | Caret, selection, IME, VoiceOver and copy/paste checks pass on folded and indented text |
371
372Phase 2 (writable editor) does not start until every gate passes.
373
374## iOS considerations
375
376| Area | Mac v1 | iOS impact |
377| --- | --- | --- |
378| Core | `OrgCore`, `OrgIndex` with no AppKit imports | Reused unchanged. CI builds them for iOS from phase 1 to catch accidental AppKit use. |
379| Editor view | TextKit 2 in an `NSTextView` wrapper | Shared layout policy (what folds, what is indented, how org markup is styled) in a platform-free module; separate AppKit and UIKit adapters for caret movement, selection, input methods and accessibility across hidden text. The phase 1 editing spike keeps that adapter boundary explicit. |
380| Input | Keymaps call commands | Same commands from a keyboard accessory bar, palette and gestures; iPad hardware keyboards use the keymaps. |
381| Storage | Bookmarks, FSEvents, `NSFilePresenter` | Bookmarks and `NSFilePresenter` carry over; Syncthing through Möbius Sync's File Provider. |
382| Babel | `Process` | Maybe. Results blocks display either way. |
383| Table formulas | Native plus Emacs fallback | Native only; Emacs-only tables show a marker. |
384| Notifications | Local notifications from the index | iOS caps pending local notifications at 64 and doesn't guarantee background time to refill them. Schedule the next 64 events, reconcile on every launch, foreground and sync, and show in the agenda how far ahead reminders are scheduled. If the app isn't opened, reminders stop after the last scheduled one, and the UI says so. Repeating requests are used only where they match org repeater semantics. |
385
386## Open questions
387
388- [ ] App name: Orgstar (chosen).
389- [x] Save model: a user setting (see Saving).
390- [x] Minimum OS: macOS 26 and iOS 26. One release behind current (macOS 27).
391- [x] License: 0BSD.
392- [x] Oracle pins: Emacs 31.1 and Org 9.8.7.
393- [x] OrgSwift moves onto `OrgCore`'s tree later, tracked in krz/org-swift#2. Out of scope for now.
docs/plans/2026-10-04-orgcore-parser.md added +1600
@@ -0,0 +1,1600 @@
1# OrgCore Parser Foundation Implementation Plan
2
3> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
4
5**Goal:** A Swift package, `OrgCore`, that turns org file bytes into a lossless syntax tree of block-level elements, with the byte and encoding contract from the design.
6
7**Architecture:** Bytes decode into `SourceText` (UTF-8 only, BOM kept, invalid input read-only). The parser splits lines without normalizing endings, classifies each line once, matches block and drawer ends, scans in-buffer TODO settings outside blocks, then builds an immutable green tree through `GreenBuilder`. `SyntaxNode` gives offset-aware red views. Every input byte ends up in exactly one token, so the tree text always equals the source.
8
9**Tech Stack:** Swift 6.2 tools, Swift Testing, Foundation only. No third-party dependencies.
10
11**Spec:** `docs/design.md` (sections "OrgCore data model", "Testing", "Phases").
12
13## Global Constraints
14
15- Platforms: macOS 26, iOS 26.
16- `OrgCore` imports Foundation only; never AppKit, UIKit or SwiftUI.
17- Ranges and offsets exposed by the tree are UTF-16 code units.
18- Supported encoding: UTF-8 with or without BOM. Anything else is read-only and never converted.
19- Round trip: tree text equals source text for every input, and encoding unchanged text returns the original bytes.
20- License 0BSD. No attribution lines in code, commits or docs.
21
22## Out of scope for this plan
23
24Inline objects (emphasis, links, timestamps), the semantic layer, incremental reparse, conformance rendering, and the private-corpus benchmarks. Each gets its own plan.
25
26## File structure
27
28| File | Responsibility |
29| --- | --- |
30| `Package.swift` | Package with `OrgCore` library and `OrgCoreTests` |
31| `Sources/OrgCore/SourceText.swift` | Byte decoding, BOM, validity, encoding back to bytes |
32| `Sources/OrgCore/Syntax/SyntaxKind.swift` | Token and node kinds |
33| `Sources/OrgCore/Syntax/GreenTree.swift` | `GreenToken`, `GreenNode`, `GreenElement`, `GreenBuilder` |
34| `Sources/OrgCore/Syntax/SyntaxNode.swift` | Red nodes, tokens, `OrgTree` |
35| `Sources/OrgCore/Parser/Lines.swift` | Line splitting and classification |
36| `Sources/OrgCore/Parser/Settings.swift` | TODO sequences, priorities, settings scan |
37| `Sources/OrgCore/Parser/Parser.swift` | Tree construction |
38| `Tests/OrgCoreTests/*.swift` | One test file per source file, plus round-trip fuzz and corpus tests |
39
40---
41
42### Task 1: Package and SourceText
43
44**Files:**
45- Create: `Package.swift`
46- Create: `Sources/OrgCore/SourceText.swift`
47- Test: `Tests/OrgCoreTests/SourceTextTests.swift`
48
49**Interfaces:**
50- Produces: `SourceText(bytes: [UInt8])`, `SourceText(_ text: String)`, properties `originalBytes`, `hasBOM`, `isValidUTF8`, `isEditable`, `text`; `encode(_ newText: String) -> [UInt8]`.
51
52- [ ] **Step 1: Create the package**
53
54```swift
55// swift-tools-version: 6.2
56import PackageDescription
57
58let package = Package(
59 name: "Orgstar",
60 platforms: [.macOS(.v26), .iOS(.v26)],
61 products: [
62 .library(name: "OrgCore", targets: ["OrgCore"])
63 ],
64 targets: [
65 .target(name: "OrgCore"),
66 .testTarget(name: "OrgCoreTests", dependencies: ["OrgCore"])
67 ]
68)
69```
70
71- [ ] **Step 2: Write the failing tests**
72
73```swift
74import Testing
75@testable import OrgCore
76
77struct SourceTextTests {
78 @Test func plainUTF8() {
79 let source = SourceText(bytes: Array("* a\n".utf8))
80 #expect(source.text == "* a\n")
81 #expect(!source.hasBOM)
82 #expect(source.isEditable)
83 }
84
85 @Test func bomIsStrippedAndRestored() {
86 let bytes: [UInt8] = [0xEF, 0xBB, 0xBF] + Array("x\n".utf8)
87 let source = SourceText(bytes: bytes)
88 #expect(source.hasBOM)
89 #expect(source.text == "x\n")
90 #expect(source.encode(source.text) == bytes)
91 #expect(source.encode("y\n") == [0xEF, 0xBB, 0xBF] + Array("y\n".utf8))
92 }
93
94 @Test func crlfAndMixedEndingsSurvive() {
95 let bytes = Array("a\r\nb\nc\r\n".utf8)
96 let source = SourceText(bytes: bytes)
97 #expect(source.encode(source.text) == bytes)
98 }
99
100 @Test func invalidUTF8IsReadOnlyAndUnchanged() {
101 let bytes: [UInt8] = [0x61, 0xFF, 0x0A]
102 let source = SourceText(bytes: bytes)
103 #expect(!source.isValidUTF8)
104 #expect(!source.isEditable)
105 #expect(source.encode(source.text) == bytes)
106 }
107
108 @Test func nonBMPAndCombiningSurvive() {
109 let bytes = Array("😀 e\u{301}\n".utf8)
110 let source = SourceText(bytes: bytes)
111 #expect(source.encode(source.text) == bytes)
112 }
113}
114```
115
116- [ ] **Step 3: Run tests to verify they fail**
117
118Run: `swift test --filter SourceTextTests`
119Expected: build failure, `cannot find 'SourceText' in scope`.
120
121- [ ] **Step 4: Implement**
122
123```swift
124import Foundation
125
126/// A file's bytes and their decoded text. Only UTF-8 (with or without a BOM) is editable;
127/// anything else decodes with replacement characters for display and is never written back.
128public struct SourceText: Sendable {
129 public let originalBytes: [UInt8]
130 public let hasBOM: Bool
131 public let isValidUTF8: Bool
132 /// Decoded text without the BOM.
133 public let text: String
134
135 private static let bom: [UInt8] = [0xEF, 0xBB, 0xBF]
136
137 public init(bytes: [UInt8]) {
138 originalBytes = bytes
139 hasBOM = bytes.starts(with: Self.bom)
140 let body = hasBOM ? Array(bytes.dropFirst(3)) : bytes
141 if let decoded = String(validating: body, as: UTF8.self) {
142 text = decoded
143 isValidUTF8 = true
144 } else {
145 text = String(decoding: body, as: UTF8.self)
146 isValidUTF8 = false
147 }
148 }
149
150 public init(_ text: String) {
151 self.init(bytes: Array(text.utf8))
152 }
153
154 public var isEditable: Bool { isValidUTF8 }
155
156 /// Bytes to write for `newText`. Unchanged text returns the original bytes. Valid UTF-8
157 /// round-trips through `String` unchanged, so untouched spans keep their exact bytes.
158 public func encode(_ newText: String) -> [UInt8] {
159 if newText == text { return originalBytes }
160 precondition(isEditable, "a read-only document cannot be re-encoded")
161 return (hasBOM ? Self.bom : []) + Array(newText.utf8)
162 }
163}
164```
165
166- [ ] **Step 5: Run tests to verify they pass**
167
168Run: `swift test --filter SourceTextTests`
169Expected: 5 tests pass.
170
171- [ ] **Step 6: Commit**
172
173```bash
174git add Package.swift Sources Tests
175git commit -m "Add OrgCore package and SourceText byte contract"
176```
177
178---
179
180### Task 2: Green and red syntax tree
181
182**Files:**
183- Create: `Sources/OrgCore/Syntax/SyntaxKind.swift`
184- Create: `Sources/OrgCore/Syntax/GreenTree.swift`
185- Create: `Sources/OrgCore/Syntax/SyntaxNode.swift`
186- Test: `Tests/OrgCoreTests/SyntaxTreeTests.swift`
187
188**Interfaces:**
189- Produces: `SyntaxKind` (enum, `String` raw values), `GreenToken(kind:text:)`, `GreenNode(kind:children:)` with `.length`, `.text`; `GreenElement` (`.node`, `.token`); internal `GreenBuilder` with `start(_:)`, `token(_:_:)`, `finish()`, `build()`; `SyntaxNode` with `kind`, `range`, `text`, `children`, `tokens`, `descendants()`; `SyntaxToken`. `OrgTree` depends on `OrgSettings` (Task 4), so it is added in Task 5.
190
191- [ ] **Step 1: Write the failing tests**
192
193```swift
194import Testing
195@testable import OrgCore
196
197struct SyntaxTreeTests {
198 func sample() -> GreenNode {
199 var b = GreenBuilder()
200 b.start(.document)
201 b.start(.paragraph)
202 b.token(.text, "hé😀")
203 b.token(.newline, "\n")
204 b.finish()
205 b.token(.newline, "\r\n")
206 b.finish()
207 return b.build()
208 }
209
210 @Test func lengthsAreUTF16() {
211 let green = sample()
212 #expect(green.length == 4 + 1 + 2)
213 #expect(green.text == "hé😀\n\r\n")
214 }
215
216 @Test func redNodesCarryOffsets() {
217 let root = SyntaxNode(green: sample(), offset: 0, parent: nil)
218 let paragraph = root.children[0]
219 #expect(paragraph.kind == .paragraph)
220 #expect(paragraph.range == 0..<5)
221 #expect(paragraph.parent === root)
222 #expect(root.tokens.map(\.range) == [5..<7])
223 #expect(paragraph.tokens.map(\.kind) == [.text, .newline])
224 }
225
226 @Test func builderSkipsEmptyTokens() {
227 var b = GreenBuilder()
228 b.start(.document)
229 b.token(.whitespace, "")
230 b.finish()
231 #expect(b.build().children.isEmpty)
232 }
233
234 @Test func descendantsArePreorder() {
235 let root = SyntaxNode(green: sample(), offset: 0, parent: nil)
236 #expect(root.descendants().map(\.kind) == [.document, .paragraph])
237 }
238}
239```
240
241- [ ] **Step 2: Run tests to verify they fail**
242
243Run: `swift test --filter SyntaxTreeTests`
244Expected: build failure, `cannot find 'GreenBuilder' in scope`.
245
246- [ ] **Step 3: Implement `SyntaxKind.swift`**
247
248```swift
249public enum SyntaxKind: String, Sendable {
250 // Tokens
251 case text, newline, whitespace
252 case stars, todoKeyword, priority, title, tags
253
254 // Nodes
255 case document, zerothSection, section, heading
256 case planning, propertyDrawer, nodeProperty, drawer, clock
257 case paragraph, plainList, item, table, tableRow, tableFormula
258 case block, dynamicBlock, keyword, affiliatedKeyword
259 case comment, fixedWidth, horizontalRule, footnoteDefinition
260}
261```
262
263- [ ] **Step 4: Implement `GreenTree.swift`**
264
265```swift
266public struct GreenToken: Sendable, Equatable {
267 public let kind: SyntaxKind
268 public let text: String
269 /// Length in UTF-16 code units.
270 public let length: Int
271
272 public init(kind: SyntaxKind, text: String) {
273 self.kind = kind
274 self.text = text
275 self.length = text.utf16.count
276 }
277}
278
279/// An immutable node. Stores only kind, children and length, so unchanged subtrees can be
280/// shared between versions of a document.
281public final class GreenNode: Sendable, Equatable {
282 public let kind: SyntaxKind
283 public let children: [GreenElement]
284 /// Length in UTF-16 code units.
285 public let length: Int
286
287 public init(kind: SyntaxKind, children: [GreenElement]) {
288 self.kind = kind
289 self.children = children
290 self.length = children.reduce(0) { $0 + $1.length }
291 }
292
293 public static func == (lhs: GreenNode, rhs: GreenNode) -> Bool {
294 lhs === rhs || (lhs.kind == rhs.kind && lhs.children == rhs.children)
295 }
296
297 public var text: String {
298 var out = ""
299 write(to: &out)
300 return out
301 }
302
303 func write(to out: inout String) {
304 for child in children {
305 switch child {
306 case .node(let node): node.write(to: &out)
307 case .token(let token): out += token.text
308 }
309 }
310 }
311}
312
313public enum GreenElement: Sendable, Equatable {
314 case node(GreenNode)
315 case token(GreenToken)
316
317 public var length: Int {
318 switch self {
319 case .node(let node): node.length
320 case .token(let token): token.length
321 }
322 }
323}
324
325struct GreenBuilder {
326 private var stack: [(kind: SyntaxKind, children: [GreenElement])] = []
327 private var root: GreenNode?
328
329 mutating func start(_ kind: SyntaxKind) {
330 stack.append((kind, []))
331 }
332
333 mutating func token(_ kind: SyntaxKind, _ text: some StringProtocol) {
334 guard !text.isEmpty else { return }
335 stack[stack.count - 1].children.append(.token(GreenToken(kind: kind, text: String(text))))
336 }
337
338 mutating func finish() {
339 let (kind, children) = stack.removeLast()
340 let node = GreenNode(kind: kind, children: children)
341 if stack.isEmpty {
342 root = node
343 } else {
344 stack[stack.count - 1].children.append(.node(node))
345 }
346 }
347
348 func build() -> GreenNode {
349 precondition(stack.isEmpty, "unfinished nodes")
350 return root!
351 }
352}
353```
354
355- [ ] **Step 5: Implement `SyntaxNode.swift`**
356
357```swift
358/// A view of a green node at an absolute offset, with a parent link. Created on demand.
359public final class SyntaxNode: Sendable {
360 public let green: GreenNode
361 public let offset: Int
362 public let parent: SyntaxNode?
363
364 init(green: GreenNode, offset: Int, parent: SyntaxNode?) {
365 self.green = green
366 self.offset = offset
367 self.parent = parent
368 }
369
370 public var kind: SyntaxKind { green.kind }
371 public var range: Range<Int> { offset..<(offset + green.length) }
372 public var text: String { green.text }
373
374 public var children: [SyntaxNode] {
375 var result: [SyntaxNode] = []
376 var at = offset
377 for child in green.children {
378 if case .node(let node) = child {
379 result.append(SyntaxNode(green: node, offset: at, parent: self))
380 }
381 at += child.length
382 }
383 return result
384 }
385
386 public var tokens: [SyntaxToken] {
387 var result: [SyntaxToken] = []
388 var at = offset
389 for child in green.children {
390 if case .token(let token) = child {
391 result.append(SyntaxToken(kind: token.kind, text: token.text, range: at..<(at + token.length)))
392 }
393 at += child.length
394 }
395 return result
396 }
397
398 /// This node and every node below it, in document order.
399 public func descendants() -> [SyntaxNode] {
400 [self] + children.flatMap { $0.descendants() }
401 }
402}
403
404public struct SyntaxToken: Sendable, Equatable {
405 public let kind: SyntaxKind
406 public let text: String
407 public let range: Range<Int>
408}
409```
410
411- [ ] **Step 6: Run tests to verify they pass**
412
413Run: `swift test --filter SyntaxTreeTests`
414Expected: 4 tests pass.
415
416- [ ] **Step 7: Commit**
417
418```bash
419git add Sources Tests
420git commit -m "Add green and red syntax tree"
421```
422
423---
424
425### Task 3: Line splitting and classification
426
427**Files:**
428- Create: `Sources/OrgCore/Parser/Lines.swift`
429- Test: `Tests/OrgCoreTests/LinesTests.swift`
430
431**Interfaces:**
432- Produces (internal): `RawLine(content: Substring, ending: Substring)`, `splitRawLines(_ text: String) -> [RawLine]`, `LineClass` enum, `ClassifiedLine(cls:indent:)`, `classifyLine(_ line: Substring) -> ClassifiedLine`, `Substring.trimmingTrailingWhitespace`.
433
434- [ ] **Step 1: Write the failing tests**
435
436```swift
437import Testing
438@testable import OrgCore
439
440struct LinesTests {
441 @Test func splitKeepsEveryEnding() {
442 let lines = splitRawLines("a\r\nb\n\nc")
443 #expect(lines.map { String($0.content) } == ["a", "b", "", "c"])
444 #expect(lines.map { String($0.ending) } == ["\r\n", "\n", "\n", ""])
445 #expect(splitRawLines("").isEmpty)
446 #expect(splitRawLines("x\n").count == 1)
447 }
448
449 @Test func splitIsLossless() {
450 let text = "\r\n\n a\r b\r\n😀\n"
451 #expect(splitRawLines(text).map { String($0.content) + String($0.ending) }.joined() == text)
452 }
453
454 @Test(arguments: [
455 ("", LineClass.blank),
456 (" \t", .blank),
457 ("* a", .heading(level: 1)),
458 ("*** ", .heading(level: 3)),
459 ("*", .heading(level: 1)),
460 ("*bold* text", .plain),
461 (" * a", .listItem),
462 ("#+BEGIN_SRC sh :results output", .blockBegin(name: "src")),
463 ("#+end_src", .blockEnd(name: "src")),
464 ("#+BEGIN: clocktable :scope file", .dynamicBegin),
465 ("#+END:", .dynamicEnd),
466 ("#+TITLE: x", .keyword(key: "TITLE")),
467 ("#+tblfm: $2=$1", .keyword(key: "TBLFM")),
468 ("# comment", .comment),
469 ("#", .comment),
470 ("#hashtag", .plain),
471 (": fixed", .fixedWidth),
472 (":", .fixedWidth),
473 (":PROPERTIES:", .drawerBegin(name: "PROPERTIES")),
474 (" :LOGBOOK:", .drawerBegin(name: "LOGBOOK")),
475 (":END:", .drawerEnd),
476 ("| a | b |", .tableRow),
477 ("-----", .horizontalRule),
478 ("----", .plain),
479 ("[fn:1] note", .footnoteDefinition),
480 ("CLOCK: [2026-10-04 Sun 10:00]", .clock),
481 ("SCHEDULED: <2026-10-04 Sun>", .planning),
482 ("- item", .listItem),
483 ("+ item", .listItem),
484 ("1. item", .listItem),
485 ("2) item", .listItem),
486 ("-", .listItem),
487 ("-x", .plain),
488 ("1.5 apples", .plain),
489 ("plain text", .plain),
490 ])
491 func classify(line: String, expected: LineClass) {
492 #expect(classifyLine(line[...]).cls == expected)
493 }
494
495 @Test func indentCountsTabsToEight() {
496 #expect(classifyLine("\t- a").indent == 8)
497 #expect(classifyLine(" \t- a").indent == 8)
498 #expect(classifyLine(" - a").indent == 3)
499 }
500}
501```
502
503- [ ] **Step 2: Run tests to verify they fail**
504
505Run: `swift test --filter LinesTests`
506Expected: build failure, `cannot find 'splitRawLines' in scope`.
507
508- [ ] **Step 3: Implement**
509
510```swift
511/// One line of source: its content and its terminator ("", "\n" or "\r\n"), both as slices of
512/// the original text.
513struct RawLine {
514 let content: Substring
515 let ending: Substring
516}
517
518/// Splits on "\n" without normalizing anything. Works on unicode scalars, because String
519/// treats "\r\n" as a single Character.
520func splitRawLines(_ text: String) -> [RawLine] {
521 var lines: [RawLine] = []
522 let scalars = text.unicodeScalars
523 var lineStart = scalars.startIndex
524 var i = lineStart
525 while i != scalars.endIndex {
526 if scalars[i] == "\n" {
527 var contentEnd = i
528 if contentEnd > lineStart, scalars[scalars.index(before: i)] == "\r" {
529 contentEnd = scalars.index(before: i)
530 }
531 let next = scalars.index(after: i)
532 lines.append(RawLine(content: text[lineStart..<contentEnd], ending: text[contentEnd..<next]))
533 lineStart = next
534 i = next
535 } else {
536 i = scalars.index(after: i)
537 }
538 }
539 if lineStart != scalars.endIndex {
540 lines.append(RawLine(content: text[lineStart...], ending: ""))
541 }
542 return lines
543}
544
545enum LineClass: Equatable {
546 case blank
547 case heading(level: Int)
548 case blockBegin(name: String)
549 case blockEnd(name: String)
550 case dynamicBegin
551 case dynamicEnd
552 case drawerBegin(name: String)
553 case drawerEnd
554 case keyword(key: String)
555 case comment
556 case fixedWidth
557 case horizontalRule
558 case tableRow
559 case footnoteDefinition
560 case clock
561 case planning
562 case listItem
563 case plain
564}
565
566struct ClassifiedLine {
567 let cls: LineClass
568 /// Column of the first non-blank character, with tabs advancing to the next multiple of 8.
569 let indent: Int
570}
571
572func classifyLine(_ line: Substring) -> ClassifiedLine {
573 var column = 0
574 var rest = line
575 while let c = rest.first, c == " " || c == "\t" {
576 column = c == "\t" ? (column / 8 + 1) * 8 : column + 1
577 rest = rest.dropFirst()
578 }
579 if rest.isEmpty { return ClassifiedLine(cls: .blank, indent: column) }
580 return ClassifiedLine(cls: lineClass(rest, columnZero: column == 0), indent: column)
581}
582
583private func lineClass(_ rest: Substring, columnZero: Bool) -> LineClass {
584 let trimmed = rest.trimmingTrailingWhitespace
585
586 if columnZero, rest.first == "*" {
587 let stars = rest.prefix { $0 == "*" }
588 let after = rest.dropFirst(stars.count)
589 if after.isEmpty || after.first == " " || after.first == "\t" {
590 return .heading(level: stars.count)
591 }
592 }
593
594 if rest.hasPrefix("#+") {
595 let lower = trimmed.lowercased()
596 if lower.hasPrefix("#+begin_") {
597 let name = lower.dropFirst(8).prefix { !$0.isWhitespace }
598 if !name.isEmpty { return .blockBegin(name: String(name)) }
599 }
600 if lower.hasPrefix("#+end_") {
601 let name = lower.dropFirst(6)
602 if !name.isEmpty, !name.contains(where: \.isWhitespace) { return .blockEnd(name: String(name)) }
603 }
604 if lower.hasPrefix("#+begin:") { return .dynamicBegin }
605 if lower == "#+end:" { return .dynamicEnd }
606 if let colon = rest.firstIndex(of: ":") {
607 let key = rest[rest.index(rest.startIndex, offsetBy: 2)..<colon]
608 if !key.isEmpty, !key.contains(where: \.isWhitespace) { return .keyword(key: key.uppercased()) }
609 }
610 }
611
612 if trimmed == "#" || rest.hasPrefix("# ") || rest.hasPrefix("#\t") { return .comment }
613
614 if rest.first == ":" {
615 if trimmed == ":" || rest.hasPrefix(": ") || rest.hasPrefix(":\t") { return .fixedWidth }
616 if trimmed.uppercased() == ":END:" { return .drawerEnd }
617 if trimmed.count >= 3, trimmed.last == ":" {
618 let name = trimmed.dropFirst().dropLast()
619 if name.allSatisfy({ $0.isLetter || $0.isNumber || $0 == "_" || $0 == "-" }) {
620 return .drawerBegin(name: String(name))
621 }
622 }
623 }
624
625 if rest.first == "|" { return .tableRow }
626 if trimmed.count >= 5, trimmed.allSatisfy({ $0 == "-" }) { return .horizontalRule }
627
628 if columnZero, rest.hasPrefix("[fn:"), let close = rest.firstIndex(of: "]"),
629 close > rest.index(rest.startIndex, offsetBy: 4) {
630 return .footnoteDefinition
631 }
632
633 if rest.hasPrefix("CLOCK:") { return .clock }
634 if rest.hasPrefix("SCHEDULED:") || rest.hasPrefix("DEADLINE:") || rest.hasPrefix("CLOSED:") { return .planning }
635 if isListBullet(rest, indented: !columnZero) { return .listItem }
636 return .plain
637}
638
639/// `-`, `+`, `*` (indented only), `1.` or `1)`, followed by whitespace or end of line.
640/// Alphabetical bullets are off, as in org's default.
641private func isListBullet(_ rest: Substring, indented: Bool) -> Bool {
642 guard let first = rest.first else { return false }
643 let afterBullet: Substring
644 if first == "-" || first == "+" || (first == "*" && indented) {
645 afterBullet = rest.dropFirst()
646 } else if first.isASCII, first.isNumber {
647 let digits = rest.prefix { $0.isASCII && $0.isNumber }
648 let tail = rest.dropFirst(digits.count)
649 guard let separator = tail.first, separator == "." || separator == ")" else { return false }
650 afterBullet = tail.dropFirst()
651 } else {
652 return false
653 }
654 return afterBullet.isEmpty || afterBullet.first == " " || afterBullet.first == "\t"
655}
656
657extension Substring {
658 var trimmingTrailingWhitespace: Substring {
659 var s = self
660 while let last = s.last, last == " " || last == "\t" { s = s.dropLast() }
661 return s
662 }
663}
664```
665
666- [ ] **Step 4: Run tests to verify they pass**
667
668Run: `swift test --filter LinesTests`
669Expected: all pass.
670
671- [ ] **Step 5: Commit**
672
673```bash
674git add Sources Tests
675git commit -m "Add line splitting and classification"
676```
677
678---
679
680### Task 4: In-buffer settings
681
682**Files:**
683- Create: `Sources/OrgCore/Parser/Settings.swift`
684- Test: `Tests/OrgCoreTests/SettingsTests.swift`
685
686**Interfaces:**
687- Consumes: `RawLine`, `ClassifiedLine`, `LineClass` (Task 3).
688- Produces: `TodoKeyword`, `TodoSequence`, `Priorities`, `OrgSettings` with `.default`, `todoKeywordNames: Set<String>`, `isDone(_:)`; internal `SettingsScanner.scan(lines:info:blockEnds:defaults:) -> OrgSettings`, where `blockEnds: [Int: Int]` maps a block's begin line index to its end line index.
689
690- [ ] **Step 1: Write the failing tests**
691
692```swift
693import Testing
694@testable import OrgCore
695
696struct SettingsTests {
697 func scan(_ text: String, blockEnds: [Int: Int] = [:]) -> OrgSettings {
698 let lines = splitRawLines(text)
699 let info = lines.map { classifyLine($0.content) }
700 return SettingsScanner.scan(lines: lines, info: info, blockEnds: blockEnds, defaults: .default)
701 }
702
703 @Test func defaultsWithoutKeywords() {
704 let settings = scan("* TODO a\n")
705 #expect(settings.todoKeywordNames == ["TODO", "DONE"])
706 #expect(settings.isDone("DONE"))
707 }
708
709 @Test func fileKeywordsReplaceDefaults() {
710 let settings = scan("#+TODO: NEXT(n) WAIT(w@/!) | DONE(d!) CANCELED(c@)\n")
711 #expect(settings.todoKeywordNames == ["NEXT", "WAIT", "DONE", "CANCELED"])
712 let sequence = settings.todoSequences[0]
713 #expect(sequence.active.map(\.name) == ["NEXT", "WAIT"])
714 #expect(sequence.done.map(\.name) == ["DONE", "CANCELED"])
715 #expect(sequence.active[1] == TodoKeyword(name: "WAIT", fastKey: "w", logOnEnter: "@", logOnLeave: "!"))
716 #expect(sequence.done[0] == TodoKeyword(name: "DONE", fastKey: "d", logOnEnter: "!", logOnLeave: nil))
717 }
718
719 @Test func lastWordIsDoneWithoutSeparator() {
720 let settings = scan("#+SEQ_TODO: A B C\n")
721 #expect(settings.todoSequences[0].active.map(\.name) == ["A", "B"])
722 #expect(settings.todoSequences[0].done.map(\.name) == ["C"])
723 }
724
725 @Test func severalLinesMakeSeveralSequences() {
726 let settings = scan("#+TODO: A | B\n#+TYP_TODO: X | Y\n")
727 #expect(settings.todoSequences.count == 2)
728 #expect(settings.todoSequences[1].kind == .type)
729 }
730
731 @Test func keywordsInsideBlocksAreIgnored() {
732 let text = "#+begin_example\n#+TODO: X | Y\n#+end_example\n"
733 #expect(scan(text, blockEnds: [0: 2]).todoKeywordNames == ["TODO", "DONE"])
734 }
735
736 @Test func priorities() {
737 #expect(scan("#+PRIORITIES: 1 10 5\n").priorities == Priorities(highest: "1", lowest: "10", default: "5"))
738 #expect(scan("").priorities == Priorities(highest: "A", lowest: "C", default: "B"))
739 }
740}
741```
742
743- [ ] **Step 2: Run tests to verify they fail**
744
745Run: `swift test --filter SettingsTests`
746Expected: build failure, `cannot find 'SettingsScanner' in scope`.
747
748- [ ] **Step 3: Implement**
749
750```swift
751public struct TodoKeyword: Sendable, Hashable {
752 public var name: String
753 public var fastKey: Character?
754 /// Logging flag when entering the state (`!` or `@`), from `NAME(k!/@)`.
755 public var logOnEnter: String?
756 /// Logging flag when leaving the state.
757 public var logOnLeave: String?
758
759 public init(name: String, fastKey: Character? = nil, logOnEnter: String? = nil, logOnLeave: String? = nil) {
760 self.name = name
761 self.fastKey = fastKey
762 self.logOnEnter = logOnEnter
763 self.logOnLeave = logOnLeave
764 }
765}
766
767public struct TodoSequence: Sendable, Equatable {
768 public enum Kind: Sendable, Equatable { case sequence, type }
769
770 public var kind: Kind
771 public var active: [TodoKeyword]
772 public var done: [TodoKeyword]
773
774 public init(kind: Kind, active: [TodoKeyword], done: [TodoKeyword]) {
775 self.kind = kind
776 self.active = active
777 self.done = done
778 }
779}
780
781public struct Priorities: Sendable, Equatable {
782 public var highest: String
783 public var lowest: String
784 public var `default`: String
785
786 public init(highest: String, lowest: String, default: String) {
787 self.highest = highest
788 self.lowest = lowest
789 self.default = `default`
790 }
791}
792
793public struct OrgSettings: Sendable, Equatable {
794 public var todoSequences: [TodoSequence]
795 public var priorities: Priorities
796
797 public init(todoSequences: [TodoSequence], priorities: Priorities) {
798 self.todoSequences = todoSequences
799 self.priorities = priorities
800 }
801
802 public static let `default` = OrgSettings(
803 todoSequences: [TodoSequence(kind: .sequence, active: [TodoKeyword(name: "TODO")], done: [TodoKeyword(name: "DONE")])],
804 priorities: Priorities(highest: "A", lowest: "C", default: "B")
805 )
806
807 public var todoKeywordNames: Set<String> {
808 Set(todoSequences.flatMap { ($0.active + $0.done).map(\.name) })
809 }
810
811 public func isDone(_ name: String) -> Bool {
812 todoSequences.contains { $0.done.contains { $0.name == name } }
813 }
814}
815
816enum SettingsScanner {
817 /// Reads `#+TODO`, `#+SEQ_TODO`, `#+TYP_TODO` and `#+PRIORITIES` outside blocks. Any TODO
818 /// line replaces the default sequences, as in org.
819 static func scan(lines: [RawLine], info: [ClassifiedLine], blockEnds: [Int: Int], defaults: OrgSettings) -> OrgSettings {
820 var sequences: [TodoSequence] = []
821 var priorities = defaults.priorities
822 var k = 0
823 while k < lines.count {
824 switch info[k].cls {
825 case .blockBegin, .dynamicBegin:
826 if let end = blockEnds[k] { k = end }
827 case .keyword(let key):
828 let value = keywordValue(lines[k].content)
829 switch key {
830 case "TODO", "SEQ_TODO":
831 if let s = todoSequence(value, kind: .sequence) { sequences.append(s) }
832 case "TYP_TODO":
833 if let s = todoSequence(value, kind: .type) { sequences.append(s) }
834 case "PRIORITIES":
835 let words = value.split(whereSeparator: \.isWhitespace)
836 if words.count == 3 {
837 priorities = Priorities(highest: String(words[0]), lowest: String(words[1]), default: String(words[2]))
838 }
839 default:
840 break
841 }
842 default:
843 break
844 }
845 k += 1
846 }
847 return OrgSettings(todoSequences: sequences.isEmpty ? defaults.todoSequences : sequences, priorities: priorities)
848 }
849
850 static func keywordValue(_ line: Substring) -> Substring {
851 guard let colon = line.firstIndex(of: ":") else { return "" }
852 return line[line.index(after: colon)...]
853 }
854
855 static func todoSequence(_ value: Substring, kind: TodoSequence.Kind) -> TodoSequence? {
856 let words = value.split(whereSeparator: \.isWhitespace)
857 guard !words.isEmpty else { return nil }
858 if let bar = words.firstIndex(of: "|") {
859 return TodoSequence(kind: kind, active: words[..<bar].map(todoKeyword), done: words[(bar + 1)...].map(todoKeyword))
860 }
861 return TodoSequence(kind: kind, active: words.dropLast().map(todoKeyword), done: [todoKeyword(words.last!)])
862 }
863
864 /// `NAME`, or `NAME(spec)` where spec is an optional fast key followed by `enter/leave`
865 /// logging flags.
866 static func todoKeyword(_ word: Substring) -> TodoKeyword {
867 guard let open = word.firstIndex(of: "("), word.last == ")" else { return TodoKeyword(name: String(word)) }
868 var spec = word[word.index(after: open)..<word.index(before: word.endIndex)]
869 var fastKey: Character?
870 if let first = spec.first, first != "!", first != "@", first != "/" {
871 fastKey = first
872 spec = spec.dropFirst()
873 }
874 let parts = spec.split(separator: "/", omittingEmptySubsequences: false)
875 let enter = parts.first.flatMap { $0.isEmpty ? nil : String($0) }
876 let leave = parts.count > 1 && !parts[1].isEmpty ? String(parts[1]) : nil
877 return TodoKeyword(name: String(word[..<open]), fastKey: fastKey, logOnEnter: enter, logOnLeave: leave)
878 }
879}
880```
881
882- [ ] **Step 4: Run tests to verify they pass**
883
884Run: `swift test --filter SettingsTests`
885Expected: all pass.
886
887- [ ] **Step 5: Commit**
888
889```bash
890git add Sources Tests
891git commit -m "Add in-buffer TODO and priority settings"
892```
893
894---
895
896### Task 5: Parser — document, sections and headings
897
898**Files:**
899- Create: `Sources/OrgCore/Parser/Parser.swift`
900- Modify: `Sources/OrgCore/Syntax/SyntaxNode.swift` (append `OrgTree`)
901- Test: `Tests/OrgCoreTests/ParserSectionTests.swift`
902
903**Interfaces:**
904- Consumes: Tasks 2–4.
905- Produces: `public enum OrgParser { static func parse(_ text: String, defaults: OrgSettings = .default) -> OrgTree }`; `public struct OrgTree { green: GreenNode; settings: OrgSettings; root: SyntaxNode; text: String }`; internal `struct Parser` with `element(limit:floor:)` that Task 6 fills in. In this task `element` handles every class as a paragraph or a blank line.
906
907- [ ] **Step 1: Write the failing tests**
908
909```swift
910import Testing
911@testable import OrgCore
912
913func nodeKinds(_ text: String) -> [SyntaxKind] {
914 OrgParser.parse(text).root.descendants().map(\.kind)
915}
916
917func tokens(of kind: SyntaxKind, in text: String) -> [SyntaxToken] {
918 OrgParser.parse(text).root.descendants().filter { $0.kind == kind }.flatMap(\.tokens)
919}
920
921struct ParserSectionTests {
922 @Test func emptyDocument() {
923 let tree = OrgParser.parse("")
924 #expect(tree.text == "")
925 #expect(nodeKinds("") == [.document])
926 }
927
928 @Test func zerothSectionHoldsPreamble() {
929 #expect(nodeKinds("text\n* a\n") == [.document, .zerothSection, .paragraph, .section, .heading])
930 }
931
932 @Test func sectionsNestByLevel() {
933 let text = "* a\n** b\n*** c\n** d\n* e\n"
934 let root = OrgParser.parse(text).root
935 let top = root.children
936 #expect(top.map(\.kind) == [.section, .section])
937 #expect(top[0].children.map(\.kind) == [.heading, .section, .section])
938 #expect(top[0].children[1].children.map(\.kind) == [.heading, .section])
939 }
940
941 @Test func headingTokens() {
942 let parts = tokens(of: .heading, in: "** TODO [#A] Write the plan :work:urgent: \n")
943 #expect(parts.map(\.kind) == [.stars, .whitespace, .todoKeyword, .whitespace, .priority, .whitespace, .title, .whitespace, .tags, .whitespace, .newline])
944 #expect(parts.first { $0.kind == .tags }?.text == ":work:urgent:")
945 #expect(parts.first { $0.kind == .title }?.text == "Write the plan")
946 }
947
948 @Test func todoKeywordsComeFromSettings() {
949 let text = "#+TODO: NEXT | DONE\n* NEXT a\n* TODO b\n"
950 let todo = tokens(of: .heading, in: text).filter { $0.kind == .todoKeyword }.map(\.text)
951 #expect(todo == ["NEXT"])
952 }
953
954 @Test func priorityNeedsValidValueAndSpace() {
955 #expect(tokens(of: .heading, in: "* [#B] x\n").contains { $0.kind == .priority })
956 #expect(tokens(of: .heading, in: "* [#10] x\n").contains { $0.kind == .priority })
957 #expect(!tokens(of: .heading, in: "* [#AB] x\n").contains { $0.kind == .priority })
958 #expect(!tokens(of: .heading, in: "* [#A]x\n").contains { $0.kind == .priority })
959 }
960
961 @Test func tagsNeedValidCharacters() {
962 #expect(tokens(of: .heading, in: "* a :b@c_1:\n").contains { $0.kind == .tags })
963 #expect(!tokens(of: .heading, in: "* a :b c:\n").contains { $0.kind == .tags })
964 #expect(!tokens(of: .heading, in: "* a :b:c\n").contains { $0.kind == .tags })
965 }
966
967 @Test func headingWithoutNewlineAtEnd() {
968 #expect(OrgParser.parse("* a").text == "* a")
969 }
970
971 @Test(arguments: ["* a\n", "*\n", "* TODO\n", "text\r\n* a\r\n** b\r\n", "\n\n* a\n\n"])
972 func roundTrip(text: String) {
973 #expect(OrgParser.parse(text).text == text)
974 }
975}
976```
977
978- [ ] **Step 2: Run tests to verify they fail**
979
980Run: `swift test --filter ParserSectionTests`
981Expected: build failure, `cannot find 'OrgParser' in scope`.
982
983- [ ] **Step 3: Append `OrgTree` to `SyntaxNode.swift`**
984
985```swift
986public struct OrgTree: Sendable {
987 public let green: GreenNode
988 public let settings: OrgSettings
989
990 public var root: SyntaxNode { SyntaxNode(green: green, offset: 0, parent: nil) }
991 public var text: String { green.text }
992}
993```
994
995- [ ] **Step 4: Implement `Parser.swift`**
996
997```swift
998public enum OrgParser {
999 public static func parse(_ text: String, defaults: OrgSettings = .default) -> OrgTree {
1000 var parser = Parser(text: text, defaults: defaults)
1001 return parser.run()
1002 }
1003}
1004
1005struct Parser {
1006 let lines: [RawLine]
1007 let info: [ClassifiedLine]
1008 /// Begin line → end line, for blocks, dynamic blocks and drawers that are closed before the
1009 /// next heading.
1010 let blockEnds: [Int: Int]
1011 let settings: OrgSettings
1012 var builder = GreenBuilder()
1013 var i = 0
1014
1015 init(text: String, defaults: OrgSettings) {
1016 lines = splitRawLines(text)
1017 info = lines.map { classifyLine($0.content) }
1018 blockEnds = Parser.matchEnds(info)
1019 settings = SettingsScanner.scan(lines: lines, info: info, blockEnds: blockEnds, defaults: defaults)
1020 }
1021
1022 static func matchEnds(_ info: [ClassifiedLine]) -> [Int: Int] {
1023 var ends: [Int: Int] = [:]
1024 var k = 0
1025 while k < info.count {
1026 let isEnd: ((LineClass) -> Bool)?
1027 switch info[k].cls {
1028 case .blockBegin(let name): isEnd = { $0 == .blockEnd(name: name) }
1029 case .dynamicBegin: isEnd = { $0 == .dynamicEnd }
1030 case .drawerBegin: isEnd = { $0 == .drawerEnd }
1031 default: isEnd = nil
1032 }
1033 if let isEnd {
1034 var j = k + 1
1035 while j < info.count {
1036 if case .heading = info[j].cls { break }
1037 if isEnd(info[j].cls) { ends[k] = j; break }
1038 j += 1
1039 }
1040 // Block contents are verbatim, so nothing inside starts another element.
1041 if let end = ends[k], !isDrawer(info[k].cls) { k = end }
1042 }
1043 k += 1
1044 }
1045 return ends
1046 }
1047
1048 static func isDrawer(_ cls: LineClass) -> Bool {
1049 if case .drawerBegin = cls { return true }
1050 return false
1051 }
1052
1053 mutating func run() -> OrgTree {
1054 builder.start(.document)
1055 if !lines.isEmpty, !isHeading(0) {
1056 builder.start(.zerothSection)
1057 parseContent(limit: lines.count)
1058 builder.finish()
1059 }
1060 while i < lines.count, case .heading(let level) = info[i].cls {
1061 parseSection(level: level)
1062 }
1063 builder.finish()
1064 return OrgTree(green: builder.build(), settings: settings)
1065 }
1066
1067 func isHeading(_ k: Int) -> Bool {
1068 if case .heading = info[k].cls { return true }
1069 return false
1070 }
1071
1072 mutating func parseSection(level: Int) {
1073 builder.start(.section)
1074 headingLine(lines[i])
1075 i += 1
1076 if i < lines.count, info[i].cls == .planning {
1077 builder.start(.planning)
1078 line(i)
1079 i += 1
1080 builder.finish()
1081 }
1082 if i < lines.count, case .drawerBegin(let name) = info[i].cls, name.uppercased() == "PROPERTIES",
1083 let end = blockEnds[i] {
1084 propertyDrawer(end: end)
1085 }
1086 parseContent(limit: lines.count)
1087 while i < lines.count, case .heading(let child) = info[i].cls, child > level {
1088 parseSection(level: child)
1089 }
1090 builder.finish()
1091 }
1092
1093 mutating func propertyDrawer(end: Int) {
1094 builder.start(.propertyDrawer)
1095 line(i)
1096 i += 1
1097 while i < end {
1098 if info[i].cls == .blank {
1099 line(i)
1100 } else {
1101 builder.start(.nodeProperty)
1102 line(i)
1103 builder.finish()
1104 }
1105 i += 1
1106 }
1107 line(i)
1108 i += 1
1109 builder.finish()
1110 }
1111
1112 /// Elements until `limit` or the next heading.
1113 mutating func parseContent(limit: Int) {
1114 while i < limit, !isHeading(i) {
1115 element(limit: limit, floor: nil)
1116 }
1117 }
1118
1119 /// One element starting at `i`. `floor` is the indent of the enclosing list item, if any:
1120 /// non-blank lines at or left of it end the element.
1121 mutating func element(limit: Int, floor: Int?) {
1122 if info[i].cls == .blank {
1123 line(i)
1124 i += 1
1125 } else {
1126 paragraph(limit: limit, floor: floor)
1127 }
1128 }
1129
1130 mutating func paragraph(limit: Int, floor: Int?) {
1131 builder.start(.paragraph)
1132 line(i)
1133 i += 1
1134 while i < limit, within(floor, i), continuesParagraph(i) {
1135 line(i)
1136 i += 1
1137 }
1138 builder.finish()
1139 }
1140
1141 func continuesParagraph(_ k: Int) -> Bool {
1142 switch info[k].cls {
1143 case .blank, .heading: return false
1144 default: return true
1145 }
1146 }
1147
1148 func within(_ floor: Int?, _ k: Int) -> Bool {
1149 guard let floor else { return true }
1150 return info[k].indent > floor
1151 }
1152
1153 // MARK: - Tokens
1154
1155 /// A whole line as leading whitespace, content and line ending.
1156 mutating func line(_ k: Int) {
1157 let content = lines[k].content
1158 let rest = whitespace(content)
1159 builder.token(.text, rest)
1160 builder.token(.newline, lines[k].ending)
1161 }
1162
1163 mutating func whitespace(_ s: Substring) -> Substring {
1164 let ws = s.prefix { $0 == " " || $0 == "\t" }
1165 builder.token(.whitespace, ws)
1166 return s.dropFirst(ws.count)
1167 }
1168
1169 mutating func headingLine(_ raw: RawLine) {
1170 builder.start(.heading)
1171 var rest = raw.content
1172 let stars = rest.prefix { $0 == "*" }
1173 builder.token(.stars, stars)
1174 rest = whitespace(rest.dropFirst(stars.count))
1175
1176 let word = rest.prefix { $0 != " " && $0 != "\t" }
1177 if !word.isEmpty, settings.todoKeywordNames.contains(String(word)) {
1178 builder.token(.todoKeyword, word)
1179 rest = whitespace(rest.dropFirst(word.count))
1180 }
1181
1182 if let cookie = priorityCookie(rest) {
1183 builder.token(.priority, cookie)
1184 rest = whitespace(rest.dropFirst(cookie.count))
1185 }
1186
1187 let parts = splitTags(rest)
1188 builder.token(.title, parts.title)
1189 builder.token(.whitespace, parts.gap)
1190 builder.token(.tags, parts.tags)
1191 builder.token(.whitespace, parts.trailing)
1192 builder.token(.newline, raw.ending)
1193 builder.finish()
1194 }
1195
1196 /// `[#A]` or `[#10]`, followed by whitespace or end of line.
1197 func priorityCookie(_ s: Substring) -> Substring? {
1198 guard s.hasPrefix("[#"), let close = s.firstIndex(of: "]") else { return nil }
1199 let value = s[s.index(s.startIndex, offsetBy: 2)..<close]
1200 let valid = (value.count == 1 && value.first!.isLetter && value.first!.isUppercase)
1201 || (!value.isEmpty && value.allSatisfy { $0.isASCII && $0.isNumber })
1202 guard valid else { return nil }
1203 let after = s[s.index(after: close)...]
1204 guard after.isEmpty || after.first == " " || after.first == "\t" else { return nil }
1205 return s[...close]
1206 }
1207
1208 func splitTags(_ s: Substring) -> (title: Substring, gap: Substring, tags: Substring, trailing: Substring) {
1209 let trimmed = s.trimmingTrailingWhitespace
1210 let trailing = s[trimmed.endIndex...]
1211 let none = (title: trimmed, gap: Substring(), tags: Substring(), trailing: trailing)
1212 guard trimmed.last == ":" else { return none }
1213 let tagStart = trimmed.lastIndex { $0 == " " || $0 == "\t" }.map { trimmed.index(after: $0) } ?? trimmed.startIndex
1214 let tags = trimmed[tagStart...]
1215 guard tags.count >= 3, tags.first == ":", isTagString(tags) else { return none }
1216 let before = trimmed[..<tagStart]
1217 let title = before.trimmingTrailingWhitespace
1218 return (title, before[title.endIndex...], tags, trailing)
1219 }
1220
1221 func isTagString(_ tags: Substring) -> Bool {
1222 tags.dropFirst().dropLast().split(separator: ":", omittingEmptySubsequences: false).allSatisfy { tag in
1223 !tag.isEmpty && tag.allSatisfy { $0.isLetter || $0.isNumber || "_@#%".contains($0) }
1224 }
1225 }
1226}
1227```
1228
1229- [ ] **Step 5: Run tests to verify they pass**
1230
1231Run: `swift test --filter ParserSectionTests`
1232Expected: all pass.
1233
1234- [ ] **Step 6: Commit**
1235
1236```bash
1237git add Sources Tests
1238git commit -m "Parse document, sections and headings"
1239```
1240
1241---
1242
1243### Task 6: Parser — elements
1244
1245**Files:**
1246- Modify: `Sources/OrgCore/Parser/Parser.swift` (replace `element`, `continuesParagraph`; add `list`, `item`, `table`, `consecutive`, `footnoteDefinition`)
1247- Test: `Tests/OrgCoreTests/ParserElementTests.swift`
1248
1249**Interfaces:**
1250- Consumes: Task 5's `Parser`.
1251- Produces: nodes `block`, `dynamicBlock`, `drawer`, `keyword`, `affiliatedKeyword`, `comment`, `fixedWidth`, `horizontalRule`, `table`/`tableRow`/`tableFormula`, `footnoteDefinition`, `clock`, `plainList`/`item`, `planning` (after heading only), `propertyDrawer`/`nodeProperty`.
1252
1253- [ ] **Step 1: Write the failing tests**
1254
1255```swift
1256import Testing
1257@testable import OrgCore
1258
1259func childKinds(_ text: String) -> [SyntaxKind] {
1260 let root = OrgParser.parse(text).root
1261 let container = root.children.first { $0.kind == .zerothSection || $0.kind == .section }!
1262 return container.children.map(\.kind)
1263}
1264
1265struct ParserElementTests {
1266 @Test func planningAndPropertiesFollowHeading() {
1267 let text = "* a\nSCHEDULED: <2026-10-04 Sun>\n:PROPERTIES:\n:ID: x\n:END:\nbody\n"
1268 #expect(childKinds(text) == [.heading, .planning, .propertyDrawer, .paragraph])
1269 }
1270
1271 @Test func planningElsewhereIsText() {
1272 #expect(childKinds("SCHEDULED: <2026-10-04 Sun>\n") == [.paragraph])
1273 }
1274
1275 @Test func blocks() {
1276 #expect(childKinds("#+begin_src sh\n,* escaped\n:END:\n#+end_src\nafter\n") == [.block, .paragraph])
1277 #expect(childKinds("#+BEGIN: clocktable\n#+END:\n") == [.dynamicBlock])
1278 }
1279
1280 @Test func headingsEndBlocks() {
1281 #expect(childKinds("#+begin_src sh\n* heading\n#+end_src\n") == [.paragraph])
1282 }
1283
1284 @Test func unclosedBlockIsParagraph() {
1285 #expect(childKinds("#+begin_src sh\necho\n") == [.paragraph])
1286 }
1287
1288 @Test func drawersHoldElements() {
1289 let root = OrgParser.parse(":LOGBOOK:\nCLOCK: [2026-10-04 Sun 10:00]\n:END:\n").root
1290 let drawer = root.children[0].children[0]
1291 #expect(drawer.kind == .drawer)
1292 #expect(drawer.children.map(\.kind) == [.clock])
1293 }
1294
1295 @Test func keywords() {
1296 #expect(childKinds("#+TITLE: x\n#+NAME: t\n#+ATTR_HTML: :width 10\n") == [.keyword, .affiliatedKeyword, .affiliatedKeyword])
1297 }
1298
1299 @Test func commentsAndFixedWidthGroup() {
1300 #expect(childKinds("# a\n# b\n: c\n: d\n-----\n") == [.comment, .fixedWidth, .horizontalRule])
1301 }
1302
1303 @Test func tableWithFormulas() {
1304 let root = OrgParser.parse("| a |\n|---|\n| 1 |\n#+TBLFM: $1=2\n#+TBLFM: $1=3\n").root
1305 let table = root.children[0].children[0]
1306 #expect(table.kind == .table)
1307 #expect(table.children.map(\.kind) == [.tableRow, .tableRow, .tableRow, .tableFormula, .tableFormula])
1308 }
1309
1310 @Test func footnoteDefinition() {
1311 #expect(childKinds("[fn:1] note\ncontinued\n\nafter\n") == [.footnoteDefinition, .paragraph])
1312 }
1313
1314 @Test func listsNestByIndent() {
1315 let text = "- a\n more\n - b\n- c\n\nafter\n"
1316 let root = OrgParser.parse(text).root
1317 let list = root.children[0].children[0]
1318 #expect(list.kind == .plainList)
1319 #expect(list.children.map(\.kind) == [.item, .item])
1320 #expect(list.children[0].children.map(\.kind) == [.paragraph, .plainList])
1321 #expect(childKinds(text) == [.plainList, .paragraph])
1322 }
1323
1324 @Test func twoBlankLinesEndAList() {
1325 #expect(childKinds("- a\n\n\n- b\n") == [.plainList, .plainList])
1326 #expect(childKinds("- a\n\n- b\n") == [.plainList])
1327 }
1328
1329 @Test func paragraphStopsAtElementStart() {
1330 #expect(childKinds("text\n| a |\n") == [.paragraph, .table])
1331 #expect(childKinds("text\n#+begin_quote\nq\n#+end_quote\n") == [.paragraph, .block])
1332 #expect(childKinds("text\n#+begin_quote\nq\n") == [.paragraph])
1333 }
1334}
1335```
1336
1337- [ ] **Step 2: Run tests to verify they fail**
1338
1339Run: `swift test --filter ParserElementTests`
1340Expected: failures; every element parses as a paragraph.
1341
1342- [ ] **Step 3: Replace `element` and `continuesParagraph`, add the element parsers**
1343
1344```swift
1345 static let affiliatedKeys: Set<String> = ["NAME", "CAPTION", "RESULTS", "HEADER", "PLOT"]
1346
1347 mutating func element(limit: Int, floor: Int?) {
1348 switch info[i].cls {
1349 case .blank:
1350 line(i)
1351 i += 1
1352 case .blockBegin, .dynamicBegin:
1353 if let end = blockEnds[i], end < limit {
1354 builder.start(info[i].cls == .dynamicBegin ? .dynamicBlock : .block)
1355 while i <= end {
1356 line(i)
1357 i += 1
1358 }
1359 builder.finish()
1360 } else {
1361 paragraph(limit: limit, floor: floor)
1362 }
1363 case .drawerBegin:
1364 if let end = blockEnds[i], end < limit {
1365 builder.start(.drawer)
1366 line(i)
1367 i += 1
1368 parseContent(limit: end)
1369 line(i)
1370 i += 1
1371 builder.finish()
1372 } else {
1373 paragraph(limit: limit, floor: floor)
1374 }
1375 case .keyword(let key):
1376 let affiliated = Self.affiliatedKeys.contains(key) || key.hasPrefix("ATTR_")
1377 single(affiliated ? .affiliatedKeyword : .keyword)
1378 case .comment:
1379 consecutive(.comment, limit: limit, floor: floor) { $0 == .comment }
1380 case .fixedWidth:
1381 consecutive(.fixedWidth, limit: limit, floor: floor) { $0 == .fixedWidth }
1382 case .horizontalRule:
1383 single(.horizontalRule)
1384 case .clock:
1385 single(.clock)
1386 case .tableRow:
1387 table(limit: limit, floor: floor)
1388 case .footnoteDefinition:
1389 footnoteDefinition(limit: limit)
1390 case .listItem:
1391 list(limit: limit, floor: floor)
1392 default:
1393 paragraph(limit: limit, floor: floor)
1394 }
1395 }
1396
1397 /// Lines that don't start an element of their own.
1398 func continuesParagraph(_ k: Int) -> Bool {
1399 switch info[k].cls {
1400 case .plain, .planning, .blockEnd, .dynamicEnd, .drawerEnd:
1401 return true
1402 case .blockBegin, .dynamicBegin, .drawerBegin:
1403 return blockEnds[k] == nil
1404 default:
1405 return false
1406 }
1407 }
1408
1409 mutating func single(_ kind: SyntaxKind) {
1410 builder.start(kind)
1411 line(i)
1412 i += 1
1413 builder.finish()
1414 }
1415
1416 mutating func consecutive(_ kind: SyntaxKind, limit: Int, floor: Int?, matching: (LineClass) -> Bool) {
1417 builder.start(kind)
1418 repeat {
1419 line(i)
1420 i += 1
1421 } while i < limit && matching(info[i].cls) && within(floor, i)
1422 builder.finish()
1423 }
1424
1425 mutating func table(limit: Int, floor: Int?) {
1426 builder.start(.table)
1427 while i < limit, info[i].cls == .tableRow, within(floor, i) {
1428 single(.tableRow)
1429 }
1430 while i < limit, info[i].cls == .keyword(key: "TBLFM"), within(floor, i) {
1431 single(.tableFormula)
1432 }
1433 builder.finish()
1434 }
1435
1436 mutating func footnoteDefinition(limit: Int) {
1437 builder.start(.footnoteDefinition)
1438 line(i)
1439 i += 1
1440 while i < limit, info[i].cls == .plain {
1441 line(i)
1442 i += 1
1443 }
1444 builder.finish()
1445 }
1446
1447 mutating func list(limit: Int, floor: Int?) {
1448 let base = info[i].indent
1449 builder.start(.plainList)
1450 while i < limit, info[i].cls == .listItem, info[i].indent == base, within(floor, i) {
1451 item(base: base, limit: limit)
1452 }
1453 builder.finish()
1454 }
1455
1456 /// An item's first line, then everything indented past its bullet. One blank line stays
1457 /// inside the item when the item or list continues after it; two end the list.
1458 mutating func item(base: Int, limit: Int) {
1459 builder.start(.item)
1460 line(i)
1461 i += 1
1462 while i < limit, !isHeading(i) {
1463 if info[i].cls == .blank {
1464 var j = i
1465 while j < limit, info[j].cls == .blank { j += 1 }
1466 guard j - i < 2, j < limit else { break }
1467 let continuesItem = info[j].indent > base
1468 let nextSibling = info[j].cls == .listItem && info[j].indent == base
1469 guard continuesItem || nextSibling else { break }
1470 line(i)
1471 i += 1
1472 if nextSibling { break }
1473 continue
1474 }
1475 guard info[i].indent > base else { break }
1476 element(limit: limit, floor: base)
1477 }
1478 builder.finish()
1479 }
1480```
1481
1482- [ ] **Step 4: Run all tests**
1483
1484Run: `swift test`
1485Expected: all pass, including Task 5's tests.
1486
1487- [ ] **Step 5: Commit**
1488
1489```bash
1490git add Sources Tests
1491git commit -m "Parse block-level elements"
1492```
1493
1494---
1495
1496### Task 7: Round-trip fuzz, encoding fixtures and corpus test
1497
1498**Files:**
1499- Test: `Tests/OrgCoreTests/RoundTripTests.swift`
1500
1501**Interfaces:**
1502- Consumes: `SourceText`, `OrgParser`, `SyntaxNode`.
1503- Produces: a seeded fuzz test (2,000 documents), a structural invariant check, and an opt-in corpus test driven by `ORGSTAR_CORPUS`.
1504
1505- [ ] **Step 1: Write the tests**
1506
1507```swift
1508import Foundation
1509import Testing
1510@testable import OrgCore
1511
1512/// SplitMix64, so failures reproduce from the seed.
1513struct SeededGenerator: RandomNumberGenerator {
1514 var state: UInt64
1515 mutating func next() -> UInt64 {
1516 state &+= 0x9E37_79B9_7F4A_7C15
1517 var z = state
1518 z = (z ^ (z >> 30)) &* 0xBF58_476D_1CE4_E5B9
1519 z = (z ^ (z >> 27)) &* 0x94D0_49BB_1331_11EB
1520 return z ^ (z >> 31)
1521 }
1522}
1523
1524let fragments = [
1525 "* ", "** TODO [#A] title :a:b:", "*** DONE", "#+TODO: NEXT | DONE", "#+begin_src sh", "#+end_src",
1526 "#+BEGIN_QUOTE", "#+end_quote", "#+BEGIN: clocktable", "#+END:", ":PROPERTIES:", ":ID: x", ":END:",
1527 ":LOGBOOK:", "CLOCK: [2026-10-04 Sun 10:00]", "SCHEDULED: <2026-10-04 Sun>", "- item", " - nested",
1528 "\t+ tab", "1. one", "| a | b |", "|---+---|", "#+TBLFM: $2=$1", "# comment", ": fixed", "-----",
1529 "[fn:1] note", "#+NAME: x", "plain text", "é", "😀", "e\u{301}", " ", "\t", "\n", "\n", "\r\n", "\r", "",
1530]
1531
1532func randomDocument(_ rng: inout SeededGenerator) -> String {
1533 (0..<Int.random(in: 0...40, using: &rng)).map { _ in
1534 fragments.randomElement(using: &rng)! + (Bool.random(using: &rng) ? "\n" : "")
1535 }.joined()
1536}
1537
1538func checkLengths(_ node: SyntaxNode) -> Bool {
1539 let sum = node.green.children.reduce(0) { $0 + $1.length }
1540 return sum == node.green.length && node.children.allSatisfy(checkLengths)
1541}
1542
1543struct RoundTripTests {
1544 @Test func fuzzedDocumentsRoundTrip() {
1545 var rng = SeededGenerator(state: 20261004)
1546 for n in 0..<2_000 {
1547 let text = randomDocument(&rng)
1548 let tree = OrgParser.parse(text)
1549 #expect(tree.text == text, "document \(n)")
1550 #expect(checkLengths(tree.root), "document \(n)")
1551 }
1552 }
1553
1554 @Test(arguments: [
1555 [0xEF, 0xBB, 0xBF] + Array("* a\r\n".utf8),
1556 Array("* a\r\n- b\n\tc".utf8),
1557 Array("* a".utf8),
1558 Array("😀 e\u{301}\n".utf8),
1559 [0x2A, 0x20, 0xFF, 0x0A] as [UInt8],
1560 ])
1561 func encodingFixturesRoundTrip(bytes: [UInt8]) {
1562 let source = SourceText(bytes: bytes)
1563 let tree = OrgParser.parse(source.text)
1564 #expect(source.encode(tree.text) == bytes)
1565 }
1566
1567 /// Private corpus: `ORGSTAR_CORPUS=/path/to/org swift test --filter corpus`.
1568 @Test(.enabled(if: ProcessInfo.processInfo.environment["ORGSTAR_CORPUS"] != nil))
1569 func corpusRoundTrips() throws {
1570 let root = URL(fileURLWithPath: ProcessInfo.processInfo.environment["ORGSTAR_CORPUS"]!)
1571 let files = FileManager.default.enumerator(at: root, includingPropertiesForKeys: nil)!
1572 .compactMap { $0 as? URL }
1573 .filter { ["org", "org_archive"].contains($0.pathExtension) }
1574 #expect(!files.isEmpty)
1575 for file in files {
1576 let bytes = try [UInt8](Data(contentsOf: file))
1577 let source = SourceText(bytes: bytes)
1578 let tree = OrgParser.parse(source.text)
1579 #expect(source.encode(tree.text) == bytes, "\(file.path)")
1580 }
1581 }
1582}
1583```
1584
1585- [ ] **Step 2: Run the tests**
1586
1587Run: `swift test`
1588Expected: all pass. The corpus test is skipped.
1589
1590- [ ] **Step 3: Run the private corpus locally**
1591
1592Run: `ORGSTAR_CORPUS=~/org swift test --filter corpusRoundTrips`
1593Expected: pass. Any failure names the file; reduce it to a fuzz fragment or a unit test before fixing.
1594
1595- [ ] **Step 4: Commit**
1596
1597```bash
1598git add Tests
1599git commit -m "Add round-trip fuzz, encoding and corpus tests"
1600```