| @@ -0,0 +1,249 @@ |
| 1 | import Foundation |
| 2 | |
| 3 | public struct TextEdit: Sendable, Equatable { |
| 4 | /// UTF-16 range in the old text, on unicode scalar boundaries. |
| 5 | public let range: Range<Int> |
| 6 | public let replacement: String |
| 7 | |
| 8 | public init(range: Range<Int>, replacement: String) { |
| 9 | self.range = range |
| 10 | self.replacement = replacement |
| 11 | } |
| 12 | |
| 13 | /// Works on unicode scalars, so an edit between "\r" and "\n" stays exact. |
| 14 | public func apply(to text: String) -> String { |
| 15 | let scalars = text.unicodeScalars |
| 16 | let start = String.Index(utf16Offset: range.lowerBound, in: text) |
| 17 | let end = String.Index(utf16Offset: range.upperBound, in: text) |
| 18 | return String(scalars[..<start]) + replacement + String(scalars[end...]) |
| 19 | } |
| 20 | } |
| 21 | |
| 22 | enum ReparseStrategy: Equatable { |
| 23 | /// One element reparsed in place; everything else reused. |
| 24 | case element |
| 25 | /// A run of top-level sections reparsed; the sections before and after reused. |
| 26 | case sections |
| 27 | case full |
| 28 | } |
| 29 | |
| 30 | extension OrgParser { |
| 31 | /// The tree for `edit.apply(to: oldText)`, reusing as much of `old` as stays valid. Always |
| 32 | /// equal to a full parse of the new text. |
| 33 | public static func reparse(_ old: OrgTree, oldText: String, edit: TextEdit, defaults: OrgSettings = .default) -> OrgTree { |
| 34 | reparseWithStrategy(old, oldText: oldText, edit: edit, defaults: defaults).tree |
| 35 | } |
| 36 | |
| 37 | static func reparseWithStrategy( |
| 38 | _ old: OrgTree, oldText: String, edit: TextEdit, defaults: OrgSettings = .default |
| 39 | ) -> (tree: OrgTree, strategy: ReparseStrategy) { |
| 40 | let newText = edit.apply(to: oldText) |
| 41 | let context = EditContext(oldText: oldText, newText: newText, edit: edit) |
| 42 | if context.touchesSettings() { return (parse(newText, defaults: defaults), .full) } |
| 43 | if let tree = context.reparseElement(old) { return (tree, .element) } |
| 44 | if let tree = context.reparseSections(old) { return (tree, .sections) } |
| 45 | return (parse(newText, defaults: defaults), .full) |
| 46 | } |
| 47 | } |
| 48 | |
| 49 | private struct EditContext { |
| 50 | let oldText: String |
| 51 | let newText: String |
| 52 | let edit: TextEdit |
| 53 | /// Change in UTF-16 length. |
| 54 | var delta: Int { edit.replacement.utf16.count - edit.range.count } |
| 55 | |
| 56 | static let settingsKeys: Set<String> = ["TODO", "SEQ_TODO", "TYP_TODO", "PRIORITIES"] |
| 57 | |
| 58 | /// Elements that parse the same in isolation as in place, as long as their lines keep |
| 59 | /// their classes. |
| 60 | static let leafKinds: Set<SyntaxKind> = [ |
| 61 | .paragraph, .heading, .tableRow, .planning, .clock, .keyword, .affiliatedKeyword, .horizontalRule, |
| 62 | .nodeProperty, .block, .comment, .fixedWidth, .footnoteDefinition, |
| 63 | ] |
| 64 | |
| 65 | /// File settings must be read again when an edited line is a settings keyword, or when it |
| 66 | /// moves a block or heading boundary in a file that has settings keywords: that can hide or |
| 67 | /// expose a keyword inside a block. |
| 68 | func touchesSettings() -> Bool { |
| 69 | let start = lineStart(oldText, edit.range.lowerBound) |
| 70 | let oldLines = classes(slice(oldText, start, lineEnd(oldText, edit.range.upperBound))) |
| 71 | let newLines = classes(slice(newText, start, lineEnd(newText, edit.range.lowerBound + edit.replacement.utf16.count))) |
| 72 | var movesBoundary = false |
| 73 | for line in oldLines + newLines { |
| 74 | switch line.cls { |
| 75 | case .keyword(let key) where Self.settingsKeys.contains(key): |
| 76 | return true |
| 77 | case .blockBegin, .blockEnd, .dynamicBegin, .dynamicEnd, .heading: |
| 78 | movesBoundary = true |
| 79 | default: |
| 80 | break |
| 81 | } |
| 82 | } |
| 83 | return movesBoundary && Self.settingsKeys.contains { newText.range(of: "#+\($0):", options: .caseInsensitive) != nil } |
| 84 | } |
| 85 | |
| 86 | // MARK: - Element |
| 87 | |
| 88 | func reparseElement(_ old: OrgTree) -> OrgTree? { |
| 89 | guard let leaf = leaf(in: old.root) else { return nil } |
| 90 | let start = lineStart(oldText, leaf.range.lowerBound) |
| 91 | // An element that starts mid-line (an item's paragraph) follows tokens that would absorb |
| 92 | // whitespace typed at its start. |
| 93 | guard start == leaf.range.lowerBound || edit.range.lowerBound > leaf.range.lowerBound else { return nil } |
| 94 | let oldEnd = leaf.range.upperBound |
| 95 | let newEnd = oldEnd + delta |
| 96 | let newLength = newText.utf16.count |
| 97 | guard newEnd > leaf.range.lowerBound, newEnd <= newLength else { return nil } |
| 98 | // The element must still end at a line end. |
| 99 | if newEnd < newLength { |
| 100 | let utf16 = newText.utf16 |
| 101 | guard utf16[utf16.index(utf16.startIndex, offsetBy: newEnd - 1)] == 0x0A else { return nil } |
| 102 | } |
| 103 | let oldClasses = classes(slice(oldText, start, oldEnd)) |
| 104 | let newClasses = classes(slice(newText, start, newEnd)) |
| 105 | guard oldClasses.count == newClasses.count, |
| 106 | zip(oldClasses, newClasses).allSatisfy({ $0.cls == $1.cls && $0.indent == $1.indent }) else { return nil } |
| 107 | let text = String(slice(newText, leaf.range.lowerBound, newEnd)) |
| 108 | guard let green = Parser.reparseElement(leaf.kind, text: text, settings: old.settings) else { return nil } |
| 109 | return OrgTree(green: replacing(leaf, with: green), settings: old.settings) |
| 110 | } |
| 111 | |
| 112 | /// The outermost leaf element containing the edit. |
| 113 | func leaf(in root: SyntaxNode) -> SyntaxNode? { |
| 114 | let range = edit.range |
| 115 | var current = root |
| 116 | while let child = current.children.first(where: { |
| 117 | $0.range.lowerBound <= range.lowerBound && range.lowerBound < $0.range.upperBound |
| 118 | && range.upperBound <= $0.range.upperBound |
| 119 | }) { |
| 120 | if Self.leafKinds.contains(child.kind) { return child } |
| 121 | current = child |
| 122 | } |
| 123 | return nil |
| 124 | } |
| 125 | |
| 126 | /// Copies the path from the root down to `target`, swapping in `green`. |
| 127 | func replacing(_ target: SyntaxNode, with green: GreenNode) -> GreenNode { |
| 128 | var node = target |
| 129 | var replacement = green |
| 130 | while let parent = node.parent { |
| 131 | var children = parent.green.children |
| 132 | let index = children.firstIndex { |
| 133 | if case .node(let child) = $0 { return child === node.green } |
| 134 | return false |
| 135 | }! |
| 136 | children[index] = .node(replacement) |
| 137 | replacement = GreenNode(kind: parent.kind, children: children) |
| 138 | node = parent |
| 139 | } |
| 140 | return replacement |
| 141 | } |
| 142 | |
| 143 | // MARK: - Sections |
| 144 | |
| 145 | /// Reparses from the top-level section before the edit up to the first later top-level |
| 146 | /// heading that still ends the reparsed run, then reuses the old sections from there. |
| 147 | /// Top-level sections parse independently: blocks and drawers never cross a heading. |
| 148 | func reparseSections(_ old: OrgTree) -> OrgTree? { |
| 149 | let units = old.root.children |
| 150 | guard !units.isEmpty else { return nil } |
| 151 | let anchor = max(0, lineStart(oldText, edit.range.lowerBound) - 1) |
| 152 | guard let startIndex = units.firstIndex(where: { $0.range.contains(anchor) }) else { return nil } |
| 153 | let start = units[startIndex].range.lowerBound |
| 154 | var endIndex = units.firstIndex { $0.range.lowerBound > edit.range.upperBound } ?? units.count |
| 155 | let newLength = newText.utf16.count |
| 156 | while true { |
| 157 | let newEnd = endIndex < units.count ? units[endIndex].range.lowerBound + delta : newLength |
| 158 | var parser = Parser(text: String(slice(newText, start, newEnd)), settings: old.settings) |
| 159 | let window = parser.run().green |
| 160 | if endIndex == units.count || endsRun(window, next: units[endIndex].green) { |
| 161 | let children = units[..<startIndex].map { GreenElement.node($0.green) } |
| 162 | + window.children |
| 163 | + units[endIndex...].map { GreenElement.node($0.green) } |
| 164 | return OrgTree(green: GreenNode(kind: .document, children: children), settings: old.settings) |
| 165 | } |
| 166 | endIndex += 1 |
| 167 | } |
| 168 | } |
| 169 | |
| 170 | /// Whether the heading of `next` ends the last section of the reparsed window. |
| 171 | func endsRun(_ window: GreenNode, next: GreenNode) -> Bool { |
| 172 | guard case .node(let last)? = window.children.last, last.kind == .section else { return true } |
| 173 | return headingLevel(next) <= headingLevel(last) |
| 174 | } |
| 175 | |
| 176 | func headingLevel(_ section: GreenNode) -> Int { |
| 177 | for case .node(let heading) in section.children where heading.kind == .heading { |
| 178 | for case .token(let token) in heading.children where token.kind == .stars { |
| 179 | return token.text.count |
| 180 | } |
| 181 | } |
| 182 | return 0 |
| 183 | } |
| 184 | |
| 185 | // MARK: - Text helpers |
| 186 | |
| 187 | func classes(_ text: Substring) -> [ClassifiedLine] { |
| 188 | splitRawLines(String(text)).map { classifyLine($0.content) } |
| 189 | } |
| 190 | |
| 191 | func slice(_ text: String, _ from: Int, _ to: Int) -> Substring { |
| 192 | text[String.Index(utf16Offset: from, in: text)..<String.Index(utf16Offset: to, in: text)] |
| 193 | } |
| 194 | |
| 195 | /// UTF-16 offset of the start of the line containing `offset`. |
| 196 | func lineStart(_ text: String, _ offset: Int) -> Int { |
| 197 | let utf16 = text.utf16 |
| 198 | var index = utf16.index(utf16.startIndex, offsetBy: offset) |
| 199 | while index > utf16.startIndex { |
| 200 | let before = utf16.index(before: index) |
| 201 | if utf16[before] == 0x0A { break } |
| 202 | index = before |
| 203 | } |
| 204 | return utf16.distance(from: utf16.startIndex, to: index) |
| 205 | } |
| 206 | |
| 207 | /// UTF-16 offset just past the newline ending the line containing `offset`. |
| 208 | func lineEnd(_ text: String, _ offset: Int) -> Int { |
| 209 | let utf16 = text.utf16 |
| 210 | var index = utf16.index(utf16.startIndex, offsetBy: offset) |
| 211 | while index < utf16.endIndex { |
| 212 | let current = utf16[index] |
| 213 | index = utf16.index(after: index) |
| 214 | if current == 0x0A { break } |
| 215 | } |
| 216 | return utf16.distance(from: utf16.startIndex, to: index) |
| 217 | } |
| 218 | } |
| 219 | |
| 220 | extension Parser { |
| 221 | /// One element of `kind` parsed from exactly `text`, or nil if `text` doesn't parse as a |
| 222 | /// single element of that kind. |
| 223 | static func reparseElement(_ kind: SyntaxKind, text: String, settings: OrgSettings) -> GreenNode? { |
| 224 | var parser = Parser(text: text, settings: settings) |
| 225 | let count = parser.lines.count |
| 226 | guard count > 0 else { return nil } |
| 227 | switch kind { |
| 228 | case .paragraph: |
| 229 | parser.paragraph(limit: count, floor: nil) |
| 230 | case .heading: |
| 231 | guard count == 1 else { return nil } |
| 232 | parser.headingLine(parser.lines[0]) |
| 233 | parser.i = 1 |
| 234 | case .tableRow: |
| 235 | guard count == 1 else { return nil } |
| 236 | parser.tableRow() |
| 237 | case .planning, .clock, .keyword, .affiliatedKeyword, .horizontalRule, .nodeProperty: |
| 238 | guard count == 1 else { return nil } |
| 239 | parser.single(kind) |
| 240 | case .block, .comment, .fixedWidth, .footnoteDefinition: |
| 241 | parser.element(limit: count, floor: nil) |
| 242 | default: |
| 243 | return nil |
| 244 | } |
| 245 | guard parser.i == count else { return nil } |
| 246 | let green = parser.builder.build() |
| 247 | return green.kind == kind ? green : nil |
| 248 | } |
| 249 | } |