Commit 5fe7e78dfd
Verified 路 cmc
Layout: unified 路 split
Sources/OrgCore/Parser/Lines.swift added +153
| @@ -0,0 +1,153 @@ | ||
| 1 | /// One line of source: its content and its terminator ("", "\n" or "\r\n"), both as slices of | |
| 2 | /// the original text. | |
| 3 | struct RawLine { | |
| 4 | let content: Substring | |
| 5 | let ending: Substring | |
| 6 | } | |
| 7 | ||
| 8 | /// Splits on "\n" without normalizing anything. Works on unicode scalars, because String | |
| 9 | /// treats "\r\n" as a single Character. | |
| 10 | func splitRawLines(_ text: String) -> [RawLine] { | |
| 11 | var lines: [RawLine] = [] | |
| 12 | let scalars = text.unicodeScalars | |
| 13 | var lineStart = scalars.startIndex | |
| 14 | var i = lineStart | |
| 15 | while i != scalars.endIndex { | |
| 16 | if scalars[i] == "\n" { | |
| 17 | var contentEnd = i | |
| 18 | if contentEnd > lineStart, scalars[scalars.index(before: i)] == "\r" { | |
| 19 | contentEnd = scalars.index(before: i) | |
| 20 | } | |
| 21 | let next = scalars.index(after: i) | |
| 22 | lines.append(RawLine(content: text[lineStart..<contentEnd], ending: text[contentEnd..<next])) | |
| 23 | lineStart = next | |
| 24 | i = next | |
| 25 | } else { | |
| 26 | i = scalars.index(after: i) | |
| 27 | } | |
| 28 | } | |
| 29 | if lineStart != scalars.endIndex { | |
| 30 | lines.append(RawLine(content: text[lineStart...], ending: "")) | |
| 31 | } | |
| 32 | return lines | |
| 33 | } | |
| 34 | ||
| 35 | enum LineClass: Equatable { | |
| 36 | case blank | |
| 37 | case heading(level: Int) | |
| 38 | case blockBegin(name: String) | |
| 39 | case blockEnd(name: String) | |
| 40 | case dynamicBegin | |
| 41 | case dynamicEnd | |
| 42 | case drawerBegin(name: String) | |
| 43 | case drawerEnd | |
| 44 | case keyword(key: String) | |
| 45 | case comment | |
| 46 | case fixedWidth | |
| 47 | case horizontalRule | |
| 48 | case tableRow | |
| 49 | case footnoteDefinition | |
| 50 | case clock | |
| 51 | case planning | |
| 52 | case listItem | |
| 53 | case plain | |
| 54 | } | |
| 55 | ||
| 56 | struct ClassifiedLine { | |
| 57 | let cls: LineClass | |
| 58 | /// Column of the first non-blank character, with tabs advancing to the next multiple of 8. | |
| 59 | let indent: Int | |
| 60 | } | |
| 61 | ||
| 62 | func classifyLine(_ line: Substring) -> ClassifiedLine { | |
| 63 | var column = 0 | |
| 64 | var rest = line | |
| 65 | while let c = rest.first, c == " " || c == "\t" { | |
| 66 | column = c == "\t" ? (column / 8 + 1) * 8 : column + 1 | |
| 67 | rest = rest.dropFirst() | |
| 68 | } | |
| 69 | if rest.isEmpty { return ClassifiedLine(cls: .blank, indent: column) } | |
| 70 | return ClassifiedLine(cls: lineClass(rest, columnZero: column == 0), indent: column) | |
| 71 | } | |
| 72 | ||
| 73 | private func lineClass(_ rest: Substring, columnZero: Bool) -> LineClass { | |
| 74 | let trimmed = rest.trimmingTrailingWhitespace | |
| 75 | ||
| 76 | if columnZero, rest.first == "*" { | |
| 77 | let stars = rest.prefix { $0 == "*" } | |
| 78 | let after = rest.dropFirst(stars.count) | |
| 79 | if after.isEmpty || after.first == " " || after.first == "\t" { | |
| 80 | return .heading(level: stars.count) | |
| 81 | } | |
| 82 | } | |
| 83 | ||
| 84 | if rest.hasPrefix("#+") { | |
| 85 | let lower = trimmed.lowercased() | |
| 86 | if lower.hasPrefix("#+begin_") { | |
| 87 | let name = lower.dropFirst(8).prefix { !$0.isWhitespace } | |
| 88 | if !name.isEmpty { return .blockBegin(name: String(name)) } | |
| 89 | } | |
| 90 | if lower.hasPrefix("#+end_") { | |
| 91 | let name = lower.dropFirst(6) | |
| 92 | if !name.isEmpty, !name.contains(where: \.isWhitespace) { return .blockEnd(name: String(name)) } | |
| 93 | } | |
| 94 | if lower.hasPrefix("#+begin:") { return .dynamicBegin } | |
| 95 | if lower == "#+end:" { return .dynamicEnd } | |
| 96 | if let colon = rest.firstIndex(of: ":") { | |
| 97 | let key = rest[rest.index(rest.startIndex, offsetBy: 2)..<colon] | |
| 98 | if !key.isEmpty, !key.contains(where: \.isWhitespace) { return .keyword(key: key.uppercased()) } | |
| 99 | } | |
| 100 | } | |
| 101 | ||
| 102 | if trimmed == "#" || rest.hasPrefix("# ") || rest.hasPrefix("#\t") { return .comment } | |
| 103 | ||
| 104 | if rest.first == ":" { | |
| 105 | if trimmed == ":" || rest.hasPrefix(": ") || rest.hasPrefix(":\t") { return .fixedWidth } | |
| 106 | if trimmed.uppercased() == ":END:" { return .drawerEnd } | |
| 107 | if trimmed.count >= 3, trimmed.last == ":" { | |
| 108 | let name = trimmed.dropFirst().dropLast() | |
| 109 | if name.allSatisfy({ $0.isLetter || $0.isNumber || $0 == "_" || $0 == "-" }) { | |
| 110 | return .drawerBegin(name: String(name)) | |
| 111 | } | |
| 112 | } | |
| 113 | } | |
| 114 | ||
| 115 | if rest.first == "|" { return .tableRow } | |
| 116 | if trimmed.count >= 5, trimmed.allSatisfy({ $0 == "-" }) { return .horizontalRule } | |
| 117 | ||
| 118 | if columnZero, rest.hasPrefix("[fn:"), let close = rest.firstIndex(of: "]"), | |
| 119 | close > rest.index(rest.startIndex, offsetBy: 4) { | |
| 120 | return .footnoteDefinition | |
| 121 | } | |
| 122 | ||
| 123 | if rest.hasPrefix("CLOCK:") { return .clock } | |
| 124 | if rest.hasPrefix("SCHEDULED:") || rest.hasPrefix("DEADLINE:") || rest.hasPrefix("CLOSED:") { return .planning } | |
| 125 | if isListBullet(rest, indented: !columnZero) { return .listItem } | |
| 126 | return .plain | |
| 127 | } | |
| 128 | ||
| 129 | /// `-`, `+`, `*` (indented only), `1.` or `1)`, followed by whitespace or end of line. | |
| 130 | /// Alphabetical bullets are off, as in org's default. | |
| 131 | private func isListBullet(_ rest: Substring, indented: Bool) -> Bool { | |
| 132 | guard let first = rest.first else { return false } | |
| 133 | let afterBullet: Substring | |
| 134 | if first == "-" || first == "+" || (first == "*" && indented) { | |
| 135 | afterBullet = rest.dropFirst() | |
| 136 | } else if first.isASCII, first.isNumber { | |
| 137 | let digits = rest.prefix { $0.isASCII && $0.isNumber } | |
| 138 | let tail = rest.dropFirst(digits.count) | |
| 139 | guard let separator = tail.first, separator == "." || separator == ")" else { return false } | |
| 140 | afterBullet = tail.dropFirst() | |
| 141 | } else { | |
| 142 | return false | |
| 143 | } | |
| 144 | return afterBullet.isEmpty || afterBullet.first == " " || afterBullet.first == "\t" | |
| 145 | } | |
| 146 | ||
| 147 | extension Substring { | |
| 148 | var trimmingTrailingWhitespace: Substring { | |
| 149 | var s = self | |
| 150 | while let last = s.last, last == " " || last == "\t" { s = s.dropLast() } | |
| 151 | return s | |
| 152 | } | |
| 153 | } | |
Tests/OrgCoreTests/LinesTests.swift added +64
| @@ -0,0 +1,64 @@ | ||
| 1 | import Testing | |
| 2 | @testable import OrgCore | |
| 3 | ||
| 4 | struct LinesTests { | |
| 5 | @Test func splitKeepsEveryEnding() { | |
| 6 | let lines = splitRawLines("a\r\nb\n\nc") | |
| 7 | #expect(lines.map { String($0.content) } == ["a", "b", "", "c"]) | |
| 8 | #expect(lines.map { String($0.ending) } == ["\r\n", "\n", "\n", ""]) | |
| 9 | #expect(splitRawLines("").isEmpty) | |
| 10 | #expect(splitRawLines("x\n").count == 1) | |
| 11 | } | |
| 12 | ||
| 13 | @Test func splitIsLossless() { | |
| 14 | let text = "\r\n\n a\r b\r\n馃榾\n" | |
| 15 | #expect(splitRawLines(text).map { String($0.content) + String($0.ending) }.joined() == text) | |
| 16 | } | |
| 17 | ||
| 18 | @Test(arguments: [ | |
| 19 | ("", LineClass.blank), | |
| 20 | (" \t", .blank), | |
| 21 | ("* a", .heading(level: 1)), | |
| 22 | ("*** ", .heading(level: 3)), | |
| 23 | ("*", .heading(level: 1)), | |
| 24 | ("*bold* text", .plain), | |
| 25 | (" * a", .listItem), | |
| 26 | ("#+BEGIN_SRC sh :results output", .blockBegin(name: "src")), | |
| 27 | ("#+end_src", .blockEnd(name: "src")), | |
| 28 | ("#+BEGIN: clocktable :scope file", .dynamicBegin), | |
| 29 | ("#+END:", .dynamicEnd), | |
| 30 | ("#+TITLE: x", .keyword(key: "TITLE")), | |
| 31 | ("#+tblfm: $2=$1", .keyword(key: "TBLFM")), | |
| 32 | ("# comment", .comment), | |
| 33 | ("#", .comment), | |
| 34 | ("#hashtag", .plain), | |
| 35 | (": fixed", .fixedWidth), | |
| 36 | (":", .fixedWidth), | |
| 37 | (":PROPERTIES:", .drawerBegin(name: "PROPERTIES")), | |
| 38 | (" :LOGBOOK:", .drawerBegin(name: "LOGBOOK")), | |
| 39 | (":END:", .drawerEnd), | |
| 40 | ("| a | b |", .tableRow), | |
| 41 | ("-----", .horizontalRule), | |
| 42 | ("----", .plain), | |
| 43 | ("[fn:1] note", .footnoteDefinition), | |
| 44 | ("CLOCK: [2026-10-04 Sun 10:00]", .clock), | |
| 45 | ("SCHEDULED: <2026-10-04 Sun>", .planning), | |
| 46 | ("- item", .listItem), | |
| 47 | ("+ item", .listItem), | |
| 48 | ("1. item", .listItem), | |
| 49 | ("2) item", .listItem), | |
| 50 | ("-", .listItem), | |
| 51 | ("-x", .plain), | |
| 52 | ("1.5 apples", .plain), | |
| 53 | ("plain text", .plain), | |
| 54 | ]) | |
| 55 | func classify(line: String, expected: LineClass) { | |
| 56 | #expect(classifyLine(line[...]).cls == expected) | |
| 57 | } | |
| 58 | ||
| 59 | @Test func indentCountsTabsToEight() { | |
| 60 | #expect(classifyLine("\t- a").indent == 8) | |
| 61 | #expect(classifyLine(" \t- a").indent == 8) | |
| 62 | #expect(classifyLine(" - a").indent == 3) | |
| 63 | } | |
| 64 | } | |