Commit 5fe7e78dfd
Verified 路 cmc
Layout: unified 路 split
Sources/OrgCore/Parser/Lines.swift added +153
| @@ -0,0 +1,153 @@ | |||
| 1 | /// One line of source: its content and its terminator ("", "\n" or "\r\n"), both as slices of | ||
| 2 | /// the original text. | ||
| 3 | struct RawLine { | ||
| 4 | let content: Substring | ||
| 5 | let ending: Substring | ||
| 6 | } | ||
| 7 | |||
| 8 | /// Splits on "\n" without normalizing anything. Works on unicode scalars, because String | ||
| 9 | /// treats "\r\n" as a single Character. | ||
| 10 | func splitRawLines(_ text: String) -> [RawLine] { | ||
| 11 | var lines: [RawLine] = [] | ||
| 12 | let scalars = text.unicodeScalars | ||
| 13 | var lineStart = scalars.startIndex | ||
| 14 | var i = lineStart | ||
| 15 | while i != scalars.endIndex { | ||
| 16 | if scalars[i] == "\n" { | ||
| 17 | var contentEnd = i | ||
| 18 | if contentEnd > lineStart, scalars[scalars.index(before: i)] == "\r" { | ||
| 19 | contentEnd = scalars.index(before: i) | ||
| 20 | } | ||
| 21 | let next = scalars.index(after: i) | ||
| 22 | lines.append(RawLine(content: text[lineStart..<contentEnd], ending: text[contentEnd..<next])) | ||
| 23 | lineStart = next | ||
| 24 | i = next | ||
| 25 | } else { | ||
| 26 | i = scalars.index(after: i) | ||
| 27 | } | ||
| 28 | } | ||
| 29 | if lineStart != scalars.endIndex { | ||
| 30 | lines.append(RawLine(content: text[lineStart...], ending: "")) | ||
| 31 | } | ||
| 32 | return lines | ||
| 33 | } | ||
| 34 | |||
| 35 | enum LineClass: Equatable { | ||
| 36 | case blank | ||
| 37 | case heading(level: Int) | ||
| 38 | case blockBegin(name: String) | ||
| 39 | case blockEnd(name: String) | ||
| 40 | case dynamicBegin | ||
| 41 | case dynamicEnd | ||
| 42 | case drawerBegin(name: String) | ||
| 43 | case drawerEnd | ||
| 44 | case keyword(key: String) | ||
| 45 | case comment | ||
| 46 | case fixedWidth | ||
| 47 | case horizontalRule | ||
| 48 | case tableRow | ||
| 49 | case footnoteDefinition | ||
| 50 | case clock | ||
| 51 | case planning | ||
| 52 | case listItem | ||
| 53 | case plain | ||
| 54 | } | ||
| 55 | |||
| 56 | struct ClassifiedLine { | ||
| 57 | let cls: LineClass | ||
| 58 | /// Column of the first non-blank character, with tabs advancing to the next multiple of 8. | ||
| 59 | let indent: Int | ||
| 60 | } | ||
| 61 | |||
| 62 | func classifyLine(_ line: Substring) -> ClassifiedLine { | ||
| 63 | var column = 0 | ||
| 64 | var rest = line | ||
| 65 | while let c = rest.first, c == " " || c == "\t" { | ||
| 66 | column = c == "\t" ? (column / 8 + 1) * 8 : column + 1 | ||
| 67 | rest = rest.dropFirst() | ||
| 68 | } | ||
| 69 | if rest.isEmpty { return ClassifiedLine(cls: .blank, indent: column) } | ||
| 70 | return ClassifiedLine(cls: lineClass(rest, columnZero: column == 0), indent: column) | ||
| 71 | } | ||
| 72 | |||
| 73 | private func lineClass(_ rest: Substring, columnZero: Bool) -> LineClass { | ||
| 74 | let trimmed = rest.trimmingTrailingWhitespace | ||
| 75 | |||
| 76 | if columnZero, rest.first == "*" { | ||
| 77 | let stars = rest.prefix { $0 == "*" } | ||
| 78 | let after = rest.dropFirst(stars.count) | ||
| 79 | if after.isEmpty || after.first == " " || after.first == "\t" { | ||
| 80 | return .heading(level: stars.count) | ||
| 81 | } | ||
| 82 | } | ||
| 83 | |||
| 84 | if rest.hasPrefix("#+") { | ||
| 85 | let lower = trimmed.lowercased() | ||
| 86 | if lower.hasPrefix("#+begin_") { | ||
| 87 | let name = lower.dropFirst(8).prefix { !$0.isWhitespace } | ||
| 88 | if !name.isEmpty { return .blockBegin(name: String(name)) } | ||
| 89 | } | ||
| 90 | if lower.hasPrefix("#+end_") { | ||
| 91 | let name = lower.dropFirst(6) | ||
| 92 | if !name.isEmpty, !name.contains(where: \.isWhitespace) { return .blockEnd(name: String(name)) } | ||
| 93 | } | ||
| 94 | if lower.hasPrefix("#+begin:") { return .dynamicBegin } | ||
| 95 | if lower == "#+end:" { return .dynamicEnd } | ||
| 96 | if let colon = rest.firstIndex(of: ":") { | ||
| 97 | let key = rest[rest.index(rest.startIndex, offsetBy: 2)..<colon] | ||
| 98 | if !key.isEmpty, !key.contains(where: \.isWhitespace) { return .keyword(key: key.uppercased()) } | ||
| 99 | } | ||
| 100 | } | ||
| 101 | |||
| 102 | if trimmed == "#" || rest.hasPrefix("# ") || rest.hasPrefix("#\t") { return .comment } | ||
| 103 | |||
| 104 | if rest.first == ":" { | ||
| 105 | if trimmed == ":" || rest.hasPrefix(": ") || rest.hasPrefix(":\t") { return .fixedWidth } | ||
| 106 | if trimmed.uppercased() == ":END:" { return .drawerEnd } | ||
| 107 | if trimmed.count >= 3, trimmed.last == ":" { | ||
| 108 | let name = trimmed.dropFirst().dropLast() | ||
| 109 | if name.allSatisfy({ $0.isLetter || $0.isNumber || $0 == "_" || $0 == "-" }) { | ||
| 110 | return .drawerBegin(name: String(name)) | ||
| 111 | } | ||
| 112 | } | ||
| 113 | } | ||
| 114 | |||
| 115 | if rest.first == "|" { return .tableRow } | ||
| 116 | if trimmed.count >= 5, trimmed.allSatisfy({ $0 == "-" }) { return .horizontalRule } | ||
| 117 | |||
| 118 | if columnZero, rest.hasPrefix("[fn:"), let close = rest.firstIndex(of: "]"), | ||
| 119 | close > rest.index(rest.startIndex, offsetBy: 4) { | ||
| 120 | return .footnoteDefinition | ||
| 121 | } | ||
| 122 | |||
| 123 | if rest.hasPrefix("CLOCK:") { return .clock } | ||
| 124 | if rest.hasPrefix("SCHEDULED:") || rest.hasPrefix("DEADLINE:") || rest.hasPrefix("CLOSED:") { return .planning } | ||
| 125 | if isListBullet(rest, indented: !columnZero) { return .listItem } | ||
| 126 | return .plain | ||
| 127 | } | ||
| 128 | |||
| 129 | /// `-`, `+`, `*` (indented only), `1.` or `1)`, followed by whitespace or end of line. | ||
| 130 | /// Alphabetical bullets are off, as in org's default. | ||
| 131 | private func isListBullet(_ rest: Substring, indented: Bool) -> Bool { | ||
| 132 | guard let first = rest.first else { return false } | ||
| 133 | let afterBullet: Substring | ||
| 134 | if first == "-" || first == "+" || (first == "*" && indented) { | ||
| 135 | afterBullet = rest.dropFirst() | ||
| 136 | } else if first.isASCII, first.isNumber { | ||
| 137 | let digits = rest.prefix { $0.isASCII && $0.isNumber } | ||
| 138 | let tail = rest.dropFirst(digits.count) | ||
| 139 | guard let separator = tail.first, separator == "." || separator == ")" else { return false } | ||
| 140 | afterBullet = tail.dropFirst() | ||
| 141 | } else { | ||
| 142 | return false | ||
| 143 | } | ||
| 144 | return afterBullet.isEmpty || afterBullet.first == " " || afterBullet.first == "\t" | ||
| 145 | } | ||
| 146 | |||
| 147 | extension Substring { | ||
| 148 | var trimmingTrailingWhitespace: Substring { | ||
| 149 | var s = self | ||
| 150 | while let last = s.last, last == " " || last == "\t" { s = s.dropLast() } | ||
| 151 | return s | ||
| 152 | } | ||
| 153 | } | ||
Tests/OrgCoreTests/LinesTests.swift added +64
| @@ -0,0 +1,64 @@ | |||
| 1 | import Testing | ||
| 2 | @testable import OrgCore | ||
| 3 | |||
| 4 | struct LinesTests { | ||
| 5 | @Test func splitKeepsEveryEnding() { | ||
| 6 | let lines = splitRawLines("a\r\nb\n\nc") | ||
| 7 | #expect(lines.map { String($0.content) } == ["a", "b", "", "c"]) | ||
| 8 | #expect(lines.map { String($0.ending) } == ["\r\n", "\n", "\n", ""]) | ||
| 9 | #expect(splitRawLines("").isEmpty) | ||
| 10 | #expect(splitRawLines("x\n").count == 1) | ||
| 11 | } | ||
| 12 | |||
| 13 | @Test func splitIsLossless() { | ||
| 14 | let text = "\r\n\n a\r b\r\n馃榾\n" | ||
| 15 | #expect(splitRawLines(text).map { String($0.content) + String($0.ending) }.joined() == text) | ||
| 16 | } | ||
| 17 | |||
| 18 | @Test(arguments: [ | ||
| 19 | ("", LineClass.blank), | ||
| 20 | (" \t", .blank), | ||
| 21 | ("* a", .heading(level: 1)), | ||
| 22 | ("*** ", .heading(level: 3)), | ||
| 23 | ("*", .heading(level: 1)), | ||
| 24 | ("*bold* text", .plain), | ||
| 25 | (" * a", .listItem), | ||
| 26 | ("#+BEGIN_SRC sh :results output", .blockBegin(name: "src")), | ||
| 27 | ("#+end_src", .blockEnd(name: "src")), | ||
| 28 | ("#+BEGIN: clocktable :scope file", .dynamicBegin), | ||
| 29 | ("#+END:", .dynamicEnd), | ||
| 30 | ("#+TITLE: x", .keyword(key: "TITLE")), | ||
| 31 | ("#+tblfm: $2=$1", .keyword(key: "TBLFM")), | ||
| 32 | ("# comment", .comment), | ||
| 33 | ("#", .comment), | ||
| 34 | ("#hashtag", .plain), | ||
| 35 | (": fixed", .fixedWidth), | ||
| 36 | (":", .fixedWidth), | ||
| 37 | (":PROPERTIES:", .drawerBegin(name: "PROPERTIES")), | ||
| 38 | (" :LOGBOOK:", .drawerBegin(name: "LOGBOOK")), | ||
| 39 | (":END:", .drawerEnd), | ||
| 40 | ("| a | b |", .tableRow), | ||
| 41 | ("-----", .horizontalRule), | ||
| 42 | ("----", .plain), | ||
| 43 | ("[fn:1] note", .footnoteDefinition), | ||
| 44 | ("CLOCK: [2026-10-04 Sun 10:00]", .clock), | ||
| 45 | ("SCHEDULED: <2026-10-04 Sun>", .planning), | ||
| 46 | ("- item", .listItem), | ||
| 47 | ("+ item", .listItem), | ||
| 48 | ("1. item", .listItem), | ||
| 49 | ("2) item", .listItem), | ||
| 50 | ("-", .listItem), | ||
| 51 | ("-x", .plain), | ||
| 52 | ("1.5 apples", .plain), | ||
| 53 | ("plain text", .plain), | ||
| 54 | ]) | ||
| 55 | func classify(line: String, expected: LineClass) { | ||
| 56 | #expect(classifyLine(line[...]).cls == expected) | ||
| 57 | } | ||
| 58 | |||
| 59 | @Test func indentCountsTabsToEight() { | ||
| 60 | #expect(classifyLine("\t- a").indent == 8) | ||
| 61 | #expect(classifyLine(" \t- a").indent == 8) | ||
| 62 | #expect(classifyLine(" - a").indent == 3) | ||
| 63 | } | ||
| 64 | } | ||