Sources/OrgCore/Semantic/TextStats.swift
112 lines · 4340 bytes
1/// Position and counts for the modeline, as Emacs computes them in an org buffer.
2public enum TextStats {
3 /// The line (from 1) and `current-column` (from 0, tabs to multiples of 8, wide
4 /// characters two columns, control characters as `^X` or `\200`) at a UTF-16 offset.
5 public static func position(in text: String, at offset: Int) -> (line: Int, column: Int) {
6 let utf16 = text.utf16
7 let end = utf16.index(utf16.startIndex, offsetBy: min(max(0, offset), utf16.count))
8 var line = 1
9 var lineStart = utf16.startIndex
10 var index = utf16.startIndex
11 while index < end {
12 if utf16[index] == 10 {
13 line += 1
14 lineStart = utf16.index(after: index)
15 }
16 index = utf16.index(after: index)
17 }
18 var column = 0
19 for character in text[lineStart..<end] {
20 let v = character.unicodeScalars.first!.value
21 if character == "\t" {
22 column = (column / 8 + 1) * 8
23 } else if v < 0x20 || v == 0x7F {
24 column += 2 // ^X
25 } else if (0x80...0x9F).contains(v) {
26 column += 4 // \200
27 } else {
28 column += displayWidth(of: character)
29 }
30 }
31 return (line, column)
32 }
33
34 /// `count-words`: runs of characters with word syntax in org-mode (`WordSyntax.swift`),
35 /// split where `forward-word` sees a boundary between scripts.
36 public static func words(in text: some StringProtocol) -> Int {
37 var count = 0
38 var previous: Unicode.Scalar?
39 for scalar in text.unicodeScalars {
40 if isWord(scalar) {
41 if let previous, !boundary(previous, scalar) {} else { count += 1 }
42 previous = scalar
43 } else {
44 previous = nil
45 }
46 }
47 return count
48 }
49
50 /// `word_boundary_p` with the default `word-combining-categories` and
51 /// `word-separating-categories`.
52 static func boundary(_ a: Unicode.Scalar, _ b: Unicode.Scalar) -> Bool {
53 if script(a) == script(b) { return isHiragana(a) && isKatakana(b) }
54 if isCombining(a) || isCombining(b) { return false }
55 return !(isHan(a) && (isHiragana(b) || isKatakana(b)))
56 }
57
58 static func script(_ scalar: Unicode.Scalar) -> UInt16 {
59 let v = scalar.value
60 var low = 0
61 var high = scripts.count
62 while low < high {
63 let mid = (low + high) / 2
64 if scripts[mid].start <= v { low = mid + 1 } else { high = mid }
65 }
66 return scripts[low - 1].script
67 }
68
69 static func isCombining(_ scalar: Unicode.Scalar) -> Bool {
70 switch scalar.properties.generalCategory {
71 case .nonspacingMark, .spacingMark, .enclosingMark: true
72 default: false
73 }
74 }
75
76 static func isHan(_ scalar: Unicode.Scalar) -> Bool {
77 let v = scalar.value
78 return (0x3400...0x4DBF).contains(v) || (0x4E00...0x9FFF).contains(v) || (0xF900...0xFAFF).contains(v) || (0x20000...0x3FFFF).contains(v)
79 }
80
81 static func isHiragana(_ scalar: Unicode.Scalar) -> Bool { (0x3040...0x309F).contains(scalar.value) }
82
83 static func isKatakana(_ scalar: Unicode.Scalar) -> Bool {
84 let v = scalar.value
85 return (0x30A0...0x30FF).contains(v) || (0x31F0...0x31FF).contains(v) || (0xFF65...0xFF9F).contains(v)
86 }
87
88 static func isWord(_ scalar: Unicode.Scalar) -> Bool {
89 let v = scalar.value
90 var low = 0
91 var high = nonWord.count
92 while low < high {
93 let mid = (low + high) / 2
94 if nonWord[mid].upperBound < v { low = mid + 1 } else { high = mid }
95 }
96 return low == nonWord.count || !nonWord[low].contains(v)
97 }
98
99 /// `count-words-region`: lines (a partial last line counts), words and characters.
100 public static func region(_ text: some StringProtocol) -> (lines: Int, words: Int, characters: Int) {
101 var lines = 0
102 var characters = 0
103 var last: Unicode.Scalar?
104 for scalar in text.unicodeScalars {
105 characters += 1
106 if scalar == "\n" { lines += 1 }
107 last = scalar
108 }
109 if let last, last != "\n" { lines += 1 }
110 return (lines, words(in: text), characters)
111 }
112}