Sources/OrgIndex/IndexStore.swift
315 lines · 14379 bytes
1import Foundation
2import GRDB
3
4/// What the index knows about a file without reading it.
5public struct FileState: Sendable, Equatable {
6 public let kind: FileKind
7 public let size: Int
8 public let mtime: Double
9 public let hash: String
10 public let settingsVersion: Int
11}
12
13/// A heading as found by a query. `contentHash` is the hash of the text the offsets refer to,
14/// so a caller can tell whether they still apply to an open buffer.
15public struct HeadingLocation: Sendable, Equatable {
16 public let path: String
17 public let ordinal: Int
18 public let title: String
19 public let start: Int
20 public let contentHash: String
21}
22
23/// One reconciliation's worth of changes, applied in a single transaction.
24public struct IndexChange: Sendable {
25 public var records: [FileRecord] = []
26 /// Unchanged content with a new modification time.
27 public var touches: [(path: String, mtime: Double)] = []
28 /// Renamed files whose content didn't change.
29 public var moves: [(from: String, to: String, mtime: Double)] = []
30 public var removals: [String] = []
31
32 public init() {}
33
34 public var isEmpty: Bool { records.isEmpty && touches.isEmpty && moves.isEmpty && removals.isEmpty }
35}
36
37/// The SQLite index. A cache: deleting it loses nothing that the files don't hold.
38public final class IndexStore: Sendable {
39 let database: DatabaseQueue
40
41 /// `path` nil opens an in-memory index.
42 public init(path: String? = nil) throws {
43 database = try path.map { try DatabaseQueue(path: $0) } ?? DatabaseQueue()
44 try Self.migrator.migrate(database)
45 }
46
47 static var migrator: DatabaseMigrator {
48 var migrator = DatabaseMigrator()
49 migrator.registerMigration("v1") { db in
50 try db.execute(sql: """
51 CREATE TABLE files (
52 id INTEGER PRIMARY KEY,
53 path TEXT NOT NULL UNIQUE,
54 root TEXT NOT NULL,
55 kind TEXT NOT NULL,
56 size INTEGER NOT NULL,
57 mtime REAL NOT NULL,
58 hash TEXT NOT NULL,
59 settings_version INTEGER NOT NULL,
60 parsed_at REAL NOT NULL
61 );
62 CREATE INDEX files_root ON files(root);
63 CREATE TABLE headings (
64 id INTEGER PRIMARY KEY,
65 file_id INTEGER NOT NULL REFERENCES files(id) ON DELETE CASCADE,
66 ordinal INTEGER NOT NULL,
67 parent_ordinal INTEGER,
68 start_offset INTEGER NOT NULL,
69 end_offset INTEGER NOT NULL,
70 level INTEGER NOT NULL,
71 todo TEXT,
72 is_done INTEGER NOT NULL,
73 priority TEXT,
74 title TEXT NOT NULL,
75 outline_path TEXT NOT NULL,
76 org_id TEXT,
77 archived INTEGER NOT NULL
78 );
79 CREATE INDEX headings_file ON headings(file_id);
80 CREATE INDEX headings_org_id ON headings(org_id);
81 CREATE TABLE tags (
82 heading_id INTEGER NOT NULL REFERENCES headings(id) ON DELETE CASCADE,
83 tag TEXT NOT NULL,
84 inherited INTEGER NOT NULL
85 );
86 CREATE INDEX tags_heading ON tags(heading_id);
87 CREATE TABLE properties (
88 heading_id INTEGER NOT NULL REFERENCES headings(id) ON DELETE CASCADE,
89 key TEXT NOT NULL,
90 value TEXT NOT NULL,
91 inherited INTEGER NOT NULL
92 );
93 CREATE INDEX properties_heading ON properties(heading_id);
94 CREATE TABLE timestamps (
95 heading_id INTEGER NOT NULL REFERENCES headings(id) ON DELETE CASCADE,
96 kind TEXT NOT NULL,
97 start_at TEXT NOT NULL,
98 end_at TEXT,
99 repeater TEXT,
100 warning TEXT
101 );
102 CREATE INDEX timestamps_heading ON timestamps(heading_id);
103 CREATE TABLE clocks (
104 heading_id INTEGER NOT NULL REFERENCES headings(id) ON DELETE CASCADE,
105 start_at TEXT NOT NULL,
106 end_at TEXT,
107 minutes INTEGER
108 );
109 CREATE INDEX clocks_heading ON clocks(heading_id);
110 CREATE TABLE links (
111 heading_id INTEGER NOT NULL REFERENCES headings(id) ON DELETE CASCADE,
112 type TEXT NOT NULL,
113 target TEXT NOT NULL
114 );
115 CREATE INDEX links_heading ON links(heading_id);
116 CREATE VIRTUAL TABLE headings_fts USING fts5(title, body, tokenize = 'unicode61 remove_diacritics 2');
117 CREATE TRIGGER headings_fts_delete AFTER DELETE ON headings BEGIN
118 DELETE FROM headings_fts WHERE rowid = old.id;
119 END;
120 """)
121 }
122 return migrator
123 }
124
125 // MARK: - Writing
126
127 public func write(_ record: FileRecord) throws {
128 var change = IndexChange()
129 change.records = [record]
130 try apply(change)
131 }
132
133 public func apply(_ change: IndexChange) throws {
134 guard !change.isEmpty else { return }
135 try database.write { db in
136 for path in change.removals {
137 try db.execute(sql: "DELETE FROM files WHERE path = ?", arguments: [path])
138 }
139 for move in change.moves {
140 try db.execute(sql: "UPDATE files SET path = ?, mtime = ? WHERE path = ?", arguments: [move.to, move.mtime, move.from])
141 }
142 for touch in change.touches {
143 try db.execute(sql: "UPDATE files SET mtime = ? WHERE path = ?", arguments: [touch.mtime, touch.path])
144 }
145 for record in change.records {
146 try insert(record, db)
147 }
148 }
149 }
150
151 private func insert(_ record: FileRecord, _ db: Database) throws {
152 try db.execute(sql: "DELETE FROM files WHERE path = ?", arguments: [record.path])
153 try db.execute(
154 sql: """
155 INSERT INTO files (path, root, kind, size, mtime, hash, settings_version, parsed_at)
156 VALUES (?, ?, ?, ?, ?, ?, ?, ?)
157 """,
158 arguments: [record.path, record.root, record.kind.rawValue, record.size, record.mtime, record.hash,
159 record.settingsVersion, Date().timeIntervalSince1970]
160 )
161 let fileID = db.lastInsertedRowID
162 for heading in record.headings {
163 try db.execute(
164 sql: """
165 INSERT INTO headings (file_id, ordinal, parent_ordinal, start_offset, end_offset, level, todo,
166 is_done, priority, title, outline_path, org_id, archived)
167 VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
168 """,
169 arguments: [fileID, heading.ordinal, heading.parent, heading.start, heading.end, heading.level, heading.todo,
170 heading.isDone, heading.priority, heading.title, heading.outlinePath.joined(separator: "\u{1F}"),
171 heading.orgID, heading.archived]
172 )
173 let id = db.lastInsertedRowID
174 try db.execute(sql: "INSERT INTO headings_fts (rowid, title, body) VALUES (?, ?, ?)", arguments: [id, heading.title, heading.body])
175 for tag in heading.tags {
176 try db.execute(sql: "INSERT INTO tags VALUES (?, ?, ?)", arguments: [id, tag.name, tag.inherited])
177 }
178 for property in heading.properties {
179 try db.execute(sql: "INSERT INTO properties VALUES (?, ?, ?, ?)", arguments: [id, property.key, property.value, property.inherited])
180 }
181 for stamp in heading.timestamps {
182 try db.execute(
183 sql: "INSERT INTO timestamps VALUES (?, ?, ?, ?, ?, ?)",
184 arguments: [id, stamp.kind.rawValue, stamp.start, stamp.end, stamp.repeater, stamp.warning]
185 )
186 }
187 for clock in heading.clocks {
188 try db.execute(sql: "INSERT INTO clocks VALUES (?, ?, ?, ?)", arguments: [id, clock.start, clock.end, clock.minutes])
189 }
190 for link in heading.links {
191 try db.execute(sql: "INSERT INTO links VALUES (?, ?, ?)", arguments: [id, link.type, link.target])
192 }
193 }
194 }
195
196 // MARK: - Reading
197
198 /// Indexed files under `root`, by path.
199 public func fileStates(root: String) throws -> [String: FileState] {
200 try database.read { db in
201 let rows = try Row.fetchAll(db, sql: "SELECT path, kind, size, mtime, hash, settings_version FROM files WHERE root = ?", arguments: [root])
202 var states: [String: FileState] = [:]
203 for row in rows {
204 states[row["path"]] = FileState(
205 kind: FileKind(rawValue: row["kind"]) ?? .org,
206 size: row["size"], mtime: row["mtime"], hash: row["hash"], settingsVersion: row["settings_version"]
207 )
208 }
209 return states
210 }
211 }
212
213 public func files() throws -> [(path: String, kind: FileKind)] {
214 try database.read { db in
215 try Row.fetchAll(db, sql: "SELECT path, kind FROM files ORDER BY path").map {
216 (path: $0["path"], kind: FileKind(rawValue: $0["kind"]) ?? .org)
217 }
218 }
219 }
220
221 // MARK: - Queries
222 //
223 // `overlay` holds records for open documents with unsaved edits, keyed by path. Their rows
224 // replace that file's indexed rows in every query.
225
226 /// Headings whose title or body contain every word of `query` as a word prefix.
227 public func search(_ query: String, overlay: [String: FileRecord] = [:], limit: Int = 50) throws -> [HeadingLocation] {
228 let terms = query.split(whereSeparator: \.isWhitespace).map(String.init)
229 guard !terms.isEmpty else { return [] }
230 let match = terms.map { "\"" + $0.replacingOccurrences(of: "\"", with: "\"\"") + "\"*" }.joined(separator: " ")
231 let indexed = try database.read { db in
232 try Row.fetchAll(
233 db,
234 sql: """
235 SELECT f.path, f.hash, h.ordinal, h.title, h.start_offset
236 FROM headings_fts
237 JOIN headings h ON h.id = headings_fts.rowid
238 JOIN files f ON f.id = h.file_id
239 WHERE headings_fts MATCH ?
240 ORDER BY bm25(headings_fts)
241 LIMIT ?
242 """,
243 arguments: [match, limit + overlay.count * 10]
244 ).map(location)
245 }
246 // Unsaved buffers get the same word-prefix matching as the full-text index.
247 let prefixes = terms.flatMap(Self.words)
248 let live = overlay.values.sorted { $0.path < $1.path }.flatMap { record in
249 record.headings.filter { heading in
250 let words = Self.words(heading.title + "\n" + heading.body)
251 return prefixes.allSatisfy { prefix in words.contains { $0.hasPrefix(prefix) } }
252 }.map { location(record, $0) }
253 }
254 return Array((live + indexed.filter { overlay[$0.path] == nil }).prefix(limit))
255 }
256
257 /// Headings with `:ID: id`. More than one means the ID is duplicated.
258 public func headings(withID id: String, overlay: [String: FileRecord] = [:]) throws -> [HeadingLocation] {
259 let indexed = try database.read { db in
260 try Row.fetchAll(
261 db,
262 sql: """
263 SELECT f.path, f.hash, h.ordinal, h.title, h.start_offset
264 FROM headings h JOIN files f ON f.id = h.file_id
265 WHERE h.org_id = ? ORDER BY f.path, h.ordinal
266 """,
267 arguments: [id]
268 ).map(location)
269 }
270 let live = overlay.values.sorted { $0.path < $1.path }.flatMap { record in
271 record.headings.filter { $0.orgID == id }.map { location(record, $0) }
272 }
273 return indexed.filter { overlay[$0.path] == nil } + live
274 }
275
276 /// IDs used by more than one heading.
277 public func duplicateIDs(overlay: [String: FileRecord] = [:]) throws -> [String: [HeadingLocation]] {
278 let indexed = try database.read { db in
279 try Row.fetchAll(
280 db,
281 sql: """
282 SELECT h.org_id, f.path, f.hash, h.ordinal, h.title, h.start_offset
283 FROM headings h JOIN files f ON f.id = h.file_id
284 WHERE h.org_id IN (SELECT org_id FROM headings WHERE org_id IS NOT NULL GROUP BY org_id HAVING count(*) > 1)
285 OR (h.org_id IS NOT NULL AND ? > 0)
286 ORDER BY f.path, h.ordinal
287 """,
288 arguments: [overlay.count]
289 ).map { (id: $0["org_id"] as String, location: location($0)) }
290 }
291 var byID: [String: [HeadingLocation]] = [:]
292 for row in indexed where overlay[row.location.path] == nil {
293 byID[row.id, default: []].append(row.location)
294 }
295 for record in overlay.values.sorted(by: { $0.path < $1.path }) {
296 for heading in record.headings {
297 if let id = heading.orgID { byID[id, default: []].append(location(record, heading)) }
298 }
299 }
300 return byID.filter { $0.value.count > 1 }
301 }
302
303 /// Lower-cased runs of letters and digits, as the unicode61 tokenizer splits them.
304 static func words(_ text: String) -> [String] {
305 text.lowercased().split { !$0.isLetter && !$0.isNumber }.map(String.init)
306 }
307
308 private func location(_ row: Row) -> HeadingLocation {
309 HeadingLocation(path: row["path"], ordinal: row["ordinal"], title: row["title"], start: row["start_offset"], contentHash: row["hash"])
310 }
311
312 private func location(_ record: FileRecord, _ heading: HeadingRecord) -> HeadingLocation {
313 HeadingLocation(path: record.path, ordinal: heading.ordinal, title: heading.title, start: heading.start, contentHash: record.hash)
314 }
315}