koan_core/db/connection.rs
1use std::cell::RefCell;
2use std::collections::HashMap;
3use std::path::Path;
4use std::rc::Rc;
5
6use rusqlite::functions::FunctionFlags;
7use rusqlite::{Connection, OpenFlags};
8use thiserror::Error;
9
10use super::schema;
11use crate::config;
12
13#[derive(Debug, Error)]
14pub enum DbError {
15 #[error("sqlite error: {0}")]
16 Sqlite(#[from] rusqlite::Error),
17 #[error("io error: {0}")]
18 Io(#[from] std::io::Error),
19 /// A bulk delete looked like a mount failure rather than an intentional
20 /// deletion, so it was refused. The library is untouched.
21 #[error("refused unsafe bulk delete: {0}")]
22 UnsafeBulkDelete(String),
23}
24
25/// Wrapper around a SQLite connection with koan's schema applied.
26pub struct Database {
27 pub conn: Connection,
28}
29
30impl Database {
31 /// Open (or create) a database at the given path, applying the schema and
32 /// pending migrations.
33 ///
34 /// This is the once-per-process path: it creates the parent directory,
35 /// tightens file permissions, checkpoints the WAL and runs the ~30-statement
36 /// DDL batch. Anything opening a connection per request wants
37 /// [`Database::open_existing`] instead.
38 pub fn open(path: &Path) -> Result<Self, DbError> {
39 if let Some(parent) = path.parent() {
40 std::fs::create_dir_all(parent)?;
41 }
42
43 let conn = Connection::open(path)?;
44
45 #[cfg(unix)]
46 {
47 use std::os::unix::fs::PermissionsExt;
48 let _ = std::fs::set_permissions(path, std::fs::Permissions::from_mode(0o600));
49 }
50
51 configure(&conn)?;
52
53 // Attempt a passive WAL checkpoint on open. This is non-blocking — it
54 // moves WAL pages back to the main DB file only if no readers/writers
55 // are active, preventing unbounded WAL growth across sessions.
56 let _ = conn.execute_batch("PRAGMA wal_checkpoint(PASSIVE)");
57
58 schema::create_tables(&conn)?;
59
60 // The planner picks between indexes by guessing how many rows each one
61 // will yield, and with no statistics it guesses the same number for all
62 // of them. That is how a partial index on the column a query filters by
63 // loses to an index that merely happens to supply the ORDER BY. Cheap
64 // after the first run, and a no-op when nothing has moved.
65 let _ = conn.execute_batch("PRAGMA optimize");
66
67 Ok(Self { conn })
68 }
69
70 /// Apply this build's schema and migrations to a snapshot of the database
71 /// at `path`, to learn whether it would open. The original is only read,
72 /// so this is safe beside a server that has it open: it is how a deploy
73 /// finds a migration that fails on the real library before replacing the
74 /// running version.
75 pub fn check_upgrade(path: &Path) -> Result<(), DbError> {
76 let snapshot = std::env::temp_dir().join(format!("koan-check-{}.db", std::process::id()));
77 let _ = std::fs::remove_file(&snapshot);
78 let result = (|| {
79 let source = Connection::open_with_flags(path, OpenFlags::SQLITE_OPEN_READ_ONLY)?;
80 source.execute("VACUUM INTO ?1", [snapshot.to_string_lossy()])?;
81 drop(source);
82 Self::open(&snapshot).map(drop)
83 })();
84 for suffix in ["", "-wal", "-shm"] {
85 let mut file = snapshot.clone().into_os_string();
86 file.push(suffix);
87 let _ = std::fs::remove_file(file);
88 }
89 result
90 }
91
92 /// Open an additional connection to a database whose schema is already
93 /// applied — pragmas only, no DDL, no checkpoint, no permission syscall.
94 ///
95 /// Callers are responsible for having run [`Database::open`] at least once
96 /// against the same path first.
97 pub fn open_existing(path: &Path) -> Result<Self, DbError> {
98 let conn = Connection::open(path)?;
99 configure(&conn)?;
100 Ok(Self { conn })
101 }
102
103 /// Open the default database at the standard data directory.
104 pub fn open_default() -> Result<Self, DbError> {
105 Self::open(&config::db_path())
106 }
107
108 /// Refresh the planner's statistics.
109 ///
110 /// Worth calling wherever the library changes size in bulk — a scan, a
111 /// remote sync — because the statistics gathered when the process started
112 /// describe a library that no longer exists, and the planner will keep
113 /// choosing for it. A no-op when nothing has moved far enough to matter.
114 pub fn optimize(&self) {
115 if let Err(e) = self.conn.execute_batch("PRAGMA optimize") {
116 log::debug!("PRAGMA optimize failed: {e}");
117 }
118 }
119}
120
121/// Connection-scoped pragmas. Every connection needs these; none of them touch
122/// the file on disk, so they are cheap enough to repeat per connection.
123fn configure(conn: &Connection) -> Result<(), DbError> {
124 // WAL mode for concurrent reads + single writer.
125 conn.pragma_update(None, "journal_mode", "wal")?;
126 conn.pragma_update(None, "foreign_keys", "on")?;
127 // Long enough to outlast a scan chunk: a writer that gives up mid-scan
128 // silently loses favourites, queue state and play counts.
129 conn.pragma_update(None, "busy_timeout", 30000)?;
130 // Slightly faster at the cost of durability on power loss (acceptable for a media DB).
131 conn.pragma_update(None, "synchronous", "normal")?;
132 // Map the whole library. A library this size fits well inside this, so
133 // reads become dereferences into a mapped region rather than syscalls —
134 // which is the useful sense in which a database can be "in memory". The
135 // page cache was already holding it; this stops copying it out per read.
136 conn.pragma_update(None, "mmap_size", 268_435_456i64)?;
137 // 32 MiB of pages, per connection. Negative means KiB rather than pages,
138 // so the figure does not change meaning with the page size.
139 conn.pragma_update(None, "cache_size", -32_000i64)?;
140 // Sorts and intermediate tables in memory. FTS and the ORDER BYs behind
141 // every library listing make temporary tables constantly.
142 conn.pragma_update(None, "temp_store", "memory")?;
143 // How many rows `PRAGMA optimize` samples per index. Bounded, so gathering
144 // statistics stays a fraction of a second on a library of any size; the
145 // planner needs the shape of the distribution, not an exact count.
146 conn.pragma_update(None, "analysis_limit", 400i64)?;
147 // Here rather than with the schema: `open_existing` skips the DDL, and a
148 // connection without this collation fails every ORDER BY that uses it.
149 register_library_collation(conn)?;
150 register_shuffle_function(conn)?;
151 Ok(())
152}
153
154/// `koan_shuffle(id, seed)` — a stable pseudo-random ordering key.
155///
156/// A shuffled listing that is read a page at a time cannot shuffle in the
157/// client: page two would be drawn from a different shuffle than page one, and
158/// records would repeat or vanish as you scrolled. Ordering by a hash of the
159/// row id and a seed gives one order that every page of the same seed agrees
160/// on, and a new seed gives a different one.
161///
162/// Registered next to the collation, and for the same reason: a connection
163/// without it fails the query outright rather than sorting some other way.
164pub(crate) fn register_shuffle_function(conn: &Connection) -> rusqlite::Result<()> {
165 conn.create_scalar_function(
166 "koan_shuffle",
167 2,
168 FunctionFlags::SQLITE_UTF8 | FunctionFlags::SQLITE_DETERMINISTIC,
169 |ctx| {
170 let id = ctx.get::<i64>(0)? as u64;
171 let seed = ctx.get::<i64>(1)? as u64;
172 Ok(splitmix64(id ^ splitmix64(seed)) as i64)
173 },
174 )
175}
176
177/// SplitMix64. Cheap, and it scatters consecutive ids — which matters, because
178/// a library's ids are consecutive in the order it was scanned.
179fn splitmix64(x: u64) -> u64 {
180 let mut z = x.wrapping_add(0x9E37_79B9_7F4A_7C15);
181 z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
182 z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
183 z ^ (z >> 31)
184}
185
186/// A collation for names the way a person reads them.
187///
188/// SQLite's default is a byte comparison, which sorts every capital before
189/// every lowercase and every accented letter after the whole ASCII range — so
190/// an artist list ran `Zebra`, then `aphex twin`, and put `Âme` at the end
191/// where nobody would look for it.
192///
193/// Case is folded, accents are folded onto their base letter (`Âme` sorts with
194/// `Ame`), and runs of digits compare by value so `Track 2` precedes
195/// `Track 10`. Ties fall back to the raw bytes, so two names that differ only
196/// in case or accent still have a stable order rather than being treated as
197/// equal.
198/// Registered by `create_tables`, so every connection has it — a query using
199/// `COLLATE LIBRARY` on a connection that skipped this fails outright rather
200/// than quietly sorting some other way.
201pub(crate) fn register_library_collation(conn: &Connection) -> rusqlite::Result<()> {
202 conn.create_collation("LIBRARY", |a, b| {
203 cached_sort_key(a).cmp(&cached_sort_key(b)).then(a.cmp(b))
204 })
205}
206
207thread_local! {
208 /// Sort keys, kept for the life of the thread.
209 ///
210 /// A collation sees the same name once per level of the sort — around two
211 /// dozen times in a five-thousand-row list — and building a key means an
212 /// NFD pass and a `Vec` of freshly allocated `String`s. Cached, each name is
213 /// folded once per thread instead of once per comparison.
214 static SORT_KEYS: RefCell<HashMap<Box<str>, Rc<[Chunk]>>> = RefCell::new(HashMap::new());
215}
216
217fn cached_sort_key(s: &str) -> Rc<[Chunk]> {
218 SORT_KEYS.with_borrow_mut(|cache| {
219 if let Some(key) = cache.get(s) {
220 return Rc::clone(key);
221 }
222 // A library's worth of names is tens of thousands of entries. Anything
223 // beyond that is a query sorting something other than names, and it
224 // should not grow this without bound.
225 if cache.len() >= 50_000 {
226 cache.clear();
227 }
228 let key: Rc<[Chunk]> = sort_key(s).into();
229 cache.insert(s.into(), Rc::clone(&key));
230 key
231 })
232}
233
234/// One comparable chunk of a name: either a run of digits, as a number, or a
235/// run of folded characters.
236#[derive(PartialEq, Eq, PartialOrd, Ord)]
237enum Chunk {
238 Number(u128),
239 Text(String),
240}
241
242fn sort_key(s: &str) -> Vec<Chunk> {
243 use unicode_normalization::UnicodeNormalization;
244
245 // NFD splits an accented letter into its base plus a combining mark; dropping
246 // the marks leaves the base letter to sort on.
247 let folded: String = s
248 .nfd()
249 .filter(|c| !matches!(*c as u32, 0x0300..=0x036F))
250 .flat_map(char::to_lowercase)
251 .collect();
252
253 let mut chunks = Vec::new();
254 let mut rest = folded.as_str();
255 while !rest.is_empty() {
256 let digits = rest
257 .find(|c: char| !c.is_ascii_digit())
258 .unwrap_or(rest.len());
259 if digits > 0 && rest.starts_with(|c: char| c.is_ascii_digit()) {
260 // Absurdly long digit runs are not numbers anyone sorts by.
261 match rest[..digits].parse::<u128>() {
262 Ok(n) => chunks.push(Chunk::Number(n)),
263 Err(_) => chunks.push(Chunk::Text(rest[..digits].to_string())),
264 }
265 rest = &rest[digits..];
266 continue;
267 }
268 let text = rest
269 .find(|c: char| c.is_ascii_digit())
270 .unwrap_or(rest.len())
271 .max(1);
272 chunks.push(Chunk::Text(rest[..text].to_string()));
273 rest = &rest[text..];
274 }
275 chunks
276}
277
278#[cfg(test)]
279mod collation_tests {
280 use super::*;
281
282 fn sorted(names: &[&str]) -> Vec<String> {
283 let conn = Connection::open_in_memory().unwrap();
284 crate::db::schema::create_tables(&conn).unwrap();
285 conn.execute_batch("CREATE TABLE t (name TEXT)").unwrap();
286 for n in names {
287 conn.execute("INSERT INTO t VALUES (?1)", [n]).unwrap();
288 }
289 let mut stmt = conn
290 .prepare("SELECT name FROM t ORDER BY name COLLATE LIBRARY")
291 .unwrap();
292 let rows = stmt.query_map([], |r| r.get::<_, String>(0)).unwrap();
293 rows.map(Result::unwrap).collect()
294 }
295
296 #[test]
297 fn lowercase_does_not_sort_after_everything() {
298 assert_eq!(
299 sorted(&["Zebra", "aphex twin", "Boards of Canada"]),
300 ["aphex twin", "Boards of Canada", "Zebra"]
301 );
302 }
303
304 #[test]
305 fn accents_sort_with_their_base_letter() {
306 // Byte order puts every non-ASCII name after `z`, which is where nobody
307 // looks for Âme.
308 assert_eq!(
309 sorted(&["Zomby", "Âme", "Alva Noto"]),
310 ["Alva Noto", "Âme", "Zomby"]
311 );
312 }
313
314 #[test]
315 fn digit_runs_compare_as_numbers() {
316 assert_eq!(
317 sorted(&["Track 10", "Track 2", "Track 1"]),
318 ["Track 1", "Track 2", "Track 10"]
319 );
320 }
321
322 #[test]
323 fn names_differing_only_in_case_keep_a_stable_order() {
324 // Folding must not make them equal, or the order flips between runs.
325 assert_eq!(
326 sorted(&["kraftwerk", "Kraftwerk"]),
327 ["Kraftwerk", "kraftwerk"]
328 );
329 }
330}
331
332#[cfg(test)]
333mod check_upgrade_tests {
334 use super::*;
335
336 #[test]
337 fn leaves_the_original_untouched() {
338 let dir = tempfile::tempdir().unwrap();
339 let path = dir.path().join("koan.db");
340 let conn = Connection::open(&path).unwrap();
341 conn.execute_batch("CREATE TABLE marker (x); INSERT INTO marker VALUES (1);")
342 .unwrap();
343 drop(conn);
344
345 Database::check_upgrade(&path).unwrap();
346
347 let conn = Connection::open(&path).unwrap();
348 let tables: i64 = conn
349 .query_row(
350 "SELECT COUNT(*) FROM sqlite_master WHERE type = 'table'",
351 [],
352 |r| r.get(0),
353 )
354 .unwrap();
355 assert_eq!(
356 tables, 1,
357 "the schema went into the snapshot, not the original"
358 );
359 }
360
361 #[test]
362 fn a_missing_database_fails() {
363 let dir = tempfile::tempdir().unwrap();
364 assert!(Database::check_upgrade(&dir.path().join("absent.db")).is_err());
365 }
366}