koan_core/db/connection.rs
1use std::cell::RefCell;
2use std::collections::HashMap;
3use std::path::Path;
4use std::rc::Rc;
5
6use rusqlite::functions::FunctionFlags;
7use rusqlite::{Connection, OpenFlags};
8use thiserror::Error;
9
10use super::schema;
11use crate::config;
12
13#[derive(Debug, Error)]
14pub enum DbError {
15 #[error("sqlite error: {0}")]
16 Sqlite(#[from] rusqlite::Error),
17 #[error("io error: {0}")]
18 Io(#[from] std::io::Error),
19 /// A bulk delete looked like a mount failure rather than an intentional
20 /// deletion, so it was refused. The library is untouched.
21 #[error("refused unsafe bulk delete: {0}")]
22 UnsafeBulkDelete(String),
23}
24
25/// Wrapper around a SQLite connection with koan's schema applied.
26pub struct Database {
27 pub conn: Connection,
28}
29
30impl Database {
31 /// Open (or create) a database at the given path, applying the schema and
32 /// pending migrations.
33 ///
34 /// This is the once-per-process path: it creates the parent directory,
35 /// tightens file permissions, checkpoints the WAL and runs the ~30-statement
36 /// DDL batch. Anything opening a connection per request wants
37 /// [`Database::open_existing`] instead.
38 pub fn open(path: &Path) -> Result<Self, DbError> {
39 if let Some(parent) = path.parent() {
40 std::fs::create_dir_all(parent)?;
41 }
42
43 let conn = Connection::open(path)?;
44
45 #[cfg(unix)]
46 {
47 use std::os::unix::fs::PermissionsExt;
48 let _ = std::fs::set_permissions(path, std::fs::Permissions::from_mode(0o600));
49 }
50
51 configure(&conn)?;
52
53 // Attempt a passive WAL checkpoint on open. This is non-blocking — it
54 // moves WAL pages back to the main DB file only if no readers/writers
55 // are active, preventing unbounded WAL growth across sessions.
56 let _ = conn.execute_batch("PRAGMA wal_checkpoint(PASSIVE)");
57
58 schema::create_tables(&conn)?;
59
60 // The planner picks between indexes by guessing how many rows each one
61 // will yield, and with no statistics it guesses the same number for all
62 // of them. That is how a partial index on the column a query filters by
63 // loses to an index that merely happens to supply the ORDER BY. Cheap
64 // after the first run, and a no-op when nothing has moved.
65 let _ = conn.execute_batch("PRAGMA optimize");
66
67 Ok(Self { conn })
68 }
69
70 /// Apply this build's schema and migrations to a snapshot of the database
71 /// at `path`, to learn whether it would open. The original is only read,
72 /// so this is safe beside a server that has it open: it is how a deploy
73 /// finds a migration that fails on the real library before replacing the
74 /// running version.
75 pub fn check_upgrade(path: &Path) -> Result<(), DbError> {
76 let snapshot = std::env::temp_dir().join(format!("koan-check-{}.db", std::process::id()));
77 let _ = std::fs::remove_file(&snapshot);
78 let result = (|| {
79 let source = Connection::open_with_flags(path, OpenFlags::SQLITE_OPEN_READ_ONLY)?;
80 source.execute("VACUUM INTO ?1", [snapshot.to_string_lossy()])?;
81 drop(source);
82 Self::open(&snapshot).map(drop)
83 })();
84 for suffix in ["", "-wal", "-shm"] {
85 let mut file = snapshot.clone().into_os_string();
86 file.push(suffix);
87 let _ = std::fs::remove_file(file);
88 }
89 result
90 }
91
92 /// Open an additional connection to a database whose schema is already
93 /// applied — pragmas only, no DDL, no checkpoint, no permission syscall.
94 ///
95 /// Callers are responsible for having run [`Database::open`] at least once
96 /// against the same path first.
97 pub fn open_existing(path: &Path) -> Result<Self, DbError> {
98 let conn = Connection::open(path)?;
99 configure(&conn)?;
100 Ok(Self { conn })
101 }
102
103 /// Open the default database at the standard data directory.
104 pub fn open_default() -> Result<Self, DbError> {
105 Self::open(&config::db_path())
106 }
107
108 /// Refresh the planner's statistics.
109 ///
110 /// Worth calling wherever the library changes size in bulk — a scan, a
111 /// remote sync — because the statistics gathered when the process started
112 /// describe a library that no longer exists, and the planner will keep
113 /// choosing for it. A no-op when nothing has moved far enough to matter.
114 pub fn optimize(&self) {
115 if let Err(e) = self.conn.execute_batch("PRAGMA optimize") {
116 log::debug!("PRAGMA optimize failed: {e}");
117 }
118 }
119}
120
121/// Connection-scoped pragmas. Every connection needs these; none of them touch
122/// the file on disk, so they are cheap enough to repeat per connection.
123fn configure(conn: &Connection) -> Result<(), DbError> {
124 // WAL mode for concurrent reads + single writer.
125 conn.pragma_update(None, "journal_mode", "wal")?;
126 conn.pragma_update(None, "foreign_keys", "on")?;
127 // Long enough to outlast a scan chunk: a writer that gives up mid-scan
128 // silently loses favourites, queue state and play counts.
129 conn.pragma_update(None, "busy_timeout", 30000)?;
130 // Slightly faster at the cost of durability on power loss (acceptable for a media DB).
131 conn.pragma_update(None, "synchronous", "normal")?;
132 // Map the whole library. A library this size fits well inside this, so
133 // reads become dereferences into a mapped region rather than syscalls —
134 // which is the useful sense in which a database can be "in memory". The
135 // page cache was already holding it; this stops copying it out per read.
136 conn.pragma_update(None, "mmap_size", 268_435_456i64)?;
137 // 32 MiB of pages, per connection. Negative means KiB rather than pages,
138 // so the figure does not change meaning with the page size.
139 conn.pragma_update(None, "cache_size", -32_000i64)?;
140 // Sorts and intermediate tables in memory. FTS and the ORDER BYs behind
141 // every library listing make temporary tables constantly.
142 conn.pragma_update(None, "temp_store", "memory")?;
143 // How many rows `PRAGMA optimize` samples per index. Bounded, so gathering
144 // statistics stays a fraction of a second on a library of any size; the
145 // planner needs the shape of the distribution, not an exact count.
146 conn.pragma_update(None, "analysis_limit", 400i64)?;
147 // Here rather than with the schema: `open_existing` skips the DDL, and a
148 // connection without this collation fails every ORDER BY that uses it.
149 register_library_collation(conn)?;
150 register_shuffle_function(conn)?;
151 Ok(())
152}
153
154/// `koan_shuffle(id, seed)` — a stable pseudo-random ordering key.
155///
156/// A shuffled listing that is read a page at a time cannot shuffle in the
157/// client: page two would be drawn from a different shuffle than page one, and
158/// records would repeat or vanish as you scrolled. Ordering by a hash of the
159/// row id and a seed gives one order that every page of the same seed agrees
160/// on, and a new seed gives a different one.
161///
162/// Registered next to the collation, and for the same reason: a connection
163/// without it fails the query outright rather than sorting some other way.
164pub(crate) fn register_shuffle_function(conn: &Connection) -> rusqlite::Result<()> {
165 conn.create_scalar_function(
166 "koan_shuffle",
167 2,
168 FunctionFlags::SQLITE_UTF8 | FunctionFlags::SQLITE_DETERMINISTIC,
169 |ctx| {
170 let id = ctx.get::<i64>(0)? as u64;
171 let seed = ctx.get::<i64>(1)? as u64;
172 Ok(splitmix64(id ^ splitmix64(seed)) as i64)
173 },
174 )
175}
176
177/// SplitMix64. Cheap, and it scatters consecutive ids — which matters, because
178/// a library's ids are consecutive in the order it was scanned.
179fn splitmix64(x: u64) -> u64 {
180 let mut z = x.wrapping_add(0x9E37_79B9_7F4A_7C15);
181 z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
182 z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
183 z ^ (z >> 31)
184}
185
186/// A collation for names the way a person reads them.
187///
188/// SQLite's default is a byte comparison, which sorts every capital before
189/// every lowercase and every accented letter after the whole ASCII range — so
190/// an artist list ran `Zebra`, then `aphex twin`, and put `Âme` at the end
191/// where nobody would look for it.
192///
193/// Case is folded, accents are folded onto their base letter (`Âme` sorts with
194/// `Ame`), and runs of digits compare by value so `Track 2` precedes
195/// `Track 10`. Ties fall back to the raw bytes, so two names that differ only
196/// in case or accent still have a stable order rather than being treated as
197/// equal.
198///
199/// Registered by `configure`, so every connection has it — a query using
200/// `COLLATE LIBRARY` on a connection that skipped this fails outright rather
201/// than quietly sorting some other way.
202pub(crate) fn register_library_collation(conn: &Connection) -> rusqlite::Result<()> {
203 conn.create_collation("LIBRARY", |a, b| {
204 cached_sort_key(a).cmp(&cached_sort_key(b)).then(a.cmp(b))
205 })
206}
207
208thread_local! {
209 /// Sort keys, kept for the life of the thread.
210 ///
211 /// A collation sees the same name once per level of the sort — around two
212 /// dozen times in a five-thousand-row list — and building a key means an
213 /// NFD pass and a `Vec` of freshly allocated `String`s. Cached, each name is
214 /// folded once per thread instead of once per comparison.
215 static SORT_KEYS: RefCell<HashMap<Box<str>, Rc<[Chunk]>>> = RefCell::new(HashMap::new());
216}
217
218fn cached_sort_key(s: &str) -> Rc<[Chunk]> {
219 SORT_KEYS.with_borrow_mut(|cache| {
220 if let Some(key) = cache.get(s) {
221 return Rc::clone(key);
222 }
223 // A library's worth of names is tens of thousands of entries. Anything
224 // beyond that is a query sorting something other than names, and it
225 // should not grow this without bound.
226 if cache.len() >= 50_000 {
227 cache.clear();
228 }
229 let key: Rc<[Chunk]> = sort_key(s).into();
230 cache.insert(s.into(), Rc::clone(&key));
231 key
232 })
233}
234
235/// One comparable chunk of a name: either a run of digits, as a number, or a
236/// run of folded characters.
237#[derive(PartialEq, Eq, PartialOrd, Ord)]
238enum Chunk {
239 Number(u128),
240 Text(String),
241}
242
243fn sort_key(s: &str) -> Vec<Chunk> {
244 use unicode_normalization::UnicodeNormalization;
245
246 // NFD splits an accented letter into its base plus a combining mark; dropping
247 // the marks leaves the base letter to sort on.
248 let folded: String = s
249 .nfd()
250 .filter(|c| !matches!(*c as u32, 0x0300..=0x036F))
251 .flat_map(char::to_lowercase)
252 .collect();
253
254 let mut chunks = Vec::new();
255 let mut rest = folded.as_str();
256 while !rest.is_empty() {
257 let digits = rest
258 .find(|c: char| !c.is_ascii_digit())
259 .unwrap_or(rest.len());
260 if digits > 0 && rest.starts_with(|c: char| c.is_ascii_digit()) {
261 // Absurdly long digit runs are not numbers anyone sorts by.
262 match rest[..digits].parse::<u128>() {
263 Ok(n) => chunks.push(Chunk::Number(n)),
264 Err(_) => chunks.push(Chunk::Text(rest[..digits].to_string())),
265 }
266 rest = &rest[digits..];
267 continue;
268 }
269 let text = rest
270 .find(|c: char| c.is_ascii_digit())
271 .unwrap_or(rest.len())
272 .max(1);
273 chunks.push(Chunk::Text(rest[..text].to_string()));
274 rest = &rest[text..];
275 }
276 chunks
277}
278
279#[cfg(test)]
280mod collation_tests {
281 use super::*;
282
283 fn sorted(names: &[&str]) -> Vec<String> {
284 let conn = Connection::open_in_memory().unwrap();
285 crate::db::schema::create_tables(&conn).unwrap();
286 conn.execute_batch("CREATE TABLE t (name TEXT)").unwrap();
287 for n in names {
288 conn.execute("INSERT INTO t VALUES (?1)", [n]).unwrap();
289 }
290 let mut stmt = conn
291 .prepare("SELECT name FROM t ORDER BY name COLLATE LIBRARY")
292 .unwrap();
293 let rows = stmt.query_map([], |r| r.get::<_, String>(0)).unwrap();
294 rows.map(Result::unwrap).collect()
295 }
296
297 #[test]
298 fn lowercase_does_not_sort_after_everything() {
299 assert_eq!(
300 sorted(&["Zebra", "aphex twin", "Boards of Canada"]),
301 ["aphex twin", "Boards of Canada", "Zebra"]
302 );
303 }
304
305 #[test]
306 fn accents_sort_with_their_base_letter() {
307 // Byte order puts every non-ASCII name after `z`, which is where nobody
308 // looks for Âme.
309 assert_eq!(
310 sorted(&["Zomby", "Âme", "Alva Noto"]),
311 ["Alva Noto", "Âme", "Zomby"]
312 );
313 }
314
315 #[test]
316 fn digit_runs_compare_as_numbers() {
317 assert_eq!(
318 sorted(&["Track 10", "Track 2", "Track 1"]),
319 ["Track 1", "Track 2", "Track 10"]
320 );
321 }
322
323 #[test]
324 fn names_differing_only_in_case_keep_a_stable_order() {
325 // Folding must not make them equal, or the order flips between runs.
326 assert_eq!(
327 sorted(&["kraftwerk", "Kraftwerk"]),
328 ["Kraftwerk", "kraftwerk"]
329 );
330 }
331}
332
333#[cfg(test)]
334mod check_upgrade_tests {
335 use super::*;
336
337 #[test]
338 fn leaves_the_original_untouched() {
339 let dir = tempfile::tempdir().unwrap();
340 let path = dir.path().join("koan.db");
341 let conn = Connection::open(&path).unwrap();
342 conn.execute_batch("CREATE TABLE marker (x); INSERT INTO marker VALUES (1);")
343 .unwrap();
344 drop(conn);
345
346 Database::check_upgrade(&path).unwrap();
347
348 let conn = Connection::open(&path).unwrap();
349 let tables: i64 = conn
350 .query_row(
351 "SELECT COUNT(*) FROM sqlite_master WHERE type = 'table'",
352 [],
353 |r| r.get(0),
354 )
355 .unwrap();
356 assert_eq!(
357 tables, 1,
358 "the schema went into the snapshot, not the original"
359 );
360 }
361
362 #[test]
363 fn a_missing_database_fails() {
364 let dir = tempfile::tempdir().unwrap();
365 assert!(Database::check_upgrade(&dir.path().join("absent.db")).is_err());
366 }
367}