recall-server 0.4.13

Recall's sync server: SQLite persistence, LLM-assisted merge, and the HTTP API
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
//! The `audit_log` table: append-only storage for the Merkle tree in
//! [`crate::audit::merkle`], and the one transactional primitive
//! ([`Store::audited`]) every authenticated state change goes through so
//! its leaf commits with it.
//!
//! Every method on [`Store`] the server uses to change a file, a device, an
//! authkey, a passkey, a bootstrap code or a merge job's outcome takes the
//! leaf it appends as an argument; none can change one without it. The
//! writes left without a leaf are deliberate, and none changes a file, a
//! credential or what a job came to: an enrolment waiting for approval
//! (anyone may ask, and unauthenticated routes append nothing), a
//! machine's poll for it, a device's `last_seen`, the sweep of
//! long-expired enrolments, an admin session started by a sign-in, used,
//! ended by a sign-out or swept once idle, a passkey's signature counter,
//! the release of a revoked worker's leases before the server claims them
//! itself ([`Store::release_open_jobs`]), and the pruning of finished jobs.
//!
//! Another process may append: `recall-server reset-passkeys` does, on the
//! host, and so do `recall-server admin`'s renames, removes and restores
//! (`store/admin.rs`). Every append first reads in any leaf the table holds
//! past this store's tree, so they all write one log rather than a fork.

use anyhow::{bail, Result};
use rusqlite::{Connection, TransactionBehavior};

use super::Store;
use crate::audit::merkle::{self, Hash, Tree};

/// Created alongside `memory_files` and the device tables, every time the
/// store opens.
///
/// `leaf` and `leaf_hash` are both stored: `leaf` is the exact bytes a
/// verifier hashes and exports, and `leaf_hash` is what the leaf hashed to
/// when it was written, which [`load`] checks it still does.
///
/// The triggers are what makes this append-only in the database itself,
/// not only by convention: even a bug — or a future migration reaching for
/// `UPDATE` out of habit — cannot rewrite, remove or skip a leaf. The
/// insert trigger is the one that stops `INSERT OR REPLACE`, whose
/// replacing delete does not fire a delete trigger, and an insert that
/// leaves a gap.
///
/// None of it stops someone holding the database file: they can drop a
/// trigger, or rewrite the table and every hash in it consistently. That
/// is what a checkpoint an owner saved elsewhere is for — a rewritten log
/// no longer extends it.
pub(super) const SCHEMA: &str = "
    CREATE TABLE IF NOT EXISTS audit_log (
        seq       INTEGER PRIMARY KEY,
        leaf      BLOB NOT NULL,
        leaf_hash BLOB NOT NULL
    );
    CREATE TRIGGER IF NOT EXISTS audit_log_no_update
        BEFORE UPDATE ON audit_log
    BEGIN
        SELECT RAISE(ABORT, 'audit_log is append-only');
    END;
    CREATE TRIGGER IF NOT EXISTS audit_log_no_delete
        BEFORE DELETE ON audit_log
    BEGIN
        SELECT RAISE(ABORT, 'audit_log is append-only');
    END;
    CREATE TRIGGER IF NOT EXISTS audit_log_next_seq
        BEFORE INSERT ON audit_log
        WHEN NEW.seq IS NOT (SELECT COALESCE(MAX(seq), -1) + 1 FROM audit_log)
    BEGIN
        SELECT RAISE(ABORT, 'audit_log is append-only: a leaf takes the next seq');
    END;
";

/// What [`load`] read back.
pub(super) struct Loaded {
    /// Every leaf's hash, as a tree.
    pub(super) tree: Tree,
    /// The newest leaf's `at`, or empty for an empty log.
    pub(super) last_at: String,
}

/// Reads every leaf back at open and rebuilds the tree from them, refusing
/// a log that is not what this server wrote: a `seq` missing or out of
/// place, or a leaf whose bytes no longer hash to its stored `leaf_hash`.
/// The triggers keep both from happening through SQL; this is for the
/// file changed some other way, by a disk or by hand.
///
/// The hashes are recomputed from the leaves rather than trusted, which
/// is most of what opening costs: measured at 8.5 seconds for a million
/// leaves (a gigabyte of them) on a small VM, against 1.8 seconds to read
/// the stored hashes alone, and 64 bytes of memory a leaf for the tree
/// kept after. A log that size is years of one owner's syncing; the check
/// is worth its wait. A server that started anyway would sign every later
/// checkpoint over a tree it had already lost, so it does not start; the
/// error says to restore the database from a backup.
pub(super) fn load(conn: &Connection) -> Result<Loaded> {
    let mut tree = Tree::new();
    read_from(conn, 0, |hash| tree.append(hash))?;
    let last_at = last_at(conn, tree.size())?;
    Ok(Loaded { tree, last_at })
}

/// Reads every leaf from `seq` `from` on, in order, checked as [`load`]
/// describes, handing each one's hash to `each`. What both opening and
/// [`Store::audited_each`]'s catching up do. Answers how many it read.
fn read_from(conn: &Connection, from: u64, each: impl FnMut(Hash)) -> Result<u64> {
    read_range(conn, from, i64::MAX as u64, each)
}

/// [`read_from`], stopping after `most` leaves.
fn read_range(conn: &Connection, from: u64, most: u64, mut each: impl FnMut(Hash)) -> Result<u64> {
    let mut stmt = conn.prepare(
        "SELECT seq, leaf, leaf_hash FROM audit_log WHERE seq >= ?1 ORDER BY seq LIMIT ?2",
    )?;
    let mut rows = stmt.query((from as i64, most.min(i64::MAX as u64) as i64))?;
    let mut want = from;
    while let Some(row) = rows.next()? {
        let seq: i64 = row.get(0)?;
        if seq != want as i64 {
            bail!(
                "the audit log is damaged: leaf {want} is missing (the next one stored is {seq}); \
                 restore the database from a backup"
            );
        }
        let leaf = row.get_ref(1)?.as_blob()?;
        let hash = merkle::hash_leaf(leaf);
        if row.get_ref(2)?.as_blob()? != hash.as_slice() {
            bail!(
                "the audit log is damaged: leaf {seq} no longer hashes to its stored leaf_hash; \
                 restore the database from a backup"
            );
        }
        each(hash);
        want += 1;
    }
    Ok(want - from)
}

/// The `at` of the newest of `size` leaves, or empty for none.
fn last_at(conn: &Connection, size: u64) -> Result<String> {
    if size == 0 {
        return Ok(String::new());
    }
    let leaf: Vec<u8> = conn.query_row(
        "SELECT leaf FROM audit_log WHERE seq = ?1",
        (size as i64 - 1,),
        |r| r.get(0),
    )?;
    // Every leaf `leaf::encode` wrote has one; one that somehow does not
    // only means the next `at` is not held to it.
    Ok(serde_json::from_slice::<serde_json::Value>(&leaf)
        .ok()
        .and_then(|v| v.get("at")?.as_str().map(str::to_string))
        .unwrap_or_default())
}

/// The leaves the table holds past the `held` this store's tree has, for
/// leaves another process appended since this one last looked:
/// `recall-server reset-passkeys` or `recall-server admin`, run on the host
/// beside a running server.
/// Their hashes, checked as [`load`] checks them, and the newest one's
/// `at`, for the tree to take once the transaction this runs in commits;
/// its lock keeps anything else from appending until then.
///
/// A table that holds fewer leaves than `held` was rolled back under a
/// running server, and nothing more is appended onto it until the server
/// restarts and reads it afresh. A backup put in place of the file of a
/// running server is mostly not seen here, since under WAL the server's
/// connection goes by its WAL and its cache rather than the file. The
/// check that the path still names the file it opened (`FileId` in the
/// store), made before this, catches a file moved aside or replaced by a
/// new one; one overwritten in place is caught by neither, which is why a
/// restore stops the server first and never copies over `recall.db`
/// (`deploy/README.md`).
fn catch_up(conn: &Connection, held: u64) -> Result<(Vec<Hash>, Option<String>)> {
    let stored: i64 =
        conn.query_row("SELECT COALESCE(MAX(seq) + 1, 0) FROM audit_log", [], |r| {
            r.get(0)
        })?;
    if stored < held as i64 {
        bail!(
            "the audit log holds {stored} leaves, fewer than the {held} this server appended: \
             the database was replaced under a running server; restart it"
        );
    }
    let mut hashes = Vec::new();
    if stored == held as i64 {
        return Ok((hashes, None));
    }
    let read = read_from(conn, held, |hash| hashes.push(hash))?;
    Ok((hashes, Some(last_at(conn, held + read)?)))
}

/// What [`Store::audited`]'s write closure hands back.
pub enum Outcome<T> {
    /// The write happened; its leaf is appended in the same transaction.
    Commit(T),
    /// Nothing happened — a refused request, such as an unknown code or one
    /// already decided. The transaction rolls back and no leaf is
    /// appended, matching "unauthenticated routes and refused requests
    /// append nothing" for the requests that do carry a credential but are
    /// refused for another reason.
    Refuse(T),
}

/// One stored leaf, as [`Store::audit_entries`] returns it: its position and
/// its exact bytes.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct AuditEntry {
    /// Its index in the tree, from 0.
    pub seq: u64,
    /// The leaf exactly as written — never re-serialized.
    pub leaf: Vec<u8>,
}

/// Why a consistency proof could not be produced.
#[derive(Debug, Clone, Copy, PartialEq, Eq, thiserror::Error)]
pub enum ConsistencyError {
    /// `first` is 0, or greater than `second`.
    #[error("first must be at least 1 and at most second")]
    BadRange,
    /// `second` is past the tree's current size.
    #[error("second is past the end of the log")]
    SecondBeyondTreeSize,
}

/// The `at` the next leaf carries: now, or the newest leaf's `at` (the
/// later of the two given) if the clock has gone back since — so `at` never
/// decreases along `seq`. Read under the store's lock, like the `seq` it
/// goes with. Being one fixed-width format, two of these compare as
/// strings.
fn next_at(one: &str, other: &str) -> String {
    let newest = one.max(other);
    let now = crate::now();
    if now.as_str() < newest {
        newest.to_string()
    } else {
        now
    }
}

impl Store {
    /// Runs `write` inside one transaction and, only if it commits,
    /// appends the leaf `build_leaf` makes from the position that leaf
    /// will hold (its `seq`), its `at`, and `write`'s own result — then,
    /// and only then, the in-memory tree learns about it.
    ///
    /// `write` is given the same `at`, so a row it stamps and the leaf
    /// that records it carry one time. Both are taken under the lock the
    /// whole transaction holds, which is what makes `at` rise with `seq`.
    ///
    /// This and [`Store::audited_each`] are the only places a leaf is ever
    /// appended, so "every authenticated state change appends its leaf, in
    /// the same transaction as the change" is true by construction: nothing
    /// calls `INSERT INTO audit_log` any other way, and a write that never
    /// commits — because `write` returned [`Outcome::Refuse`] or an error —
    /// leaves neither the state change nor a leaf behind.
    pub fn audited<T>(
        &self,
        write: impl FnOnce(&rusqlite::Transaction, &str) -> Result<Outcome<T>>,
        build_leaf: impl FnOnce(u64, &str, &T) -> Vec<u8>,
    ) -> Result<T> {
        self.audited_each(write, |seq, at, value| vec![build_leaf(seq, at, value)])
    }

    /// [`Store::audited`], for a write that records any number of changes
    /// at once: `build_leaves` returns one leaf per change, the first
    /// taking `seq` and each after it the next, all appended in `write`'s
    /// transaction. The sweep is the one user: every device it removes has
    /// a leaf, and they commit together with the removals.
    pub fn audited_each<T>(
        &self,
        write: impl FnOnce(&rusqlite::Transaction, &str) -> Result<Outcome<T>>,
        build_leaves: impl FnOnce(u64, &str, &T) -> Vec<Vec<u8>>,
    ) -> Result<T> {
        self.audited_each_as(
            anyhow::Error::from,
            anyhow::Error::from,
            write,
            build_leaves,
        )
    }

    /// [`Store::audited_each`], with the error that starting the
    /// transaction or committing it becomes chosen by the caller:
    /// `recall-server admin` tells the owner which of the two it was, and
    /// that the database stayed locked, since those are the failures they
    /// can do something about (wait, and run it again).
    pub(crate) fn audited_each_as<T>(
        &self,
        begin_failed: impl FnOnce(rusqlite::Error) -> anyhow::Error,
        commit_failed: impl FnOnce(rusqlite::Error) -> anyhow::Error,
        write: impl FnOnce(&rusqlite::Transaction, &str) -> Result<Outcome<T>>,
        build_leaves: impl FnOnce(u64, &str, &T) -> Vec<Vec<u8>>,
    ) -> Result<T> {
        let mut state = self.lock();
        // Before anything is written: is the path still the file this
        // connection opened? See `FileId` in the store.
        if let Some(opened) = state.file {
            if super::file_id(&state.conn) != Some(opened) {
                let why = format!(
                    "the database file {} was moved or replaced under this running server, \
                     which would otherwise go on writing into the file it had opened. Nothing \
                     was written. Restart the server, so it opens the file that is there now",
                    state.conn.path().unwrap_or("")
                );
                // Every write from here on is refused with this, as a 500
                // the owner may never see; the server's log says it once.
                if !std::mem::replace(&mut state.moved_said, true) && state.log_moved {
                    eprintln!("{why}");
                }
                bail!(why);
            }
        }
        let (held, held_at) = (state.audit.size(), state.audit_at.clone());
        // `IMMEDIATE`: the database's write lock from the start, so another
        // process (`reset-passkeys`, `admin`) cannot append between the
        // catch-up below and this transaction's own leaves. The transaction
        // borrows the whole state, so the tree cannot learn of anything
        // until it commits.
        let tx = state
            .conn
            .transaction_with_behavior(TransactionBehavior::Immediate)
            .map_err(begin_failed)?;
        let (caught, caught_at) = catch_up(&tx, held)?;
        // The position the first leaf will hold if this commits: the size
        // of the tree the table holds, read under the lock the transaction
        // holds for its whole duration, so no other request or process can
        // claim this seq first.
        let seq = held + caught.len() as u64;
        let at = next_at(caught_at.as_deref().unwrap_or(""), &held_at);
        let value = match write(&tx, &at)? {
            Outcome::Commit(value) => value,
            Outcome::Refuse(value) => return Ok(value), // `tx` drops here: rolled back.
        };
        let leaves = build_leaves(seq, &at, &value);
        let mut hashes = Vec::with_capacity(leaves.len());
        for (i, leaf) in leaves.iter().enumerate() {
            let leaf_hash = merkle::hash_leaf(leaf);
            tx.execute(
                "INSERT INTO audit_log (seq, leaf, leaf_hash) VALUES (?1, ?2, ?3)",
                (seq as i64 + i as i64, leaf, leaf_hash.as_slice()),
            )?;
            hashes.push(leaf_hash);
        }
        tx.commit().map_err(commit_failed)?;
        for leaf_hash in caught.into_iter().chain(hashes) {
            state.audit.append(leaf_hash);
        }
        if !leaves.is_empty() {
            state.audit_at = at;
        } else if let Some(caught_at) = caught_at.filter(|c| *c > state.audit_at) {
            state.audit_at = caught_at;
        }
        Ok(value)
    }

    /// Reads into the tree every leaf the table holds past it, outside any
    /// write transaction and a few thousand leaves at a time, for a process
    /// that opened the database without reading its log: `recall-server
    /// admin`, about to append.
    ///
    /// Only to be quick about it. [`Store::audited_each`] catches up with
    /// the table by itself, but it does so holding the write lock, which a
    /// running server then waits on for no more than its busy timeout: 5
    /// seconds, against 8.5 to read a million leaves. Read here first, the
    /// catch-up under the lock is only the leaves appended since, and each
    /// read here holds a read lock for a fraction of a second. The leaves
    /// are checked as [`load`] checks them; the log is append-only, so what
    /// this reads stays true.
    pub(crate) fn audit_read_ahead(&self) -> Result<()> {
        self.audit_read_ahead_by(4096)
    }

    /// [`Store::audit_read_ahead`], `chunk` leaves at a time.
    fn audit_read_ahead_by(&self, chunk: u64) -> Result<()> {
        let mut state = self.lock();
        loop {
            let held = state.audit.size();
            let mut hashes = Vec::new();
            let read = read_range(&state.conn, held, chunk, |hash| hashes.push(hash))?;
            if read == 0 {
                return Ok(());
            }
            let at = last_at(&state.conn, held + read)?;
            for hash in hashes {
                state.audit.append(hash);
            }
            if at > state.audit_at {
                state.audit_at = at;
            }
        }
    }

    /// Appends a leaf with no other state to change: a pull, the server's
    /// own `start`. Still one transaction (of one statement), so it shares
    /// [`Store::audited`]'s all-or-nothing behaviour rather than being a
    /// special case.
    ///
    /// Returns the leaf's `seq`, read out of the same closure `audited`
    /// calls under its lock — not by asking the tree its size again
    /// afterwards, which another append could have moved on by then.
    pub fn audit_append(&self, build_leaf: impl FnOnce(u64, &str) -> Vec<u8>) -> Result<u64> {
        let assigned = std::cell::Cell::new(0u64);
        self.audited(
            |_tx, _at| Ok(Outcome::Commit(())),
            |seq, at, ()| {
                assigned.set(seq);
                build_leaf(seq, at)
            },
        )?;
        Ok(assigned.get())
    }

    /// The tree's size and root, for `GET /v1/audit/checkpoint` and the
    /// `Recall-Audit-Checkpoint` header every pull carries.
    pub fn audit_checkpoint(&self) -> (u64, Hash) {
        let state = self.lock();
        (state.audit.size(), state.audit.root())
    }

    /// Leaves `start` to `end - 1`, stopping early once they come to more
    /// than `max_bytes` together — though never before the first, however
    /// large, so a caller paging on from the last `seq` it got always
    /// moves. The caller (the route handler) is responsible for
    /// `end <= tree_size` and the 1,000-row page limit.
    pub fn audit_entries(&self, start: u64, end: u64, max_bytes: usize) -> Result<Vec<AuditEntry>> {
        let state = self.lock();
        let mut stmt = state
            .conn
            .prepare("SELECT seq, leaf FROM audit_log WHERE seq >= ?1 AND seq < ?2 ORDER BY seq")?;
        let mut rows = stmt.query((start as i64, end as i64))?;
        let (mut out, mut bytes) = (Vec::new(), 0usize);
        while let Some(row) = rows.next()? {
            let leaf: Vec<u8> = row.get(1)?;
            bytes += leaf.len();
            if bytes > max_bytes && !out.is_empty() {
                break;
            }
            out.push(AuditEntry {
                seq: row.get::<_, i64>(0)? as u64,
                leaf,
            });
        }
        Ok(out)
    }

    /// The RFC 9162 §2.1.4 proof that the tree at `second` extends the one
    /// at `first`, from the tree in memory: a few hundred hashes at most,
    /// and no read of the database, however long the log.
    pub fn audit_consistency(
        &self,
        first: u64,
        second: u64,
    ) -> Result<Result<Vec<Hash>, ConsistencyError>> {
        let state = self.lock();
        if first == 0 || first > second {
            return Ok(Err(ConsistencyError::BadRange));
        }
        if second > state.audit.size() {
            return Ok(Err(ConsistencyError::SecondBeyondTreeSize));
        }
        Ok(Ok(state.audit.consistency(first, second)))
    }
}

#[cfg(test)]
mod tests {
    use super::*;
    use crate::audit::leaf;

    fn store() -> Store {
        Store::open_in_memory().unwrap()
    }

    fn push_leaf(seq: u64) -> Vec<u8> {
        leaf::encode(
            seq,
            "2026-01-01T00:00:00.000Z",
            leaf::action::PUSH,
            &leaf::Actor::Operator,
            leaf::subject_file(&leaf::FileChange {
                project_key: "acme/app",
                file_path: "a.md",
                deleted: false,
                stored_sha256: "abc",
                base_sha256: None,
                merged: false,
                merge_job: None,
            }),
            None,
        )
    }

    fn append(st: &Store) {
        st.audited(
            |_tx, _| Ok(Outcome::Commit(())),
            |seq, _, ()| push_leaf(seq),
        )
        .unwrap();
    }

    const INSERT_FILE: &str = "INSERT INTO memory_files \
         (project_key, file_path, content, source_env, updated_at) VALUES ('a','b','c','d','e')";

    #[test]
    fn a_committed_write_appends_exactly_one_leaf() {
        let st = store();
        st.audited(
            |tx, _| {
                tx.execute(INSERT_FILE, [])?;
                Ok(Outcome::Commit(()))
            },
            |seq, _, ()| push_leaf(seq),
        )
        .unwrap();
        let (size, _) = st.audit_checkpoint();
        assert_eq!(size, 1);
        assert_eq!(st.audit_entries(0, 1, usize::MAX).unwrap().len(), 1);
    }

    /// The atomicity the mutation table asks for: a write that returns an
    /// error after already changing something rolls the whole transaction
    /// back, leaf included.
    #[test]
    fn a_failing_write_appends_no_leaf_and_keeps_no_change() {
        let st = store();
        let result = st.audited(
            |tx, _| {
                tx.execute(INSERT_FILE, [])?;
                Err(anyhow::Error::from(rusqlite::Error::ExecuteReturnedResults))
            },
            |seq, _, ()| push_leaf(seq),
        );
        assert!(result.is_err());
        assert_eq!(
            st.audit_checkpoint().0,
            0,
            "no leaf from a rolled-back write"
        );
        assert!(st.get("a", "b").unwrap().is_none(), "no row either");
    }

    /// And the other way round: a leaf that cannot be written takes the
    /// change down with it. The leaf goes in inside the change's own
    /// transaction, before it commits, so there is never a moment when the
    /// change is stored and its leaf is not.
    #[test]
    fn a_leaf_that_cannot_be_written_undoes_the_change() {
        let st = store();
        st.with_raw(|c| {
            c.execute_batch(
                "CREATE TEMP TRIGGER no_leaves BEFORE INSERT ON audit_log
                 BEGIN SELECT RAISE(ABORT, 'no leaves today'); END;",
            )
        })
        .unwrap();
        let result = st.audited(
            |tx, _| {
                tx.execute(INSERT_FILE, [])?;
                Ok(Outcome::Commit(()))
            },
            |seq, _, ()| push_leaf(seq),
        );
        assert!(result.is_err(), "the leaf's insert failed");
        assert!(
            st.get("a", "b").unwrap().is_none(),
            "so the row is not there"
        );
        assert_eq!(st.audit_checkpoint().0, 0);
    }

    /// A refused request — nothing wrong at the database level, just a
    /// decision not to proceed — is the same all-or-nothing story: no leaf.
    #[test]
    fn a_refused_write_appends_no_leaf() {
        let st = store();
        let refusal: &str = st
            .audited(
                |_tx, _| Ok(Outcome::Refuse("no such code")),
                |seq, _, _| push_leaf(seq),
            )
            .unwrap();
        assert_eq!(refusal, "no such code");
        assert_eq!(st.audit_checkpoint().0, 0);
    }

    /// Several leaves from one write take consecutive seqs, and commit or
    /// roll back together.
    #[test]
    fn one_write_can_append_several_leaves_in_order() {
        let st = store();
        append(&st);
        st.audited_each(
            |_tx, _| Ok(Outcome::Commit(3u64)),
            |seq, _, n| (seq..seq + n).map(push_leaf).collect(),
        )
        .unwrap();
        let seqs: Vec<u64> = st
            .audit_entries(0, 4, usize::MAX)
            .unwrap()
            .iter()
            .map(|e| e.seq)
            .collect();
        assert_eq!(seqs, vec![0, 1, 2, 3]);
        assert_eq!(
            st.audit_entries(1, 2, usize::MAX).unwrap()[0].leaf,
            push_leaf(1)
        );
    }

    /// `at` is taken under the lock, and never goes back: a leaf written
    /// after the clock has moved backwards carries the newest `at` already
    /// in the log, not an earlier one.
    #[test]
    fn at_never_decreases_along_seq() {
        let st = store();
        let mut ats = Vec::new();
        for _ in 0..3 {
            st.audit_append(|seq, at| {
                ats.push(at.to_string());
                push_leaf(seq)
            })
            .unwrap();
        }
        assert!(ats.windows(2).all(|w| w[0] <= w[1]), "{ats:?}");
        assert_eq!(ats[0].len(), 24, "the API's timestamp format");

        // A newest leaf from the future: the clock went back after it.
        st.lock().audit_at = "2999-01-01T00:00:00.000Z".into();
        let mut got = String::new();
        st.audit_append(|seq, at| {
            got = at.to_string();
            push_leaf(seq)
        })
        .unwrap();
        assert_eq!(got, "2999-01-01T00:00:00.000Z");
    }

    #[test]
    fn checkpoint_matches_the_merkle_root_of_every_leaf() {
        let st = store();
        for _ in 0..5 {
            append(&st);
        }
        let (size, root) = st.audit_checkpoint();
        assert_eq!(size, 5);
        let leaves: Vec<Hash> = (0..5)
            .map(push_leaf)
            .map(|l| merkle::hash_leaf(&l))
            .collect();
        assert_eq!(root, merkle::root(&leaves));
    }

    #[test]
    fn entries_pages_by_seq_and_by_bytes() {
        let st = store();
        for _ in 0..3 {
            append(&st);
        }
        let got = st.audit_entries(1, 3, usize::MAX).unwrap();
        assert_eq!(got.iter().map(|e| e.seq).collect::<Vec<_>>(), vec![1, 2]);
        assert_eq!(got[0].leaf, push_leaf(1));

        let one = push_leaf(0).len();
        let seqs = |max| -> Vec<u64> {
            st.audit_entries(0, 3, max)
                .unwrap()
                .iter()
                .map(|e| e.seq)
                .collect()
        };
        assert_eq!(seqs(2 * one), vec![0, 1], "two fit exactly");
        assert_eq!(seqs(2 * one - 1), vec![0], "the second would overflow");
        assert_eq!(seqs(1), vec![0], "never fewer than one");
    }

    #[test]
    fn consistency_matches_merkle_and_rejects_bad_ranges() {
        let st = store();
        for _ in 0..8 {
            append(&st);
        }
        let proof = st.audit_consistency(3, 8).unwrap().unwrap();
        let leaves: Vec<Hash> = (0..8)
            .map(push_leaf)
            .map(|l| merkle::hash_leaf(&l))
            .collect();
        assert_eq!(proof, merkle::consistency(3, 8, &leaves));

        assert_eq!(
            st.audit_consistency(0, 8).unwrap(),
            Err(ConsistencyError::BadRange)
        );
        assert_eq!(
            st.audit_consistency(5, 3).unwrap(),
            Err(ConsistencyError::BadRange)
        );
        assert_eq!(
            st.audit_consistency(1, 100).unwrap(),
            Err(ConsistencyError::SecondBeyondTreeSize)
        );
    }

    /// A proof comes from the tree in memory: the table can be out of reach
    /// entirely and it is still served, which is what keeps the store's one
    /// lock from being held over a read of every leaf.
    #[test]
    fn a_proof_does_not_read_the_table() {
        let st = store();
        for _ in 0..40 {
            append(&st);
        }
        let want = st.audit_consistency(7, 40).unwrap().unwrap();
        st.with_raw(|c| c.execute_batch("ALTER TABLE audit_log RENAME TO audit_log_away"))
            .unwrap();
        assert_eq!(st.audit_consistency(7, 40).unwrap().unwrap(), want);
    }

    /// The triggers, not just application discipline: a raw `UPDATE`,
    /// `DELETE`, upsert, `INSERT OR REPLACE` over a leaf, or an insert that
    /// skips a seq, aborts.
    #[test]
    fn nothing_but_the_next_leaf_can_be_written() {
        let st = store();
        append(&st);
        append(&st);

        let refused = |sql: &str| {
            assert!(st.with_raw(|c| c.execute(sql, [])).is_err(), "{sql}");
        };
        refused("UPDATE audit_log SET seq = 99 WHERE seq = 0");
        refused("DELETE FROM audit_log WHERE seq = 0");
        refused(
            "INSERT OR REPLACE INTO audit_log (seq, leaf, leaf_hash) \
             VALUES (0, CAST('forged' AS BLOB), zeroblob(32))",
        );
        refused(
            "INSERT INTO audit_log (seq, leaf, leaf_hash) VALUES (0, x'01', zeroblob(32)) \
             ON CONFLICT(seq) DO UPDATE SET leaf = excluded.leaf",
        );
        refused("INSERT INTO audit_log (seq, leaf, leaf_hash) VALUES (100, x'02', zeroblob(32))");
        refused("INSERT INTO audit_log (leaf, leaf_hash) VALUES (x'02', zeroblob(32))");
        assert_eq!(
            st.audit_entries(0, 2, usize::MAX).unwrap()[0].leaf,
            push_leaf(0),
            "leaf 0 is what was written"
        );
        assert_eq!(st.audit_checkpoint().0, 2);

        // The next seq is still accepted — the trigger stops only the rest.
        st.with_raw(|c| {
            c.execute(
                "INSERT INTO audit_log (seq, leaf, leaf_hash) VALUES (2, x'03', zeroblob(32))",
                [],
            )
        })
        .unwrap();
    }

    /// The tree is rebuilt at open from what the file holds, so a reopened
    /// store answers with the same checkpoint and goes on from the same seq.
    #[test]
    fn reopening_rebuilds_the_same_tree() {
        let dir = tempfile::tempdir().unwrap();
        let path = dir.path().join("r.db");
        let before = {
            let st = Store::open(&path).unwrap();
            for _ in 0..5 {
                append(&st);
            }
            st.audit_checkpoint()
        };
        let st = Store::open(&path).unwrap();
        assert_eq!(st.audit_checkpoint(), before);
        assert_eq!(st.audit_append(|seq, _| push_leaf(seq)).unwrap(), 5);
    }

    /// A leaf another process appends to the same file, as
    /// `recall-server reset-passkeys` does beside a running server, is read
    /// into this one's tree before its next append, which then takes the
    /// seq after it; the two agree on the tree after. A log that went back
    /// under a running store stops its appends rather than growing a fork.
    #[test]
    fn a_leaf_another_process_appended_is_read_in_before_the_next() {
        let dir = tempfile::tempdir().unwrap();
        let path = dir.path().join("r.db");
        let server = Store::open(&path).unwrap();
        append(&server);
        append(&server);
        let host = Store::open(&path).unwrap();
        assert_eq!(host.audit_append(|seq, _| push_leaf(seq)).unwrap(), 2);
        assert_eq!(server.audit_checkpoint().0, 2, "not read until it appends");
        assert_eq!(server.audit_append(|seq, _| push_leaf(seq)).unwrap(), 3);
        assert_eq!(
            server.audit_checkpoint(),
            Store::open(&path).unwrap().audit_checkpoint()
        );

        // A store that has never looked at the log reads all of it in first.
        let blank = Store::with_connection(Connection::open(&path).unwrap()).unwrap();
        blank.lock().audit = Tree::new();
        assert_eq!(blank.audit_append(|seq, _| push_leaf(seq)).unwrap(), 4);

        // The log gone back under the running store. Done through SQLite,
        // not by copying an older file over this one: under WAL the store
        // would not see a copy at all (its WAL and cache still describe the
        // file it had) and would write on over it, which is why a restore
        // stops the server and moves its WAL aside first.
        Connection::open(&path)
            .unwrap()
            .execute_batch("DROP TRIGGER audit_log_no_delete; DELETE FROM audit_log WHERE seq >= 1")
            .unwrap();
        let err = server.audit_append(|seq, _| push_leaf(seq)).unwrap_err();
        assert!(format!("{err:#}").contains("restart it"), "{err:#}");
    }

    /// Reading ahead, a chunk at a time, leaves the tree where the catch-up
    /// under the write lock would: the same checkpoint, the same next seq,
    /// and the newest `at` to hold the next one to. A damaged leaf is
    /// refused here as at open.
    #[test]
    fn reading_ahead_builds_the_tree_catching_up_would() {
        let dir = tempfile::tempdir().unwrap();
        let path = dir.path().join("r.db");
        let server = Store::open(&path).unwrap();
        for _ in 0..5 {
            append(&server);
        }
        let host = Store::with_connection(Connection::open(&path).unwrap()).unwrap();
        host.lock().audit = Tree::new();
        host.lock().audit_at = String::new();
        host.audit_read_ahead_by(2).unwrap();
        assert_eq!(host.audit_checkpoint(), server.audit_checkpoint());
        assert_eq!(host.lock().audit_at, "2026-01-01T00:00:00.000Z");
        append(&server);
        host.audit_read_ahead_by(2).unwrap();
        assert_eq!(host.audit_checkpoint(), server.audit_checkpoint());
        assert_eq!(host.audit_append(|seq, _| push_leaf(seq)).unwrap(), 6);
        assert_eq!(server.audit_append(|seq, _| push_leaf(seq)).unwrap(), 7);
        assert_eq!(host.audit_append(|seq, _| push_leaf(seq)).unwrap(), 8);

        let conn = Connection::open(&path).unwrap();
        conn.execute_batch(
            "DROP TRIGGER audit_log_no_update;
             UPDATE audit_log SET leaf = CAST('forged' AS BLOB) WHERE seq = 3",
        )
        .unwrap();
        let blank =
            Store::with_connection(Connection::open(dir.path().join("x.db")).unwrap()).unwrap();
        blank.lock().conn = Connection::open(&path).unwrap();
        let err = blank.audit_read_ahead_by(2).unwrap_err();
        assert!(
            format!("{err:#}").contains("leaf 3 no longer hashes"),
            "{err:#}"
        );
    }

    /// Opening refuses a log changed behind the triggers' back: a leaf
    /// rewritten without its hash, one rewritten with it but the hash row
    /// left, or a gap.
    #[test]
    fn opening_refuses_a_damaged_log() {
        let damaged = |how: &str| -> String {
            let dir = tempfile::tempdir().unwrap();
            let path = dir.path().join("r.db");
            {
                let st = Store::open(&path).unwrap();
                for _ in 0..4 {
                    append(&st);
                }
            }
            let conn = Connection::open(&path).unwrap();
            conn.execute_batch(&format!(
                "DROP TRIGGER audit_log_no_update; DROP TRIGGER audit_log_no_delete; {how}"
            ))
            .unwrap();
            drop(conn);
            match Store::open(&path) {
                Ok(_) => panic!("opened a log damaged by {how}"),
                Err(e) => format!("{e:#}"),
            }
        };
        let e = damaged("UPDATE audit_log SET leaf = CAST('forged' AS BLOB) WHERE seq = 1");
        assert!(e.contains("leaf 1 no longer hashes"), "{e}");
        let e = damaged("DELETE FROM audit_log WHERE seq = 2");
        assert!(e.contains("leaf 2 is missing"), "{e}");
        let e = damaged("UPDATE audit_log SET leaf_hash = zeroblob(32) WHERE seq = 3");
        assert!(e.contains("leaf 3 no longer hashes"), "{e}");
    }
}