Skip to main content

recall_server/store/
audit.rs

1//! The `audit_log` table: append-only storage for the Merkle tree in
2//! [`crate::audit::merkle`], and the one transactional primitive
3//! ([`Store::audited`]) every authenticated state change goes through so
4//! its leaf commits with it.
5//!
6//! Every method on [`Store`] the server uses to change a file, a device, an
7//! authkey, a passkey, a bootstrap code or a merge job's outcome takes the
8//! leaf it appends as an argument; none can change one without it. The
9//! writes left without a leaf are deliberate, and none changes a file, a
10//! credential or what a job came to: an enrolment waiting for approval
11//! (anyone may ask, and unauthenticated routes append nothing), a
12//! machine's poll for it, a device's `last_seen`, the sweep of
13//! long-expired enrolments, an admin session started by a sign-in, used,
14//! ended by a sign-out or swept once idle, a passkey's signature counter,
15//! the release of a revoked worker's leases before the server claims them
16//! itself ([`Store::release_open_jobs`]), and the pruning of finished jobs.
17//!
18//! Another process may append: `recall-server reset-passkeys` does, on the
19//! host, and so do `recall-server admin`'s renames, removes and restores
20//! (`store/admin.rs`). Every append first reads in any leaf the table holds
21//! past this store's tree, so they all write one log rather than a fork.
22
23use anyhow::{bail, Result};
24use rusqlite::{Connection, TransactionBehavior};
25
26use super::Store;
27use crate::audit::merkle::{self, Hash, Tree};
28
29/// Created alongside `memory_files` and the device tables, every time the
30/// store opens.
31///
32/// `leaf` and `leaf_hash` are both stored: `leaf` is the exact bytes a
33/// verifier hashes and exports, and `leaf_hash` is what the leaf hashed to
34/// when it was written, which [`load`] checks it still does.
35///
36/// The triggers are what makes this append-only in the database itself,
37/// not only by convention: even a bug — or a future migration reaching for
38/// `UPDATE` out of habit — cannot rewrite, remove or skip a leaf. The
39/// insert trigger is the one that stops `INSERT OR REPLACE`, whose
40/// replacing delete does not fire a delete trigger, and an insert that
41/// leaves a gap.
42///
43/// None of it stops someone holding the database file: they can drop a
44/// trigger, or rewrite the table and every hash in it consistently. That
45/// is what a checkpoint an owner saved elsewhere is for — a rewritten log
46/// no longer extends it.
47pub(super) const SCHEMA: &str = "
48    CREATE TABLE IF NOT EXISTS audit_log (
49        seq       INTEGER PRIMARY KEY,
50        leaf      BLOB NOT NULL,
51        leaf_hash BLOB NOT NULL
52    );
53    CREATE TRIGGER IF NOT EXISTS audit_log_no_update
54        BEFORE UPDATE ON audit_log
55    BEGIN
56        SELECT RAISE(ABORT, 'audit_log is append-only');
57    END;
58    CREATE TRIGGER IF NOT EXISTS audit_log_no_delete
59        BEFORE DELETE ON audit_log
60    BEGIN
61        SELECT RAISE(ABORT, 'audit_log is append-only');
62    END;
63    CREATE TRIGGER IF NOT EXISTS audit_log_next_seq
64        BEFORE INSERT ON audit_log
65        WHEN NEW.seq IS NOT (SELECT COALESCE(MAX(seq), -1) + 1 FROM audit_log)
66    BEGIN
67        SELECT RAISE(ABORT, 'audit_log is append-only: a leaf takes the next seq');
68    END;
69";
70
71/// What [`load`] read back.
72pub(super) struct Loaded {
73    /// Every leaf's hash, as a tree.
74    pub(super) tree: Tree,
75    /// The newest leaf's `at`, or empty for an empty log.
76    pub(super) last_at: String,
77}
78
79/// Reads every leaf back at open and rebuilds the tree from them, refusing
80/// a log that is not what this server wrote: a `seq` missing or out of
81/// place, or a leaf whose bytes no longer hash to its stored `leaf_hash`.
82/// The triggers keep both from happening through SQL; this is for the
83/// file changed some other way, by a disk or by hand.
84///
85/// The hashes are recomputed from the leaves rather than trusted, which
86/// is most of what opening costs: measured at 8.5 seconds for a million
87/// leaves (a gigabyte of them) on a small VM, against 1.8 seconds to read
88/// the stored hashes alone, and 64 bytes of memory a leaf for the tree
89/// kept after. A log that size is years of one owner's syncing; the check
90/// is worth its wait. A server that started anyway would sign every later
91/// checkpoint over a tree it had already lost, so it does not start; the
92/// error says to restore the database from a backup.
93pub(super) fn load(conn: &Connection) -> Result<Loaded> {
94    let mut tree = Tree::new();
95    read_from(conn, 0, |hash| tree.append(hash))?;
96    let last_at = last_at(conn, tree.size())?;
97    Ok(Loaded { tree, last_at })
98}
99
100/// Reads every leaf from `seq` `from` on, in order, checked as [`load`]
101/// describes, handing each one's hash to `each`. What both opening and
102/// [`Store::audited_each`]'s catching up do. Answers how many it read.
103fn read_from(conn: &Connection, from: u64, each: impl FnMut(Hash)) -> Result<u64> {
104    read_range(conn, from, i64::MAX as u64, each)
105}
106
107/// [`read_from`], stopping after `most` leaves.
108fn read_range(conn: &Connection, from: u64, most: u64, mut each: impl FnMut(Hash)) -> Result<u64> {
109    let mut stmt = conn.prepare(
110        "SELECT seq, leaf, leaf_hash FROM audit_log WHERE seq >= ?1 ORDER BY seq LIMIT ?2",
111    )?;
112    let mut rows = stmt.query((from as i64, most.min(i64::MAX as u64) as i64))?;
113    let mut want = from;
114    while let Some(row) = rows.next()? {
115        let seq: i64 = row.get(0)?;
116        if seq != want as i64 {
117            bail!(
118                "the audit log is damaged: leaf {want} is missing (the next one stored is {seq}); \
119                 restore the database from a backup"
120            );
121        }
122        let leaf = row.get_ref(1)?.as_blob()?;
123        let hash = merkle::hash_leaf(leaf);
124        if row.get_ref(2)?.as_blob()? != hash.as_slice() {
125            bail!(
126                "the audit log is damaged: leaf {seq} no longer hashes to its stored leaf_hash; \
127                 restore the database from a backup"
128            );
129        }
130        each(hash);
131        want += 1;
132    }
133    Ok(want - from)
134}
135
136/// The `at` of the newest of `size` leaves, or empty for none.
137fn last_at(conn: &Connection, size: u64) -> Result<String> {
138    if size == 0 {
139        return Ok(String::new());
140    }
141    let leaf: Vec<u8> = conn.query_row(
142        "SELECT leaf FROM audit_log WHERE seq = ?1",
143        (size as i64 - 1,),
144        |r| r.get(0),
145    )?;
146    // Every leaf `leaf::encode` wrote has one; one that somehow does not
147    // only means the next `at` is not held to it.
148    Ok(serde_json::from_slice::<serde_json::Value>(&leaf)
149        .ok()
150        .and_then(|v| v.get("at")?.as_str().map(str::to_string))
151        .unwrap_or_default())
152}
153
154/// The leaves the table holds past the `held` this store's tree has, for
155/// leaves another process appended since this one last looked:
156/// `recall-server reset-passkeys` or `recall-server admin`, run on the host
157/// beside a running server.
158/// Their hashes, checked as [`load`] checks them, and the newest one's
159/// `at`, for the tree to take once the transaction this runs in commits;
160/// its lock keeps anything else from appending until then.
161///
162/// A table that holds fewer leaves than `held` was rolled back under a
163/// running server (a backup restored without stopping it), and nothing
164/// more is appended onto it until the server restarts and reads it afresh.
165fn catch_up(conn: &Connection, held: u64) -> Result<(Vec<Hash>, Option<String>)> {
166    let stored: i64 =
167        conn.query_row("SELECT COALESCE(MAX(seq) + 1, 0) FROM audit_log", [], |r| {
168            r.get(0)
169        })?;
170    if stored < held as i64 {
171        bail!(
172            "the audit log holds {stored} leaves, fewer than the {held} this server appended: \
173             the database was replaced under a running server; restart it"
174        );
175    }
176    let mut hashes = Vec::new();
177    if stored == held as i64 {
178        return Ok((hashes, None));
179    }
180    let read = read_from(conn, held, |hash| hashes.push(hash))?;
181    Ok((hashes, Some(last_at(conn, held + read)?)))
182}
183
184/// What [`Store::audited`]'s write closure hands back.
185pub enum Outcome<T> {
186    /// The write happened; its leaf is appended in the same transaction.
187    Commit(T),
188    /// Nothing happened — a refused request, such as an unknown code or one
189    /// already decided. The transaction rolls back and no leaf is
190    /// appended, matching "unauthenticated routes and refused requests
191    /// append nothing" for the requests that do carry a credential but are
192    /// refused for another reason.
193    Refuse(T),
194}
195
196/// One stored leaf, as [`Store::audit_entries`] returns it: its position and
197/// its exact bytes.
198#[derive(Debug, Clone, PartialEq, Eq)]
199pub struct AuditEntry {
200    /// Its index in the tree, from 0.
201    pub seq: u64,
202    /// The leaf exactly as written — never re-serialized.
203    pub leaf: Vec<u8>,
204}
205
206/// Why a consistency proof could not be produced.
207#[derive(Debug, Clone, Copy, PartialEq, Eq, thiserror::Error)]
208pub enum ConsistencyError {
209    /// `first` is 0, or greater than `second`.
210    #[error("first must be at least 1 and at most second")]
211    BadRange,
212    /// `second` is past the tree's current size.
213    #[error("second is past the end of the log")]
214    SecondBeyondTreeSize,
215}
216
217/// The `at` the next leaf carries: now, or the newest leaf's `at` (the
218/// later of the two given) if the clock has gone back since — so `at` never
219/// decreases along `seq`. Read under the store's lock, like the `seq` it
220/// goes with. Being one fixed-width format, two of these compare as
221/// strings.
222fn next_at(one: &str, other: &str) -> String {
223    let newest = one.max(other);
224    let now = crate::now();
225    if now.as_str() < newest {
226        newest.to_string()
227    } else {
228        now
229    }
230}
231
232impl Store {
233    /// Runs `write` inside one transaction and, only if it commits,
234    /// appends the leaf `build_leaf` makes from the position that leaf
235    /// will hold (its `seq`), its `at`, and `write`'s own result — then,
236    /// and only then, the in-memory tree learns about it.
237    ///
238    /// `write` is given the same `at`, so a row it stamps and the leaf
239    /// that records it carry one time. Both are taken under the lock the
240    /// whole transaction holds, which is what makes `at` rise with `seq`.
241    ///
242    /// This and [`Store::audited_each`] are the only places a leaf is ever
243    /// appended, so "every authenticated state change appends its leaf, in
244    /// the same transaction as the change" is true by construction: nothing
245    /// calls `INSERT INTO audit_log` any other way, and a write that never
246    /// commits — because `write` returned [`Outcome::Refuse`] or an error —
247    /// leaves neither the state change nor a leaf behind.
248    pub fn audited<T>(
249        &self,
250        write: impl FnOnce(&rusqlite::Transaction, &str) -> Result<Outcome<T>>,
251        build_leaf: impl FnOnce(u64, &str, &T) -> Vec<u8>,
252    ) -> Result<T> {
253        self.audited_each(write, |seq, at, value| vec![build_leaf(seq, at, value)])
254    }
255
256    /// [`Store::audited`], for a write that records any number of changes
257    /// at once: `build_leaves` returns one leaf per change, the first
258    /// taking `seq` and each after it the next, all appended in `write`'s
259    /// transaction. The sweep is the one user: every device it removes has
260    /// a leaf, and they commit together with the removals.
261    pub fn audited_each<T>(
262        &self,
263        write: impl FnOnce(&rusqlite::Transaction, &str) -> Result<Outcome<T>>,
264        build_leaves: impl FnOnce(u64, &str, &T) -> Vec<Vec<u8>>,
265    ) -> Result<T> {
266        self.audited_each_as(
267            anyhow::Error::from,
268            anyhow::Error::from,
269            write,
270            build_leaves,
271        )
272    }
273
274    /// [`Store::audited_each`], with the error that starting the
275    /// transaction or committing it becomes chosen by the caller:
276    /// `recall-server admin` tells the owner which of the two it was, and
277    /// that the database stayed locked, since those are the failures they
278    /// can do something about (wait, and run it again).
279    pub(crate) fn audited_each_as<T>(
280        &self,
281        begin_failed: impl FnOnce(rusqlite::Error) -> anyhow::Error,
282        commit_failed: impl FnOnce(rusqlite::Error) -> anyhow::Error,
283        write: impl FnOnce(&rusqlite::Transaction, &str) -> Result<Outcome<T>>,
284        build_leaves: impl FnOnce(u64, &str, &T) -> Vec<Vec<u8>>,
285    ) -> Result<T> {
286        let mut state = self.lock();
287        let (held, held_at) = (state.audit.size(), state.audit_at.clone());
288        // `IMMEDIATE`: the database's write lock from the start, so another
289        // process (`reset-passkeys`, `admin`) cannot append between the
290        // catch-up below and this transaction's own leaves. The transaction
291        // borrows the whole state, so the tree cannot learn of anything
292        // until it commits.
293        let tx = state
294            .conn
295            .transaction_with_behavior(TransactionBehavior::Immediate)
296            .map_err(begin_failed)?;
297        let (caught, caught_at) = catch_up(&tx, held)?;
298        // The position the first leaf will hold if this commits: the size
299        // of the tree the table holds, read under the lock the transaction
300        // holds for its whole duration, so no other request or process can
301        // claim this seq first.
302        let seq = held + caught.len() as u64;
303        let at = next_at(caught_at.as_deref().unwrap_or(""), &held_at);
304        let value = match write(&tx, &at)? {
305            Outcome::Commit(value) => value,
306            Outcome::Refuse(value) => return Ok(value), // `tx` drops here: rolled back.
307        };
308        let leaves = build_leaves(seq, &at, &value);
309        let mut hashes = Vec::with_capacity(leaves.len());
310        for (i, leaf) in leaves.iter().enumerate() {
311            let leaf_hash = merkle::hash_leaf(leaf);
312            tx.execute(
313                "INSERT INTO audit_log (seq, leaf, leaf_hash) VALUES (?1, ?2, ?3)",
314                (seq as i64 + i as i64, leaf, leaf_hash.as_slice()),
315            )?;
316            hashes.push(leaf_hash);
317        }
318        tx.commit().map_err(commit_failed)?;
319        for leaf_hash in caught.into_iter().chain(hashes) {
320            state.audit.append(leaf_hash);
321        }
322        if !leaves.is_empty() {
323            state.audit_at = at;
324        } else if let Some(caught_at) = caught_at.filter(|c| *c > state.audit_at) {
325            state.audit_at = caught_at;
326        }
327        Ok(value)
328    }
329
330    /// Reads into the tree every leaf the table holds past it, outside any
331    /// write transaction and a few thousand leaves at a time, for a process
332    /// that opened the database without reading its log: `recall-server
333    /// admin`, about to append.
334    ///
335    /// Only to be quick about it. [`Store::audited_each`] catches up with
336    /// the table by itself, but it does so holding the write lock, which a
337    /// running server then waits on for no more than its busy timeout: 5
338    /// seconds, against 8.5 to read a million leaves. Read here first, the
339    /// catch-up under the lock is only the leaves appended since, and each
340    /// read here holds a read lock for a fraction of a second. The leaves
341    /// are checked as [`load`] checks them; the log is append-only, so what
342    /// this reads stays true.
343    pub(crate) fn audit_read_ahead(&self) -> Result<()> {
344        self.audit_read_ahead_by(4096)
345    }
346
347    /// [`Store::audit_read_ahead`], `chunk` leaves at a time.
348    fn audit_read_ahead_by(&self, chunk: u64) -> Result<()> {
349        let mut state = self.lock();
350        loop {
351            let held = state.audit.size();
352            let mut hashes = Vec::new();
353            let read = read_range(&state.conn, held, chunk, |hash| hashes.push(hash))?;
354            if read == 0 {
355                return Ok(());
356            }
357            let at = last_at(&state.conn, held + read)?;
358            for hash in hashes {
359                state.audit.append(hash);
360            }
361            if at > state.audit_at {
362                state.audit_at = at;
363            }
364        }
365    }
366
367    /// Appends a leaf with no other state to change: a pull, the server's
368    /// own `start`. Still one transaction (of one statement), so it shares
369    /// [`Store::audited`]'s all-or-nothing behaviour rather than being a
370    /// special case.
371    ///
372    /// Returns the leaf's `seq`, read out of the same closure `audited`
373    /// calls under its lock — not by asking the tree its size again
374    /// afterwards, which another append could have moved on by then.
375    pub fn audit_append(&self, build_leaf: impl FnOnce(u64, &str) -> Vec<u8>) -> Result<u64> {
376        let assigned = std::cell::Cell::new(0u64);
377        self.audited(
378            |_tx, _at| Ok(Outcome::Commit(())),
379            |seq, at, ()| {
380                assigned.set(seq);
381                build_leaf(seq, at)
382            },
383        )?;
384        Ok(assigned.get())
385    }
386
387    /// The tree's size and root, for `GET /v1/audit/checkpoint` and the
388    /// `Recall-Audit-Checkpoint` header every pull carries.
389    pub fn audit_checkpoint(&self) -> (u64, Hash) {
390        let state = self.lock();
391        (state.audit.size(), state.audit.root())
392    }
393
394    /// Leaves `start` to `end - 1`, stopping early once they come to more
395    /// than `max_bytes` together — though never before the first, however
396    /// large, so a caller paging on from the last `seq` it got always
397    /// moves. The caller (the route handler) is responsible for
398    /// `end <= tree_size` and the 1,000-row page limit.
399    pub fn audit_entries(&self, start: u64, end: u64, max_bytes: usize) -> Result<Vec<AuditEntry>> {
400        let state = self.lock();
401        let mut stmt = state
402            .conn
403            .prepare("SELECT seq, leaf FROM audit_log WHERE seq >= ?1 AND seq < ?2 ORDER BY seq")?;
404        let mut rows = stmt.query((start as i64, end as i64))?;
405        let (mut out, mut bytes) = (Vec::new(), 0usize);
406        while let Some(row) = rows.next()? {
407            let leaf: Vec<u8> = row.get(1)?;
408            bytes += leaf.len();
409            if bytes > max_bytes && !out.is_empty() {
410                break;
411            }
412            out.push(AuditEntry {
413                seq: row.get::<_, i64>(0)? as u64,
414                leaf,
415            });
416        }
417        Ok(out)
418    }
419
420    /// The RFC 9162 §2.1.4 proof that the tree at `second` extends the one
421    /// at `first`, from the tree in memory: a few hundred hashes at most,
422    /// and no read of the database, however long the log.
423    pub fn audit_consistency(
424        &self,
425        first: u64,
426        second: u64,
427    ) -> Result<Result<Vec<Hash>, ConsistencyError>> {
428        let state = self.lock();
429        if first == 0 || first > second {
430            return Ok(Err(ConsistencyError::BadRange));
431        }
432        if second > state.audit.size() {
433            return Ok(Err(ConsistencyError::SecondBeyondTreeSize));
434        }
435        Ok(Ok(state.audit.consistency(first, second)))
436    }
437}
438
439#[cfg(test)]
440mod tests {
441    use super::*;
442    use crate::audit::leaf;
443
444    fn store() -> Store {
445        Store::open_in_memory().unwrap()
446    }
447
448    fn push_leaf(seq: u64) -> Vec<u8> {
449        leaf::encode(
450            seq,
451            "2026-01-01T00:00:00.000Z",
452            leaf::action::PUSH,
453            &leaf::Actor::Operator,
454            leaf::subject_file(&leaf::FileChange {
455                project_key: "acme/app",
456                file_path: "a.md",
457                deleted: false,
458                stored_sha256: "abc",
459                base_sha256: None,
460                merged: false,
461                merge_job: None,
462            }),
463            None,
464        )
465    }
466
467    fn append(st: &Store) {
468        st.audited(
469            |_tx, _| Ok(Outcome::Commit(())),
470            |seq, _, ()| push_leaf(seq),
471        )
472        .unwrap();
473    }
474
475    const INSERT_FILE: &str = "INSERT INTO memory_files \
476         (project_key, file_path, content, source_env, updated_at) VALUES ('a','b','c','d','e')";
477
478    #[test]
479    fn a_committed_write_appends_exactly_one_leaf() {
480        let st = store();
481        st.audited(
482            |tx, _| {
483                tx.execute(INSERT_FILE, [])?;
484                Ok(Outcome::Commit(()))
485            },
486            |seq, _, ()| push_leaf(seq),
487        )
488        .unwrap();
489        let (size, _) = st.audit_checkpoint();
490        assert_eq!(size, 1);
491        assert_eq!(st.audit_entries(0, 1, usize::MAX).unwrap().len(), 1);
492    }
493
494    /// The atomicity the mutation table asks for: a write that returns an
495    /// error after already changing something rolls the whole transaction
496    /// back, leaf included.
497    #[test]
498    fn a_failing_write_appends_no_leaf_and_keeps_no_change() {
499        let st = store();
500        let result = st.audited(
501            |tx, _| {
502                tx.execute(INSERT_FILE, [])?;
503                Err(anyhow::Error::from(rusqlite::Error::ExecuteReturnedResults))
504            },
505            |seq, _, ()| push_leaf(seq),
506        );
507        assert!(result.is_err());
508        assert_eq!(
509            st.audit_checkpoint().0,
510            0,
511            "no leaf from a rolled-back write"
512        );
513        assert!(st.get("a", "b").unwrap().is_none(), "no row either");
514    }
515
516    /// And the other way round: a leaf that cannot be written takes the
517    /// change down with it. The leaf goes in inside the change's own
518    /// transaction, before it commits, so there is never a moment when the
519    /// change is stored and its leaf is not.
520    #[test]
521    fn a_leaf_that_cannot_be_written_undoes_the_change() {
522        let st = store();
523        st.with_raw(|c| {
524            c.execute_batch(
525                "CREATE TEMP TRIGGER no_leaves BEFORE INSERT ON audit_log
526                 BEGIN SELECT RAISE(ABORT, 'no leaves today'); END;",
527            )
528        })
529        .unwrap();
530        let result = st.audited(
531            |tx, _| {
532                tx.execute(INSERT_FILE, [])?;
533                Ok(Outcome::Commit(()))
534            },
535            |seq, _, ()| push_leaf(seq),
536        );
537        assert!(result.is_err(), "the leaf's insert failed");
538        assert!(
539            st.get("a", "b").unwrap().is_none(),
540            "so the row is not there"
541        );
542        assert_eq!(st.audit_checkpoint().0, 0);
543    }
544
545    /// A refused request — nothing wrong at the database level, just a
546    /// decision not to proceed — is the same all-or-nothing story: no leaf.
547    #[test]
548    fn a_refused_write_appends_no_leaf() {
549        let st = store();
550        let refusal: &str = st
551            .audited(
552                |_tx, _| Ok(Outcome::Refuse("no such code")),
553                |seq, _, _| push_leaf(seq),
554            )
555            .unwrap();
556        assert_eq!(refusal, "no such code");
557        assert_eq!(st.audit_checkpoint().0, 0);
558    }
559
560    /// Several leaves from one write take consecutive seqs, and commit or
561    /// roll back together.
562    #[test]
563    fn one_write_can_append_several_leaves_in_order() {
564        let st = store();
565        append(&st);
566        st.audited_each(
567            |_tx, _| Ok(Outcome::Commit(3u64)),
568            |seq, _, n| (seq..seq + n).map(push_leaf).collect(),
569        )
570        .unwrap();
571        let seqs: Vec<u64> = st
572            .audit_entries(0, 4, usize::MAX)
573            .unwrap()
574            .iter()
575            .map(|e| e.seq)
576            .collect();
577        assert_eq!(seqs, vec![0, 1, 2, 3]);
578        assert_eq!(
579            st.audit_entries(1, 2, usize::MAX).unwrap()[0].leaf,
580            push_leaf(1)
581        );
582    }
583
584    /// `at` is taken under the lock, and never goes back: a leaf written
585    /// after the clock has moved backwards carries the newest `at` already
586    /// in the log, not an earlier one.
587    #[test]
588    fn at_never_decreases_along_seq() {
589        let st = store();
590        let mut ats = Vec::new();
591        for _ in 0..3 {
592            st.audit_append(|seq, at| {
593                ats.push(at.to_string());
594                push_leaf(seq)
595            })
596            .unwrap();
597        }
598        assert!(ats.windows(2).all(|w| w[0] <= w[1]), "{ats:?}");
599        assert_eq!(ats[0].len(), 24, "the API's timestamp format");
600
601        // A newest leaf from the future: the clock went back after it.
602        st.lock().audit_at = "2999-01-01T00:00:00.000Z".into();
603        let mut got = String::new();
604        st.audit_append(|seq, at| {
605            got = at.to_string();
606            push_leaf(seq)
607        })
608        .unwrap();
609        assert_eq!(got, "2999-01-01T00:00:00.000Z");
610    }
611
612    #[test]
613    fn checkpoint_matches_the_merkle_root_of_every_leaf() {
614        let st = store();
615        for _ in 0..5 {
616            append(&st);
617        }
618        let (size, root) = st.audit_checkpoint();
619        assert_eq!(size, 5);
620        let leaves: Vec<Hash> = (0..5)
621            .map(push_leaf)
622            .map(|l| merkle::hash_leaf(&l))
623            .collect();
624        assert_eq!(root, merkle::root(&leaves));
625    }
626
627    #[test]
628    fn entries_pages_by_seq_and_by_bytes() {
629        let st = store();
630        for _ in 0..3 {
631            append(&st);
632        }
633        let got = st.audit_entries(1, 3, usize::MAX).unwrap();
634        assert_eq!(got.iter().map(|e| e.seq).collect::<Vec<_>>(), vec![1, 2]);
635        assert_eq!(got[0].leaf, push_leaf(1));
636
637        let one = push_leaf(0).len();
638        let seqs = |max| -> Vec<u64> {
639            st.audit_entries(0, 3, max)
640                .unwrap()
641                .iter()
642                .map(|e| e.seq)
643                .collect()
644        };
645        assert_eq!(seqs(2 * one), vec![0, 1], "two fit exactly");
646        assert_eq!(seqs(2 * one - 1), vec![0], "the second would overflow");
647        assert_eq!(seqs(1), vec![0], "never fewer than one");
648    }
649
650    #[test]
651    fn consistency_matches_merkle_and_rejects_bad_ranges() {
652        let st = store();
653        for _ in 0..8 {
654            append(&st);
655        }
656        let proof = st.audit_consistency(3, 8).unwrap().unwrap();
657        let leaves: Vec<Hash> = (0..8)
658            .map(push_leaf)
659            .map(|l| merkle::hash_leaf(&l))
660            .collect();
661        assert_eq!(proof, merkle::consistency(3, 8, &leaves));
662
663        assert_eq!(
664            st.audit_consistency(0, 8).unwrap(),
665            Err(ConsistencyError::BadRange)
666        );
667        assert_eq!(
668            st.audit_consistency(5, 3).unwrap(),
669            Err(ConsistencyError::BadRange)
670        );
671        assert_eq!(
672            st.audit_consistency(1, 100).unwrap(),
673            Err(ConsistencyError::SecondBeyondTreeSize)
674        );
675    }
676
677    /// A proof comes from the tree in memory: the table can be out of reach
678    /// entirely and it is still served, which is what keeps the store's one
679    /// lock from being held over a read of every leaf.
680    #[test]
681    fn a_proof_does_not_read_the_table() {
682        let st = store();
683        for _ in 0..40 {
684            append(&st);
685        }
686        let want = st.audit_consistency(7, 40).unwrap().unwrap();
687        st.with_raw(|c| c.execute_batch("ALTER TABLE audit_log RENAME TO audit_log_away"))
688            .unwrap();
689        assert_eq!(st.audit_consistency(7, 40).unwrap().unwrap(), want);
690    }
691
692    /// The triggers, not just application discipline: a raw `UPDATE`,
693    /// `DELETE`, upsert, `INSERT OR REPLACE` over a leaf, or an insert that
694    /// skips a seq, aborts.
695    #[test]
696    fn nothing_but_the_next_leaf_can_be_written() {
697        let st = store();
698        append(&st);
699        append(&st);
700
701        let refused = |sql: &str| {
702            assert!(st.with_raw(|c| c.execute(sql, [])).is_err(), "{sql}");
703        };
704        refused("UPDATE audit_log SET seq = 99 WHERE seq = 0");
705        refused("DELETE FROM audit_log WHERE seq = 0");
706        refused(
707            "INSERT OR REPLACE INTO audit_log (seq, leaf, leaf_hash) \
708             VALUES (0, CAST('forged' AS BLOB), zeroblob(32))",
709        );
710        refused(
711            "INSERT INTO audit_log (seq, leaf, leaf_hash) VALUES (0, x'01', zeroblob(32)) \
712             ON CONFLICT(seq) DO UPDATE SET leaf = excluded.leaf",
713        );
714        refused("INSERT INTO audit_log (seq, leaf, leaf_hash) VALUES (100, x'02', zeroblob(32))");
715        refused("INSERT INTO audit_log (leaf, leaf_hash) VALUES (x'02', zeroblob(32))");
716        assert_eq!(
717            st.audit_entries(0, 2, usize::MAX).unwrap()[0].leaf,
718            push_leaf(0),
719            "leaf 0 is what was written"
720        );
721        assert_eq!(st.audit_checkpoint().0, 2);
722
723        // The next seq is still accepted — the trigger stops only the rest.
724        st.with_raw(|c| {
725            c.execute(
726                "INSERT INTO audit_log (seq, leaf, leaf_hash) VALUES (2, x'03', zeroblob(32))",
727                [],
728            )
729        })
730        .unwrap();
731    }
732
733    /// The tree is rebuilt at open from what the file holds, so a reopened
734    /// store answers with the same checkpoint and goes on from the same seq.
735    #[test]
736    fn reopening_rebuilds_the_same_tree() {
737        let dir = tempfile::tempdir().unwrap();
738        let path = dir.path().join("r.db");
739        let before = {
740            let st = Store::open(&path).unwrap();
741            for _ in 0..5 {
742                append(&st);
743            }
744            st.audit_checkpoint()
745        };
746        let st = Store::open(&path).unwrap();
747        assert_eq!(st.audit_checkpoint(), before);
748        assert_eq!(st.audit_append(|seq, _| push_leaf(seq)).unwrap(), 5);
749    }
750
751    /// A leaf another process appends to the same file, as
752    /// `recall-server reset-passkeys` does beside a running server, is read
753    /// into this one's tree before its next append, which then takes the
754    /// seq after it; the two agree on the tree after. A log that went back
755    /// under a running store stops its appends rather than growing a fork.
756    #[test]
757    fn a_leaf_another_process_appended_is_read_in_before_the_next() {
758        let dir = tempfile::tempdir().unwrap();
759        let path = dir.path().join("r.db");
760        let server = Store::open(&path).unwrap();
761        append(&server);
762        append(&server);
763        let host = Store::open(&path).unwrap();
764        assert_eq!(host.audit_append(|seq, _| push_leaf(seq)).unwrap(), 2);
765        assert_eq!(server.audit_checkpoint().0, 2, "not read until it appends");
766        assert_eq!(server.audit_append(|seq, _| push_leaf(seq)).unwrap(), 3);
767        assert_eq!(
768            server.audit_checkpoint(),
769            Store::open(&path).unwrap().audit_checkpoint()
770        );
771
772        // A store that has never looked at the log reads all of it in first.
773        let blank = Store::with_connection(Connection::open(&path).unwrap()).unwrap();
774        blank.lock().audit = Tree::new();
775        assert_eq!(blank.audit_append(|seq, _| push_leaf(seq)).unwrap(), 4);
776
777        // The file replaced by an older copy under the running store.
778        let older = dir.path().join("older.db");
779        {
780            let st = Store::open(&older).unwrap();
781            append(&st);
782        }
783        std::fs::copy(&older, &path).unwrap();
784        let err = server.audit_append(|seq, _| push_leaf(seq)).unwrap_err();
785        assert!(format!("{err:#}").contains("restart it"), "{err:#}");
786    }
787
788    /// Reading ahead, a chunk at a time, leaves the tree where the catch-up
789    /// under the write lock would: the same checkpoint, the same next seq,
790    /// and the newest `at` to hold the next one to. A damaged leaf is
791    /// refused here as at open.
792    #[test]
793    fn reading_ahead_builds_the_tree_catching_up_would() {
794        let dir = tempfile::tempdir().unwrap();
795        let path = dir.path().join("r.db");
796        let server = Store::open(&path).unwrap();
797        for _ in 0..5 {
798            append(&server);
799        }
800        let host = Store::with_connection(Connection::open(&path).unwrap()).unwrap();
801        host.lock().audit = Tree::new();
802        host.lock().audit_at = String::new();
803        host.audit_read_ahead_by(2).unwrap();
804        assert_eq!(host.audit_checkpoint(), server.audit_checkpoint());
805        assert_eq!(host.lock().audit_at, "2026-01-01T00:00:00.000Z");
806        append(&server);
807        host.audit_read_ahead_by(2).unwrap();
808        assert_eq!(host.audit_checkpoint(), server.audit_checkpoint());
809        assert_eq!(host.audit_append(|seq, _| push_leaf(seq)).unwrap(), 6);
810        assert_eq!(server.audit_append(|seq, _| push_leaf(seq)).unwrap(), 7);
811        assert_eq!(host.audit_append(|seq, _| push_leaf(seq)).unwrap(), 8);
812
813        let conn = Connection::open(&path).unwrap();
814        conn.execute_batch(
815            "DROP TRIGGER audit_log_no_update;
816             UPDATE audit_log SET leaf = CAST('forged' AS BLOB) WHERE seq = 3",
817        )
818        .unwrap();
819        let blank =
820            Store::with_connection(Connection::open(dir.path().join("x.db")).unwrap()).unwrap();
821        blank.lock().conn = Connection::open(&path).unwrap();
822        let err = blank.audit_read_ahead_by(2).unwrap_err();
823        assert!(
824            format!("{err:#}").contains("leaf 3 no longer hashes"),
825            "{err:#}"
826        );
827    }
828
829    /// Opening refuses a log changed behind the triggers' back: a leaf
830    /// rewritten without its hash, one rewritten with it but the hash row
831    /// left, or a gap.
832    #[test]
833    fn opening_refuses_a_damaged_log() {
834        let damaged = |how: &str| -> String {
835            let dir = tempfile::tempdir().unwrap();
836            let path = dir.path().join("r.db");
837            {
838                let st = Store::open(&path).unwrap();
839                for _ in 0..4 {
840                    append(&st);
841                }
842            }
843            let conn = Connection::open(&path).unwrap();
844            conn.execute_batch(&format!(
845                "DROP TRIGGER audit_log_no_update; DROP TRIGGER audit_log_no_delete; {how}"
846            ))
847            .unwrap();
848            drop(conn);
849            match Store::open(&path) {
850                Ok(_) => panic!("opened a log damaged by {how}"),
851                Err(e) => format!("{e:#}"),
852            }
853        };
854        let e = damaged("UPDATE audit_log SET leaf = CAST('forged' AS BLOB) WHERE seq = 1");
855        assert!(e.contains("leaf 1 no longer hashes"), "{e}");
856        let e = damaged("DELETE FROM audit_log WHERE seq = 2");
857        assert!(e.contains("leaf 2 is missing"), "{e}");
858        let e = damaged("UPDATE audit_log SET leaf_hash = zeroblob(32) WHERE seq = 3");
859        assert!(e.contains("leaf 3 no longer hashes"), "{e}");
860    }
861}