Skip to main content

sqlite_core/
lib.rs

1//! `sqlite-core` — native, read-only, panic-free `SQLite` file-format reader.
2//!
3//! Parses the 100-byte file header (magic + page size), walks table b-trees
4//! (interior + leaf) yielding rows as typed [`Value`]s, reassembles
5//! overflow-page chains for large payloads, walks the freelist
6//! ([`Database::freelist_pages`]), and applies a read-only `-wal` overlay
7//! ([`Database::open_with_wal`]) — all bounds-checked and panic-free on crafted
8//! input. [`Database::carve_cells`] recognizes record-shaped cells in
9//! free/unallocated space for the analyzer's deleted-record recovery. The bespoke
10//! [`WalTimeline`] ([`Database::wal_timeline`]) models a `-wal` as a salt-bounded
11//! segment of materializable [`CommitSnapshot`]s for "carve all snapshots".
12//!
13//! Format constants are consumed from [`forensicnomicon::sqlite`] (the KNOWLEDGE
14//! leaf), including the page-1 header field offsets (reserved-space 20, in-header
15//! DB-size 28, freelist-count 36, text-encoding 56) promoted there in §3.1.
16//! Index-b-tree LEAF reading is a foundation
17//! ([`Database::index_leaf_cells`], roadmap §1.4) — the second substrate for a
18//! table's data and the storage of `WITHOUT ROWID` rows; carving DELETED index
19//! entries and following index-key overflow remain follow-ups.
20//! (UTF-16 text decoding and WAL frame-checksum verification are implemented.)
21
22#![cfg_attr(test, allow(clippy::unwrap_used, clippy::expect_used))]
23
24pub mod attribution;
25pub mod rebuild;
26pub mod row_history;
27pub mod sqlcipher;
28
29// The page-1 header field offsets are consumed from the KNOWLEDGE leaf
30// (forensicnomicon::sqlite ≥ 1.5.0); the previously-local duplicates were promoted
31// there (roadmap §3.1). Aliased to the historical local names so every use site is
32// unchanged and the names read naturally in context.
33use forensicnomicon::sqlite::{
34    SQLITE_DB_SIZE_OFFSET as DB_SIZE_IN_PAGES_OFFSET,
35    SQLITE_FREELIST_COUNT_OFFSET as FREELIST_COUNT_OFFSET, SQLITE_FREELIST_TRUNK_OFFSET,
36    SQLITE_HEADER_SIZE, SQLITE_MAGIC, SQLITE_PAGE_SIZE_OFFSET,
37    SQLITE_RESERVED_SPACE_OFFSET as RESERVED_SPACE_OFFSET,
38    SQLITE_TEXT_ENCODING_OFFSET as TEXT_ENCODING_OFFSET,
39};
40
41/// Errors that can arise while reading a `SQLite` database, all recoverable —
42/// the reader never panics on malformed input.
43#[derive(Debug, Clone, PartialEq, Eq)]
44pub enum Error {
45    /// File is shorter than the 100-byte header.
46    TooShort,
47    /// First 16 bytes are not the `SQLite format 3\0` magic.
48    BadMagic,
49    /// Page-size field is not a power of two in `[512, 65536]`.
50    BadPageSize(u32),
51    /// A page number referenced by the b-tree is out of range for the file.
52    PageOutOfRange(u32),
53    /// A b-tree page had an unexpected type byte where a table page was required.
54    NotATablePage(u8),
55    /// A cell pointer or payload ran past the end of its page.
56    TruncatedCell,
57    /// The b-tree was deeper / wider than the safety cap allows.
58    TooManyPages,
59    /// The freelist trunk chain cycled or exceeded the file's page count.
60    MalformedFreelist,
61    /// An overflow-page chain cycled or exceeded the file's page count.
62    MalformedOverflow,
63    /// A rollback-journal page size was not a power of two in `[512, 65536]`.
64    /// Carries the offending value (Show-the-unrecognized-value).
65    BadJournalPageSize(u32),
66    /// A rollback journal was applied to a database opened WAL-applied, or whose
67    /// page size disagrees with the journal's. WAL and rollback-journal modes are
68    /// mutually exclusive timelines and must not be overlaid.
69    JournalModeConflict,
70    /// The file could not be opened or read (an I/O failure via
71    /// [`Database::open_path`], not a malformed database). Carries the
72    /// [`std::io::ErrorKind`] (show-the-unrecognized-value).
73    Io(std::io::ErrorKind),
74    /// `SQLCipher` decryption failed (wrong key, unsupported cipher parameters, or
75    /// a failed page authentication) via [`Database::open_encrypted`]. Carries
76    /// the underlying [`sqlcipher::DecryptError`] (show-the-unrecognized-value).
77    Decrypt(sqlcipher::DecryptError),
78}
79
80impl From<std::io::Error> for Error {
81    fn from(e: std::io::Error) -> Self {
82        Error::Io(e.kind())
83    }
84}
85
86impl From<sqlcipher::DecryptError> for Error {
87    fn from(e: sqlcipher::DecryptError) -> Self {
88        Error::Decrypt(e)
89    }
90}
91
92/// A freed overflow-page chain could not be followed to a complete, trustworthy
93/// payload (task #73): a chain page that is not a freelist leaf (live / trunk /
94/// unreachable), a cycle, a premature terminator with bytes still owed, an
95/// out-of-range page, or a declared payload exceeding the freelist's capacity.
96/// Carries no detail by design — any break is a uniform "this chain is not
97/// recoverable as a Tier-1 row", and the candidate degrades to a Tier-2 fragment.
98#[derive(Debug, Clone, Copy, PartialEq, Eq)]
99pub struct ChainBreak;
100
101/// A single decoded column value from a table row. Mirrors `SQLite`'s storage
102/// classes.
103#[derive(Debug, Clone, PartialEq)]
104pub enum Value {
105    Null,
106    Integer(i64),
107    Real(f64),
108    Text(String),
109    Blob(Vec<u8>),
110}
111
112/// One table row: its rowid plus decoded column values, in column order.
113#[derive(Debug, Clone, PartialEq)]
114pub struct Row {
115    pub rowid: i64,
116    pub values: Vec<Value>,
117}
118
119/// A live user table dumped for export: its name, the column header to present,
120/// and every live row in rowid order. Produced by [`Database::live_table_rows`].
121///
122/// `column_names` are the table's **real** column names parsed from its
123/// `CREATE TABLE` when available, falling back to generic `c0..c{N-1}` (sized to
124/// the widest row) when the schema parse was low-confidence — so a header is
125/// always present and never a fabricated guess. `rows` preserves b-tree order,
126/// which for an integer-rowid table is ascending rowid order.
127#[derive(Debug, Clone, PartialEq)]
128pub struct LiveTableDump {
129    /// Table name from `sqlite_master.name`.
130    pub name: String,
131    /// Header column names: real names from the schema, or `c0..c{N-1}`.
132    pub column_names: Vec<String>,
133    /// Every live row (rowid + decoded values), in b-tree (rowid) order.
134    pub rows: Vec<Row>,
135}
136
137/// A `WITHOUT ROWID` user table's live rows, produced by
138/// [`Database::without_rowid_table_rows`]. Such a table's data lives entirely in
139/// an index b-tree (there is no rowid), so `rows` holds the decoded index records
140/// in the table's declared column order, in index (primary-key) order.
141#[derive(Debug, Clone, PartialEq)]
142pub struct WithoutRowidTable {
143    /// Table name from `sqlite_master.name`.
144    pub name: String,
145    /// Every live row's decoded column values, in the table's column order.
146    pub rows: Vec<Vec<Value>>,
147}
148
149/// A record-shaped cell recovered from unallocated / free space by
150/// [`Database::carve_cells`]. Carries the decoded row plus enough provenance for
151/// the analyzer to grade it as a "consistent with a deleted row" observation.
152#[derive(Debug, Clone, PartialEq)]
153pub struct CarvedCell {
154    /// Byte offset of the cell within the page slice that was scanned.
155    pub offset: usize,
156    /// Total bytes the candidate cell occupies (cell header + payload), so the
157    /// scanner can skip past a recovered record.
158    pub byte_len: usize,
159    /// Decoded rowid varint.
160    pub rowid: i64,
161    /// Decoded column values, in column order.
162    pub values: Vec<Value>,
163    /// Heuristic confidence in `(0.0, 1.0]` that these bytes are a real record
164    /// rather than a coincidental match.
165    pub confidence: f32,
166}
167
168/// A **partial** deleted record salvaged from a freed-cell reconstruction that
169/// failed full-row validation: the maximal decodable column prefix at a
170/// structural anchor [`Database::reconstruct_freeblock_records`] already trusts.
171///
172/// Deliberately NOT a [`CarvedCell`]: it has no rowid (clobbered) and an
173/// incomplete value set, so the type system keeps it out of the full-row output
174/// — a fragment can never be silently rendered as a recovered row. Emitted only
175/// at an anchor where full reconstruction failed but at least one *distinctive*
176/// cell (TEXT ≥ 4 bytes of valid UTF-8, or REAL) decoded cleanly, so a lone
177/// coincidental integer pattern never anchors a fragment. Graded
178/// `FRAGMENT_CONFIDENCE` — strictly below every full-row class.
179#[derive(Debug, Clone, PartialEq)]
180pub struct CellFragment {
181    /// Byte offset of the failed cell's anchor within the scanned page slice.
182    pub offset: usize,
183    /// Bytes covered by the decoded prefix (anchor to the last decoded body byte).
184    pub byte_len: usize,
185    /// `(column_index, value)` for each column that decoded cleanly, ascending by
186    /// index. Column indexes come from the page's schema template, so they are
187    /// meaningful against the table's column order.
188    pub surviving: Vec<(usize, Value)>,
189    /// Number of the template's columns that did NOT decode (`column_count` minus
190    /// the number of surviving columns).
191    pub missing: usize,
192    /// Always `FRAGMENT_CONFIDENCE` for now; the field is kept so future
193    /// per-fragment grading does not change the public type.
194    pub confidence: f32,
195}
196
197/// A freed table-leaf cell whose declared payload **spills onto an overflow-page
198/// chain** (task #73). Recognized by `try_carve_spilled_cell_at` from the
199/// cell's intact local prefix; the chain itself is resolved separately
200/// ([`Database::read_freed_overflow_chain`]) because that needs whole-database
201/// access. A `SpilledCell` is deliberately NOT a [`CarvedCell`]: until its chain
202/// is walked and validated it cannot masquerade as a recovered row (secure by
203/// design — the type system keeps an unresolved spill out of the full-row output).
204#[derive(Debug, Clone, PartialEq)]
205pub struct SpilledCell {
206    /// Byte offset of the cell within the scanned slice.
207    pub offset: usize,
208    /// On-page footprint of the cell prefix: `n1 + n2 + local_len + 4`.
209    pub byte_len: usize,
210    /// Declared total payload length `P` (header + full body).
211    pub payload_len: usize,
212    /// Decoded rowid varint (intact-prefix anchors); `0` when the prefix was
213    /// clobbered and the rowid is unrecoverable (template path).
214    pub rowid: i64,
215    /// Full serial-type array, decoded from the local record header.
216    pub serials: Vec<i64>,
217    /// Local payload bytes kept on the leaf page (`local_payload_len(P, usable)`).
218    pub local_len: usize,
219    /// Offset, within the scanned slice, at which the local payload begins.
220    pub local_payload_off: usize,
221    /// First overflow-page number (big-endian u32 at `local_payload_off + local_len`).
222    pub first_overflow: u32,
223}
224
225/// Database text encoding (file-format §1.3, header byte 56). Determines how
226/// `TEXT` column bytes are decoded; a fixed property set at database creation.
227#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
228pub enum TextEncoding {
229    /// `1` (and `0`, an unwritten database): UTF-8.
230    #[default]
231    Utf8,
232    /// `2`: UTF-16 little-endian.
233    Utf16Le,
234    /// `3`: UTF-16 big-endian.
235    Utf16Be,
236}
237
238impl TextEncoding {
239    /// Decode a `TEXT` value's raw bytes per this encoding. Lossy so a corrupt
240    /// byte sequence yields U+FFFD rather than a panic or an error.
241    fn decode(self, bytes: &[u8]) -> String {
242        match self {
243            Self::Utf8 => String::from_utf8_lossy(bytes).into_owned(),
244            Self::Utf16Le => Self::decode_utf16(bytes, u16::from_le_bytes),
245            Self::Utf16Be => Self::decode_utf16(bytes, u16::from_be_bytes),
246        }
247    }
248
249    fn decode_utf16(bytes: &[u8], conv: fn([u8; 2]) -> u16) -> String {
250        // The DB-encoding path keeps its lossy-by-default contract: it discards
251        // the flag, so a truncated or corrupt unit still yields U+FFFD as before.
252        // The pairing itself lives in `decode_utf16_units` (DRY — the Local
253        // Storage decode reuses it and keeps the flag).
254        decode_utf16_units(bytes, conv).0
255    }
256}
257
258/// Shared UTF-16 → `String` pairing: pairs 2-byte code units via `conv`, resolves
259/// surrogate pairs, and reports whether the decode was **lossy**. A trailing odd
260/// byte (half a code unit) or an unpaired surrogate emits U+FFFD and sets the
261/// flag; it never panics or errors. Endianness is the caller's via `conv`.
262fn decode_utf16_units(bytes: &[u8], conv: fn([u8; 2]) -> u16) -> (String, bool) {
263    // An odd trailing byte is half a code unit — real data was truncated. It is
264    // dropped by `chunks_exact`; the flag records that a byte was lost.
265    let mut lossy = bytes.len() % 2 != 0;
266    let units = bytes.chunks_exact(2).map(|c| conv([c[0], c[1]]));
267    let mut text = String::new();
268    for unit in char::decode_utf16(units) {
269        if let Ok(c) = unit {
270            text.push(c);
271        } else {
272            lossy = true;
273            text.push(char::REPLACEMENT_CHARACTER);
274        }
275    }
276    (text, lossy)
277}
278
279/// A WebKit/Chrome Local Storage `ItemTable.value` decoded to text, plus whether
280/// the decode was lossy. `lossy` is a struct field, not a side-channel warning,
281/// so a caller cannot render a lossy value as if it were faithfully recovered
282/// (secure by design).
283#[derive(Debug, Clone, PartialEq, Eq, Default)]
284pub struct LocalStorageValue {
285    /// The decoded string; any code unit that could not be decoded is a U+FFFD.
286    pub text: String,
287    /// `true` when at least one input byte/unit could not be decoded cleanly (an
288    /// odd-length BLOB or an unpaired surrogate).
289    pub lossy: bool,
290}
291
292/// Decode a WebKit/Chromium Local Storage `ItemTable.value` BLOB to a `String`.
293///
294/// A `.localstorage` file is a standard `SQLite` database this crate already
295/// reads; the one artifact-specific quirk is that the `value` column is a BLOB
296/// holding the string as raw **UTF-16 little-endian** code units — no BOM, no
297/// type-prefix byte — so a normal dump surfaces it as opaque hex. This turns
298/// such a BLOB back into readable text.
299///
300/// Panic-free and lossy-by-report: an odd-length BLOB (a trailing half code
301/// unit) or an unpaired surrogate yields U+FFFD and sets
302/// [`LocalStorageValue::lossy`] rather than erroring or panicking. An empty BLOB
303/// decodes to the empty string with `lossy == false`.
304#[must_use]
305pub fn decode_localstorage_value(blob: &[u8]) -> LocalStorageValue {
306    let (text, lossy) = decode_utf16_units(blob, u16::from_le_bytes);
307    LocalStorageValue { text, lossy }
308}
309
310/// Recognize the WebKit/Chromium Local Storage `ItemTable(key TEXT, value BLOB)`
311/// table, so a caller knows when [`decode_localstorage_value`] applies to a
312/// dumped table's `value` column.
313///
314/// Keyed on the distinctive table name `ItemTable` — the name WebKit/Chromium
315/// create for Local Storage. The column names are deliberately NOT part of the
316/// test: the real schema declares them with `ON CONFLICT` clauses
317/// (`key TEXT UNIQUE ON CONFLICT REPLACE, value BLOB NOT NULL ON CONFLICT FAIL`)
318/// that a lightweight `CREATE TABLE` parse does not always split cleanly, so a
319/// name match is the robust signal. The row shape (a TEXT key, a BLOB value)
320/// still surfaces positionally in each [`Row`].
321#[must_use]
322pub fn is_local_storage_item_table(table_name: &str) -> bool {
323    table_name == "ItemTable"
324}
325
326/// Parsed 100-byte `SQLite` file header.
327#[derive(Debug, Clone, Copy, PartialEq, Eq)]
328pub struct Header {
329    /// Logical page size in bytes (512..=65536).
330    pub page_size: u32,
331    /// Reserved bytes at the end of each page (usually 0).
332    pub reserved: u8,
333    /// Text encoding for `TEXT` columns (header byte 56).
334    pub text_encoding: TextEncoding,
335}
336
337impl Header {
338    /// Usable bytes per page = `page_size` − reserved (file-format §1.3.4).
339    #[must_use]
340    pub fn usable_size(self) -> u32 {
341        self.page_size.saturating_sub(u32::from(self.reserved))
342    }
343}
344
345/// A read-only view over the raw bytes of a `SQLite` database file.
346///
347/// Holds the whole file in memory — adequate for the spike and for browser
348/// evidence DBs (tens of MB). A `Read + Seek` / mmap backend is a later
349/// refinement and does not change the parsing logic proven here.
350pub struct Database {
351    /// Page byte source: the whole file in memory ([`Database::open`]) or a
352    /// paged, LRU-cached file reader ([`Database::open_path`], roadmap §3.1).
353    source: ByteSource,
354    /// The 100-byte file header, kept resident so fixed-offset header-field reads
355    /// (page count, freelist count/trunk) never touch the byte source.
356    head: Box<[u8]>,
357    header: Header,
358    /// Read-only WAL overlay: newest committed page versions from a `-wal`
359    /// sidecar, applied without checkpointing (never mutates the main file).
360    /// `None` when opened without a WAL.
361    wal: Option<WalOverlay>,
362}
363
364/// A page image handed back by the byte source: a slice borrowed from an
365/// in-memory buffer, or a reference-counted page from the paged LRU cache.
366/// Derefs to `[u8]` so callers treat it as a page slice regardless of origin.
367///
368/// A page-*handle* rather than a `with_page(|bytes| …)` closure because the walk
369/// uses `&dyn PageSource` (a generic closure method would make that trait
370/// non-object-safe) and the recursive b-tree descent cannot hold a pinning
371/// closure across its own recursion. The `Shared` variant keeps a cached page
372/// alive while held, so LRU eviction can never dangle it.
373pub enum PageBytes<'a> {
374    /// Borrowed from an in-memory buffer (the `open` / WAL-overlay path).
375    Borrowed(&'a [u8]),
376    /// Shared out of the paged LRU cache (the `open_path` path).
377    Shared(std::rc::Rc<[u8]>),
378}
379
380impl std::ops::Deref for PageBytes<'_> {
381    type Target = [u8];
382    fn deref(&self) -> &[u8] {
383        match self {
384            PageBytes::Borrowed(s) => s,
385            PageBytes::Shared(r) => r,
386        }
387    }
388}
389
390/// Where a [`Database`]'s page bytes come from.
391enum ByteSource {
392    /// The whole file resident in memory.
393    Mem(Vec<u8>),
394    /// A file read page-by-page through a bounded LRU cache.
395    Paged(Paged),
396}
397
398impl ByteSource {
399    /// Total byte length of the underlying file.
400    fn len(&self) -> usize {
401        match self {
402            ByteSource::Mem(b) => b.len(),
403            ByteSource::Paged(p) => p.len,
404        }
405    }
406
407    /// The 1-based `page`'s bytes, or `None` for page 0 / out of range / an I/O
408    /// error. Bounded and panic-free.
409    fn page(&self, page: u32, page_size: usize) -> Option<PageBytes<'_>> {
410        let start = (page as usize).checked_sub(1)?.checked_mul(page_size)?;
411        let end = start.checked_add(page_size)?;
412        match self {
413            ByteSource::Mem(b) => b.get(start..end).map(PageBytes::Borrowed),
414            ByteSource::Paged(p) if end <= p.len => {
415                p.read_page(start, page_size).map(PageBytes::Shared)
416            }
417            ByteSource::Paged(_) => None,
418        }
419    }
420
421    /// The whole file as one slice when resident in memory; `None` for a paged
422    /// source (which never materializes the whole file). Used only on the
423    /// WAL-overlay path, which is in-memory by construction.
424    fn whole(&self) -> Option<&[u8]> {
425        match self {
426            ByteSource::Mem(b) => Some(b),
427            ByteSource::Paged(_) => None, // cov:unreachable: WAL overlay is in-memory only
428        }
429    }
430}
431
432/// A file read page-by-page through a small LRU cache, so resident memory stays
433/// bounded regardless of file size (roadmap §3.1).
434struct Paged {
435    file: std::cell::RefCell<std::fs::File>,
436    len: usize,
437    cache: std::cell::RefCell<PageCache>,
438}
439
440impl Paged {
441    /// Read `page_size` bytes at `start`, serving from and populating the LRU
442    /// cache. `None` on any I/O error (panic-free).
443    fn read_page(&self, start: usize, page_size: usize) -> Option<std::rc::Rc<[u8]>> {
444        use std::io::{Read, Seek, SeekFrom};
445        if let Some(hit) = self.cache.borrow_mut().get(start) {
446            return Some(hit);
447        }
448        let mut buf = vec![0u8; page_size];
449        {
450            let mut file = self.file.borrow_mut();
451            file.seek(SeekFrom::Start(start as u64)).ok()?;
452            file.read_exact(&mut buf).ok()?;
453        }
454        let rc: std::rc::Rc<[u8]> = std::rc::Rc::from(buf);
455        self.cache.borrow_mut().put(start, std::rc::Rc::clone(&rc));
456        Some(rc)
457    }
458}
459
460/// A tiny bounded LRU of page images keyed by file offset, capping resident
461/// memory to [`PageCache::CAP`] pages so a multi-GB database never loads whole.
462struct PageCache {
463    map: std::collections::HashMap<usize, std::rc::Rc<[u8]>>,
464    order: std::collections::VecDeque<usize>,
465}
466
467impl PageCache {
468    /// Maximum resident pages (`CAP` × `page_size` bytes; 256 pages is about one
469    /// megabyte at a 4-kilobyte page), so a multi-gigabyte database never loads whole.
470    const CAP: usize = 256;
471
472    fn new() -> Self {
473        Self {
474            map: std::collections::HashMap::new(),
475            order: std::collections::VecDeque::new(),
476        }
477    }
478
479    fn get(&mut self, key: usize) -> Option<std::rc::Rc<[u8]>> {
480        let hit = self.map.get(&key).map(std::rc::Rc::clone)?;
481        self.touch(key);
482        Some(hit)
483    }
484
485    fn put(&mut self, key: usize, value: std::rc::Rc<[u8]>) {
486        if self.map.insert(key, value).is_some() {
487            self.touch(key);
488        } else {
489            self.order.push_back(key);
490            if self.order.len() > Self::CAP {
491                if let Some(evicted) = self.order.pop_front() {
492                    self.map.remove(&evicted);
493                }
494            }
495        }
496    }
497
498    fn touch(&mut self, key: usize) {
499        if let Some(pos) = self.order.iter().position(|&k| k == key) {
500            self.order.remove(pos);
501            self.order.push_back(key);
502        }
503    }
504}
505
506/// The newest committed version of each WAL page, materialized into owned bytes.
507///
508/// Built once at open; `page_slice` consults it before the main file so a table
509/// walk transparently sees the WAL-applied view. Read-only: building it copies
510/// frame data out of the `-wal` sidecar and never writes back to either file.
511struct WalOverlay {
512    /// page number (1-based) → that page's newest committed contents.
513    pages: std::collections::BTreeMap<u32, Vec<u8>>,
514    /// Every committed frame's page image, in file order, with provenance. Unlike
515    /// `pages` (newest version per page, the consistent view), this keeps EACH
516    /// committed frame so the carver can recover deleted residue that a later
517    /// frame for the same page superseded in `pages` but that still survives in an
518    /// earlier frame's slack — the genuinely-different records an on-disk-only
519    /// carve cannot see.
520    frames: Vec<WalFramePage>,
521    /// The original `-wal` sidecar bytes, retained so [`Database::wal_timeline`]
522    /// can re-parse them into the richer segmented temporal model without the
523    /// caller re-supplying the file. Held read-only; never mutated.
524    raw: Vec<u8>,
525}
526
527/// One committed WAL frame's full page image plus its provenance, exposed by
528/// [`Database::wal_frame_pages`] so the deleted-record carver can scan the
529/// uncheckpointed WAL frames the main file does not yet reflect.
530///
531/// The `(salt1, salt2, frame_index)` triple is the WAL log-sequence identity that
532/// task #55 will formalize: `salt1`/`salt2` pin the checkpoint generation and
533/// `frame_index` the position within it.
534#[derive(Debug, Clone, PartialEq, Eq)]
535pub struct WalFramePage {
536    /// 0-based position of this frame within the `-wal` file (its LSN ordinal).
537    pub frame_index: usize,
538    /// 1-based database page number this frame rewrites.
539    pub page_no: u32,
540    /// WAL header salt-1 (checkpoint generation), shared by every live frame.
541    pub salt1: u32,
542    /// WAL header salt-2 (checkpoint generation), shared by every live frame.
543    pub salt2: u32,
544    /// Whether this is a COMMIT frame (`db_size_after_commit != 0`).
545    pub is_commit: bool,
546    /// The frame's full page image (`page_size` bytes).
547    pub page: Vec<u8>,
548}
549
550/// Hard cap on b-tree pages visited in one table walk, to bound work on a
551/// crafted file with cyclic interior pointers.
552const MAX_PAGES_PER_WALK: usize = 1_000_000;
553
554/// Minimum column count accepted when **inferring** a record's width during
555/// dropped-table carving. A coincidental byte run can look like a self-consistent
556/// 1-column record far too easily; requiring at least two columns (the smallest a
557/// real rowid table with a non-rowid column has) suppresses that false-positive
558/// class without losing real records.
559const MIN_INFERRED_COLUMNS: usize = 2;
560
561/// Confidence multiplier applied to records carved from an allocated page's
562/// in-page free space. Such residue is more often partially overwritten (its
563/// freeblock may have been reused) than whole-page freelist recovery, so it is
564/// graded a notch lower even when it parses cleanly.
565const IN_PAGE_CONFIDENCE_FACTOR: f32 = 0.8;
566
567/// Confidence multiplier applied to a **chain-reassembled overflow** full row
568/// (task #73, [`Database::carve_overflow_records`]). Overflow Tier-1 is NOT part
569/// of the structural 0-false-positive guarantee (Codex ruling #1): a freelist
570/// *leaf* page can be stale — allocated, overwritten, freed, now a leaf holding
571/// unrelated bytes that happen to decode. The freelist-leaf requirement plus the
572/// strict-UTF-8 gate make a clean decode strong evidence, but one indirection
573/// weaker than a contiguous in-page span, so it is graded below the in-page
574/// full-row tier (0.9 × this factor). The residual stale-leaf risk is documented
575/// and the row remains a "consistent with a deleted row" observation, never a
576/// verdict.
577const OVERFLOW_CHAIN_CONFIDENCE_FACTOR: f32 = 0.75;
578
579/// Confidence assigned to a record rebuilt by **freeblock reconstruction**
580/// ([`Database::reconstruct_freeblock_records`]). The cell's first four bytes
581/// (payload-length + rowid varints, the record `header_len`, and the leading
582/// serial type) were destroyed by freeblock conversion, so the record is rebuilt
583/// from its surviving serial-type tail plus a schema-derived header template — a
584/// weaker reconstruction than an intact-header carve, hence graded LOW (a
585/// "consistent with a deleted row" lead the examiner weighs, never a certainty).
586const FREEBLOCK_RECONSTRUCT_CONFIDENCE: f32 = 0.4;
587
588/// Confidence assigned to a Tier-2 [`CellFragment`] — a partial recovery whose
589/// full row could not be reconstructed but at least one distinctive cell survived.
590/// Flat 0.2 = the `MinConfidence::Low` threshold, one notch below freeblock
591/// reconstruction's 0.4 (= Medium): a fragment is the weakest lead in the ladder,
592/// "consistent with a partial deleted row", never a recovered row.
593const FRAGMENT_CONFIDENCE: f32 = 0.2;
594
595/// Upper bound on the number of freeblocks walked on a single page, to cap work
596/// on a crafted file whose freeblock `next` pointers form a long or cyclic chain.
597/// Real pages hold at most a few hundred cells.
598const MAX_FREEBLOCKS_PER_PAGE: usize = 4096;
599
600/// WAL magic, big-endian variant (native byte order in the page checksums; the
601/// little-endian variant `0x377f_0683` differs only in checksum endianness,
602/// which the overlay does not verify). file-format §4.1.
603const WAL_MAGIC_BE: u32 = 0x377f_0682;
604/// WAL magic, little-endian-checksum variant.
605const WAL_MAGIC_LE: u32 = 0x377f_0683;
606
607/// Byte order in which the WAL checksum reads its 32-bit words (file-format
608/// §4.2). NOT the same as the constant names above: per the spec, magic
609/// `0x377f0683` selects **big-endian** words and `0x377f0682` **little-endian**
610/// words. (The legacy `WAL_MAGIC_*` constant names predate this checksum work
611/// and are used only as a "valid magic" set; this enum is the spec-faithful
612/// source of truth for checksum endianness.)
613#[derive(Debug, Clone, Copy, PartialEq, Eq)]
614enum WalChecksumEndian {
615    Big,
616    Little,
617}
618
619impl WalChecksumEndian {
620    /// The checksum word order selected by the WAL header magic (offset 0), or
621    /// `None` for a magic that is neither WAL variant (file-format §4.2).
622    fn from_magic(magic: u32) -> Option<Self> {
623        match magic {
624            0x377f_0683 => Some(Self::Big),
625            0x377f_0682 => Some(Self::Little),
626            _ => None,
627        }
628    }
629
630    /// Read one 32-bit word from `b` (exactly 4 bytes) in this endianness.
631    fn read_word(self, b: [u8; 4]) -> u32 {
632        match self {
633            Self::Big => u32::from_be_bytes(b),
634            Self::Little => u32::from_le_bytes(b),
635        }
636    }
637}
638
639/// Advance the cumulative WAL checksum `(s0, s1)` over `data` (file-format
640/// §4.2). `data` is interpreted as 32-bit words in the given endianness and
641/// consumed 8 bytes (two words) at a time via the Fibonacci-weighted recurrence
642///   `s0 += x[i] + s1;  s1 += x[i+1] + s0;`
643/// using wrapping (u32) arithmetic. A trailing partial group (< 8 bytes) is
644/// ignored — the spec defines the checksum only over an even number of words,
645/// and every real WAL input (24-byte header prefix, 8-byte frame-header prefix,
646/// page data) is a multiple of 8 bytes.
647fn wal_checksum(endian: WalChecksumEndian, mut s0: u32, mut s1: u32, data: &[u8]) -> (u32, u32) {
648    let mut chunks = data.chunks_exact(8);
649    for c in &mut chunks {
650        let x0 = endian.read_word([c[0], c[1], c[2], c[3]]);
651        let x1 = endian.read_word([c[4], c[5], c[6], c[7]]);
652        s0 = s0.wrapping_add(x0).wrapping_add(s1);
653        s1 = s1.wrapping_add(x1).wrapping_add(s0);
654    }
655    (s0, s1)
656}
657
658impl Database {
659    /// Parse the file header and validate magic + page size. No WAL overlay.
660    pub fn open(bytes: Vec<u8>) -> Result<Self, Error> {
661        let header = parse_header(&bytes)?;
662        let head = header_prefix(&bytes);
663        Ok(Self {
664            source: ByteSource::Mem(bytes),
665            head,
666            header,
667            wal: None,
668        })
669    }
670
671    /// Decrypt a **`SQLCipher`** database with `key` and open the resulting
672    /// plaintext, detecting the cipher version automatically (see
673    /// [`sqlcipher::decrypt`]). The reader then consumes the decrypted byte
674    /// stream exactly as for a plaintext file — the encryption is transparent
675    /// past this call.
676    ///
677    /// Secure-by-default and read-only: a wrong key or unsupported cipher
678    /// parameters is a loud [`Error::Decrypt`], never a silently-misread
679    /// database; nothing is written back to the evidence file.
680    pub fn open_encrypted(bytes: &[u8], key: &sqlcipher::SqlCipherKey) -> Result<Self, Error> {
681        let decrypted = sqlcipher::decrypt(bytes, key)?;
682        Self::open(decrypted.plaintext)
683    }
684
685    /// Open a database from a filesystem path with a **bounded-memory paged
686    /// read** (roadmap §3.1): pages are streamed on demand through a small LRU
687    /// cache instead of loading the whole file into a `Vec<u8>`, so a multi-GB
688    /// database opens without proportional RAM. Main file only — for the
689    /// WAL-applied view use [`Database::open_with_wal`] (WAL sidecars are small
690    /// and stay in memory).
691    ///
692    /// Read-only and panic-free: an unreadable file or a malformed header is a
693    /// typed [`Error`] ([`Error::Io`] carries the [`std::io::ErrorKind`]); nothing
694    /// is written back.
695    pub fn open_path<P: AsRef<std::path::Path>>(path: P) -> Result<Self, Error> {
696        use std::io::{Read, Seek, SeekFrom};
697        let mut file = std::fs::File::open(path)?;
698        let len = file.metadata()?.len();
699        // Read just the header prefix to parse page size / encoding; the rest of
700        // the file is read page-by-page on demand.
701        let prefix_len = usize::try_from(len)
702            .unwrap_or(usize::MAX)
703            .min(SQLITE_HEADER_SIZE);
704        let mut head = vec![0u8; prefix_len];
705        file.seek(SeekFrom::Start(0))?;
706        file.read_exact(&mut head)?;
707        let header = parse_header(&head)?;
708        let source = ByteSource::Paged(Paged {
709            file: std::cell::RefCell::new(file),
710            len: usize::try_from(len).unwrap_or(usize::MAX),
711            cache: std::cell::RefCell::new(PageCache::new()),
712        });
713        Ok(Self {
714            source,
715            head: head.into(),
716            header,
717            wal: None,
718        })
719    }
720
721    /// Parse the main database plus a `-wal` sidecar, overlaying the newest
722    /// **committed** page versions from the WAL on top of the main file.
723    ///
724    /// This is the forensic-safe alternative to libsqlite checkpointing: neither
725    /// file is mutated. The resulting [`Database`] answers `read_table` with the
726    /// WAL-applied view (use [`Database::open`] for the main-only view). Frames
727    /// past the last commit frame, or whose salt does not match the WAL header,
728    /// are ignored — they are uncommitted / superseded and not part of the
729    /// consistent snapshot.
730    pub fn open_with_wal(bytes: Vec<u8>, wal: &[u8]) -> Result<Self, Error> {
731        let header = parse_header(&bytes)?;
732        let overlay = WalOverlay::parse(wal, header.page_size)?;
733        let head = header_prefix(&bytes);
734        Ok(Self {
735            source: ByteSource::Mem(bytes),
736            head,
737            header,
738            wal: overlay,
739        })
740    }
741
742    /// Materialize the single pre-transaction state from a rollback `-journal`,
743    /// binding it to THIS database (design §5). The journal's page images (the
744    /// bytes BEFORE the last transaction) are overlaid on the live pages, yielding
745    /// a [`PriorSnapshot`] — a DISTINCT read-only view, never a [`Database`], so a
746    /// prior/deleted row can never be read as "live" (secure-by-design).
747    ///
748    /// The main db's page size is authoritative (a PERSIST journal has a zeroed
749    /// header). **Errors with [`Error::JournalModeConflict`]** when `self` was
750    /// opened WAL-applied ([`Database::open_with_wal`]): WAL and rollback-journal
751    /// modes are mutually exclusive timelines and must not be overlaid.
752    ///
753    /// Robust and panic-free: a malformed/truncated journal yields a prior
754    /// snapshot with fewer overlaid pages (degrading toward the live image), never
755    /// a panic; a non-power-of-two page size is a typed
756    /// [`Error::BadJournalPageSize`].
757    pub fn rollback_prior(&self, journal: &[u8]) -> Result<PriorSnapshot, Error> {
758        if self.wal_applied() {
759            return Err(Error::JournalModeConflict);
760        }
761        let page_size = self.header.page_size;
762        let parsed = RollbackJournal::parse(journal, page_size)?;
763
764        // Start from the live main pages, then overlay the journal's prior images.
765        let main_pages = self.file_page_count();
766        let mut overlaid: std::collections::BTreeMap<u32, Vec<u8>> =
767            std::collections::BTreeMap::new();
768        for pgno in 1..=main_pages {
769            if let Some(slice) = self.raw_page(pgno) {
770                overlaid.insert(pgno, slice.to_vec());
771            }
772        }
773        let mut grew_db = false;
774        for img in parsed.page_images() {
775            if img.pgno > main_pages {
776                grew_db = true;
777            }
778            overlaid.insert(img.pgno, img.bytes.clone());
779        }
780
781        // Usable bytes per page from the PRIOR page-1 header (reserved byte @ 20),
782        // so a reserved-space change in the last txn is honored. Fall back to the
783        // live header when page 1 is not in the snapshot.
784        let reserved = overlaid
785            .get(&1)
786            .and_then(|p| p.get(RESERVED_SPACE_OFFSET).copied())
787            .unwrap_or(self.header.reserved);
788        let usable = page_size.saturating_sub(u32::from(reserved));
789        let page_bound = overlaid.keys().copied().next_back().unwrap_or(main_pages);
790
791        Ok(PriorSnapshot {
792            overlaid,
793            usable,
794            page_bound,
795            grew_db,
796        })
797    }
798
799    /// Whether a non-empty WAL overlay is in effect (at least one committed
800    /// frame was applied on top of the main file).
801    #[must_use]
802    pub fn wal_applied(&self) -> bool {
803        self.wal.as_ref().is_some_and(|w| !w.pages.is_empty())
804    }
805
806    /// Every committed `-wal` frame's page image, in file order, with provenance.
807    ///
808    /// Empty when the database was opened without a WAL (or the WAL held no
809    /// committed frames). The carver scans these page images for deleted-cell
810    /// residue that lives ONLY in the uncheckpointed WAL — the genuinely-different
811    /// records the on-disk pages do not hold — tagging each with the
812    /// `(salt1, salt2, frame_index)` log-sequence identity.
813    #[must_use]
814    pub fn wal_frame_pages(&self) -> &[WalFramePage] {
815        self.wal.as_ref().map_or(&[], |w| w.frames.as_slice())
816    }
817
818    /// Build the bespoke, format-exact [`WalTimeline`] for this database's `-wal`
819    /// sidecar, if one was supplied to [`Database::open_with_wal`].
820    ///
821    /// Returns `None` when the database was opened without a WAL, or the WAL held
822    /// no committed frame (no materializable state). The timeline enumerates the
823    /// segment's [`CommitSnapshot`]s — the only materializable database states —
824    /// each addressable by [`CommitId`]; see [`WalTimeline`].
825    ///
826    /// This consults the original `-wal` bytes retained at open time, re-parsing
827    /// them into the richer temporal model (the on-open `WalOverlay` keeps only
828    /// the consistent-view pages; the timeline keeps every segment, snapshot, and
829    /// residue tail). A page-size mismatch or malformed header surfaces as `None`
830    /// here — use [`Database::wal_timeline_from`] when you need the typed
831    /// [`WalValidationError`].
832    #[must_use]
833    pub fn wal_timeline(&self) -> Option<WalTimeline> {
834        let raw = self.wal.as_ref()?.raw.as_slice();
835        WalTimeline::parse(self.source.whole()?, raw, self.header.page_size).ok()
836    }
837
838    /// Parse a main database + `-wal` sidecar directly into a [`WalTimeline`],
839    /// surfacing the typed [`WalValidationError`] when the WAL is malformed.
840    ///
841    /// This is the validation-tier entry point: a page-size mismatch between the DB
842    /// header and the WAL header is a HARD STOP ([`WalValidationError::PageSizeMismatch`]),
843    /// not a silently mis-sliced overlay; a bad magic / unparsable header is
844    /// [`WalValidationError::BadMagic`]. Both are caught at the physical-validation
845    /// tier before any replay.
846    pub fn wal_timeline_from(bytes: &[u8], wal: &[u8]) -> Result<WalTimeline, WalValidationError> {
847        let header = parse_header(bytes).map_err(WalValidationError::Header)?;
848        WalTimeline::parse(bytes, wal, header.page_size)
849    }
850
851    #[must_use]
852    pub fn header(&self) -> Header {
853        self.header
854    }
855
856    /// Number of pages in the database file.
857    ///
858    /// Prefers the in-header DB size (offset 28) when it is a valid, non-zero
859    /// value that is consistent with the file length; otherwise falls back to
860    /// `file_len / page_size`. A mismatch between the two is itself a forensic
861    /// signal (see [`Database::header_page_count`] / [`Database::file_page_count`]).
862    #[must_use]
863    pub fn page_count(&self) -> u32 {
864        let header = self.header_page_count();
865        let file = self.file_page_count();
866        if header != 0 && header == file {
867            header
868        } else {
869            file
870        }
871    }
872
873    /// The page count recorded in the file header (offset 28). May be 0 (legacy
874    /// "size not valid" sentinel) or disagree with the file length after an
875    /// out-of-band truncation/extension.
876    #[must_use]
877    pub fn header_page_count(&self) -> u32 {
878        be_u32(&self.head, DB_SIZE_IN_PAGES_OFFSET)
879    }
880
881    /// The page count implied by the raw file length (`file_len / page_size`).
882    #[must_use]
883    pub fn file_page_count(&self) -> u32 {
884        let ps = self.header.page_size as usize;
885        u32::try_from(self.source.len() / ps).unwrap_or(u32::MAX)
886    }
887
888    /// The freelist page **count** recorded in the file header (offset 36).
889    #[must_use]
890    pub fn freelist_count(&self) -> u32 {
891        be_u32(&self.head, FREELIST_COUNT_OFFSET)
892    }
893
894    /// Walk the freelist trunk/leaf chain and return every free (unallocated)
895    /// page number, in trunk order. Free pages retain the bytes of whatever they
896    /// last held — on a `secure_delete=OFF` database that includes deleted
897    /// records, which the analyzer can carve.
898    ///
899    /// Bounded against crafted cyclic trunk chains: a page already visited, an
900    /// out-of-range page, or a leaf-pointer count larger than a trunk page can
901    /// hold aborts with [`Error::MalformedFreelist`] rather than looping.
902    pub fn freelist_pages(&self) -> Result<Vec<u32>, Error> {
903        let (leaves, trunks) = self.freelist_pages_split()?;
904        // Preserve the historical order: each trunk's leaves, then the trunk.
905        // The split sets are ordered, which is sufficient for every caller (they
906        // treat the result as a set), and keeps a single source of truth.
907        let mut free: Vec<u32> = leaves.into_iter().collect();
908        free.extend(trunks);
909        Ok(free)
910    }
911
912    /// Walk the freelist and return its **leaf** and **trunk** page numbers
913    /// separately (task #73). The distinction is load-bearing for chain-aware
914    /// overflow recovery: a freed page that became a freelist *leaf* keeps its
915    /// former content byte-for-byte, while a *trunk* page has its head
916    /// (next-trunk pointer + leaf count + leaf-number array) written over the
917    /// former content (file-format §"The Freelist"). Only leaves are
918    /// content-preserving, so [`Database::read_freed_overflow_chain`] accepts a
919    /// chain page only when it is a leaf.
920    ///
921    /// Bounded identically to [`Database::freelist_pages`]: a cyclic trunk chain,
922    /// an out-of-range page, or an over-large leaf count aborts with
923    /// [`Error::MalformedFreelist`] rather than looping.
924    pub fn freelist_pages_split(
925        &self,
926    ) -> Result<
927        (
928            std::collections::BTreeSet<u32>,
929            std::collections::BTreeSet<u32>,
930        ),
931        Error,
932    > {
933        let mut leaves = std::collections::BTreeSet::new();
934        let mut trunks = std::collections::BTreeSet::new();
935        let mut trunk = be_u32(&self.head, SQLITE_FREELIST_TRUNK_OFFSET);
936        let total_pages = self.file_page_count();
937        // Each trunk page holds at most (page_size/4 - 2) leaf pointers.
938        let max_leaves = (self.header.page_size as usize / 4).saturating_sub(2);
939        let mut visited = 0usize;
940        let cap = total_pages as usize + 1;
941
942        while trunk != 0 {
943            visited += 1;
944            if visited > cap {
945                return Err(Error::MalformedFreelist);
946            }
947            if trunk > total_pages {
948                return Err(Error::MalformedFreelist);
949            }
950            let slice = self.page_slice(trunk)?;
951            let slice = &*slice;
952            let next = be_u32(slice, 0);
953            let leaf_count = be_u32(slice, 4) as usize;
954            if leaf_count > max_leaves {
955                return Err(Error::MalformedFreelist);
956            }
957            for i in 0..leaf_count {
958                let leaf = be_u32(slice, 8 + i * 4);
959                if leaf == 0 || leaf > total_pages {
960                    return Err(Error::MalformedFreelist);
961                }
962                leaves.insert(leaf);
963            }
964            trunks.insert(trunk);
965            trunk = next;
966        }
967        Ok((leaves, trunks))
968    }
969
970    /// Follow a **freed** overflow-page chain starting at `first`, reading raw
971    /// main-file pages only (carving wants on-disk residue, not the WAL view),
972    /// and assemble up to `remaining` content bytes (task #73). The carve-side
973    /// dual of `Database::read_overflow_chain`, with one extra discipline that
974    /// makes it the 0-FP-relevant guard: **every chain page must be a freelist
975    /// leaf** (`freed_leaves`). A page that is not a leaf is live, a trunk, or
976    /// unreachable — following its pointer would risk reading reused or clobbered
977    /// content, so it is a [`ChainBreak`] (Codex ruling #2: the leaf requirement,
978    /// not the UTF-8 gate, is what rejects a destroyed chain).
979    ///
980    /// Returns the assembled content and the ordered list of chain pages on
981    /// success. Robustness (Paranoid Gatekeeper, design §4.2): the anti-bomb cap
982    /// rejects upfront any `remaining` above what the freelist leaves can deliver
983    /// (`(usable - 4) × freed_leaves.len()`), so an attacker-declared huge
984    /// payload dies before any allocation; cycles are caught by a visited set;
985    /// a premature `next == 0` with bytes still wanted, an out-of-range page, or
986    /// page 0 mid-chain all break. Never panics — every read is bounds-checked.
987    pub fn read_freed_overflow_chain(
988        &self,
989        first: u32,
990        remaining: usize,
991        usable: usize,
992        freed_leaves: &std::collections::BTreeSet<u32>,
993    ) -> Result<(Vec<u8>, Vec<u32>), ChainBreak> {
994        let per_page = usable.checked_sub(4).filter(|&p| p > 0).ok_or(ChainBreak)?;
995        // Anti-bomb cap: the chain can deliver at most this many bytes. Reject an
996        // absurd declared payload before allocating (design §4.2).
997        let max_deliverable = per_page.checked_mul(freed_leaves.len()).ok_or(ChainBreak)?;
998        if remaining > max_deliverable {
999            return Err(ChainBreak);
1000        }
1001        let total_pages = self.file_page_count();
1002        let mut content = Vec::with_capacity(remaining);
1003        let mut chain = Vec::new();
1004        let mut visited = std::collections::BTreeSet::new();
1005        let mut page = first;
1006        let mut left = remaining;
1007        while left > 0 {
1008            if page == 0 || page > total_pages {
1009                return Err(ChainBreak);
1010            }
1011            // The load-bearing guard: a chain page must be a freelist LEAF.
1012            if !freed_leaves.contains(&page) {
1013                return Err(ChainBreak);
1014            }
1015            if !visited.insert(page) {
1016                return Err(ChainBreak); // cycle
1017            }
1018            let slice = self.raw_page(page).ok_or(ChainBreak)?;
1019            let slice = &*slice;
1020            let next = be_u32(slice, 0);
1021            let take = left.min(per_page);
1022            let chunk = slice.get(4..4 + take).ok_or(ChainBreak)?;
1023            content.extend_from_slice(chunk);
1024            chain.push(page);
1025            left -= take;
1026            page = next;
1027        }
1028        Ok((content, chain))
1029    }
1030
1031    /// Raw bytes of the 1-based `page` from the **main file only**, ignoring any
1032    /// WAL overlay. Carving wants the on-disk page (where deleted residue lives),
1033    /// not the WAL-applied view. Returns `None` for page 0 or out-of-range pages.
1034    #[must_use]
1035    pub fn raw_page(&self, page: u32) -> Option<PageBytes<'_>> {
1036        if page == 0 {
1037            return None;
1038        }
1039        self.source.page(page, self.header.page_size as usize)
1040    }
1041
1042    /// Scan a slice of page bytes for record-shaped table-leaf cells of exactly
1043    /// `column_count` columns, recovering each as a [`CarvedCell`].
1044    ///
1045    /// This is the carving primitive the forensic analyzer drives over free /
1046    /// unallocated regions: at every byte offset it speculatively parses a
1047    /// `payload_len` varint, a `rowid` varint, and a record header, accepting the
1048    /// candidate only when the serial-type count matches `column_count`, the
1049    /// declared lengths stay within the slice, and every value decodes. Strict
1050    /// validation keeps the false-positive rate low; `confidence` reflects how
1051    /// strongly the bytes are record-shaped. Bounded: each offset does O(record)
1052    /// work and the scan is linear in the slice length.
1053    #[must_use]
1054    pub fn carve_cells(&self, page_bytes: &[u8], column_count: usize) -> Vec<CarvedCell> {
1055        let mut out = Vec::new();
1056        if column_count == 0 {
1057            return out;
1058        }
1059        let mut off = 0usize;
1060        while off < page_bytes.len() {
1061            if let Some(cell) = try_carve_cell_at(
1062                page_bytes,
1063                off,
1064                Some(column_count),
1065                self.header.text_encoding,
1066            ) {
1067                // Skip past this record to avoid re-reporting sub-slices of it.
1068                off += cell.byte_len.max(1);
1069                out.push(cell);
1070            } else {
1071                off += 1;
1072            }
1073        }
1074        out
1075    }
1076
1077    /// Carve record-shaped cells from a page slice **inferring** each record's
1078    /// column count from its own serial-type array, instead of requiring a fixed
1079    /// count. This is what makes **dropped-table / schema-gone** recovery
1080    /// possible: the page's table was `DROP`ped, so `sqlite_master` no longer
1081    /// records a column count, but each record still self-describes its columns.
1082    ///
1083    /// Inferring the count removes one validity check, so the remaining
1084    /// self-consistency checks are kept strict to hold the false-positive rate
1085    /// down: `header_len + body_len == payload_len`, every serial type legal,
1086    /// `rowid > 0`, the payload fully in-bounds, and at least
1087    /// `MIN_INFERRED_COLUMNS` columns. Records carved this way are graded a
1088    /// notch lower in confidence than fixed-count carving.
1089    #[must_use]
1090    pub fn carve_cells_inferred(&self, page_bytes: &[u8]) -> Vec<CarvedCell> {
1091        let mut out = Vec::new();
1092        let mut off = 0usize;
1093        while off < page_bytes.len() {
1094            if let Some(cell) = try_carve_cell_at(page_bytes, off, None, self.header.text_encoding)
1095            {
1096                off += cell.byte_len.max(1);
1097                out.push(cell);
1098            } else {
1099                off += 1;
1100            }
1101        }
1102        out
1103    }
1104
1105    /// Decode **every cell present in a table-leaf page image** (type `0x0D`) by
1106    /// walking its cell-pointer array, inferring each record's column count from
1107    /// its own serial-type array. Unlike [`Database::carve_free_regions`] (which
1108    /// scans only free space and excludes live cells), this returns the cells the
1109    /// page itself records as allocated.
1110    ///
1111    /// This is the primitive WAL-frame recovery needs: a `-wal` frame is a full
1112    /// page snapshot at one point in time, so a cell that is allocated in an
1113    /// EARLIER frame's image but absent from the final WAL-applied view is a row
1114    /// that was deleted later and survives ONLY in that superseded frame. The
1115    /// caller filters the returned cells against the final live view to isolate
1116    /// exactly those genuinely-deleted rows (so a still-live row is never
1117    /// re-surfaced — the filter is the caller's responsibility, mirroring the
1118    /// freeblock-reconstruction discipline).
1119    ///
1120    /// Bounded and panic-free: a malformed cell pointer or record simply yields
1121    /// fewer cells. Non-leaf pages yield nothing.
1122    #[must_use]
1123    pub fn carve_leaf_cells(&self, page_bytes: &[u8]) -> Vec<CarvedCell> {
1124        let hdr_off = if page_bytes.starts_with(SQLITE_MAGIC) {
1125            SQLITE_HEADER_SIZE
1126        } else {
1127            0
1128        };
1129        let Some(&page_type) = page_bytes.get(hdr_off) else {
1130            return Vec::new();
1131        };
1132        if page_type != 0x0d {
1133            return Vec::new(); // only table-leaf pages hold decodable cells here
1134        }
1135        let cell_count = be_u16(page_bytes, hdr_off + 3) as usize;
1136        let cell_ptr_array = hdr_off + 8; // leaf b-tree header is 8 bytes
1137        let mut out = Vec::new();
1138        for i in 0..cell_count {
1139            let cell_off = be_u16(page_bytes, cell_ptr_array + i * 2) as usize;
1140            if cell_off == 0 || cell_off >= page_bytes.len() {
1141                continue; // cov:unreachable: a valid leaf points cells within page
1142            }
1143            if let Some(cell) =
1144                try_carve_cell_at(page_bytes, cell_off, None, self.header.text_encoding)
1145            {
1146                out.push(cell);
1147            }
1148        }
1149        out
1150    }
1151
1152    /// Carve deleted records from the **free (unallocated) regions** of an
1153    /// allocated table-leaf page (type `0x0D`), never re-surfacing a live cell.
1154    ///
1155    /// On an allocated leaf, deleted-cell residue survives in two places: the
1156    /// unallocated gap between the cell-pointer array and the cell-content area,
1157    /// and the slack between/after live cells (a former freeblock whose chain
1158    /// pointer may already be gone). This method computes the exact byte ranges
1159    /// occupied by **live** cells and carves only the complement — so a live
1160    /// (allocated) cell can never be returned as a deleted record. That is the
1161    /// 0-false-positive guarantee, enforced structurally rather than by a filter.
1162    ///
1163    /// `page_bytes` is one whole page. `column_count_hint`, when non-zero, is the
1164    /// table's known column count (matched exactly); pass 0 to infer the count
1165    /// per record (for a page whose schema is gone). Non-leaf pages yield nothing.
1166    #[must_use]
1167    pub fn carve_free_regions(
1168        &self,
1169        page_bytes: &[u8],
1170        column_count_hint: usize,
1171    ) -> Vec<CarvedCell> {
1172        // Page 1 carries the 100-byte file header before its b-tree header; for a
1173        // standalone page slice we assume hdr_off 0 unless it starts with the
1174        // file magic (page 1 passed whole).
1175        let hdr_off = if page_bytes.starts_with(SQLITE_MAGIC) {
1176            SQLITE_HEADER_SIZE
1177        } else {
1178            0
1179        };
1180        let Some(&page_type) = page_bytes.get(hdr_off) else {
1181            return Vec::new();
1182        };
1183        if page_type != 0x0d {
1184            return Vec::new(); // only table-leaf pages have carvable cell residue
1185        }
1186        // Carve each maximal free region (complement of the live cell extents),
1187        // within the cell-content area only — so no allocated cell is ever
1188        // re-surfaced (the 0-false-positive guarantee, enforced structurally).
1189        let mut out = Vec::new();
1190        let regions = self.free_regions_of_leaf(page_bytes, hdr_off);
1191        for (lo, hi) in regions {
1192            let Some(region) = page_bytes.get(lo..hi) else {
1193                continue; // cov:unreachable: free_regions yields in-bounds spans
1194            };
1195            let cells = if column_count_hint == 0 {
1196                self.carve_cells_inferred(region)
1197            } else {
1198                self.carve_cells(region, column_count_hint)
1199            };
1200            for mut cell in cells {
1201                // Translate the offset from region-local to page-local, and grade
1202                // in-page recovery a notch lower (residue here is more often
1203                // partially overwritten than freed-page recovery).
1204                cell.offset += lo;
1205                cell.confidence *= IN_PAGE_CONFIDENCE_FACTOR;
1206                out.push(cell);
1207            }
1208        }
1209        out
1210    }
1211
1212    /// Recover **spilled** deleted records on a table-leaf page whose payload
1213    /// continued onto a freed overflow-page chain (task #73). Scans the page's
1214    /// free regions (the complement of the live cells — same discipline as
1215    /// [`Database::carve_free_regions`], so a live cell is never re-surfaced) for
1216    /// a [`SpilledCell`], then resolves each chain through freelist **leaf** pages
1217    /// only and assembles the full payload.
1218    ///
1219    /// A resolved record is returned only when ALL hold (design §5):
1220    /// 1. the chain is intact through freelist leaves (Codex ruling #2: the leaf
1221    ///    requirement is the load-bearing 0-FP guard — a trunk/live/off-freelist
1222    ///    chain page is rejected);
1223    /// 2. the assembled bytes total exactly the declared `P` and decode cleanly;
1224    /// 3. **strict UTF-8 on chain-resident TEXT** — an EXTRA reject signal, not a
1225    ///    correctness proof (Codex ruling #2: a clobbered chain can still be valid
1226    ///    UTF-8, so this cannot prove integrity; it only catches the cases where
1227    ///    the lossy decoder would otherwise mask an overwrite as `U+FFFD`).
1228    ///
1229    /// Each returned tuple is `(cell, chain)` where `chain` is the ordered list of
1230    /// overflow pages the bytes came from (for provenance). Confidence is graded
1231    /// BELOW the in-page full-row tier (Codex ruling #1: overflow Tier-1 is a
1232    /// graded recovery, NOT part of the structural 0-FP guarantee — a freelist
1233    /// leaf can be stale, holding unrelated bytes that happen to decode). Bounded
1234    /// and panic-free; a malformed page or chain simply yields fewer records.
1235    #[must_use]
1236    pub fn carve_overflow_records(&self, page_bytes: &[u8]) -> Vec<(CarvedCell, Vec<u32>)> {
1237        let hdr_off = if page_bytes.starts_with(SQLITE_MAGIC) {
1238            SQLITE_HEADER_SIZE
1239        } else {
1240            0
1241        };
1242        let Some(&page_type) = page_bytes.get(hdr_off) else {
1243            return Vec::new();
1244        };
1245        if page_type != 0x0d {
1246            return Vec::new(); // only table-leaf pages carry spilled-cell residue
1247        }
1248        let Ok((freed_leaves, _trunks)) = self.freelist_pages_split() else {
1249            return Vec::new();
1250        };
1251        let usable = self.header.usable_size() as usize;
1252
1253        let mut out = Vec::new();
1254        let regions = self.free_regions_of_leaf(page_bytes, hdr_off);
1255        for (lo, hi) in regions {
1256            let Some(region) = page_bytes.get(lo..hi) else {
1257                continue; // cov:unreachable: free_regions yields in-bounds spans
1258            };
1259            // Scan every offset for a spilled cell (recognizer abstains on in-page
1260            // payloads, so the two carve classes never overlap).
1261            let mut off = 0usize;
1262            while off < region.len() {
1263                let Some(sc) = try_carve_spilled_cell_at(region, off, usable, None) else {
1264                    off += 1;
1265                    continue;
1266                };
1267                if let Some((mut cell, chain)) =
1268                    self.resolve_spilled(region, &sc, usable, &freed_leaves)
1269                {
1270                    // Translate the region-local offset to page-local.
1271                    cell.offset = lo + sc.offset;
1272                    out.push((cell, chain));
1273                    off += sc.byte_len.max(1);
1274                } else {
1275                    off += 1;
1276                }
1277            }
1278        }
1279        out
1280    }
1281
1282    /// Resolve a recognized [`SpilledCell`] to a full [`CarvedCell`] by walking
1283    /// its freed overflow chain and decoding the assembled payload, applying the
1284    /// strict-UTF-8 chain gate. Returns `Some((cell, chain))` on a fully-validated
1285    /// recovery, `None` on any chain break or gate failure (the candidate then
1286    /// degrades to a Tier-2 fragment elsewhere).
1287    fn resolve_spilled(
1288        &self,
1289        region: &[u8],
1290        sc: &SpilledCell,
1291        usable: usize,
1292        freed_leaves: &std::collections::BTreeSet<u32>,
1293    ) -> Option<(CarvedCell, Vec<u32>)> {
1294        let remaining = sc.payload_len.checked_sub(sc.local_len)?;
1295        let local_payload =
1296            region.get(sc.local_payload_off..sc.local_payload_off + sc.local_len)?;
1297        let (chain_content, chain) = self
1298            .read_freed_overflow_chain(sc.first_overflow, remaining, usable, freed_leaves)
1299            .ok()?;
1300        let mut payload = Vec::with_capacity(sc.payload_len);
1301        payload.extend_from_slice(local_payload);
1302        payload.extend_from_slice(&chain_content);
1303        if payload.len() != sc.payload_len {
1304            return None; // cov:unreachable: chain delivers exactly `remaining` bytes
1305        }
1306
1307        let values = decode_record(
1308            &payload,
1309            sc.serials.len(),
1310            sc.rowid,
1311            self.header.text_encoding,
1312        )
1313        .ok()?;
1314        if values.len() != sc.serials.len() {
1315            return None; // cov:unreachable: decode_record yields one value per serial
1316        }
1317        // Strict-UTF-8 gate on chain-resident TEXT (extra reject signal): the
1318        // lossy decoder turns a clobbered byte into U+FFFD, so any replacement
1319        // char in a decoded TEXT value means the chain-supplied bytes did not
1320        // decode cleanly — reject. NOT a proof of integrity (a stale leaf can hold
1321        // valid UTF-8); the freelist-leaf requirement is the load-bearing guard.
1322        let any_replacement = values.iter().any(|v| match v {
1323            Value::Text(t) => t.contains('\u{FFFD}'),
1324            _ => false,
1325        });
1326        if any_replacement {
1327            return None;
1328        }
1329        // Require at least one distinctive column so a coincidental decode of stale
1330        // bytes does not anchor a full row (the same identity bar as fragments).
1331        if !values.iter().any(is_distinctive) {
1332            return None; // cov:unreachable: the spilled corpus rows carry distinctive TEXT
1333        }
1334
1335        let cell = CarvedCell {
1336            offset: sc.offset,
1337            byte_len: sc.byte_len,
1338            rowid: sc.rowid,
1339            values,
1340            // Graded below the in-page full-row tier (0.9): an overflow chain adds
1341            // one indirection of stale-leaf exposure (Codex ruling #1).
1342            confidence: 0.9 * OVERFLOW_CHAIN_CONFIDENCE_FACTOR,
1343        };
1344        Some((cell, chain))
1345    }
1346
1347    /// Reconstruct **freeblock-clobbered spilled** cells (task #73, design §2.2 /
1348    /// Codex ruling #5). When a freed cell whose payload spilled is also
1349    /// freeblock-clobbered, its declared `P` is destroyed but **re-derivable** from
1350    /// the surviving structure: `P = header_len + Σ serial_body_len` over the full
1351    /// (template + surviving) serial array. When that `P` exceeds `usable - 35` the
1352    /// record is spilled by construction, so we read the 4-byte first-overflow
1353    /// pointer that follows the local payload and resolve the chain through
1354    /// freelist leaves, exactly as the intact-prefix path does — but with
1355    /// `rowid = 0` (the prefix's rowid varint was clobbered, never invented).
1356    ///
1357    /// UNPROVEN-BY-CORPUS (Codex ruling #5): no real Nemetz `0E` cell is *both*
1358    /// freeblock-clobbered *and* spilled — every measured spilled cell kept an
1359    /// intact prefix in the unallocated gap. This path is therefore validated
1360    /// against a **synthetic** fixture only; it is the general solution the
1361    /// no-special-case rule requires (it applies the same spill formula to the
1362    /// clobbered class), but its real-data behavior is not yet observed.
1363    ///
1364    /// Returns `(cell, chain)` per fully-resolved record. Bounded and panic-free.
1365    #[must_use]
1366    pub fn carve_overflow_template_records(
1367        &self,
1368        page_bytes: &[u8],
1369    ) -> Vec<(CarvedCell, Vec<u32>)> {
1370        let hdr_off = if page_bytes.starts_with(SQLITE_MAGIC) {
1371            SQLITE_HEADER_SIZE
1372        } else {
1373            0
1374        };
1375        if page_bytes.get(hdr_off) != Some(&0x0d) {
1376            return Vec::new();
1377        }
1378        let Some(template) = freeblock_template(page_bytes, hdr_off, self.header.text_encoding)
1379        else {
1380            return Vec::new();
1381        };
1382        let Ok((freed_leaves, _trunks)) = self.freelist_pages_split() else {
1383            return Vec::new();
1384        };
1385        let usable = self.header.usable_size() as usize;
1386
1387        let mut out = Vec::new();
1388        // Walk the freeblock chain; at each freeblock head, try a clobbered-spill
1389        // reconstruction (the chain pass reaches the clobbered prefix the
1390        // intact-prefix recognizer cannot read).
1391        let first_freeblock = be_u16(page_bytes, hdr_off + 1) as usize;
1392        let mut fb = first_freeblock;
1393        let mut walked = 0usize;
1394        let mut visited = std::collections::BTreeSet::new();
1395        while fb != 0 && walked < MAX_FREEBLOCKS_PER_PAGE {
1396            walked += 1;
1397            if !visited.insert(fb) {
1398                break; // cyclic next pointer
1399            }
1400            let next = be_u16(page_bytes, fb) as usize;
1401            if let Some((cell, chain)) =
1402                template.reconstruct_spilled(self, page_bytes, fb, usable, &freed_leaves)
1403            {
1404                out.push((cell, chain));
1405            }
1406            fb = next;
1407        }
1408        out
1409    }
1410
1411    /// Tier-2 salvage for **spilled** cells whose overflow chain is broken (task
1412    /// #73, Codex ruling #4): when [`Database::carve_overflow_records`] rejects a
1413    /// recognized spilled cell because its chain failed (a trunk-clobbered or
1414    /// reused chain page), the cell's intact LOCAL prefix still holds the columns
1415    /// whose bodies fit entirely on the leaf page. Those are salvaged as a
1416    /// [`CellFragment`] — the same Tier-2 surface freeblock reconstruction uses.
1417    ///
1418    /// Only columns whose body lies wholly within the local payload are kept; the
1419    /// chain-resident columns are lost (untrusted by definition — the chain that
1420    /// would supply them is the thing that failed). A fragment is emitted only
1421    /// when the salvaged prefix carries ≥ 1 distinctive cell (TEXT ≥ 4 bytes of
1422    /// valid UTF-8, or REAL — the §3.1 gate), so a lone integer prefix never
1423    /// anchors one. Bounded and panic-free.
1424    #[must_use]
1425    pub fn carve_overflow_fragments(&self, page_bytes: &[u8]) -> Vec<CellFragment> {
1426        let hdr_off = if page_bytes.starts_with(SQLITE_MAGIC) {
1427            SQLITE_HEADER_SIZE
1428        } else {
1429            0
1430        };
1431        let Some(&page_type) = page_bytes.get(hdr_off) else {
1432            return Vec::new();
1433        };
1434        if page_type != 0x0d {
1435            return Vec::new();
1436        }
1437        let Ok((freed_leaves, _trunks)) = self.freelist_pages_split() else {
1438            return Vec::new();
1439        };
1440        let usable = self.header.usable_size() as usize;
1441
1442        let mut out = Vec::new();
1443        let regions = self.free_regions_of_leaf(page_bytes, hdr_off);
1444        for (lo, hi) in regions {
1445            let Some(region) = page_bytes.get(lo..hi) else {
1446                continue; // cov:unreachable: free_regions yields in-bounds spans
1447            };
1448            let mut off = 0usize;
1449            while off < region.len() {
1450                let Some(sc) = try_carve_spilled_cell_at(region, off, usable, None) else {
1451                    off += 1;
1452                    continue;
1453                };
1454                // Only broken chains degrade to a fragment — an intact chain is a
1455                // Tier-1 row (handled by carve_overflow_records), never both.
1456                let remaining = sc.payload_len.saturating_sub(sc.local_len);
1457                let chain_ok = self
1458                    .read_freed_overflow_chain(sc.first_overflow, remaining, usable, &freed_leaves)
1459                    .is_ok();
1460                if !chain_ok {
1461                    if let Some(mut frag) =
1462                        salvage_local_prefix(region, &sc, self.header.text_encoding)
1463                    {
1464                        frag.offset += lo;
1465                        out.push(frag);
1466                    }
1467                }
1468                off += sc.byte_len.max(1);
1469            }
1470        }
1471        out
1472    }
1473
1474    /// Reconstruct deleted records from the **freeblock chain** of an allocated
1475    /// table-leaf page (type `0x0d`) — the records a forward parse cannot recover
1476    /// because their first four bytes were destroyed by freeblock conversion.
1477    ///
1478    /// When SQLite frees an in-page cell it converts it into a **freeblock**
1479    /// (file-format §1.6): the cell's first two bytes become the next-freeblock
1480    /// offset and the next two the freeblock size, **overwriting the cell's
1481    /// payload-length + rowid varints, the record `header_len` varint, and the
1482    /// leading serial type(s)**. The record's surviving serial-type tail and its
1483    /// whole value body remain intact *after* those four bytes.
1484    ///
1485    /// This method rebuilds each freed cell from that surviving tail plus a
1486    /// **schema template** derived from a LIVE cell on the same page (the table's
1487    /// column count, header length, and the serial types of the leading columns
1488    /// that fall inside the clobbered prefix). The destroyed rowid is surfaced as
1489    /// unknown (`0`) — never invented — and the record is graded LOW.
1490    ///
1491    /// Precision discipline (task #56): a candidate is emitted only when its body
1492    /// decodes cleanly with every serial type legal AND the record fits within
1493    /// the freeblock's `[offset, offset + size)` bounds. Implausible or
1494    /// out-of-bounds candidates are rejected, so reconstruction does not
1495    /// manufacture phantom rows. (The forensic layer additionally drops any
1496    /// reconstruction whose values match a live row, so a live row is never
1497    /// re-surfaced.)
1498    ///
1499    /// Bounded and panic-free: every freeblock pointer, size, and serial length
1500    /// is range-checked against the page before use, and the chain walk is capped
1501    /// at `MAX_FREEBLOCKS_PER_PAGE` to defeat a crafted cyclic `next` chain.
1502    /// Non-leaf pages, pages with no freeblock chain, and pages with no usable
1503    /// schema template yield an empty result.
1504    #[must_use]
1505    pub fn reconstruct_freeblock_records(&self, page_bytes: &[u8]) -> Vec<CarvedCell> {
1506        // Tier-1 cells are the `.0` of the shared two-tier walker, so the full-row
1507        // output and the fragment output ([`Database::reconstruct_freeblock_fragments`])
1508        // can never diverge. The walk (freeblock-chain pass + unallocated-gap pass)
1509        // and its precision discipline live in [`reconstruct_freeblock_inner`].
1510        let _ = self;
1511        reconstruct_freeblock_inner(page_bytes, self.header.text_encoding).0
1512    }
1513
1514    /// Tier-2 partial salvage: the [`CellFragment`]s abandoned by
1515    /// [`Database::reconstruct_freeblock_records`] on this page.
1516    ///
1517    /// At every anchor where full reconstruction failed — an illegal serial in
1518    /// the surviving tail, a tail that overruns the span, or a body that does not
1519    /// fit — the columns that DID decode cleanly before the failure are salvaged
1520    /// as the maximal decodable prefix. A fragment is emitted only when that
1521    /// prefix contains at least one *distinctive* cell (TEXT ≥ 4 bytes of valid
1522    /// UTF-8, or REAL): a lone surviving integer pattern is coincidence-prone and
1523    /// never anchors a fragment.
1524    ///
1525    /// Mutually exclusive with the full reconstructions of
1526    /// [`Database::reconstruct_freeblock_records`] **by construction**: an anchor
1527    /// yields a cell or a fragment, never both. Inherits the same anchor
1528    /// discipline — no sliding scan, no strings-style hunt — so Tier-2 carries
1529    /// Tier-1's precision architecture. Bounded and panic-free identically.
1530    #[must_use]
1531    pub fn reconstruct_freeblock_fragments(&self, page_bytes: &[u8]) -> Vec<CellFragment> {
1532        let _ = self;
1533        reconstruct_freeblock_inner(page_bytes, self.header.text_encoding).1
1534    }
1535
1536    /// Parse the LIVE cells of an index-b-tree **leaf** page (type `0x0a`) into
1537    /// their decoded key records (roadmap §1.4 foundation). A regular index on a
1538    /// rowid table stores each entry as `(indexed columns…, rowid)`; a
1539    /// `WITHOUT ROWID` table stores its whole row here (the row IS the key). This
1540    /// is the structural read every later index-carve / `WITHOUT ROWID` recovery
1541    /// builds on — the second substrate for a table's data, where key columns
1542    /// survive even when the table-leaf residue is gone.
1543    ///
1544    /// Reads live cells only (via the cell-pointer array); returns empty for any
1545    /// non-index-leaf page, so a table page is never mis-read. Bounded and
1546    /// panic-free — every read is bounds-checked; a cell whose payload does not
1547    /// decode is skipped rather than panicking.
1548    ///
1549    /// SCOPE (foundation): decodes the LOCAL payload only. An index key large
1550    /// enough to spill onto an overflow-page chain is decoded up to its on-page
1551    /// bytes (the leading key columns still resolve); full overflow following, and
1552    /// carving DELETED index entries from index-page freeblocks, are follow-ups.
1553    #[must_use]
1554    pub fn index_leaf_cells(&self, page_bytes: &[u8]) -> Vec<Vec<Value>> {
1555        let hdr_off = if page_bytes.starts_with(SQLITE_MAGIC) {
1556            SQLITE_HEADER_SIZE
1557        } else {
1558            0
1559        };
1560        if page_bytes.get(hdr_off) != Some(&0x0a) {
1561            return Vec::new(); // only index-b-tree leaf pages carry index cells
1562        }
1563        let cell_count = be_u16(page_bytes, hdr_off + 3) as usize;
1564        let cell_ptr_array = hdr_off + 8; // an index-leaf header is 8 bytes
1565        let mut out = Vec::with_capacity(cell_count);
1566        for i in 0..cell_count {
1567            let ptr_off = cell_ptr_array + i * 2;
1568            if ptr_off + 1 >= page_bytes.len() {
1569                break;
1570            }
1571            let cell_off = be_u16(page_bytes, ptr_off) as usize;
1572            if cell_off == 0 || cell_off >= page_bytes.len() {
1573                continue;
1574            }
1575            // An index-leaf cell is [payload-length varint][payload][overflow?].
1576            if let Some(values) = self.index_record_at(page_bytes, cell_off) {
1577                out.push(values);
1578            }
1579        }
1580        out
1581    }
1582
1583    /// Decode the index record whose `[payload-length varint][payload]` begins at
1584    /// `off` within `page_bytes`, or `None` if it does not decode. Shared by the
1585    /// leaf read ([`index_leaf_cells`](Self::index_leaf_cells)) and the interior
1586    /// walk (whose cells also carry a key record, after the 4-byte child pointer).
1587    /// Decodes the LOCAL payload only — a key spilled to an overflow chain is
1588    /// decoded up to its on-page bytes (the leading key columns still resolve).
1589    fn index_record_at(&self, page_bytes: &[u8], off: usize) -> Option<Vec<Value>> {
1590        let (payload_len, n) = read_varint(page_bytes, off).ok()?;
1591        let payload_start = off + n;
1592        let payload_len = usize::try_from(payload_len).ok()?;
1593        let end = payload_start
1594            .saturating_add(payload_len)
1595            .min(page_bytes.len());
1596        let payload = page_bytes.get(payload_start..end)?;
1597        decode_index_payload(payload, self.header.text_encoding).ok()
1598    }
1599
1600    /// The live rows of every `WITHOUT ROWID` user table (roadmap §1.4).
1601    ///
1602    /// A `WITHOUT ROWID` table stores its whole row in an **index b-tree** — there
1603    /// is no separate table b-tree and no rowid — so the ordinary
1604    /// [`read_table`](Self::read_table) reader (which walks table pages 0x0d/0x05)
1605    /// is blind to it. This resolves each such table from `sqlite_master`, walks
1606    /// its index b-tree (interior 0x02 → leaf 0x0a), and returns its live rows,
1607    /// keyed by table name. Ordinary rowid tables are not returned.
1608    ///
1609    /// Bounded and panic-free: a malformed/cyclic b-tree stops the walk (visited
1610    /// set + page cap) rather than looping; an unreadable schema yields an empty
1611    /// result. Rows are the decoded index records, in the table's column order.
1612    #[must_use]
1613    pub fn without_rowid_table_rows(&self) -> Vec<WithoutRowidTable> {
1614        let Ok(schema) = self.read_table(1, 5) else {
1615            return Vec::new(); // cov:unreachable: a validly-opened DB has a readable page-1 schema
1616        };
1617        let mut out = Vec::new();
1618        for row in schema {
1619            // sqlite_master row: (type, name, tbl_name, rootpage, sql).
1620            let is_table = matches!(row.values.first(), Some(Value::Text(t)) if t == "table");
1621            if !is_table {
1622                continue;
1623            }
1624            let Some(Value::Text(name)) = row.values.get(1) else {
1625                continue; // cov:unreachable: a 'table' schema row has a TEXT name
1626            };
1627            if name.starts_with("sqlite_") {
1628                continue;
1629            }
1630            let sql = match row.values.get(4) {
1631                Some(Value::Text(s)) => s.as_str(),
1632                _ => "", // cov:unreachable: a 'table' schema row carries its CREATE TABLE sql
1633            };
1634            if !without_rowid_sql(sql) {
1635                continue; // ordinary rowid table — read_table handles those
1636            }
1637            let Some(Value::Integer(root)) = row.values.get(3) else {
1638                continue; // cov:unreachable: a 'table' schema row has an integer rootpage
1639            };
1640            let Ok(root) = u32::try_from(*root) else {
1641                continue; // cov:unreachable: a real rootpage is a small positive page number
1642            };
1643            let mut rows = Vec::new();
1644            let mut seen = std::collections::BTreeSet::new();
1645            self.collect_index_rows(root, &mut rows, &mut seen);
1646            out.push(WithoutRowidTable {
1647                name: name.clone(),
1648                rows,
1649            });
1650        }
1651        out
1652    }
1653
1654    /// Walk the index b-tree rooted at `page`, appending every leaf cell's decoded
1655    /// record to `rows`. Interior pages (0x02) recurse through their child pointers
1656    /// and rightmost child; leaf pages (0x0a) yield their cells via
1657    /// [`index_leaf_cells`](Self::index_leaf_cells). Bounded identically to
1658    /// [`collect_rows`](Self::collect_rows): a page is visited at most once and the
1659    /// walk is capped, so a crafted cyclic/oversized tree cannot loop.
1660    fn collect_index_rows(
1661        &self,
1662        page: u32,
1663        rows: &mut Vec<Vec<Value>>,
1664        seen: &mut std::collections::BTreeSet<u32>,
1665    ) {
1666        if page == 0 || seen.len() > MAX_PAGES_PER_WALK || !seen.insert(page) {
1667            return;
1668        }
1669        let Ok(slice) = self.page_slice(page) else {
1670            return; // cov:unreachable: schema rootpages and their children are in range
1671        };
1672        let slice = &*slice;
1673        let hdr_off = if page == 1 { SQLITE_HEADER_SIZE } else { 0 };
1674        let Some(&page_type) = slice.get(hdr_off) else {
1675            return; // cov:unreachable: a full page slice always has its header byte
1676        };
1677        match page_type {
1678            0x0a => rows.extend(self.index_leaf_cells(slice)),
1679            0x02 => {
1680                let cell_count = be_u16(slice, hdr_off + 3) as usize;
1681                let cell_ptr_array = hdr_off + 12; // an index-interior header is 12 bytes
1682                for i in 0..cell_count {
1683                    let cell_off = be_u16(slice, cell_ptr_array + i * 2) as usize;
1684                    // An interior cell is [4-byte left-child page][key record]. In
1685                    // an INDEX b-tree the key IS a real entry (a WITHOUT ROWID row),
1686                    // so decode it too — not just the child pointer, unlike a table
1687                    // b-tree where interior cells are pure navigation.
1688                    let child = be_u32(slice, cell_off);
1689                    self.collect_index_rows(child, rows, seen);
1690                    if let Some(values) = self.index_record_at(slice, cell_off + 4) {
1691                        rows.push(values);
1692                    }
1693                }
1694                let right = be_u32(slice, hdr_off + 8);
1695                self.collect_index_rows(right, rows, seen);
1696            }
1697            _ => {} // cov:unreachable: a WITHOUT ROWID b-tree page is index leaf (0x0a) or interior (0x02)
1698        }
1699    }
1700
1701    /// The maximal FREE (unallocated) byte ranges of a table-leaf page — the
1702    /// complement of its live cells within the cell-content area. Shared by
1703    /// [`Database::carve_free_regions`] and
1704    /// [`Database::reconstruct_freeblock_records`] so both scan exactly the same
1705    /// ranges and never touch a live cell. Returns empty for a non-leaf page.
1706    fn free_regions_of_leaf(&self, page_bytes: &[u8], hdr_off: usize) -> Vec<(usize, usize)> {
1707        if page_bytes.get(hdr_off) != Some(&0x0d) {
1708            return Vec::new(); // cov:unreachable: callers gate on page_type == 0x0d
1709        }
1710        let cell_count = be_u16(page_bytes, hdr_off + 3) as usize;
1711        let cell_ptr_array = hdr_off + 8; // leaf header is 8 bytes
1712        let usable = self.header.usable_size() as usize;
1713        let mut live: Vec<(usize, usize)> = Vec::with_capacity(cell_count);
1714        for i in 0..cell_count {
1715            let cell_off = be_u16(page_bytes, cell_ptr_array + i * 2) as usize;
1716            if cell_off == 0 || cell_off >= page_bytes.len() {
1717                continue; // cov:unreachable: a valid leaf points cells within page
1718            }
1719            if let Some(len) = live_cell_len(page_bytes, cell_off, usable) {
1720                live.push((cell_off, cell_off.saturating_add(len)));
1721            }
1722        }
1723        live.sort_unstable_by_key(|&(s, _)| s);
1724        let content_lo = cell_ptr_array + cell_count * 2;
1725        free_regions(&live, content_lo, page_bytes.len())
1726    }
1727
1728    /// Whether `sqlite_master` (the schema table rooted at page 1) lists at least
1729    /// one **user** table — i.e. a `type='table'` row whose name is not an
1730    /// internal `sqlite_*` table. A database where every table was `DROP`ped (or
1731    /// that never had one) returns `false`; the forensic carver uses this to label
1732    /// freed content as dropped-table residue. Errors (unreadable schema) are
1733    /// treated as "no user table" so the carver degrades safely.
1734    #[must_use]
1735    pub fn has_user_table(&self) -> bool {
1736        // sqlite_master is a 5-column table: (type, name, tbl_name, rootpage, sql).
1737        let Ok(rows) = self.read_table(1, 5) else {
1738            return false; // cov:unreachable: a validly-opened DB has a readable page-1 schema
1739        };
1740        rows.iter().any(|row| {
1741            let is_table = matches!(row.values.first(), Some(Value::Text(t)) if t == "table");
1742            let user = matches!(
1743                row.values.get(1),
1744                Some(Value::Text(n)) if !n.starts_with("sqlite_")
1745            );
1746            is_table && user
1747        })
1748    }
1749
1750    /// Collect the rowids of every **currently-live** row across all user table
1751    /// b-trees (the roots listed in `sqlite_master`). The forensic carver uses
1752    /// this to drop any carved "deleted" record whose rowid is in fact still live
1753    /// — a stale copy of a live row can linger in free space after a b-tree
1754    /// rebalance moved the row to another page, and reporting it as deleted would
1755    /// be a false positive. Rowid collection ignores the column count (the rowid
1756    /// is in the cell prefix), so it works even when a schema row is malformed.
1757    ///
1758    /// Bounded and panic-free: unreadable schema or a malformed b-tree yields a
1759    /// partial (possibly empty) set rather than an error.
1760    #[must_use]
1761    pub fn live_rowids(&self) -> std::collections::BTreeSet<i64> {
1762        let mut ids = std::collections::BTreeSet::new();
1763        let Ok(schema) = self.read_table(1, 5) else {
1764            return ids; // cov:unreachable: a validly-opened DB has a readable page-1 schema
1765        };
1766        for row in schema {
1767            // sqlite_master row: (type, name, tbl_name, rootpage, sql).
1768            let is_table = matches!(row.values.first(), Some(Value::Text(t)) if t == "table");
1769            if !is_table {
1770                continue; // cov:unreachable: the test fixtures' schemas hold only table rows
1771            }
1772            let Some(Value::Integer(root)) = row.values.get(3) else {
1773                continue; // cov:unreachable: a 'table' schema row always has an integer rootpage
1774            };
1775            let Ok(root) = u32::try_from(*root) else {
1776                continue; // cov:unreachable: a real rootpage is a small positive page number
1777            };
1778            let mut seen = std::collections::BTreeSet::new();
1779            self.collect_rowids(root, &mut ids, &mut seen);
1780        }
1781        ids
1782    }
1783
1784    /// Collect every **currently-live** row's decoded column values, keyed by
1785    /// rowid, across all user table b-trees. This is the value-aware companion to
1786    /// [`Database::live_rowids`]: the forensic carver uses it to tell a stale
1787    /// rebalance copy (same rowid AND same values → drop) from a deleted prior
1788    /// version (same rowid but DIFFERENT values → recover, e.g. an edited message
1789    /// or a changed amount).
1790    ///
1791    /// Column values are decoded by inferring the column count from each live
1792    /// cell's own serial-type array (the same self-describing record format the
1793    /// carver uses), so no schema column count is required. Best-effort,
1794    /// bounded, and panic-free: a malformed b-tree yields a partial map.
1795    #[must_use]
1796    pub fn live_rows(&self) -> std::collections::BTreeMap<i64, Vec<Value>> {
1797        let mut rows = std::collections::BTreeMap::new();
1798        let Ok(schema) = self.read_table(1, 5) else {
1799            return rows; // cov:unreachable: a validly-opened DB has a readable page-1 schema
1800        };
1801        for row in schema {
1802            let is_table = matches!(row.values.first(), Some(Value::Text(t)) if t == "table");
1803            if !is_table {
1804                continue; // cov:unreachable: the test fixtures' schemas hold only table rows
1805            }
1806            let Some(Value::Integer(root)) = row.values.get(3) else {
1807                continue; // cov:unreachable: a 'table' schema row always has an integer rootpage
1808            };
1809            let Ok(root) = u32::try_from(*root) else {
1810                continue; // cov:unreachable: a real rootpage is a small positive page number
1811            };
1812            let mut seen = std::collections::BTreeSet::new();
1813            self.collect_rows(root, &mut rows, &mut seen);
1814        }
1815        rows
1816    }
1817
1818    /// Decode every **currently-live** `sqlite_master` row (the schema table
1819    /// rooted at page 1) into its column values: `(type, name, tbl_name,
1820    /// rootpage, sql)`. This is the schema-table companion to
1821    /// [`Database::live_rows`], which collects only USER-table b-trees and so
1822    /// never sees the schema rows themselves.
1823    ///
1824    /// The forensic carver folds these into the same value-based live set it uses
1825    /// to drop stale copies of live user rows: a record carved from a materialized
1826    /// page 1 whose values equal a CURRENT schema row is the live schema entry
1827    /// re-surfaced (drop it), whereas a genuinely-deleted PRIOR schema version has
1828    /// different values (e.g. an old `CREATE TABLE`) and is still recovered.
1829    ///
1830    /// Best-effort, bounded, and panic-free: an unreadable schema yields an empty
1831    /// vector rather than an error.
1832    #[must_use]
1833    pub fn live_schema_rows(&self) -> Vec<Vec<Value>> {
1834        match self.read_table(1, 5) {
1835            Ok(rows) => rows.into_iter().map(|row| row.values).collect(),
1836            Err(_) => Vec::new(), // cov:unreachable: a validly-opened DB has a readable page-1 schema
1837        }
1838    }
1839
1840    /// Every live (schema-present) **user** table, as [`attribution::LiveTable`]:
1841    /// name, rootpage, parsed column names (or `None` when low-confidence), and
1842    /// declared column affinities. Internal `sqlite_*` tables are excluded.
1843    ///
1844    /// The forensic attribution step uses this to know each table's real column
1845    /// names (Tier-1) and its shape signature (Tier-2). Best-effort, bounded,
1846    /// panic-free: an unreadable schema yields an empty vector.
1847    #[must_use]
1848    pub fn live_tables(&self) -> Vec<attribution::LiveTable> {
1849        let mut tables = Vec::new();
1850        let Ok(schema) = self.read_table(1, 5) else {
1851            return tables; // cov:unreachable: a validly-opened DB has a readable page-1 schema
1852        };
1853        for row in schema {
1854            // sqlite_master row: (type, name, tbl_name, rootpage, sql).
1855            let is_table = matches!(row.values.first(), Some(Value::Text(t)) if t == "table");
1856            if !is_table {
1857                continue;
1858            }
1859            let Some(Value::Text(name)) = row.values.get(1) else {
1860                continue; // cov:unreachable: a 'table' schema row always has a TEXT name
1861            };
1862            if name.starts_with("sqlite_") {
1863                continue;
1864            }
1865            let Some(Value::Integer(root)) = row.values.get(3) else {
1866                continue; // cov:unreachable: a 'table' schema row always has an integer rootpage
1867            };
1868            let Ok(rootpage) = u32::try_from(*root) else {
1869                continue; // cov:unreachable: a real rootpage is a small positive page number
1870            };
1871            // The CREATE TABLE statement (column 5). A non-TEXT/absent sql is
1872            // possible on a damaged schema — degrade to no parsed columns.
1873            let sql = match row.values.get(4) {
1874                Some(Value::Text(s)) => s.as_str(),
1875                _ => "", // cov:unreachable: a 'table' schema row carries its CREATE TABLE sql
1876            };
1877            let defs = attribution::column_defs(sql);
1878            let affinities = defs.as_ref().map_or_else(Vec::new, |d| {
1879                d.iter()
1880                    .map(|(_, ty)| attribution::column_affinity(ty))
1881                    .collect()
1882            });
1883            // Only trust parsed names; if parsing failed, the caller uses c0..cN.
1884            let column_names = defs.map(|d| d.into_iter().map(|(n, _)| n).collect());
1885            tables.push(attribution::LiveTable {
1886                name: name.clone(),
1887                rootpage,
1888                column_names,
1889                affinities,
1890                create_sql: sql.to_string(),
1891            });
1892        }
1893        tables
1894    }
1895
1896    /// The live `sqlite_master` as a `name -> CREATE SQL` map for every **user**
1897    /// table (internal `sqlite_*` tables excluded) — the CURRENT-schema half of
1898    /// the Detector-B sidecar schema-change comparison
1899    /// (`docs/design/drop-recreate-attribution.md`).
1900    ///
1901    /// Reads the same page-1 schema b-tree as [`Self::live_tables`] but keeps the
1902    /// raw CREATE SQL text (not just parsed columns), so a caller can compare the
1903    /// verbatim schema against a sidecar's prior `sqlite_master`. Best-effort,
1904    /// bounded, panic-free: an unreadable schema yields an empty map.
1905    #[must_use]
1906    pub fn schema_sql(&self) -> std::collections::BTreeMap<String, String> {
1907        let mut out = std::collections::BTreeMap::new();
1908        let Ok(schema) = self.read_table(1, 5) else {
1909            return out; // cov:unreachable: a validly-opened DB has a readable page-1 schema
1910        };
1911        for row in schema {
1912            schema_sql_insert(&mut out, &row.values);
1913        }
1914        out
1915    }
1916
1917    /// Per-table, per-rowid VERSION HISTORY reconstructed from this database's WAL
1918    /// temporal model (or just the live view when no `-wal` is present).
1919    ///
1920    /// See [`row_history`] for the full model. Walks each salt epoch's commit
1921    /// snapshots in commit order, then the final live view, and emits — per rowid
1922    /// — the sequence of distinct record values it held (insert / update / delete /
1923    /// reinsert), with evidence-based [`row_history::ViewState`] and NO timestamps.
1924    /// Degrades cleanly to live-only history when [`Database::wal_timeline`] is
1925    /// `None`. `WITHOUT ROWID` tables are recorded with `without_rowid = true` and
1926    /// no versions (they have no rowid to key a history on).
1927    #[must_use]
1928    pub fn row_histories(&self) -> Vec<row_history::TableHistory> {
1929        use row_history::{RowView, VersionOrigin};
1930
1931        // Live tables: name, header columns, live rows, and a WITHOUT ROWID flag
1932        // read from the live schema (a WITHOUT ROWID table has no rowid history).
1933        let live_dumps = self.live_table_rows();
1934        let without_rowid = self.live_without_rowid_map();
1935        // WITHOUT ROWID tables' live rows (index-b-tree read); folded into each
1936        // matching history below (§1.4).
1937        let wr_rows = self.without_rowid_table_rows();
1938
1939        // Per table, build the chronological views: each WAL commit snapshot (in
1940        // epoch order, commit_seq = per-epoch ordinal) then the final live view.
1941        let mut histories = Vec::with_capacity(live_dumps.len());
1942        for dump in live_dumps {
1943            let wr = without_rowid.get(&dump.name).copied().unwrap_or(false);
1944            let mut views: Vec<RowView> = Vec::new();
1945
1946            // Historical views from the WAL timeline, if any.
1947            if let Some(timeline) = self.wal_timeline() {
1948                // commit_seq is monotonic WITHIN a salt epoch only — count per
1949                // segment, never one global sequence spanning a salt reset.
1950                let mut seq_in_segment: std::collections::BTreeMap<WalSegmentId, u32> =
1951                    std::collections::BTreeMap::new();
1952                for snapshot in timeline.commit_snapshots() {
1953                    let seg = snapshot.id().segment;
1954                    let seq = seq_in_segment.entry(seg).or_insert(0);
1955                    let commit_seq = *seq;
1956                    *seq += 1;
1957
1958                    // Resolve THIS table from the snapshot's OWN schema (a rootpage
1959                    // can be reused by a different table across commits).
1960                    let snap_tables = snapshot.tables();
1961                    let Some(st) = snap_tables.iter().find(|t| t.name == dump.name) else {
1962                        continue; // table did not exist at this commit
1963                    };
1964                    if st.without_rowid {
1965                        continue; // no rowid history for a WITHOUT ROWID table
1966                    }
1967                    // schema_known: the snapshot's CREATE TABLE parsed to columns.
1968                    let schema_known = !st.columns.is_empty();
1969                    let rows = match snapshot.read_table(st.rootpage, st.columns.len()) {
1970                        Ok(rows) => rows.into_iter().collect(),
1971                        // An unreadable historical b-tree contributes no rows but
1972                        // must not abort the whole history.
1973                        Err(_) => std::collections::BTreeMap::new(),
1974                    };
1975                    views.push(RowView {
1976                        commit_seq: Some(commit_seq),
1977                        is_final: false,
1978                        checksum_valid: snapshot.checksum_valid(),
1979                        schema_known,
1980                        origin: VersionOrigin::Commit(snapshot.id()),
1981                        rows,
1982                    });
1983                }
1984            }
1985
1986            // The final live view (current on-disk ⊕ WAL state).
1987            let live_rows: std::collections::BTreeMap<i64, Vec<Value>> = dump
1988                .rows
1989                .iter()
1990                .map(|r| (r.rowid, r.values.clone()))
1991                .collect();
1992            views.push(RowView {
1993                commit_seq: None,
1994                is_final: true,
1995                checksum_valid: true,
1996                schema_known: true,
1997                origin: VersionOrigin::Live,
1998                rows: live_rows,
1999            });
2000
2001            let mut history = row_history::table_history(dump.name, dump.column_names, wr, &views);
2002            // A WITHOUT ROWID table has no rowid version history, but its live rows
2003            // live in the index b-tree (§1.4) — read them so the carve output shows
2004            // the table's data, not just a "not version-tracked" note.
2005            if wr {
2006                if let Some(t) = wr_rows.iter().find(|t| t.name == history.table) {
2007                    history.without_rowid_rows.clone_from(&t.rows);
2008                }
2009            }
2010            histories.push(history);
2011        }
2012        histories
2013    }
2014
2015    /// Map each live user table's name to whether it is a `WITHOUT ROWID` table,
2016    /// read from the live `sqlite_master` schema. Best-effort and panic-free.
2017    fn live_without_rowid_map(&self) -> std::collections::BTreeMap<String, bool> {
2018        let mut map = std::collections::BTreeMap::new();
2019        let Ok(schema) = self.read_table(1, 5) else {
2020            return map; // cov:unreachable: a validly-opened DB has a readable page-1 schema
2021        };
2022        for row in schema {
2023            let is_table = matches!(row.values.first(), Some(Value::Text(t)) if t == "table");
2024            if !is_table {
2025                continue;
2026            }
2027            let Some(Value::Text(name)) = row.values.get(1) else {
2028                continue; // cov:unreachable: a 'table' schema row has a TEXT name
2029            };
2030            if name.starts_with("sqlite_") {
2031                continue;
2032            }
2033            let sql = match row.values.get(4) {
2034                Some(Value::Text(s)) => s.as_str(),
2035                _ => "", // cov:unreachable: a 'table' schema row carries its CREATE TABLE sql
2036            };
2037            map.insert(name.clone(), without_rowid_sql(sql));
2038        }
2039        map
2040    }
2041
2042    /// The `sqlite_sequence` table `SQLite` maintains for `AUTOINCREMENT` tables,
2043    /// as `name → seq` — `seq` being the highest rowid ever assigned to that table
2044    /// (its monotonic INSERT high-water mark).
2045    ///
2046    /// `sqlite_sequence` exists **only** once at least one `AUTOINCREMENT` table
2047    /// has been created; a database with none returns an **empty** map (never a
2048    /// fabricated `seq = 0`), so a caller can distinguish "no high-water mark" from
2049    /// "high-water mark of 0". Best-effort, bounded, panic-free: an unreadable
2050    /// `sqlite_sequence` b-tree, or a malformed row, is omitted rather than
2051    /// erroring. Note `sqlite_sequence` is a mutable user table — `seq` tracks the
2052    /// INSERT high-water mark, not live rowid assignment — so this is a forensic
2053    /// HINT input, not proof of any row's provenance.
2054    #[must_use]
2055    pub fn sqlite_sequence(&self) -> std::collections::BTreeMap<String, i64> {
2056        let mut map = std::collections::BTreeMap::new();
2057        let Ok(schema) = self.read_table(1, 5) else {
2058            return map; // cov:unreachable: a validly-opened DB has a readable page-1 schema
2059        };
2060        // Locate the sqlite_sequence table's rootpage from the schema.
2061        let mut rootpage: Option<u32> = None;
2062        for row in &schema {
2063            let is_table = matches!(row.values.first(), Some(Value::Text(t)) if t == "table");
2064            if !is_table {
2065                continue;
2066            }
2067            if !matches!(row.values.get(1), Some(Value::Text(n)) if n == "sqlite_sequence") {
2068                continue;
2069            }
2070            if let Some(Value::Integer(root)) = row.values.get(3) {
2071                rootpage = u32::try_from(*root).ok();
2072            }
2073            break;
2074        }
2075        let Some(root) = rootpage else {
2076            return map; // no AUTOINCREMENT table ⟹ no sqlite_sequence ⟹ empty
2077        };
2078        let Ok(rows) = self.read_table(root, 2) else {
2079            return map; // cov:unreachable: a present sqlite_sequence has a readable b-tree
2080        };
2081        for row in rows {
2082            // sqlite_sequence row: (name TEXT, seq INTEGER). A malformed row (wrong
2083            // types) is skipped — never a fabricated entry.
2084            let (Some(Value::Text(name)), Some(Value::Integer(seq))) =
2085                (row.values.first(), row.values.get(1))
2086            else {
2087                continue;
2088            };
2089            map.insert(name.clone(), *seq);
2090        }
2091        map
2092    }
2093
2094    /// Dump every live user table for export: name, header columns, and all live
2095    /// rows in rowid order. The base layer the combined live + recovered workbook
2096    /// is built over.
2097    ///
2098    /// For each [`Database::live_tables`] entry, the b-tree is read via
2099    /// [`Database::read_table`] (so rows arrive in ascending-rowid b-tree order).
2100    /// The header is the table's **real** column names when the schema parse was
2101    /// confident, otherwise generic `c0..c{N-1}` sized to the widest row — a
2102    /// header is always present and never a fabricated name. Best-effort and
2103    /// panic-free: a table whose b-tree is unreadable contributes an empty row set
2104    /// rather than erroring.
2105    #[must_use]
2106    pub fn live_table_rows(&self) -> Vec<LiveTableDump> {
2107        self.live_tables()
2108            .into_iter()
2109            .map(|table| {
2110                // `read_table`'s column_count drives only the INTEGER PRIMARY KEY
2111                // rowid-alias rule; use the declared arity when known, else 0
2112                // (no alias substitution) so a low-confidence schema still dumps.
2113                let declared = table.column_names.as_ref().map_or(0, Vec::len);
2114                let rows = self
2115                    .read_table(table.rootpage, declared)
2116                    .unwrap_or_default();
2117                let widest = rows.iter().map(|r| r.values.len()).max().unwrap_or(0);
2118                let column_names = match table.column_names {
2119                    // Confident schema parse: use the table's real column names.
2120                    // Live rows legitimately omit trailing NULLs, so `widest` may
2121                    // be < declared — the real header still governs (a recovered
2122                    // row pads/truncates to it).
2123                    Some(names) => names,
2124                    // Low-confidence parse (malformed/unparseable CREATE TABLE):
2125                    // generic header sized to the widest row, never a fabricated
2126                    // real name. This is the schema-damage robustness guard.
2127                    None => (0..widest).map(|i| format!("c{i}")).collect(),
2128                };
2129                LiveTableDump {
2130                    name: table.name,
2131                    column_names,
2132                    rows,
2133                }
2134            })
2135            .collect()
2136    }
2137
2138    /// A map from each **allocated** page that belongs to a live table's b-tree
2139    /// to that table's name. Built by walking every live table's b-tree page set
2140    /// from its rootpage (interior + leaf pages). A page carved as Tier-1
2141    /// in-page residue resolves to its owning table through this map.
2142    ///
2143    /// Best-effort and bounded, mirroring `live_rowids`'s b-tree walk: a
2144    /// malformed b-tree contributes fewer entries rather than erroring.
2145    #[must_use]
2146    pub fn page_to_table_map(&self) -> std::collections::BTreeMap<u32, String> {
2147        let mut map = std::collections::BTreeMap::new();
2148        for table in self.live_tables() {
2149            let mut pages = std::collections::BTreeSet::new();
2150            let mut visited = 0usize;
2151            self.collect_pages(table.rootpage, &mut pages, &mut visited);
2152            for page in pages {
2153                map.insert(page, table.name.clone());
2154            }
2155        }
2156        map
2157    }
2158
2159    /// Walk the table b-tree rooted at `page`, inserting every page it visits
2160    /// (interior + leaf) into `pages`. Best-effort and bounded, mirroring
2161    /// `collect_rowids`.
2162    fn collect_pages(
2163        &self,
2164        page: u32,
2165        pages: &mut std::collections::BTreeSet<u32>,
2166        visited: &mut usize,
2167    ) {
2168        *visited += 1;
2169        if *visited > MAX_PAGES_PER_WALK {
2170            return; // cov:unreachable: test b-trees are far below the 1M-page cap
2171        }
2172        if page == 0 || !pages.insert(page) {
2173            return; // page 0 sentinel, or already visited (cycle guard)
2174        }
2175        let Ok(slice) = self.page_slice(page) else {
2176            return; // cov:unreachable: schema rootpages and their children are in range
2177        };
2178        let slice = &*slice; // PageBytes -> &[u8]; body below is source-agnostic
2179        let hdr_off = if page == 1 { SQLITE_HEADER_SIZE } else { 0 };
2180        let Some(&page_type) = slice.get(hdr_off) else {
2181            return; // cov:unreachable: a full page slice always has its header byte
2182        };
2183        if page_type != 0x05 {
2184            return; // leaf (0x0d) or non-interior: no children to descend
2185        }
2186        let cell_count = be_u16(slice, hdr_off + 3) as usize;
2187        let cell_ptr_array = hdr_off + 12;
2188        for i in 0..cell_count {
2189            let cell_off = be_u16(slice, cell_ptr_array + i * 2) as usize;
2190            let child = be_u32(slice, cell_off);
2191            self.collect_pages(child, pages, visited);
2192        }
2193        let right = be_u32(slice, hdr_off + 8);
2194        self.collect_pages(right, pages, visited);
2195    }
2196
2197    /// Walk the table b-tree rooted at `page`, decoding every live leaf cell's
2198    /// values (column count inferred per cell) into `rows` keyed by rowid.
2199    /// Best-effort and bounded, mirroring [`Database::collect_rowids`].
2200    fn collect_rows(
2201        &self,
2202        page: u32,
2203        rows: &mut std::collections::BTreeMap<i64, Vec<Value>>,
2204        seen: &mut std::collections::BTreeSet<u32>,
2205    ) {
2206        // Visit each page at most once. A manipulated interior left-child or
2207        // right-most pointer (anti-forensic corpus category 12) can point back
2208        // into an already-visited page, and a counter-only guard would still
2209        // recurse a million frames deep before stopping — a stack overflow. The
2210        // visited-set bounds recursion DEPTH to the number of distinct pages,
2211        // mirroring `collect_pages`'s cycle guard.
2212        if page == 0 || seen.len() > MAX_PAGES_PER_WALK || !seen.insert(page) {
2213            return;
2214        }
2215        let Ok(slice) = self.page_slice(page) else {
2216            return; // cov:unreachable: schema rootpages and their children are in range
2217        };
2218        let slice = &*slice; // PageBytes -> &[u8]; body below is source-agnostic
2219        let hdr_off = if page == 1 { SQLITE_HEADER_SIZE } else { 0 };
2220        let Some(&page_type) = slice.get(hdr_off) else {
2221            return; // cov:unreachable: a full page slice always has its header byte
2222        };
2223        let cell_count = be_u16(slice, hdr_off + 3) as usize;
2224        match page_type {
2225            0x0d => {
2226                let cell_ptr_array = hdr_off + 8;
2227                for i in 0..cell_count {
2228                    let cell_off = be_u16(slice, cell_ptr_array + i * 2) as usize;
2229                    // Decode the live cell with an inferred column count; on any
2230                    // parse hiccup (e.g. a table narrower than MIN_INFERRED_COLUMNS),
2231                    // fall back to the rowid alone (empty values) so the row is
2232                    // still known to be live.
2233                    if let Some(cell) =
2234                        try_carve_cell_at(slice, cell_off, None, self.header.text_encoding)
2235                    {
2236                        rows.insert(cell.rowid, cell.values);
2237                    } else if let Some(rowid) = live_cell_rowid(slice, cell_off) {
2238                        rows.entry(rowid).or_default(); // cov:unreachable: a >=2-col live cell always decodes above
2239                    }
2240                }
2241            }
2242            0x05 => {
2243                let cell_ptr_array = hdr_off + 12;
2244                for i in 0..cell_count {
2245                    let cell_off = be_u16(slice, cell_ptr_array + i * 2) as usize;
2246                    let child = be_u32(slice, cell_off);
2247                    self.collect_rows(child, rows, seen);
2248                }
2249                let right = be_u32(slice, hdr_off + 8);
2250                self.collect_rows(right, rows, seen);
2251            }
2252            _ => {} // cov:unreachable: a table b-tree root/child is leaf (0x0d) or interior (0x05)
2253        }
2254    }
2255
2256    /// Walk the table b-tree rooted at `page`, inserting every live leaf cell's
2257    /// rowid into `ids`. Best-effort and bounded: a malformed/cyclic structure
2258    /// stops the walk rather than erroring or looping.
2259    fn collect_rowids(
2260        &self,
2261        page: u32,
2262        ids: &mut std::collections::BTreeSet<i64>,
2263        seen: &mut std::collections::BTreeSet<u32>,
2264    ) {
2265        // Visit each page at most once (see `collect_rows` for the rationale): a
2266        // manipulated child pointer that revisits a page must not recurse
2267        // unboundedly. The visited-set bounds recursion depth to distinct pages.
2268        if page == 0 || seen.len() > MAX_PAGES_PER_WALK || !seen.insert(page) {
2269            return;
2270        }
2271        let Ok(slice) = self.page_slice(page) else {
2272            return; // cov:unreachable: schema rootpages and their children are in range
2273        };
2274        let slice = &*slice; // PageBytes -> &[u8]; body below is source-agnostic
2275        let hdr_off = if page == 1 { SQLITE_HEADER_SIZE } else { 0 };
2276        let Some(&page_type) = slice.get(hdr_off) else {
2277            return; // cov:unreachable: a full page slice always has its header byte
2278        };
2279        let cell_count = be_u16(slice, hdr_off + 3) as usize;
2280        match page_type {
2281            0x0d => {
2282                let cell_ptr_array = hdr_off + 8;
2283                for i in 0..cell_count {
2284                    let cell_off = be_u16(slice, cell_ptr_array + i * 2) as usize;
2285                    if let Some(rowid) = live_cell_rowid(slice, cell_off) {
2286                        ids.insert(rowid);
2287                    }
2288                }
2289            }
2290            0x05 => {
2291                let cell_ptr_array = hdr_off + 12;
2292                for i in 0..cell_count {
2293                    let cell_off = be_u16(slice, cell_ptr_array + i * 2) as usize;
2294                    let child = be_u32(slice, cell_off);
2295                    self.collect_rowids(child, ids, seen);
2296                }
2297                let right = be_u32(slice, hdr_off + 8);
2298                self.collect_rowids(right, ids, seen);
2299            }
2300            _ => {} // cov:unreachable: a table b-tree root/child is leaf (0x0d) or interior (0x05)
2301        }
2302    }
2303
2304    /// Walk a single table b-tree rooted at `root_page` (1-based) and collect
2305    /// every leaf row as typed values. `column_count` is the table's declared
2306    /// column count, used to apply the `INTEGER PRIMARY KEY` rowid-alias rule.
2307    ///
2308    /// Shares ONE b-tree/overflow walk with the snapshot-scoped read
2309    /// ([`CommitSnapshot::read_table`]) via an internal page-source abstraction, so
2310    /// the live and historical paths can never diverge.
2311    pub fn read_table(&self, root_page: u32, column_count: usize) -> Result<Vec<Row>, Error> {
2312        read_table_via(self, root_page, column_count)
2313    }
2314
2315    /// Bytes of the 1-based `page` number, or `PageOutOfRange`.
2316    ///
2317    /// When a WAL overlay is in effect and holds a committed version of this
2318    /// page, the overlaid bytes are returned in preference to the main file —
2319    /// this is what makes a table walk see the WAL-applied view. The main file
2320    /// is never mutated.
2321    fn page_slice(&self, page: u32) -> Result<PageBytes<'_>, Error> {
2322        if page == 0 {
2323            return Err(Error::PageOutOfRange(0));
2324        }
2325        if let Some(wal) = &self.wal {
2326            if let Some(overlaid) = wal.pages.get(&page) {
2327                return Ok(PageBytes::Borrowed(overlaid.as_slice()));
2328            }
2329        }
2330        self.source
2331            .page(page, self.header.page_size as usize)
2332            .ok_or(Error::PageOutOfRange(page))
2333    }
2334}
2335
2336/// A source of page images for the shared b-tree / overflow walk — the seam that
2337/// lets the live [`Database`] (main file ⊕ WAL overlay) and a historical
2338/// [`CommitSnapshot`] (materialized commit pages) share ONE table-read
2339/// implementation instead of forking parallel copies.
2340///
2341/// All page numbers are 1-based. Implementations resolve page 1 with the
2342/// 100-byte file header in place (so the walk reads the b-tree header at offset
2343/// `SQLITE_HEADER_SIZE` for page 1, 0 otherwise).
2344trait PageSource {
2345    /// The 1-based `page`'s full image, or `None` for page 0 / out of range.
2346    fn page(&self, page: u32) -> Option<PageBytes<'_>>;
2347    /// Usable bytes per page (`page_size` − reserved-space), for the overflow and
2348    /// local-payload computations.
2349    fn usable(&self) -> usize;
2350    /// The highest valid 1-based page number (the cycle/over-range bound).
2351    fn page_bound(&self) -> u32;
2352    /// The database text encoding, for decoding TEXT values.
2353    fn encoding(&self) -> TextEncoding;
2354}
2355
2356impl PageSource for Database {
2357    fn page(&self, page: u32) -> Option<PageBytes<'_>> {
2358        self.page_slice(page).ok()
2359    }
2360    fn usable(&self) -> usize {
2361        self.header.usable_size() as usize
2362    }
2363    fn page_bound(&self) -> u32 {
2364        self.file_page_count()
2365    }
2366    fn encoding(&self) -> TextEncoding {
2367        self.header.text_encoding
2368    }
2369}
2370
2371impl PageSource for CommitSnapshot {
2372    fn page(&self, page: u32) -> Option<PageBytes<'_>> {
2373        self.overlaid
2374            .get(&page)
2375            .map(|v| PageBytes::Borrowed(v.as_slice()))
2376    }
2377    fn usable(&self) -> usize {
2378        self.usable as usize
2379    }
2380    fn page_bound(&self) -> u32 {
2381        // The committed page count at this snapshot — the cycle/over-range bound
2382        // for an overflow walk over the snapshot's materialized pages.
2383        self.id.db_size_after_commit
2384    }
2385    fn encoding(&self) -> TextEncoding {
2386        // Text encoding from the snapshot's OWN page-1 header (byte 56), so a
2387        // historical read decodes TEXT per the encoding as of this commit.
2388        self.overlaid
2389            .get(&1)
2390            .map(|p| match be_u32(p, TEXT_ENCODING_OFFSET) {
2391                2 => TextEncoding::Utf16Le,
2392                3 => TextEncoding::Utf16Be,
2393                _ => TextEncoding::Utf8,
2394            })
2395            .unwrap_or_default()
2396    }
2397}
2398
2399/// Walk a single table b-tree rooted at `root_page` over any [`PageSource`],
2400/// collecting every leaf row as typed values. The one implementation shared by
2401/// the live and snapshot-scoped reads.
2402/// Insert a `sqlite_master` row's `name -> CREATE SQL` into `out` when the row is
2403/// a **user** table (`type='table'`, name not `sqlite_*`). Shared by
2404/// [`Database::schema_sql`] and [`PriorSnapshot::schema_sql`] so the live and
2405/// prior reads classify schema rows identically. A row that is not a user-table
2406/// row (an index/view/trigger, an internal table, or a malformed row) is skipped.
2407fn schema_sql_insert(out: &mut std::collections::BTreeMap<String, String>, values: &[Value]) {
2408    // sqlite_master row: (type, name, tbl_name, rootpage, sql).
2409    let is_table = matches!(values.first(), Some(Value::Text(t)) if t == "table");
2410    if !is_table {
2411        return;
2412    }
2413    let Some(Value::Text(name)) = values.get(1) else {
2414        return; // cov:unreachable: a 'table' schema row has a TEXT name
2415    };
2416    if name.starts_with("sqlite_") {
2417        return;
2418    }
2419    let sql = match values.get(4) {
2420        Some(Value::Text(s)) => s.clone(),
2421        _ => String::new(), // cov:unreachable: a 'table' schema row carries its CREATE TABLE sql
2422    };
2423    out.insert(name.clone(), sql);
2424}
2425
2426fn read_table_via(
2427    src: &dyn PageSource,
2428    root_page: u32,
2429    column_count: usize,
2430) -> Result<Vec<Row>, Error> {
2431    let mut rows = Vec::new();
2432    let mut seen = std::collections::BTreeSet::new();
2433    walk_table_page(src, root_page, column_count, &mut rows, &mut seen)?;
2434    Ok(rows)
2435}
2436
2437fn walk_table_page(
2438    src: &dyn PageSource,
2439    page: u32,
2440    column_count: usize,
2441    rows: &mut Vec<Row>,
2442    seen: &mut std::collections::BTreeSet<u32>,
2443) -> Result<(), Error> {
2444    // Visit each page at most once. A manipulated interior child pointer
2445    // (anti-forensic corpus category 12) can revisit an already-walked page; a
2446    // counter-only guard still recurses up to the cap deep before stopping,
2447    // overflowing the stack. The visited-set bounds recursion DEPTH to the
2448    // number of distinct pages. A revisited page is silently skipped (Ok) so a
2449    // crafted cycle yields the partial-but-valid rows already collected rather
2450    // than an error.
2451    if seen.len() > MAX_PAGES_PER_WALK {
2452        return Err(Error::TooManyPages);
2453    }
2454    if !seen.insert(page) {
2455        return Ok(());
2456    }
2457    let slice = src.page(page).ok_or(Error::PageOutOfRange(page))?;
2458    let slice = &*slice;
2459
2460    // Page 1 carries the 100-byte file header before its b-tree header.
2461    let hdr_off = if page == 1 { SQLITE_HEADER_SIZE } else { 0 };
2462
2463    let page_type = *slice.get(hdr_off).ok_or(Error::TruncatedCell)?;
2464    let cell_count = be_u16(slice, hdr_off + 3) as usize;
2465
2466    match page_type {
2467        0x0d => read_leaf_cells(src, slice, hdr_off, cell_count, column_count, rows),
2468        0x05 => {
2469            // Interior table page: 12-byte header; cell = 4-byte child ptr +
2470            // varint key. Recurse into every child plus the right-most ptr.
2471            let cell_ptr_array = hdr_off + 12;
2472            for i in 0..cell_count {
2473                let p = cell_ptr_array + i * 2;
2474                let cell_off = be_u16(slice, p) as usize;
2475                let child = be_u32(slice, cell_off);
2476                walk_table_page(src, child, column_count, rows, seen)?;
2477            }
2478            let right = be_u32(slice, hdr_off + 8);
2479            walk_table_page(src, right, column_count, rows, seen)
2480        }
2481        other => Err(Error::NotATablePage(other)),
2482    }
2483}
2484
2485fn read_leaf_cells(
2486    src: &dyn PageSource,
2487    slice: &[u8],
2488    hdr_off: usize,
2489    cell_count: usize,
2490    column_count: usize,
2491    rows: &mut Vec<Row>,
2492) -> Result<(), Error> {
2493    let cell_ptr_array = hdr_off + 8; // leaf b-tree header is 8 bytes
2494    for i in 0..cell_count {
2495        let p = cell_ptr_array + i * 2;
2496        let cell_off = be_u16(slice, p) as usize;
2497        let row = decode_leaf_cell(src, slice, cell_off, column_count)?;
2498        rows.push(row);
2499    }
2500    Ok(())
2501}
2502
2503/// Decode one table-leaf cell at `off` into a [`Row`], reassembling the payload
2504/// from its overflow-page chain (resolved through the SAME [`PageSource`]) when
2505/// it spills past the leaf page.
2506fn decode_leaf_cell(
2507    src: &dyn PageSource,
2508    slice: &[u8],
2509    off: usize,
2510    column_count: usize,
2511) -> Result<Row, Error> {
2512    let (payload_len, n1) = read_varint(slice, off)?;
2513    let (rowid, n2) = read_varint(slice, off + n1)?;
2514    let payload_start = off + n1 + n2;
2515    let total = usize::try_from(payload_len).map_err(|_| Error::TruncatedCell)?;
2516
2517    let usable = src.usable();
2518    let local = local_payload_len(total, usable);
2519
2520    let payload = if local >= total {
2521        // Whole payload is on the leaf page (no spill).
2522        slice
2523            .get(payload_start..payload_start + total)
2524            .ok_or(Error::TruncatedCell)?
2525            .to_vec()
2526    } else {
2527        // Spilled: `local` bytes on the leaf, then a 4-byte overflow page
2528        // pointer, then the remainder follows the overflow chain.
2529        let head = slice
2530            .get(payload_start..payload_start + local)
2531            .ok_or(Error::TruncatedCell)?;
2532        let first_overflow = be_u32(slice, payload_start + local);
2533        // Cap the pre-allocation against the untrusted payload length: a payload
2534        // cannot exceed the bytes the file can physically supply — the `local`
2535        // bytes on the leaf plus the content bytes of every page reachable
2536        // through the overflow chain (`per_page * page_bound`). A crafted cell
2537        // that declares a multi-exabyte `payload_len` would otherwise reach
2538        // `Vec::with_capacity(total)` and abort the process with an allocation
2539        // bomb. This is the same over-range condition `read_overflow_chain`
2540        // rejects, pulled ahead of the allocation.
2541        let per_page = usable.saturating_sub(4);
2542        let max_overflow = per_page.saturating_mul(src.page_bound() as usize);
2543        let max_payload = local.saturating_add(max_overflow);
2544        if total > max_payload {
2545            return Err(Error::MalformedOverflow);
2546        }
2547        let mut buf = Vec::with_capacity(total);
2548        buf.extend_from_slice(head);
2549        read_overflow_chain(src, first_overflow, total - local, &mut buf)?;
2550        buf
2551    };
2552
2553    let values = decode_record(&payload, column_count, rowid, src.encoding())?;
2554    Ok(Row { rowid, values })
2555}
2556
2557/// Follow an overflow-page chain starting at `first` (1-based page number) over
2558/// a [`PageSource`], appending up to `remaining` payload bytes to `buf`. Each
2559/// overflow page is a 4-byte big-endian "next page" pointer (0 ends the chain)
2560/// followed by up to `usable - 4` content bytes.
2561///
2562/// Bounded against cyclic/over-long chains via [`Error::MalformedOverflow`].
2563fn read_overflow_chain(
2564    src: &dyn PageSource,
2565    first: u32,
2566    mut remaining: usize,
2567    buf: &mut Vec<u8>,
2568) -> Result<(), Error> {
2569    let usable = src.usable();
2570    let per_page = usable.saturating_sub(4);
2571    if per_page == 0 {
2572        return Err(Error::MalformedOverflow);
2573    }
2574    let total_pages = src.page_bound();
2575    let cap = total_pages as usize + 1;
2576
2577    let mut page = first;
2578    let mut visited = 0usize;
2579    while remaining > 0 {
2580        if page == 0 || page > total_pages {
2581            return Err(Error::MalformedOverflow);
2582        }
2583        visited += 1;
2584        if visited > cap {
2585            return Err(Error::MalformedOverflow);
2586        }
2587        let slice = src.page(page).ok_or(Error::PageOutOfRange(page))?;
2588        let slice = &*slice;
2589        let next = be_u32(slice, 0);
2590        let take = remaining.min(per_page);
2591        let chunk = slice.get(4..4 + take).ok_or(Error::TruncatedCell)?;
2592        buf.extend_from_slice(chunk);
2593        remaining -= take;
2594        page = next;
2595    }
2596    Ok(())
2597}
2598
2599/// Number of payload bytes stored locally on a table-leaf page for a record of
2600/// `total` bytes, given the page's `usable` size (file-format §1.6 overflow
2601/// rule). When the return value equals `total`, the record does not spill.
2602pub(crate) fn local_payload_len(total: usize, usable: usize) -> usize {
2603    let max_local = usable - 35; // X: largest payload kept entirely local
2604    if total <= max_local {
2605        return total;
2606    }
2607    let min_local = (usable - 12) * 32 / 255 - 23; // M
2608    let k = min_local + (total - min_local) % (usable - 4);
2609    if k <= max_local {
2610        k
2611    } else {
2612        min_local
2613    }
2614}
2615
2616impl WalOverlay {
2617    /// Parse a `-wal` sidecar into the newest committed page versions.
2618    ///
2619    /// Returns `Ok(None)` when `wal` is absent of a usable header / has no
2620    /// frames (a no-op overlay). Iterates frames in file order, accumulating the
2621    /// page data of each frame whose salt matches the WAL header; on reaching a
2622    /// COMMIT frame (`db_size_after_commit != 0`) the accumulated pages are
2623    /// promoted into the committed snapshot. Frames after the last commit are
2624    /// uncommitted and dropped. Bounds-checked and breadth-capped against a
2625    /// crafted WAL (a frame whose declared page data runs past the file ends the
2626    /// scan rather than panicking).
2627    fn parse(wal: &[u8], page_size: u32) -> Result<Option<Self>, Error> {
2628        use forensicnomicon::sqlite::{SQLITE_WAL_FRAME_HEADER_SIZE, SQLITE_WAL_HEADER_SIZE};
2629
2630        // No header → no overlay (treat a too-short WAL as empty, not an error:
2631        // a missing/zero-length sidecar is normal and must not fail the open).
2632        let Some(hdr) = wal.get(..SQLITE_WAL_HEADER_SIZE) else {
2633            return Ok(None);
2634        };
2635        let magic = be_u32(hdr, 0);
2636        if magic != WAL_MAGIC_BE && magic != WAL_MAGIC_LE {
2637            return Ok(None);
2638        }
2639        // The WAL records its own page size (offset 8); trust the DB header's
2640        // page size but require agreement to avoid mis-slicing frames.
2641        let wal_page_size = be_u32(hdr, 8);
2642        if wal_page_size != page_size {
2643            return Ok(None);
2644        }
2645        // WAL header layout (file-format §4.1): salt-1 at offset 16, salt-2 at
2646        // offset 20 (the two checksum words follow at 24 and 28).
2647        let salt1 = be_u32(hdr, 16);
2648        let salt2 = be_u32(hdr, 20);
2649
2650        let ps = page_size as usize;
2651        let frame_stride = SQLITE_WAL_FRAME_HEADER_SIZE + ps;
2652
2653        let mut committed: std::collections::BTreeMap<u32, Vec<u8>> =
2654            std::collections::BTreeMap::new();
2655        let mut pending: std::collections::BTreeMap<u32, Vec<u8>> =
2656            std::collections::BTreeMap::new();
2657        // Every committed frame's page image (file order), and the pending frames
2658        // not yet promoted by a COMMIT. Mirrors the page promotion above so
2659        // uncommitted trailing frames are dropped from BOTH the view and the carve.
2660        let mut frames: Vec<WalFramePage> = Vec::new();
2661        let mut pending_frames: Vec<WalFramePage> = Vec::new();
2662
2663        let mut off = SQLITE_WAL_HEADER_SIZE;
2664        // One frame per page in the file is the natural breadth cap; allow a
2665        // generous multiple for repeated rewrites, but keep it bounded.
2666        let max_frames = wal.len() / frame_stride + 1;
2667        let mut frame_no = 0usize;
2668
2669        while let Some(frame) = wal.get(off..off + frame_stride) {
2670            frame_no += 1;
2671            if frame_no > max_frames {
2672                break; // cov:unreachable: the slice walk already bounds frame_no
2673            }
2674            let page_no = be_u32(frame, 0);
2675            let db_size = be_u32(frame, 4);
2676            let fsalt1 = be_u32(frame, 8);
2677            let fsalt2 = be_u32(frame, 12);
2678            // A frame from a different checkpoint generation (salt mismatch) is
2679            // stale residue, not part of this WAL's live content — stop here.
2680            if fsalt1 != salt1 || fsalt2 != salt2 {
2681                break;
2682            }
2683            if page_no == 0 {
2684                break; // malformed frame; stop rather than mis-index
2685            }
2686            let data = frame
2687                .get(SQLITE_WAL_FRAME_HEADER_SIZE..)
2688                .ok_or(Error::TruncatedCell)?;
2689            pending.insert(page_no, data.to_vec());
2690            let is_commit = db_size != 0;
2691            pending_frames.push(WalFramePage {
2692                frame_index: frame_no - 1, // 0-based file order
2693                page_no,
2694                salt1,
2695                salt2,
2696                is_commit,
2697                page: data.to_vec(),
2698            });
2699
2700            if is_commit {
2701                // COMMIT frame: promote everything pending into the snapshot AND
2702                // into the committed frame list (keeping every frame, not just the
2703                // newest version of each page).
2704                for (p, d) in std::mem::take(&mut pending) {
2705                    committed.insert(p, d);
2706                }
2707                frames.append(&mut pending_frames);
2708            }
2709            off += frame_stride;
2710        }
2711
2712        if committed.is_empty() {
2713            Ok(None)
2714        } else {
2715            Ok(Some(WalOverlay {
2716                pages: committed,
2717                frames,
2718                raw: wal.to_vec(),
2719            }))
2720        }
2721    }
2722}
2723
2724// ===========================================================================
2725// Bespoke, format-exact WAL temporal model (task #55)
2726// ===========================================================================
2727//
2728// A `-wal` sidecar is NOT an open-ended event log. It is a BOUNDED SEGMENT under a
2729// single salt epoch: every live frame shares the WAL header's (salt1, salt2). A
2730// checkpoint reset renumbers frames and rolls the salts — a DISCONTINUITY, not a
2731// continuation. The only materializable database states are the COMMIT snapshots:
2732// the replay of all valid frames up to a commit frame. A frame BETWEEN commits is
2733// not independently materializable, so it is never surfaced as a snapshot. Tails
2734// past the last commit, or after a salt reset, are WAL residue — forensic leads,
2735// never committed history.
2736//
2737// This model is self-contained in sqlite-core. The future state-history-forensic
2738// [H] adapter attaches at the seam exposed here (WalLsn + CohortTopology +
2739// `checksums_are_tamper_evident`), but sqlite-core does NOT depend on it.
2740
2741/// Cap on the number of salt segments and frames the timeline parser will walk on a
2742/// crafted `-wal`, bounding work against an attacker-supplied file. A real WAL holds
2743/// one segment with at most a few frames per database page.
2744const MAX_WAL_SEGMENTS: usize = 1024;
2745
2746/// Identity of one salt epoch within a `-wal` file: its 0-based segment ordinal.
2747/// A fresh segment begins at file start and after every checkpoint salt reset.
2748#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)]
2749pub struct WalSegmentId(pub usize);
2750
2751/// One salt epoch of a `-wal` file — a single bounded segment.
2752///
2753/// A `-wal` is a bounded segment, not an open-ended log: every live frame here shares
2754/// `(salt1, salt2)`. A checkpoint reset (salt change + frame renumber) starts a NEW
2755/// `WalSegment`; it is a discontinuity, never another epoch of the same segment.
2756#[derive(Debug, Clone, PartialEq, Eq)]
2757pub struct WalSegment {
2758    /// This segment's ordinal within the WAL (0 = the segment at file start).
2759    pub id: WalSegmentId,
2760    /// WAL salt-1 (checkpoint generation), shared by every frame in the segment.
2761    pub salt1: u32,
2762    /// WAL salt-2 (checkpoint generation), shared by every frame in the segment.
2763    pub salt2: u32,
2764    /// Page size declared by the segment's frames (bytes).
2765    pub page_size: u32,
2766    /// Number of frames belonging to this segment.
2767    pub frame_count: usize,
2768    /// The checkpoint sequence number recorded in the WAL header (offset 12). For a
2769    /// segment discovered after a reset within the same file this is the header's
2770    /// value; per-segment sequence is otherwise not separately recorded.
2771    pub checkpoint_seq: u32,
2772}
2773
2774/// Address of a materializable database state: the replay of all valid frames up to
2775/// a COMMIT frame. `CommitId = (segment, commit_frame_index, db_size_after_commit)`.
2776#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)]
2777pub struct CommitId {
2778    /// The salt segment this commit belongs to.
2779    pub segment: WalSegmentId,
2780    /// 0-based file-order index of the COMMIT frame within the segment.
2781    pub commit_frame_index: usize,
2782    /// `db_size_after_commit` recorded in the COMMIT frame header — the database's
2783    /// page count once this commit is materialized.
2784    pub db_size_after_commit: u32,
2785}
2786
2787/// The salt-qualified log-sequence identity of a WAL position — the seam the future
2788/// `state-history-forensic` `[H]` adapter maps onto `LsnKind::SqliteWal`.
2789///
2790/// A bare `frame_index` is meaningless across checkpoint resets (frames renumber), so
2791/// ordering is ALWAYS qualified by `(salt1, salt2)`. The adapter must reconstruct
2792/// `LsnKind::SqliteWal { salt1, salt2, frame_index }` from exactly this triple — never
2793/// from a bare index.
2794#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)]
2795pub struct WalLsn {
2796    /// Salt-1 of the owning segment (checkpoint generation).
2797    pub salt1: u32,
2798    /// Salt-2 of the owning segment (checkpoint generation).
2799    pub salt2: u32,
2800    /// 0-based frame index within that segment.
2801    pub frame_index: usize,
2802}
2803
2804/// Topology of the temporal cohort the WAL exposes — the shape the `[H]` adapter maps
2805/// to `state-history-forensic::CohortTopology`.
2806#[derive(Debug, Clone, Copy, PartialEq, Eq)]
2807pub enum CohortTopology {
2808    /// A single salt epoch: the commit snapshots form one linearly-ordered chain.
2809    LinearSegment,
2810    /// Multiple salt epochs (checkpoint resets) with no replay continuity between
2811    /// them — each segment is linear internally but the segments are disconnected.
2812    Disconnected,
2813}
2814
2815/// One page's image at a particular [`CommitSnapshot`].
2816#[derive(Debug, Clone, PartialEq, Eq)]
2817pub struct CommittedPageVersion {
2818    /// 1-based database page number.
2819    pub page_no: u32,
2820    /// The page's full image (`page_size` bytes) as of this commit.
2821    pub bytes: Vec<u8>,
2822}
2823
2824/// A materializable database state: the replay of all valid frames up to a COMMIT.
2825///
2826/// This is the ONLY independently-materializable WAL state. `page_version` resolves a
2827/// page to its image as of this commit (the newest frame ≤ this commit that rewrote
2828/// the page, else the acquired base image). A frame between commits is never a
2829/// snapshot.
2830#[derive(Debug, Clone, PartialEq, Eq)]
2831pub struct CommitSnapshot {
2832    id: CommitId,
2833    /// Salt-1 of the owning segment, carried so [`CommitSnapshot::lsn`] is
2834    /// self-contained without a back-reference to the segment.
2835    salt1: u32,
2836    /// Salt-2 of the owning segment.
2837    salt2: u32,
2838    /// The materialized page images at this commit: base image overlaid with every
2839    /// committed frame up to and including this commit (newest version per page),
2840    /// capped to `db_size_after_commit` pages. `page_version` reads from this map.
2841    overlaid: std::collections::BTreeMap<u32, Vec<u8>>,
2842    /// Whether the whole frame chain up to and including this commit's COMMIT frame
2843    /// passed the WAL cumulative checksum (file-format §4.2). `false` marks a commit
2844    /// the salt+commit-marker admission would otherwise accept but whose checksum
2845    /// chain is broken (post-reset residue, tampering, or corruption) — kept, not
2846    /// dropped, so the forensic layer can label it.
2847    checksum_valid: bool,
2848    /// Usable bytes per page (`page_size` − reserved), parsed from the snapshot's
2849    /// OWN page-1 header, so a snapshot-scoped read uses the reserved-space value
2850    /// as of this commit rather than the live database's.
2851    usable: u32,
2852}
2853
2854/// One user table as of a [`CommitSnapshot`] — its schema parsed from the
2855/// snapshot's OWN materialized page 1, NOT from the live database. A rootpage can
2856/// be dropped and reused by a different table across commits, so reading the
2857/// schema from the snapshot is the only correct way to interpret its b-trees.
2858#[derive(Debug, Clone, PartialEq, Eq)]
2859pub struct SnapshotTable {
2860    /// The table's `sqlite_master.name`.
2861    pub name: String,
2862    /// 1-based root page of the table's b-tree as of this commit.
2863    pub rootpage: u32,
2864    /// Parsed column names from the table's `CREATE TABLE`, in declared order.
2865    /// Empty when the schema SQL could not be parsed with confidence.
2866    pub columns: Vec<String>,
2867    /// Whether this is a `WITHOUT ROWID` table (file-format §2.4). Such a table
2868    /// uses an INDEX b-tree with no rowid key, so the rowid-based snapshot read
2869    /// does not apply — flagged so a caller never mis-reads it as a rowid table.
2870    pub without_rowid: bool,
2871}
2872
2873/// Whether a `CREATE TABLE` statement declares a `WITHOUT ROWID` table
2874/// (file-format §2.4). Detection keys off the trailing `WITHOUT ROWID` clause,
2875/// case-insensitively and tolerant of internal whitespace, while ignoring any
2876/// occurrence inside a quoted identifier/string so a column literally named
2877/// "without rowid" is not a false positive.
2878/// A `CREATE TABLE` statement with quoted spans removed and whitespace collapsed,
2879/// uppercased — so a clause search sees only unquoted SQL tokens. Strips
2880/// `'...'` / `"..."` / `` `...` `` / `[...]` spans (the four `SQLite` identifier /
2881/// string quotings) exactly as the clause detectors require, so the keyword
2882/// appearing inside a quoted identifier or string literal can never false-match.
2883fn normalized_unquoted_sql(create_sql: &str) -> String {
2884    let bytes = create_sql.as_bytes();
2885    let mut unquoted = String::with_capacity(create_sql.len());
2886    let mut quote: Option<u8> = None;
2887    for &c in bytes {
2888        match quote {
2889            Some(q) => {
2890                if c == q {
2891                    quote = None;
2892                }
2893            }
2894            None => match c {
2895                b'\'' | b'"' | b'`' => quote = Some(c),
2896                b'[' => quote = Some(b']'),
2897                _ => unquoted.push(c as char),
2898            },
2899        }
2900    }
2901    unquoted
2902        .split_whitespace()
2903        .collect::<Vec<_>>()
2904        .join(" ")
2905        .to_ascii_uppercase()
2906}
2907
2908fn without_rowid_sql(create_sql: &str) -> bool {
2909    // Look for the clause as a discrete token sequence, ignoring quoted spans and
2910    // case/whitespace (file-format §2.4).
2911    normalized_unquoted_sql(create_sql).contains("WITHOUT ROWID")
2912}
2913
2914/// Whether `create_sql` declares an ordinary rowid table with an
2915/// `INTEGER PRIMARY KEY AUTOINCREMENT` column — the only form for which `SQLite`
2916/// maintains a monotonic `sqlite_sequence` high-water mark.
2917///
2918/// Per the file format, `AUTOINCREMENT` is valid **only** immediately after
2919/// `INTEGER PRIMARY KEY`, and **never** on a `WITHOUT ROWID` table (which has no
2920/// rowid to auto-increment). So this is true iff the normalized, unquoted CREATE
2921/// text contains the exact token run `INTEGER PRIMARY KEY AUTOINCREMENT` and does
2922/// NOT carry the `WITHOUT ROWID` clause. Quoted identifiers / string literals /
2923/// comments are stripped first (mirroring `without_rowid_sql`), so a column
2924/// merely named `"autoincrement"`, or the keyword inside a string, never matches.
2925///
2926/// This is a HINT input only: a true result means the table has an AUTOINCREMENT
2927/// high-water mark the forensic layer can reconcile against, not that any
2928/// particular row predates the current instance.
2929#[must_use]
2930pub fn is_autoincrement(create_sql: &str) -> bool {
2931    let normalized = normalized_unquoted_sql(create_sql);
2932    normalized.contains("INTEGER PRIMARY KEY AUTOINCREMENT")
2933        && !normalized.contains("WITHOUT ROWID")
2934}
2935
2936impl CommitSnapshot {
2937    /// This snapshot's [`CommitId`].
2938    #[must_use]
2939    pub fn id(&self) -> CommitId {
2940        self.id
2941    }
2942
2943    /// The database page count once this commit is materialized.
2944    #[must_use]
2945    pub fn db_size_after_commit(&self) -> u32 {
2946        self.id.db_size_after_commit
2947    }
2948
2949    /// Whether the WAL frame chain up to and including this commit's COMMIT frame
2950    /// validated against the cumulative WAL checksum (file-format §4.2).
2951    ///
2952    /// `true` is the spec-conformant case: every frame's stored `(checksum1,
2953    /// checksum2)` equalled the running checksum advanced over the frame's first
2954    /// 8 header bytes plus its full page data, seeded from the WAL header
2955    /// checksum. `false` means the chain broke at or before this commit — the
2956    /// salt + commit-marker admission accepted it, but it is residue (post-reset
2957    /// leftover, tampering, or corruption). Such a commit is deliberately KEPT
2958    /// (not dropped) so the forensic layer can mark it; a consumer that wants only
2959    /// trustworthy state filters on this flag.
2960    #[must_use]
2961    pub fn checksum_valid(&self) -> bool {
2962        self.checksum_valid
2963    }
2964
2965    /// The salt-qualified [`WalLsn`] of this commit (the `[H]` adapter seam).
2966    #[must_use]
2967    pub fn lsn(&self) -> WalLsn {
2968        WalLsn {
2969            salt1: self.salt1,
2970            salt2: self.salt2,
2971            frame_index: self.id.commit_frame_index,
2972        }
2973    }
2974
2975    /// The 1-based page numbers this commit materialized (base ∪ committed frames
2976    /// up to this commit, capped to `db_size_after_commit`), ascending.
2977    ///
2978    /// The carve-at-snapshot primitive iterates these to drive the carving
2979    /// primitives over each page image, WITHOUT assuming the pages form a
2980    /// contiguous `1..=db_size` range (a truncating commit or a sparse base image
2981    /// can leave gaps). Every returned page resolves via [`Self::page_version`].
2982    #[must_use]
2983    pub fn page_numbers(&self) -> Vec<u32> {
2984        self.overlaid.keys().copied().collect()
2985    }
2986
2987    /// The image of `page_no` as of this commit, or `None` for a page beyond the
2988    /// committed database size that the WAL never rewrote.
2989    #[must_use]
2990    pub fn page_version(&self, page_no: u32) -> Option<CommittedPageVersion> {
2991        let bytes = self.overlaid.get(&page_no)?.clone();
2992        Some(CommittedPageVersion { page_no, bytes })
2993    }
2994
2995    /// The user tables AS OF this commit, parsed from the snapshot's OWN page 1
2996    /// (the `sqlite_master` b-tree), NOT from the live database.
2997    ///
2998    /// A rootpage can be dropped and reused by a different table across commits,
2999    /// so the schema MUST come from the snapshot itself — reading today's live
3000    /// schema would mis-attribute a historical b-tree. Returns one
3001    /// [`SnapshotTable`] per `type='table'` row whose name is not an internal
3002    /// `sqlite_*` table, carrying its rootpage, parsed column names, and a
3003    /// `WITHOUT ROWID` flag (file-format §2.4). Best-effort and panic-free: an
3004    /// unreadable page-1 schema yields an empty vector.
3005    #[must_use]
3006    pub fn tables(&self) -> Vec<SnapshotTable> {
3007        // sqlite_master is a 5-column table rooted at page 1:
3008        // (type, name, tbl_name, rootpage, sql). Walk it through THIS snapshot's
3009        // pages via the shared b-tree reader.
3010        let Ok(schema) = read_table_via(self, 1, 5) else {
3011            return Vec::new(); // cov:unreachable: a committed snapshot has a readable page 1
3012        };
3013        let mut out = Vec::new();
3014        for row in schema {
3015            let is_table = matches!(row.values.first(), Some(Value::Text(t)) if t == "table");
3016            if !is_table {
3017                continue;
3018            }
3019            let Some(Value::Text(name)) = row.values.get(1) else {
3020                continue; // cov:unreachable: a 'table' schema row has a TEXT name
3021            };
3022            if name.starts_with("sqlite_") {
3023                continue;
3024            }
3025            let Some(Value::Integer(root)) = row.values.get(3) else {
3026                continue; // cov:unreachable: a 'table' schema row has an integer rootpage
3027            };
3028            let Ok(rootpage) = u32::try_from(*root) else {
3029                continue; // cov:unreachable: a real rootpage is a small positive page number
3030            };
3031            let sql = match row.values.get(4) {
3032                Some(Value::Text(s)) => s.as_str(),
3033                _ => "", // cov:unreachable: a 'table' schema row carries its CREATE TABLE sql
3034            };
3035            let columns = attribution::column_names(sql).unwrap_or_default();
3036            out.push(SnapshotTable {
3037                name: name.clone(),
3038                rootpage,
3039                columns,
3040                without_rowid: without_rowid_sql(sql),
3041            });
3042        }
3043        out
3044    }
3045
3046    /// Read every row of the table b-tree rooted at `rootpage` AS OF this commit,
3047    /// resolving overflow chains through the snapshot's OWN materialized pages, in
3048    /// rowid order.
3049    ///
3050    /// This is the snapshot-scoped counterpart to [`Database::read_table`]: it
3051    /// shares the SAME b-tree/overflow walk via an internal page-source
3052    /// abstraction, so a large row
3053    /// decodes with the page content as of this commit (not stale/future content
3054    /// the live view would supply). `column_count` drives only the
3055    /// `INTEGER PRIMARY KEY` rowid-alias rule (pass the table's declared arity,
3056    /// e.g. `SnapshotTable::columns.len()`). Returns `(rowid, values)` per row.
3057    ///
3058    /// Bounded and panic-free on hostile input, exactly as the live path: a
3059    /// cyclic/over-deep b-tree or overflow chain surfaces a typed [`Error`] rather
3060    /// than looping or panicking.
3061    pub fn read_table(
3062        &self,
3063        rootpage: u32,
3064        column_count: usize,
3065    ) -> Result<Vec<(i64, Vec<Value>)>, Error> {
3066        let rows = read_table_via(self, rootpage, column_count)?;
3067        Ok(rows.into_iter().map(|r| (r.rowid, r.values)).collect())
3068    }
3069}
3070
3071/// A page-level delta between two materialized states.
3072#[derive(Debug, Clone, PartialEq, Eq)]
3073pub struct WalDiff {
3074    changed: Vec<u32>,
3075}
3076
3077impl WalDiff {
3078    /// The 1-based page numbers whose bytes differ between the two states, ascending.
3079    #[must_use]
3080    pub fn changed_pages(&self) -> &[u32] {
3081        &self.changed
3082    }
3083}
3084
3085/// A stale WAL tail surfaced for forensics — NOT committed history.
3086///
3087/// Frames past the last COMMIT of a segment, frames after a salt reset that cannot be
3088/// replayed into the current segment, or a header/page-size break: all are residue.
3089/// The examiner weighs them; they are never part of a consistent snapshot.
3090#[derive(Debug, Clone, PartialEq, Eq)]
3091pub struct WalResidue {
3092    /// The segment the residue trails (the segment whose last commit it follows).
3093    pub segment: WalSegmentId,
3094    /// 0-based frame index (within the file) of the first residual frame.
3095    pub first_frame_index: usize,
3096    /// Number of residual frames.
3097    pub frame_count: usize,
3098    /// Why these frames are residue rather than committed history.
3099    pub reason: ResidueReason,
3100}
3101
3102/// Why a WAL tail is [`WalResidue`] (an invalidated-frame candidate), not history.
3103#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3104pub enum ResidueReason {
3105    /// Frames written after the segment's last COMMIT (uncommitted tail).
3106    BeyondLastCommit,
3107    /// Frames whose salt no longer matches the segment header (post-reset residue).
3108    SaltReset,
3109}
3110
3111/// Validation tier a WAL has cleared — strictly increasing assurance.
3112///
3113/// `PhysicalValidation` < `CommitValidation` < `ReplaySafe`. The timeline reports the
3114/// highest tier reached; a page-size mismatch never even produces a timeline (it is a
3115/// hard stop at parse, surfaced as [`WalValidationError::PageSizeMismatch`]).
3116#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)]
3117pub enum MaterializationSafety {
3118    /// Header magic / format / page-size / salts / frame boundaries are well-formed,
3119    /// but no committed snapshot was found (nothing to replay).
3120    PhysicalValidated,
3121    /// A last valid commit and committed frame ranges were established, but the
3122    /// read-only replay overlay was not (or could not be) built.
3123    CommitValidated,
3124    /// A read-only replay overlay to the last commit is available — safe to
3125    /// materialize without mutating either file.
3126    ReplaySafe,
3127}
3128
3129/// A WAL that cannot be admitted to the timeline at all (physical-validation hard
3130/// stops). Distinct from "no committed snapshot", which is a valid empty timeline.
3131#[derive(Debug, Clone, PartialEq, Eq)]
3132pub enum WalValidationError {
3133    /// The `-wal` is shorter than its 32-byte header, or carries the wrong magic.
3134    BadMagic,
3135    /// The WAL header's page size disagrees with the DB header's — a HARD STOP, since
3136    /// every frame would be mis-sliced. `db` and `wal` are the two declared sizes.
3137    PageSizeMismatch { db: u32, wal: u32 },
3138    /// The main database header itself failed to parse.
3139    Header(Error),
3140}
3141
3142/// The bespoke, format-exact temporal model of a `-wal` sidecar.
3143///
3144/// Enumerates the salt segments, the materializable [`CommitSnapshot`]s within them
3145/// (CommitId-addressable), and the [`WalResidue`] tails. Materialize a snapshot's page
3146/// images via [`CommitSnapshot::page_version`]; diff the acquired base against the last
3147/// valid commit via [`WalTimeline::diff_base_to_last_commit`].
3148#[derive(Debug, Clone, PartialEq, Eq)]
3149pub struct WalTimeline {
3150    page_size: u32,
3151    base_pages: std::collections::BTreeMap<u32, Vec<u8>>,
3152    segments: Vec<WalSegment>,
3153    snapshots: Vec<CommitSnapshot>,
3154    residue: Vec<WalResidue>,
3155    safety: MaterializationSafety,
3156}
3157
3158impl WalTimeline {
3159    /// Physical-validation tier: header magic + format check.
3160    ///
3161    /// Parses `bytes` (the acquired main DB) and `wal` (the `-wal` sidecar) into the
3162    /// segmented temporal model. A page-size mismatch between the DB header and the
3163    /// WAL header is a HARD STOP; a bad/short header is [`WalValidationError::BadMagic`].
3164    fn parse(bytes: &[u8], wal: &[u8], page_size: u32) -> Result<Self, WalValidationError> {
3165        use forensicnomicon::sqlite::{SQLITE_WAL_FRAME_HEADER_SIZE, SQLITE_WAL_HEADER_SIZE};
3166
3167        // --- PhysicalValidation: header magic / format / page-size / salts -------
3168        let hdr = wal
3169            .get(..SQLITE_WAL_HEADER_SIZE)
3170            .ok_or(WalValidationError::BadMagic)?;
3171        let magic = be_u32(hdr, 0);
3172        if magic != WAL_MAGIC_BE && magic != WAL_MAGIC_LE {
3173            return Err(WalValidationError::BadMagic);
3174        }
3175        let wal_page_size = be_u32(hdr, 8);
3176        if wal_page_size != page_size {
3177            return Err(WalValidationError::PageSizeMismatch {
3178                db: page_size,
3179                wal: wal_page_size,
3180            });
3181        }
3182        let checkpoint_seq = be_u32(hdr, 12);
3183        let mut salt1 = be_u32(hdr, 16);
3184        let mut salt2 = be_u32(hdr, 20);
3185
3186        // Checksum chain seed (file-format §4.2): the running (s0, s1) starts from
3187        // the WAL header's stored checksum (bytes 24..32, always big-endian),
3188        // which is itself the checksum over the first 24 header bytes. The word
3189        // endianness for advancing over frames comes from the magic. `from_magic`
3190        // cannot return None here — the magic was admitted above.
3191        let endian = WalChecksumEndian::from_magic(magic).unwrap_or(WalChecksumEndian::Big);
3192        let header_s0 = be_u32(hdr, 24);
3193        let header_s1 = be_u32(hdr, 28);
3194        // Per-segment running checksum state and whether the chain is still valid.
3195        let mut run_s0 = header_s0;
3196        let mut run_s1 = header_s1;
3197        let mut chain_valid = true;
3198
3199        let ps = page_size as usize;
3200        let frame_stride = SQLITE_WAL_FRAME_HEADER_SIZE + ps;
3201
3202        // The acquired main DB image: the pre-WAL base for replay within the current
3203        // validated segment (NOT "epoch 0" — just the base each commit overlays onto).
3204        let mut base_pages: std::collections::BTreeMap<u32, Vec<u8>> =
3205            std::collections::BTreeMap::new();
3206        // `chunks_exact` yields only whole pages (infallible by construction — no
3207        // out-of-bounds slice to guard); cap at `u32::MAX` pages so the 1-based page
3208        // number never overflows on a pathologically large image.
3209        for (idx, page) in bytes
3210            .chunks_exact(ps)
3211            .take(u32::MAX as usize - 1)
3212            .enumerate()
3213        {
3214            let pno = idx as u32 + 1; // 1-based page number
3215            base_pages.insert(pno, page.to_vec());
3216        }
3217
3218        let mut segments: Vec<WalSegment> = Vec::new();
3219        let mut snapshots: Vec<CommitSnapshot> = Vec::new();
3220        let mut residue: Vec<WalResidue> = Vec::new();
3221
3222        // Per-segment running state.
3223        let mut seg_ordinal = 0usize;
3224        let mut seg_frame_count = 0usize;
3225        // Cumulative newest-page map across all COMMITTED frames of the segment, so a
3226        // snapshot's `overlaid` is base ∪ committed-up-to-this-commit.
3227        let mut committed_pages: std::collections::BTreeMap<u32, Vec<u8>> = base_pages.clone();
3228        let mut pending: std::collections::BTreeMap<u32, Vec<u8>> =
3229            std::collections::BTreeMap::new();
3230        let mut last_commit_global_frame: Option<usize> = None;
3231        let mut uncommitted_tail_start: Option<usize> = None;
3232
3233        let mut off = SQLITE_WAL_HEADER_SIZE;
3234        let max_frames = wal.len() / frame_stride + 1;
3235        let mut frame_no = 0usize;
3236
3237        while let Some(frame) = wal.get(off..off + frame_stride) {
3238            if frame_no >= max_frames {
3239                break; // cov:unreachable: the slice walk already bounds frame_no
3240            }
3241            let page_no = be_u32(frame, 0);
3242            let db_size = be_u32(frame, 4);
3243            let fsalt1 = be_u32(frame, 8);
3244            let fsalt2 = be_u32(frame, 12);
3245
3246            // A salt change opens a NEW segment (checkpoint reset = discontinuity).
3247            // Anything between the prior segment's last commit and here is residue.
3248            if fsalt1 != salt1 || fsalt2 != salt2 {
3249                if segments.len() >= MAX_WAL_SEGMENTS {
3250                    break; // cov:unreachable: real WALs hold far fewer than 1024 salt epochs
3251                }
3252                // Close the current segment, recording its residue tail (if any).
3253                Self::close_segment(
3254                    &mut segments,
3255                    &mut residue,
3256                    WalSegmentId(seg_ordinal),
3257                    salt1,
3258                    salt2,
3259                    page_size,
3260                    checkpoint_seq,
3261                    seg_frame_count,
3262                    uncommitted_tail_start,
3263                );
3264                // Begin the next segment under the new salts. Its base for replay is
3265                // the prior committed view (a checkpoint would have flushed it, but on
3266                // a forensic image we keep what we can replay).
3267                seg_ordinal += 1;
3268                salt1 = fsalt1;
3269                salt2 = fsalt2;
3270                seg_frame_count = 0;
3271                pending.clear();
3272                uncommitted_tail_start = None;
3273                // The post-reset frames replay onto the latest committed view.
3274                // committed_pages carries forward.
3275                // The checksum chain for a post-reset segment threads from a WAL
3276                // header we do NOT hold (the new generation's own 32-byte header
3277                // was overwritten), so its frames cannot be validated against our
3278                // seed. Mark the chain broken for this segment: its commits are
3279                // checksum-residue, surfaced for forensics but not trusted.
3280                chain_valid = false;
3281            }
3282
3283            if page_no == 0 {
3284                break; // malformed frame; stop rather than mis-index
3285            }
3286            let data = match frame.get(SQLITE_WAL_FRAME_HEADER_SIZE..) {
3287                Some(d) => d.to_vec(),
3288                None => break, // cov:unreachable: frame slice is exactly frame_stride
3289            };
3290
3291            // Advance the cumulative checksum over this frame (file-format §4.2):
3292            // the first 8 bytes of the frame header (page-no ++ db-size) followed
3293            // by the full page data — NOT the salt/checksum bytes (frame[8..24]).
3294            // Then compare against the frame's stored checksum (frame[16..24], big-
3295            // endian). A mismatch breaks the chain for the rest of the segment.
3296            // Only advance while the chain is still intact (a post-reset segment is
3297            // pre-marked broken and is not re-seedable from our header).
3298            if chain_valid {
3299                let (n0, n1) = wal_checksum(endian, run_s0, run_s1, &frame[0..8]);
3300                let (n0, n1) = wal_checksum(endian, n0, n1, &data);
3301                run_s0 = n0;
3302                run_s1 = n1;
3303                let stored0 = be_u32(frame, 16);
3304                let stored1 = be_u32(frame, 20);
3305                if stored0 != run_s0 || stored1 != run_s1 {
3306                    chain_valid = false;
3307                }
3308            }
3309
3310            let frame_index_in_seg = seg_frame_count;
3311            seg_frame_count += 1;
3312            pending.insert(page_no, data);
3313            let is_commit = db_size != 0;
3314
3315            if is_commit {
3316                for (p, d) in std::mem::take(&mut pending) {
3317                    committed_pages.insert(p, d);
3318                }
3319                // Drop base/committed pages beyond the committed size so a snapshot
3320                // reflects the database's page count at that commit. `db_size` is
3321                // non-zero here (that is what makes this a COMMIT frame).
3322                committed_pages.retain(|&p, _| p <= db_size);
3323                let id = CommitId {
3324                    segment: WalSegmentId(seg_ordinal),
3325                    commit_frame_index: frame_index_in_seg,
3326                    db_size_after_commit: db_size,
3327                };
3328                let overlaid = committed_pages.clone();
3329                // Usable bytes per page from the snapshot's OWN page-1 header
3330                // (reserved-space byte at offset 20), so a snapshot-scoped read
3331                // honors the reserved value as of this commit. Page 1 is always
3332                // materialized; a missing/short page-1 image degrades to 0 reserved.
3333                let reserved = overlaid
3334                    .get(&1)
3335                    .and_then(|p| p.get(RESERVED_SPACE_OFFSET).copied())
3336                    .unwrap_or(0);
3337                let usable = page_size.saturating_sub(u32::from(reserved));
3338                snapshots.push(CommitSnapshot {
3339                    id,
3340                    overlaid,
3341                    salt1,
3342                    salt2,
3343                    checksum_valid: chain_valid,
3344                    usable,
3345                });
3346                last_commit_global_frame = Some(frame_no);
3347                uncommitted_tail_start = None;
3348            } else if uncommitted_tail_start.is_none() {
3349                uncommitted_tail_start = Some(frame_index_in_seg);
3350            }
3351
3352            frame_no += 1;
3353            off += frame_stride;
3354        }
3355
3356        // Close the final segment (it may have an uncommitted tail).
3357        Self::close_segment(
3358            &mut segments,
3359            &mut residue,
3360            WalSegmentId(seg_ordinal),
3361            salt1,
3362            salt2,
3363            page_size,
3364            checkpoint_seq,
3365            seg_frame_count,
3366            uncommitted_tail_start,
3367        );
3368
3369        let safety = if snapshots.is_empty() {
3370            MaterializationSafety::PhysicalValidated
3371        } else if last_commit_global_frame.is_some() {
3372            MaterializationSafety::ReplaySafe
3373        } else {
3374            MaterializationSafety::CommitValidated // cov:unreachable: a snapshot implies a commit
3375        };
3376
3377        Ok(Self {
3378            page_size,
3379            base_pages,
3380            segments,
3381            snapshots,
3382            residue,
3383            safety,
3384        })
3385    }
3386
3387    #[allow(clippy::too_many_arguments)]
3388    fn close_segment(
3389        segments: &mut Vec<WalSegment>,
3390        residue: &mut Vec<WalResidue>,
3391        id: WalSegmentId,
3392        salt1: u32,
3393        salt2: u32,
3394        page_size: u32,
3395        checkpoint_seq: u32,
3396        frame_count: usize,
3397        uncommitted_tail_start: Option<usize>,
3398    ) {
3399        if frame_count == 0 {
3400            return;
3401        }
3402        segments.push(WalSegment {
3403            id,
3404            salt1,
3405            salt2,
3406            page_size,
3407            frame_count,
3408            checkpoint_seq,
3409        });
3410        if let Some(start) = uncommitted_tail_start {
3411            residue.push(WalResidue {
3412                segment: id,
3413                first_frame_index: start,
3414                frame_count: frame_count - start,
3415                reason: ResidueReason::BeyondLastCommit,
3416            });
3417        }
3418    }
3419
3420    /// The salt segments of this WAL, in file order (one per salt epoch).
3421    #[must_use]
3422    pub fn segments(&self) -> &[WalSegment] {
3423        &self.segments
3424    }
3425
3426    /// Every materializable [`CommitSnapshot`] across all segments, in commit order.
3427    #[must_use]
3428    pub fn commit_snapshots(&self) -> &[CommitSnapshot] {
3429        &self.snapshots
3430    }
3431
3432    /// The stale WAL tails surfaced for forensics (not committed history).
3433    #[must_use]
3434    pub fn residue(&self) -> &[WalResidue] {
3435        &self.residue
3436    }
3437
3438    /// Resolve a [`CommitId`] back to its [`CommitSnapshot`].
3439    #[must_use]
3440    pub fn snapshot_at(&self, id: CommitId) -> Option<&CommitSnapshot> {
3441        self.snapshots.iter().find(|s| s.id == id)
3442    }
3443
3444    /// The highest validation tier this WAL cleared (see [`MaterializationSafety`]).
3445    #[must_use]
3446    pub fn safety(&self) -> MaterializationSafety {
3447        self.safety
3448    }
3449
3450    /// The temporal-cohort topology — `LinearSegment` for one salt epoch, else
3451    /// `Disconnected` across checkpoint resets. The `[H]` adapter maps this onto
3452    /// `state-history-forensic::CohortTopology`.
3453    #[must_use]
3454    pub fn topology(&self) -> CohortTopology {
3455        if self.segments.len() <= 1 {
3456            CohortTopology::LinearSegment
3457        } else {
3458            CohortTopology::Disconnected
3459        }
3460    }
3461
3462    /// Whether the WAL's integrity checks are tamper-EVIDENT. Always `false`: WAL
3463    /// frame checksums are non-cryptographic (corruption detection, not tamper proof),
3464    /// so the `[H]` adapter must record `tamper_resistance = LOW`.
3465    #[must_use]
3466    pub fn checksums_are_tamper_evident(&self) -> bool {
3467        false
3468    }
3469
3470    /// Diff the acquired base image against the last valid commit snapshot, returning
3471    /// the page numbers whose bytes changed. `None` when there is no committed snapshot.
3472    #[must_use]
3473    pub fn diff_base_to_last_commit(&self) -> Option<WalDiff> {
3474        let last = self.snapshots.last()?;
3475        let mut changed = Vec::new();
3476        let mut pages: std::collections::BTreeSet<u32> = std::collections::BTreeSet::new();
3477        pages.extend(self.base_pages.keys().copied());
3478        pages.extend(last.overlaid.keys().copied());
3479        for p in pages {
3480            let base = self.base_pages.get(&p);
3481            let now = last.overlaid.get(&p);
3482            if base != now {
3483                changed.push(p);
3484            }
3485        }
3486        Some(WalDiff { changed })
3487    }
3488
3489    /// The page size (bytes) common to the base image and the WAL frames.
3490    #[must_use]
3491    pub fn page_size(&self) -> u32 {
3492        self.page_size
3493    }
3494
3495    /// Map this WAL timeline onto the canonical `forensicnomicon::history` cohort
3496    /// vocabulary — the `[H]` adapter (#43 / WS-F).
3497    ///
3498    /// Each materializable [`CommitSnapshot`] becomes one `TemporalState<CommitId>`:
3499    /// - **ordering key** — a salt-qualified `LsnKind::SqliteWalFrame` (`frame_seq` is the
3500    ///   COMMIT frame index; `commit_seq` is the 0-based commit ordinal within the salt
3501    ///   segment). The `(salt1, salt2)` pair keeps the key meaningful across a checkpoint
3502    ///   reset, which renumbers frames and rolls the salts.
3503    /// - **clock + safety** — the canonical SQLite-WAL profile, single-sourced from
3504    ///   [`forensicnomicon::history::profiles`], so no consumer re-asserts the four
3505    ///   classifications locally.
3506    /// - **handle** — the snapshot's [`CommitId`]; resolve it back via [`Self::snapshot_at`].
3507    ///
3508    /// The topology is uniformly `SubJournalCommits`: every state is a committed
3509    /// transaction, and a checkpoint reset is visible as a salt change *inside* the
3510    /// ordering key — there is no separate "disconnected" topology to special-case. The
3511    /// cohort is `PathStable` (a `-wal` belongs to exactly one database path), so the
3512    /// caller supplies the path identity via `artifact`.
3513    #[must_use]
3514    pub fn to_temporal_cohort(
3515        &self,
3516        artifact: forensicnomicon::history::identity::ArtifactRef,
3517    ) -> forensicnomicon::history::cohort::TemporalCohort<CommitId> {
3518        use forensicnomicon::history::cohort::{TemporalCohort, TemporalState};
3519        use forensicnomicon::history::epoch::{CohortTopology, EpochTag, LsnKind};
3520        use forensicnomicon::history::identity::IdentityDiscipline;
3521        use forensicnomicon::history::profiles;
3522
3523        // One canonical profile drives every state's clock + safety — read from
3524        // forensicnomicon, never re-asserted here, so the fleet cannot drift.
3525        let profile = profiles::SourceTemporalProfile::sqlite_wal();
3526        let mut commit_seq_in_segment: std::collections::HashMap<WalSegmentId, u32> =
3527            std::collections::HashMap::new();
3528
3529        let states = self
3530            .snapshots
3531            .iter()
3532            .map(|snap| {
3533                let id = snap.id();
3534                let lsn = snap.lsn();
3535                let seq = commit_seq_in_segment.entry(id.segment).or_insert(0);
3536                let commit_seq = *seq;
3537                *seq += 1;
3538
3539                // Deterministic and collision-free within a cohort: the
3540                // (salt1, salt2, commit_frame_index, db_size_after_commit) quadruple is
3541                // unique per commit state. Packed big-endian into the leading 16 bytes.
3542                let mut tag = [0u8; 32];
3543                tag[0..4].copy_from_slice(&lsn.salt1.to_be_bytes());
3544                tag[4..8].copy_from_slice(&lsn.salt2.to_be_bytes());
3545                tag[8..12].copy_from_slice(&(id.commit_frame_index as u32).to_be_bytes());
3546                tag[12..16].copy_from_slice(&id.db_size_after_commit.to_be_bytes());
3547
3548                TemporalState {
3549                    epoch: EpochTag::from_bytes(tag),
3550                    ordering_key: Some(LsnKind::SqliteWalFrame {
3551                        salt1: lsn.salt1,
3552                        salt2: lsn.salt2,
3553                        frame_seq: lsn.frame_index as u32,
3554                        commit_seq,
3555                    }),
3556                    wall_time: None,
3557                    clock: profile.clock.clone(),
3558                    safety: profile.safety.clone(),
3559                    handle: id,
3560                }
3561            })
3562            .collect();
3563
3564        TemporalCohort {
3565            artifact,
3566            discipline: IdentityDiscipline::PathStable,
3567            topology: CohortTopology::SubJournalCommits,
3568            states,
3569        }
3570    }
3571}
3572
3573/// Whether a decoded [`Value`] is **distinctive** enough to anchor a Tier-2
3574/// fragment emission (the §3.1 gate): TEXT of ≥ 4 bytes of valid UTF-8 (no
3575/// replacement char), or a REAL. Bare integers (1–8-byte serial patterns),
3576/// NULL, and BLOBs are NOT distinctive alone — a short integer byte-pattern
3577/// coincides far too often in a 4 `KiB` page to serve as identity, so it can ride
3578/// along inside a fragment but never justify emitting one.
3579fn is_distinctive(value: &Value) -> bool {
3580    match value {
3581        Value::Text(t) => t.len() >= 4 && !t.contains('\u{FFFD}'),
3582        Value::Real(_) => true,
3583        Value::Null | Value::Integer(_) | Value::Blob(_) => false,
3584    }
3585}
3586
3587/// The body byte-width of a serial type (file-format §2.1), or `None` for a
3588/// serial value that cannot legally appear in a record body.
3589fn serial_body_len(serial: i64) -> Option<usize> {
3590    match serial {
3591        0 | 8 | 9 | 10 | 11 => Some(0),
3592        1 => Some(1),
3593        2 => Some(2),
3594        3 => Some(3),
3595        4 => Some(4),
3596        5 => Some(6),
3597        6 | 7 => Some(8),
3598        n if n >= 12 => Some(((n - 12) / 2) as usize),
3599        _ => None, // negative serial: impossible
3600    }
3601}
3602
3603/// Byte length of a **live** table-leaf cell at `off`, for computing the byte
3604/// extent the cell occupies (so [`Database::carve_free_regions`] can exclude it).
3605/// Returns `None` if the cell header does not parse in bounds.
3606///
3607/// Mirrors the live cell layout: payload-length varint, rowid varint, then the
3608/// local payload (capped at the spill threshold) plus a 4-byte overflow pointer
3609/// when the payload spills. We only need the on-page footprint, so for a spilled
3610/// cell that is `local + 4` bytes, not the full reassembled payload.
3611fn live_cell_len(buf: &[u8], off: usize, usable: usize) -> Option<usize> {
3612    let (payload_len, n1) = read_varint(buf, off).ok()?;
3613    let (_rowid, n2) = read_varint(buf, off + n1).ok()?;
3614    let total = usize::try_from(payload_len).ok()?;
3615    let local = local_payload_len(total, usable);
3616    let on_page = if local >= total {
3617        n1 + n2 + total
3618    } else {
3619        n1 + n2 + local + 4 // 4-byte first-overflow-page pointer
3620    };
3621    Some(on_page)
3622}
3623
3624/// The rowid of a table-leaf cell at `off` — its 2nd varint (after the
3625/// payload-length varint). `None` if either varint is out of bounds. Used to
3626/// identify a live row even when its full record cannot be decoded.
3627fn live_cell_rowid(buf: &[u8], off: usize) -> Option<i64> {
3628    let (_payload_len, n1) = read_varint(buf, off).ok()?;
3629    let (rowid, _) = read_varint(buf, off + n1).ok()?;
3630    Some(rowid)
3631}
3632
3633/// Given the sorted byte extents of live cells, return the maximal **free**
3634/// (unallocated) spans within `[lo, hi)` — the complement of the live extents.
3635/// These are the only ranges [`Database::carve_free_regions`] scans, so a live
3636/// cell can never be re-surfaced.
3637fn free_regions(live: &[(usize, usize)], lo: usize, hi: usize) -> Vec<(usize, usize)> {
3638    let mut regions = Vec::new();
3639    // An inverted or empty range (lo >= hi) has no free regions. Guard before the
3640    // `clamp(lo, hi)` calls below, which panic when lo > hi (untrusted-input path).
3641    if lo >= hi {
3642        return regions;
3643    }
3644    let mut cursor = lo;
3645    for &(s, e) in live {
3646        let s = s.clamp(lo, hi);
3647        let e = e.clamp(lo, hi);
3648        if s > cursor {
3649            regions.push((cursor, s));
3650        }
3651        if e > cursor {
3652            cursor = e;
3653        }
3654    }
3655    if cursor < hi {
3656        regions.push((cursor, hi));
3657    }
3658    regions
3659}
3660
3661/// Derive a [`FreeblockTemplate`] from the first live cell on a table-leaf page:
3662/// the record's header length, its serial-type array, and the byte width of the
3663/// cell prefix (payload-length + rowid varints) that the freeblock header
3664/// overwrites. Returns `None` when no live cell parses or the prefix is wider
3665/// than the 4 bytes a freeblock header clobbers (the simple template cannot then
3666/// place the surviving serial tail).
3667/// Shared internal walker producing BOTH recovery tiers in one pass so the cell
3668/// and fragment outputs can never diverge: `(full_cells, fragments)`.
3669/// [`Database::reconstruct_freeblock_records`] takes `.0`,
3670/// [`Database::reconstruct_freeblock_fragments`] takes `.1`. A free function (it
3671/// needs no `Database` state — only the page bytes and the page-derived
3672/// template), keeping the two public entry points a thin projection of one walk.
3673fn reconstruct_freeblock_inner(
3674    page_bytes: &[u8],
3675    enc: TextEncoding,
3676) -> (Vec<CarvedCell>, Vec<CellFragment>) {
3677    let mut cells = Vec::new();
3678    let mut frags = Vec::new();
3679    let hdr_off = if page_bytes.starts_with(SQLITE_MAGIC) {
3680        SQLITE_HEADER_SIZE
3681    } else {
3682        0
3683    };
3684    let Some(&page_type) = page_bytes.get(hdr_off) else {
3685        return (cells, frags);
3686    };
3687    if page_type != 0x0d {
3688        return (cells, frags); // only table-leaf pages have freeblock residue
3689    }
3690    let Some(template) = freeblock_template(page_bytes, hdr_off, enc) else {
3691        return (cells, frags);
3692    };
3693
3694    let first_freeblock = be_u16(page_bytes, hdr_off + 1) as usize;
3695    let mut fb = first_freeblock;
3696    let mut walked = 0usize;
3697    let mut visited = std::collections::BTreeSet::new();
3698    while fb != 0 && walked < MAX_FREEBLOCKS_PER_PAGE {
3699        walked += 1;
3700        if !visited.insert(fb) {
3701            break; // cyclic next pointer
3702        }
3703        let next = be_u16(page_bytes, fb) as usize;
3704        let size = be_u16(page_bytes, fb + 2) as usize;
3705        let Some(fb_end) = fb.checked_add(size) else {
3706            break; // cov:unreachable: usize add of two u16-range values
3707        };
3708        if size >= 4 && fb_end <= page_bytes.len() {
3709            if template.known_lead_serials.is_empty() {
3710                // Empty-lead (2-byte-rowid) page: each freeblock is a single freed
3711                // cell whose serial array fully survives. Reconstruct it ONLY if
3712                // the record tiles the freeblock exactly — the precision gate that
3713                // rejects the misaligned runs a loose walk would manufacture.
3714                cells.extend(template.reconstruct_span_exact(page_bytes, fb, fb_end));
3715            } else {
3716                template
3717                    .reconstruct_span_tiered(page_bytes, fb, fb_end, false, &mut cells, &mut frags);
3718            }
3719        }
3720        fb = next;
3721    }
3722
3723    let cell_count = be_u16(page_bytes, hdr_off + 3) as usize;
3724    let cptr_end = hdr_off + 8 + cell_count * 2;
3725    let cca = be_u16(page_bytes, hdr_off + 5) as usize;
3726    // The unallocated-gap pass anchors off a surviving forward cell and a known
3727    // leading serial; it is meaningful only for the (non-empty-lead) span-walk
3728    // templates. Empty-lead pages recover solely through the exact-tile chain pass.
3729    if !template.known_lead_serials.is_empty() && cca > cptr_end && cca <= page_bytes.len() {
3730        for anchor_off in cptr_end..cca {
3731            let Some(anchor) =
3732                try_carve_cell_at(page_bytes, anchor_off, Some(template.column_count), enc)
3733            else {
3734                continue;
3735            };
3736            let has_text = anchor
3737                .values
3738                .iter()
3739                .any(|v| matches!(v, Value::Text(t) if !t.is_empty() && !t.contains('\u{FFFD}')));
3740            if !has_text {
3741                continue;
3742            }
3743            let tail_start = anchor.offset + anchor.byte_len;
3744            template
3745                .reconstruct_span_tiered(page_bytes, tail_start, cca, true, &mut cells, &mut frags);
3746            break; // one anchored run per page — the contiguous freed tail
3747        }
3748    }
3749    (cells, frags)
3750}
3751
3752fn freeblock_template(
3753    page_bytes: &[u8],
3754    hdr_off: usize,
3755    enc: TextEncoding,
3756) -> Option<FreeblockTemplate> {
3757    let cell_count = be_u16(page_bytes, hdr_off + 3) as usize;
3758    let cell_ptr_array = hdr_off + 8;
3759    for i in 0..cell_count {
3760        let cell_off = be_u16(page_bytes, cell_ptr_array + i * 2) as usize;
3761        if cell_off == 0 || cell_off >= page_bytes.len() {
3762            continue;
3763        }
3764        // Prefix: payload-length varint, rowid varint.
3765        let Ok((_payload_len, n1)) = read_varint(page_bytes, cell_off) else {
3766            continue; // cov:unreachable: a live cell-pointer addresses an in-bounds prefix
3767        };
3768        let Ok((_rowid, n2)) = read_varint(page_bytes, cell_off + n1) else {
3769            continue; // cov:unreachable: the rowid varint follows the payload-len varint in-page
3770        };
3771        let prefix_len = n1 + n2;
3772        // The freeblock header overwrites exactly 4 bytes. If the prefix alone is
3773        // wider, no record-header byte is clobbered in a way this simple template
3774        // handles — skip (those tables keep an intact header tail the forward
3775        // carver already reaches).
3776        if prefix_len > 4 {
3777            continue; // cov:unreachable: the corpus tables all encode a <=4-byte cell prefix
3778        }
3779        let payload_start = cell_off + n1 + n2;
3780        let Ok((header_len, hn)) = read_varint(page_bytes, payload_start) else {
3781            continue; // cov:unreachable: a live cell's record header follows its prefix in-page
3782        };
3783        let header_len = usize::try_from(header_len).ok()?;
3784        if header_len < hn {
3785            continue; // cov:unreachable: a live record's header_len covers its own varint
3786        }
3787        // Read the template's serial-type array, recording each serial's byte
3788        // offset within the header so we can split clobbered vs surviving.
3789        let mut serials = Vec::new();
3790        let mut hpos = hn;
3791        let mut ok = true;
3792        while hpos < header_len {
3793            let Ok((s, used)) = read_varint(page_bytes, payload_start + hpos) else {
3794                ok = false; // cov:unreachable: header_len bounds the serial array within the page
3795                break; // cov:unreachable: paired with the read failure above
3796            };
3797            serials.push((s, hpos, used));
3798            hpos += used;
3799        }
3800        if !ok || hpos != header_len || serials.len() < MIN_INFERRED_COLUMNS {
3801            continue; // cov:unreachable: a live cell's header parses cleanly with >= 2 columns
3802        }
3803        return FreeblockTemplate::build(prefix_len, header_len, hn, &serials, enc);
3804    }
3805    None
3806}
3807
3808/// A record-header template derived from a live cell on a table-leaf page, used
3809/// to rebuild freeblock-clobbered records (see
3810/// [`Database::reconstruct_freeblock_records`]).
3811///
3812/// Freeblock conversion overwrites the freed cell's first four bytes — the
3813/// payload-length + rowid varints, the record `header_len`, and the leading
3814/// serial type(s). The surviving serial-type tail and the value body remain. The
3815/// template supplies what was destroyed: the total column count, the serial types
3816/// of the leading (clobbered) columns, and the page offset, relative to the
3817/// freeblock start, at which the surviving serial tail begins.
3818struct FreeblockTemplate {
3819    /// Total number of columns in a record of this table.
3820    column_count: usize,
3821    /// Serial types of the leading columns whose header bytes the freeblock
3822    /// header clobbered (taken from the template; e.g. the fixed-width `id`).
3823    known_lead_serials: Vec<i64>,
3824    /// Offset, relative to the freeblock start, at which the **surviving** serial
3825    /// tail begins (== `prefix_len + first_surviving_serial_header_offset`).
3826    surviving_serials_off: usize,
3827    /// Text encoding of the owning database, so reconstructed text decodes per
3828    /// the header (UTF-8 / UTF-16) rather than assuming UTF-8.
3829    text_encoding: TextEncoding,
3830}
3831
3832impl FreeblockTemplate {
3833    /// Build a template from a parsed live-cell header. `serials` is the list of
3834    /// `(serial_type, header_offset, varint_width)` tuples for every column.
3835    /// Returns `None` when the 4-byte freeblock clobber boundary cannot be
3836    /// resolved to a clean split between leading and surviving serials.
3837    fn build(
3838        prefix_len: usize,
3839        _header_len: usize,
3840        _hn: usize,
3841        serials: &[(i64, usize, usize)],
3842        enc: TextEncoding,
3843    ) -> Option<FreeblockTemplate> {
3844        // Bytes of the record header the 4-byte freeblock header destroys.
3845        let clobbered_header_bytes = 4usize.checked_sub(prefix_len)?;
3846        // The first column whose header bytes survive intact is the first serial
3847        // whose header offset is at or beyond the clobber boundary. Everything
3848        // before it is supplied from the template.
3849        let mut known_lead = Vec::new();
3850        let mut surviving_serials_off = None;
3851        for &(serial, hpos, _used) in serials {
3852            if hpos >= clobbered_header_bytes {
3853                surviving_serials_off = Some(prefix_len + hpos);
3854                break;
3855            }
3856            known_lead.push(serial);
3857        }
3858        // At least one serial must survive to anchor the reconstruction. The
3859        // leading (clobbered) serial list MAY be empty: a 2-byte-or-wider rowid
3860        // varint (rowid >= 128) widens the cell prefix so the 4-byte freeblock
3861        // clobber stops at `header_len`, destroying NO serial type — the whole
3862        // serial array survives. Such pages reconstruct via the exact-tile
3863        // single-cell path (`reconstruct_freeblock_inner` routes on
3864        // `known_lead_serials.is_empty()`), which requires each freed cell to fill
3865        // its freeblock exactly; that precision check keeps the empty-lead case
3866        // phantom-free where a loose span walk would mis-align columns.
3867        let surviving_serials_off = surviving_serials_off?;
3868        Some(FreeblockTemplate {
3869            column_count: serials.len(),
3870            known_lead_serials: known_lead,
3871            surviving_serials_off,
3872            text_encoding: enc,
3873        })
3874    }
3875
3876    /// Reconstruct **every** clobbered cell coalesced into the free span
3877    /// `[lo, hi)` — a chained freeblock or a page's unallocated gap — and append
3878    /// each to `out`.
3879    ///
3880    /// When SQLite frees adjacent cells it coalesces them into one freeblock whose
3881    /// interior still holds the freed cells back-to-back, **each** prefixed by a
3882    /// stale 4-byte freeblock header (`next`/`size`) that clobbers that cell's
3883    /// payload-length + rowid varints and leading serial(s). A single-shot
3884    /// reconstruction at `lo` recovers only the span's first cell; the trailing
3885    /// cells are intact records sitting at the previous record's end. This walks
3886    /// the template across the span: reconstruct at `lo`, advance to that record's
3887    /// end, repeat to `hi`. Every value is derived from the span bounds and the
3888    /// page's own schema template — no per-cell or per-database constant.
3889    ///
3890    /// Each candidate is validated identically to the single-cell case (legal
3891    /// serial types, record fits within `[cell_start, hi)`). The walk is
3892    /// **structural, not a sliding scan**: SQLite coalesces freed cells exactly
3893    /// back-to-back (each freed record's end abuts the next freed cell's clobbered
3894    /// 4-byte prefix), so the next cell begins precisely at the previous record's
3895    /// end. The walk therefore reconstructs at `lo`, advances to that record's
3896    /// end, and repeats — and STOPS the moment a position does not reconstruct
3897    /// cleanly. It never slides forward byte-by-byte hunting for the next cell:
3898    /// that fallback would synthesize a record from any run of bytes that happens
3899    /// to satisfy the legal-serial + fits-in-span checks, manufacturing phantoms
3900    /// in non-cell free space. Anchoring every cell at the prior record's exact
3901    /// end is what keeps the broader span-walk at single-cell precision. Bounded:
3902    /// the walk strictly advances (a record is non-empty) and is capped at
3903    /// [`MAX_FREEBLOCKS_PER_PAGE`] reconstructions per span.
3904    ///
3905    /// Follower precision (the coalesced-freeblock signature): the span's FIRST
3906    /// cell at `lo` is reconstructed unconditionally — `lo` is a real boundary (a
3907    /// freeblock-chain entry, or the gap anchor's first follower). Every SUBSEQUENT
3908    /// follower must carry the structural mark of a freed-and-coalesced cell: its
3909    /// clobbered 4-byte prefix is a stale freeblock header whose 2-byte `next`
3910    /// field is `0x0000` (a terminal/orphaned freeblock — what SQLite leaves when
3911    /// it coalesces freed cells back-to-back). A position whose leading two bytes
3912    /// are non-zero is a byte-shifted remnant, not a coalesced cell, so the run
3913    /// ends there. This is the check that separates a true coalesced tail (0D-06's
3914    /// `00 00 NN NN`-prefixed followers) from a misaligned fragment (0B-02's
3915    /// `24 09 …` remnant), keeping the gap pass phantom-free.
3916    ///
3917    /// `enforce_follower_mark` is `true` for the unallocated-gap pass, where the
3918    /// span is bounded only by `cellContentArea` (not by a page-recorded freeblock
3919    /// size) and so a byte-shifted remnant could otherwise be mistaken for a
3920    /// follower: there EVERY position must carry the `next == 0` mark. It is `false`
3921    /// for the freeblock-chain pass, whose span bounds are the page-recorded
3922    /// `[fb, fb + size)` — a strong boundary that already pins the coalesced run, so
3923    /// the interior followers (whose clobbered bytes are the original record's own
3924    /// varints, not necessarily `00 00 …`) are accepted on the fit-in-span check
3925    /// alone.
3926    ///
3927    /// Tiered walk: it pushes each reconstructed full cell into `cells`, and at the
3928    /// anchor where `reconstruct_one` would `break` it salvages the maximal
3929    /// decodable column prefix into `frags` as a [`CellFragment`] (when the §3.1
3930    /// distinctiveness gate passes) before stopping. Fragment salvage does NOT
3931    /// extend the walk — it stops at exactly the position the full walk does,
3932    /// preserving Tier-1's phantom discipline. Callers that want only the full
3933    /// cells (the Tier-1 [`Database::reconstruct_freeblock_records`]) discard
3934    /// `frags`; both tiers therefore come from one walk and can never diverge.
3935    fn reconstruct_span_tiered(
3936        &self,
3937        page: &[u8],
3938        lo: usize,
3939        hi: usize,
3940        enforce_follower_mark: bool,
3941        cells: &mut Vec<CarvedCell>,
3942        frags: &mut Vec<CellFragment>,
3943    ) {
3944        let mut cell_start = lo;
3945        let mut built = 0usize;
3946        while cell_start < hi && built < MAX_FREEBLOCKS_PER_PAGE {
3947            if enforce_follower_mark && be_u16(page, cell_start) != 0 {
3948                break; // not a coalesced freeblock follower — the contiguous run ends
3949            }
3950            let Some((cell, record_end)) = self.reconstruct_one(page, cell_start, hi) else {
3951                // Full reconstruction failed at this anchor; try to salvage the
3952                // decodable prefix as a fragment, then stop (do not extend the
3953                // walk past the failed anchor).
3954                if let Some(frag) = self.salvage_fragment(page, cell_start, hi) {
3955                    frags.push(frag);
3956                }
3957                break;
3958            };
3959            cells.push(cell);
3960            built += 1;
3961            cell_start = record_end;
3962        }
3963    }
3964
3965    /// Salvage the maximal decodable column prefix at `cell_start` (bounded by
3966    /// `span_end`) when full reconstruction failed there. Walks the template +
3967    /// surviving serial array forward, decoding each column's body while it fits
3968    /// in the span; the first illegal serial, out-of-bounds read, or body that
3969    /// overruns the span ends the prefix. Returns a [`CellFragment`] **only** when
3970    /// the salvaged prefix contains at least one distinctive cell (TEXT ≥ 4 bytes
3971    /// of valid UTF-8, or REAL) — the §3.1 emission gate — otherwise `None`.
3972    fn salvage_fragment(
3973        &self,
3974        page: &[u8],
3975        cell_start: usize,
3976        span_end: usize,
3977    ) -> Option<CellFragment> {
3978        let surviving_count = self.column_count - self.known_lead_serials.len();
3979        let tail_start = cell_start.checked_add(self.surviving_serials_off)?;
3980
3981        // Read as many legal surviving serials as decode in-bounds within the span.
3982        // The template's leading serials are always legal (they came from a live
3983        // cell), so the full serial array is `known_lead ++ legal_surviving`.
3984        let mut serials = self.known_lead_serials.clone();
3985        let mut pos = tail_start;
3986        for _ in 0..surviving_count {
3987            let Ok((s, used)) = read_varint(page, pos) else {
3988                break; // cov:unreachable: the surviving serials sit near the cell start, inside the freeblock/gap span the inner walker already bounds to the page; this read mirrors reconstruct_one's bounds guard so a truncated tail ends the prefix rather than panicking
3989            };
3990            if serial_body_len(s).is_none() {
3991                break; // cov:unreachable: serial_body_len is None only for a negative serial, which read_varint yields only from a crafted 9-byte varint; kept as a defence-in-depth guard so a malformed surviving tail ends the prefix rather than mis-decoding
3992            }
3993            let Some(next) = pos.checked_add(used) else {
3994                break; // cov:unreachable: usize add of an in-page varint width
3995            };
3996            if next > span_end {
3997                break; // serial tail overran the span
3998            }
3999            serials.push(s);
4000            pos = next;
4001        }
4002
4003        // Decode column bodies left-to-right, keeping each whose body ends within
4004        // the span. The body begins right after the surviving serial tail.
4005        let body_start = pos;
4006        let mut surviving: Vec<(usize, Value)> = Vec::new();
4007        let mut bpos = body_start;
4008        for (idx, &s) in serials.iter().enumerate() {
4009            let Some(blen) = serial_body_len(s) else {
4010                break; // cov:unreachable: only legal serials were pushed above
4011            };
4012            let Some(body_end) = bpos.checked_add(blen) else {
4013                break; // cov:unreachable: usize add of an in-page body length
4014            };
4015            if body_end > span_end {
4016                break; // this column's body overruns the span — prefix ends here
4017            }
4018            let Some(body) = page.get(bpos..body_end) else {
4019                break; // cov:unreachable: body_end <= span_end <= page.len()
4020            };
4021            let Ok((val, _)) = decode_value(body, 0, s, self.text_encoding) else {
4022                break; // cov:unreachable: serial_body_len-legal serials decode in-bounds
4023            };
4024            surviving.push((idx, val));
4025            bpos = body_end;
4026        }
4027
4028        // Emission gate: at least one distinctive cell (TEXT >= 4 UTF-8 bytes, or
4029        // REAL). A lone integer/NULL/blob prefix is coincidence-prone — no fragment.
4030        if !surviving.iter().any(|(_, v)| is_distinctive(v)) {
4031            return None;
4032        }
4033        let last_body_end = bpos;
4034        Some(CellFragment {
4035            offset: cell_start,
4036            byte_len: last_body_end.saturating_sub(cell_start),
4037            missing: self.column_count - surviving.len(),
4038            surviving,
4039            confidence: FRAGMENT_CONFIDENCE,
4040        })
4041    }
4042
4043    /// Rebuild the single record whose clobbered cell begins at `cell_start`,
4044    /// bounded by the enclosing span end `span_end`: read the surviving serial
4045    /// tail, prepend the template's leading serials, decode the body, and validate
4046    /// the whole record fits within `[cell_start, span_end)`. Returns the carved
4047    /// cell and the record's end offset (the next coalesced cell's start), or
4048    /// `None` on any out-of-bounds or implausible parse.
4049    fn reconstruct_one(
4050        &self,
4051        page: &[u8],
4052        cell_start: usize,
4053        span_end: usize,
4054    ) -> Option<(CarvedCell, usize)> {
4055        let surviving_count = self.column_count - self.known_lead_serials.len();
4056        let tail_start = cell_start.checked_add(self.surviving_serials_off)?;
4057
4058        // Read the surviving serial tail from the freeblock.
4059        let mut serials = self.known_lead_serials.clone();
4060        let mut pos = tail_start;
4061        for _ in 0..surviving_count {
4062            let (s, used) = read_varint(page, pos).ok()?;
4063            // A serial type must be legal; reject the candidate otherwise.
4064            serial_body_len(s)?;
4065            serials.push(s);
4066            pos = pos.checked_add(used)?;
4067            if pos > span_end {
4068                return None;
4069            }
4070        }
4071
4072        // The body begins right after the surviving serial tail. Compute its
4073        // length from the full (template + surviving) serial array.
4074        let mut body_len = 0usize;
4075        for &s in &serials {
4076            body_len = body_len.checked_add(serial_body_len(s)?)?;
4077        }
4078        let body_start = pos;
4079        let record_end = body_start.checked_add(body_len)?;
4080        // The reconstructed record MUST fit within the enclosing span — the core
4081        // precision check that rejects coincidental/garbage reconstructions.
4082        if record_end > span_end {
4083            return None;
4084        }
4085
4086        // Synthesize a record payload (header + body) for the shared decoder so
4087        // values are decoded with the same storage-class fidelity as live rows.
4088        // The rowid is destroyed; pass 0 so a serial-0 column reads as NULL rather
4089        // than a fabricated rowid.
4090        let body = page.get(body_start..record_end)?;
4091        let values = decode_synthetic_record(&serials, body, self.text_encoding)?;
4092        if values.len() != self.column_count {
4093            return None; // cov:unreachable: one value per serial by construction
4094        }
4095
4096        Some((
4097            CarvedCell {
4098                offset: cell_start,
4099                byte_len: record_end - cell_start,
4100                rowid: 0, // destroyed by freeblock conversion — surfaced as unknown
4101                values,
4102                confidence: FREEBLOCK_RECONSTRUCT_CONFIDENCE,
4103            },
4104            record_end,
4105        ))
4106    }
4107
4108    /// Reconstruct ONE freeblock-clobbered empty-leading-serial cell at
4109    /// `cell_start` — a 2-byte-or-wider rowid, so the 4-byte clobber destroyed no
4110    /// serial type and the whole serial array survives at
4111    /// `cell_start + surviving_serials_off`. Returns the carved cell (rowid
4112    /// destroyed → 0) **and the record's end offset**, or `None` on any
4113    /// out-of-bounds parse or a record that overruns `span_end`. Does NOT enforce
4114    /// an exact tile — the span walker [`Self::reconstruct_span_exact`] does.
4115    fn reconstruct_cell_empty_lead(
4116        &self,
4117        page: &[u8],
4118        cell_start: usize,
4119        span_end: usize,
4120    ) -> Option<(CarvedCell, usize)> {
4121        let tail_start = cell_start.checked_add(self.surviving_serials_off)?;
4122        // The whole serial array survives (no clobbered leading serial); read all
4123        // `column_count` serials from the freeblock.
4124        let mut serials = Vec::with_capacity(self.column_count);
4125        let mut pos = tail_start;
4126        for _ in 0..self.column_count {
4127            let (s, used) = read_varint(page, pos).ok()?;
4128            serial_body_len(s)?;
4129            serials.push(s);
4130            pos = pos.checked_add(used)?;
4131            if pos > span_end {
4132                return None;
4133            }
4134        }
4135        let mut body_len = 0usize;
4136        for &s in &serials {
4137            body_len = body_len.checked_add(serial_body_len(s)?)?;
4138        }
4139        let body_start = pos;
4140        let record_end = body_start.checked_add(body_len)?;
4141        if record_end > span_end {
4142            return None;
4143        }
4144        let body = page.get(body_start..record_end)?;
4145        let values = decode_synthetic_record(&serials, body, self.text_encoding)?;
4146        if values.len() != self.column_count {
4147            return None; // cov:unreachable: one value per serial by construction
4148        }
4149        Some((
4150            CarvedCell {
4151                offset: cell_start,
4152                byte_len: record_end - cell_start,
4153                rowid: 0, // destroyed by freeblock conversion — surfaced as unknown
4154                values,
4155                confidence: FREEBLOCK_RECONSTRUCT_CONFIDENCE,
4156            },
4157            record_end,
4158        ))
4159    }
4160
4161    /// Reconstruct every empty-leading-serial cell coalesced into the freeblock
4162    /// `[lo, hi)`, returned ONLY when they tile the freeblock **exactly** (the
4163    /// walk reaches `hi` with no leftover bytes).
4164    ///
4165    /// A single freed cell fills its freeblock exactly; adjacent deletions
4166    /// coalesce into one freeblock whose interior holds the freed cells
4167    /// back-to-back, each clobbered in its first 4 bytes. Walking cell-to-cell and
4168    /// requiring the run to land precisely on `hi` is the precision gate: a
4169    /// misaligned read (a deleted cell whose destroyed rowid width differs from the
4170    /// template's) fails to reach `hi` exactly, so the whole span is rejected
4171    /// rather than emitted as column-shifted phantoms. Bounded by
4172    /// [`MAX_FREEBLOCKS_PER_PAGE`]; a record always advances `cell_start`.
4173    fn reconstruct_span_exact(&self, page: &[u8], lo: usize, hi: usize) -> Vec<CarvedCell> {
4174        let mut cells = Vec::new();
4175        let mut cell_start = lo;
4176        let mut guard = 0usize;
4177        while cell_start < hi && guard < MAX_FREEBLOCKS_PER_PAGE {
4178            guard += 1;
4179            let Some((cell, record_end)) = self.reconstruct_cell_empty_lead(page, cell_start, hi)
4180            else {
4181                return Vec::new(); // a cell did not reconstruct → not a clean tiling
4182            };
4183            if record_end <= cell_start {
4184                return Vec::new(); // cov:unreachable: a non-empty record advances cell_start
4185            }
4186            cells.push(cell);
4187            cell_start = record_end;
4188        }
4189        // Exact tile: leftover bytes (or a walk stopped by the bound) mean a
4190        // misaligned run — emit nothing.
4191        if cell_start == hi {
4192            cells
4193        } else {
4194            Vec::new()
4195        }
4196    }
4197
4198    /// Reconstruct a freeblock-clobbered **spilled** cell at `cell_start` (task
4199    /// #73, design §2.2). A spilled cell always carries a multi-byte
4200    /// `payload_len` varint, so the 4-byte freeblock clobber destroys the
4201    /// `payload_len` + `rowid` varints and the record's `header_len` varint —
4202    /// **but not the serial-type array**, which survives intact immediately after
4203    /// the clobber. We therefore read the full serial array directly from
4204    /// `cell_start + CLOBBER` (using the template only for the column count),
4205    /// re-derive `header_len` and `P = header_len + Σ serial_body_len`, and — when
4206    /// `P > usable - 35` — resolve the spill: `local_payload_len(P, usable)` bytes
4207    /// of payload sit locally (the destroyed header counted within them), the
4208    /// 4-byte first-overflow pointer follows, and the chain is resolved through
4209    /// freelist leaves. Returns `(cell, chain)` with `rowid = 0`, or `None`.
4210    ///
4211    /// UNPROVEN-BY-CORPUS (Codex ruling #5): synthetic-fixture validation only.
4212    /// No real Nemetz cell is both freeblock-clobbered and spilled.
4213    fn reconstruct_spilled(
4214        &self,
4215        db: &Database,
4216        page: &[u8],
4217        cell_start: usize,
4218        usable: usize,
4219        freed_leaves: &std::collections::BTreeSet<u32>,
4220    ) -> Option<(CarvedCell, Vec<u32>)> {
4221        // The freeblock header clobbers exactly 4 bytes. For a spilled cell those
4222        // 4 bytes are payload_len(>=2) + rowid(>=1) + header_len(>=1) varints, so
4223        // the serial array begins right after the clobber.
4224        const CLOBBER: usize = 4;
4225        let serials_start = cell_start.checked_add(CLOBBER)?;
4226        let mut serials = Vec::with_capacity(self.column_count);
4227        let mut pos = serials_start;
4228        for _ in 0..self.column_count {
4229            let (s, used) = read_varint(page, pos).ok()?;
4230            serial_body_len(s)?;
4231            serials.push(s);
4232            pos = pos.checked_add(used)?;
4233        }
4234
4235        // Re-derive the record header bytes that were destroyed: header_len is a
4236        // varint counting itself plus the serial array.
4237        let mut serial_bytes_len = 0usize;
4238        for &s in &serials {
4239            serial_bytes_len += varint_len(s);
4240        }
4241        let mut header_len = serial_bytes_len + 1;
4242        while varint_len(header_len as i64) + serial_bytes_len != header_len {
4243            header_len += 1;
4244        }
4245        // The clobber removed `header_len`'s own varint plus the prefix; verify the
4246        // surviving serial array aligns with the reconstructed header (the bytes
4247        // from serials_start to `pos` are the serial array, length serial_bytes_len).
4248        if pos.checked_sub(serials_start)? != serial_bytes_len {
4249            return None; // cov:unreachable: read_varint widths sum to serial_bytes_len
4250        }
4251        let mut body_len = 0usize;
4252        for &s in &serials {
4253            body_len = body_len.checked_add(serial_body_len(s)?)?;
4254        }
4255        let payload_len = header_len.checked_add(body_len)?;
4256        // Only the spilled class — an in-page payload is the existing template path.
4257        if payload_len <= usable.checked_sub(35)? {
4258            return None;
4259        }
4260        let local_len = local_payload_len(payload_len, usable);
4261
4262        // The body starts right after the surviving serial array. The local payload
4263        // spans `local_len` bytes of (header ++ body); the destroyed header is
4264        // `header_len` of those, so `local_len - header_len` body bytes are present
4265        // locally before the 4-byte first-overflow pointer.
4266        let body_start = pos;
4267        let local_body = local_len.checked_sub(header_len)?;
4268        let local_body_end = body_start.checked_add(local_body)?;
4269        let ptr_off = local_body_end;
4270        let ptr_slice = page.get(ptr_off..ptr_off + 4)?;
4271        let first_overflow =
4272            u32::from_be_bytes([ptr_slice[0], ptr_slice[1], ptr_slice[2], ptr_slice[3]]);
4273        let local_body_bytes = page.get(body_start..local_body_end)?;
4274
4275        let remaining = payload_len - local_len;
4276        let (chain_content, chain) = db
4277            .read_freed_overflow_chain(first_overflow, remaining, usable, freed_leaves)
4278            .ok()?;
4279
4280        // Assemble the full payload: reconstructed header ++ local body ++ chain.
4281        let mut header = enc_varint_into(header_len);
4282        for &s in &serials {
4283            header.extend(enc_varint_into(usize::try_from(s).ok()?));
4284        }
4285        if header.len() != header_len {
4286            return None; // cov:unreachable: header_len was solved to this width
4287        }
4288        let mut payload = Vec::with_capacity(payload_len);
4289        payload.extend_from_slice(&header);
4290        payload.extend_from_slice(local_body_bytes);
4291        payload.extend_from_slice(&chain_content);
4292        if payload.len() != payload_len {
4293            return None; // cov:unreachable: local_body + chain == body_len by construction
4294        }
4295
4296        let values = decode_record(&payload, self.column_count, 0, db.header.text_encoding).ok()?;
4297        if values.len() != self.column_count {
4298            return None; // cov:unreachable: one value per serial
4299        }
4300        let any_replacement = values.iter().any(|v| match v {
4301            Value::Text(t) => t.contains('\u{FFFD}'),
4302            _ => false,
4303        });
4304        if any_replacement {
4305            return None;
4306        }
4307        if !values.iter().any(is_distinctive) {
4308            return None;
4309        }
4310
4311        Some((
4312            CarvedCell {
4313                offset: cell_start,
4314                byte_len: ptr_off + 4 - cell_start,
4315                rowid: 0,
4316                values,
4317                confidence: FREEBLOCK_RECONSTRUCT_CONFIDENCE * OVERFLOW_CHAIN_CONFIDENCE_FACTOR,
4318            },
4319            chain,
4320        ))
4321    }
4322}
4323
4324/// Decode a record body given an explicit serial-type array (the freeblock
4325/// reconstructor supplies the array; the on-disk `header_len` + leading serials
4326/// were destroyed). Mirrors [`decode_record`]'s body pass. Returns `None` on any
4327/// out-of-bounds read so a malformed reconstruction is rejected, never panics.
4328fn decode_synthetic_record(serials: &[i64], body: &[u8], enc: TextEncoding) -> Option<Vec<Value>> {
4329    let mut values = Vec::with_capacity(serials.len());
4330    let mut bpos = 0usize;
4331    for &serial in serials {
4332        let (val, size) = decode_value(body, bpos, serial, enc).ok()?;
4333        values.push(val);
4334        bpos = bpos.checked_add(size)?;
4335    }
4336    Some(values)
4337}
4338
4339/// Attempt to recognize a table-leaf cell at `off` in `buf` as a record.
4340///
4341/// `expected_columns` is `Some(n)` to require exactly `n` columns (fixed-schema
4342/// carving), or `None` to **infer** the column count from the record's own
4343/// serial-type array (dropped-table / schema-gone carving). Returns a
4344/// [`CarvedCell`] only when the bytes are self-consistently record-shaped;
4345/// otherwise `None`. Never panics — every access is bounds-checked.
4346fn try_carve_cell_at(
4347    buf: &[u8],
4348    off: usize,
4349    expected_columns: Option<usize>,
4350    enc: TextEncoding,
4351) -> Option<CarvedCell> {
4352    // Cell prefix: payload_len varint, rowid varint.
4353    let (payload_len, n1) = read_varint(buf, off).ok()?;
4354    let payload_len = usize::try_from(payload_len).ok()?;
4355    if payload_len == 0 {
4356        return None;
4357    }
4358    let (rowid, n2) = read_varint(buf, off + n1).ok()?;
4359    // A negative rowid is legal but vanishingly rare for browser tables; treat a
4360    // non-positive rowid as a non-match to suppress coincidental hits.
4361    if rowid <= 0 {
4362        return None;
4363    }
4364    let payload_start = off + n1 + n2;
4365    let payload = buf.get(payload_start..payload_start + payload_len)?;
4366
4367    // Record header: header_len varint, then one serial type per column.
4368    let (header_len, hn) = read_varint(payload, 0).ok()?;
4369    let header_len = usize::try_from(header_len).ok()?;
4370    if header_len > payload.len() || header_len < hn {
4371        return None;
4372    }
4373    let cap = expected_columns.unwrap_or(0);
4374    let mut serials = Vec::with_capacity(cap);
4375    let mut hpos = hn;
4376    while hpos < header_len {
4377        let (s, used) = read_varint(payload, hpos).ok()?;
4378        serials.push(s);
4379        hpos += used;
4380    }
4381    // The header must consume cleanly, and match the expected column count when
4382    // one was given. When inferring, require a minimum plausible column count to
4383    // suppress coincidental 1-column matches.
4384    if hpos != header_len {
4385        return None;
4386    }
4387    match expected_columns {
4388        Some(n) if serials.len() != n => return None,
4389        None if serials.len() < MIN_INFERRED_COLUMNS => return None,
4390        _ => {}
4391    }
4392    let column_count = serials.len();
4393
4394    // Body length implied by the serial types must equal payload_len - header_len
4395    // — a strong self-consistency check that rejects coincidental matches.
4396    let mut body_len = 0usize;
4397    for &s in &serials {
4398        // Checked: a serial from free-space bytes can declare a body length near
4399        // usize::MAX; summing must reject (None) on overflow, never panic/wrap.
4400        body_len = body_len.checked_add(serial_body_len(s)?)?;
4401    }
4402    if header_len + body_len != payload_len {
4403        return None;
4404    }
4405
4406    // Decode the record (reusing the live decoder for storage-class fidelity).
4407    let values = decode_record(payload, column_count, rowid, enc).ok()?;
4408    if values.len() != column_count {
4409        return None; // cov:unreachable: decode_record yields one value per serial
4410    }
4411
4412    // Confidence: a fully self-consistent record already passed strong checks;
4413    // raise confidence when at least one column is a non-empty, valid-UTF-8 TEXT
4414    // (record-shaped *and* human-meaningful), which coincidental byte runs rarely
4415    // satisfy.
4416    let has_real_text = values.iter().any(|v| match v {
4417        Value::Text(t) => !t.is_empty() && !t.contains('\u{FFFD}'),
4418        _ => false,
4419    });
4420    let confidence = if has_real_text { 0.9 } else { 0.6 };
4421
4422    Some(CarvedCell {
4423        offset: off,
4424        byte_len: (payload_start + payload_len) - off,
4425        rowid,
4426        values,
4427        confidence,
4428    })
4429}
4430
4431/// Recognize a freed **spilled** table-leaf cell at `off` whose payload exceeds
4432/// the in-page threshold (`usable - 35`) and therefore continues on an
4433/// overflow-page chain (task #73). The sibling of [`try_carve_cell_at`] for the
4434/// overflow class: the two partition the candidate space by the spec spill
4435/// threshold, so a cell is recognized by exactly one of them.
4436///
4437/// `expected_columns` is `Some(n)` to require exactly `n` columns, or `None` to
4438/// infer the count (≥ [`MIN_INFERRED_COLUMNS`]). Returns a [`SpilledCell`]
4439/// (recognition only — the chain is resolved later) when the local prefix is
4440/// self-consistent: header fits in the local payload, the serial array consumes
4441/// the header cleanly, `header_len + Σ serial_body_len == P` (length closure
4442/// over the *declared* P), and the local payload plus its 4-byte overflow
4443/// pointer are in-bounds. Never panics — every access is bounds-checked.
4444fn try_carve_spilled_cell_at(
4445    buf: &[u8],
4446    off: usize,
4447    usable: usize,
4448    expected_columns: Option<usize>,
4449) -> Option<SpilledCell> {
4450    let (payload_len, n1) = read_varint(buf, off).ok()?;
4451    let payload_len = usize::try_from(payload_len).ok()?;
4452    // Only the overflow class — in-page payloads belong to `try_carve_cell_at`.
4453    if payload_len <= usable.checked_sub(35)? {
4454        return None;
4455    }
4456    let (rowid, n2) = read_varint(buf, off + n1).ok()?;
4457    if rowid <= 0 {
4458        return None;
4459    }
4460    let payload_start = off + n1 + n2;
4461    let local_len = local_payload_len(payload_len, usable);
4462    // The local payload prefix plus the 4-byte first-overflow pointer must be in
4463    // bounds of the scanned slice.
4464    let prefix = buf.get(payload_start..payload_start + local_len + 4)?;
4465
4466    // The record header must fit entirely within the local prefix — otherwise the
4467    // serial array is not addressable locally and we abstain rather than guess.
4468    let (header_len, hn) = read_varint(prefix, 0).ok()?;
4469    let header_len = usize::try_from(header_len).ok()?;
4470    if header_len > local_len || header_len < hn {
4471        return None;
4472    }
4473    let mut serials = Vec::new();
4474    let mut hpos = hn;
4475    while hpos < header_len {
4476        let (s, used) = read_varint(prefix, hpos).ok()?;
4477        serials.push(s);
4478        hpos += used;
4479    }
4480    if hpos != header_len {
4481        return None;
4482    }
4483    match expected_columns {
4484        Some(n) if serials.len() != n => return None,
4485        None if serials.len() < MIN_INFERRED_COLUMNS => return None,
4486        _ => {}
4487    }
4488
4489    // Length closure over the DECLARED payload: header + body must equal P.
4490    let mut body_len = 0usize;
4491    for &s in &serials {
4492        // Checked: a serial from free-space bytes can declare a body length near
4493        // usize::MAX; summing must reject (None) on overflow, never panic/wrap.
4494        body_len = body_len.checked_add(serial_body_len(s)?)?;
4495    }
4496    if header_len + body_len != payload_len {
4497        return None;
4498    }
4499
4500    let first_overflow = be_u32(prefix, local_len);
4501    Some(SpilledCell {
4502        offset: off,
4503        byte_len: n1 + n2 + local_len + 4,
4504        payload_len,
4505        rowid,
4506        serials,
4507        local_len,
4508        local_payload_off: payload_start,
4509        first_overflow,
4510    })
4511}
4512
4513/// Salvage the columns of a recognized [`SpilledCell`] whose bodies lie wholly
4514/// within the local payload (task #73, Codex ruling #4): the chain-resident
4515/// columns are dropped (the chain that would supply them failed), and the
4516/// surviving local columns become a [`CellFragment`]. Returns `None` unless the
4517/// salvaged prefix carries ≥ 1 distinctive cell (the §3.1 emission gate). The
4518/// returned fragment's `offset` is region-local; the caller translates it.
4519fn salvage_local_prefix(
4520    region: &[u8],
4521    sc: &SpilledCell,
4522    enc: TextEncoding,
4523) -> Option<CellFragment> {
4524    // The body begins right after the local header; decode each column while its
4525    // body ends within the local payload bytes (`local_payload_off + local_len`).
4526    let local_end = sc.local_payload_off.checked_add(sc.local_len)?;
4527    // Recompute the record header length to find where the body starts.
4528    let (header_len, _hn) = read_varint(region, sc.local_payload_off).ok()?;
4529    let header_len = usize::try_from(header_len).ok()?;
4530    let mut bpos = sc.local_payload_off.checked_add(header_len)?;
4531
4532    let mut surviving: Vec<(usize, Value)> = Vec::new();
4533    for (idx, &serial) in sc.serials.iter().enumerate() {
4534        let Some(blen) = serial_body_len(serial) else {
4535            break; // cov:unreachable: recognizer accepted only legal serials
4536        };
4537        let Some(body_end) = bpos.checked_add(blen) else {
4538            break; // cov:unreachable: usize add of an in-page body length
4539        };
4540        if body_end > local_end {
4541            break; // this column's body spills into the chain — local prefix ends
4542        }
4543        let Some(body) = region.get(bpos..body_end) else {
4544            break; // cov:unreachable: body_end <= local_end <= region.len()
4545        };
4546        // Column 0 of a rowid-alias table reads as the rowid when serial 0; here a
4547        // spilled cell's id column is a stored integer, so decode it directly.
4548        let Ok((val, _)) = decode_value(body, 0, serial, enc) else {
4549            break; // cov:unreachable: legal serials decode in-bounds
4550        };
4551        surviving.push((idx, val));
4552        bpos = body_end;
4553    }
4554
4555    if !surviving.iter().any(|(_, v)| is_distinctive(v)) {
4556        return None;
4557    }
4558    Some(CellFragment {
4559        offset: sc.offset,
4560        byte_len: bpos.saturating_sub(sc.local_payload_off),
4561        missing: sc.serials.len() - surviving.len(),
4562        surviving,
4563        confidence: FRAGMENT_CONFIDENCE,
4564    })
4565}
4566
4567/// Parse + validate the 100-byte file header.
4568/// The first up-to-100 bytes (the SQLite header region), kept resident so
4569/// fixed-offset header-field reads never touch the byte source.
4570fn header_prefix(bytes: &[u8]) -> Box<[u8]> {
4571    let n = bytes.len().min(SQLITE_HEADER_SIZE);
4572    bytes[..n].into()
4573}
4574
4575fn parse_header(bytes: &[u8]) -> Result<Header, Error> {
4576    let head = bytes.get(..SQLITE_HEADER_SIZE).ok_or(Error::TooShort)?;
4577    if !head.starts_with(SQLITE_MAGIC) {
4578        return Err(Error::BadMagic);
4579    }
4580    let raw = be_u16(head, SQLITE_PAGE_SIZE_OFFSET);
4581    let page_size: u32 = if raw == 1 { 65536 } else { u32::from(raw) };
4582    let valid = (512..=65536).contains(&page_size) && page_size.is_power_of_two();
4583    if !valid {
4584        return Err(Error::BadPageSize(page_size));
4585    }
4586    let reserved = *head.get(RESERVED_SPACE_OFFSET).ok_or(Error::TooShort)?;
4587    // Header byte 56 (BE u32): 1/0 = UTF-8, 2 = UTF-16LE, 3 = UTF-16BE
4588    // (file-format §1.3.1). Tolerant: an unexpected value degrades to UTF-8
4589    // rather than rejecting the database.
4590    let text_encoding = match be_u32(head, TEXT_ENCODING_OFFSET) {
4591        2 => TextEncoding::Utf16Le,
4592        3 => TextEncoding::Utf16Be,
4593        _ => TextEncoding::Utf8,
4594    };
4595    Ok(Header {
4596        page_size,
4597        reserved,
4598        text_encoding,
4599    })
4600}
4601
4602/// Decode a record (payload) into values. Serial type 0 on the first column of
4603/// a rowid table is the `INTEGER PRIMARY KEY` alias → the cell's rowid.
4604fn decode_record(
4605    payload: &[u8],
4606    _column_count: usize,
4607    rowid: i64,
4608    enc: TextEncoding,
4609) -> Result<Vec<Value>, Error> {
4610    // A table-b-tree record: column 0 is the INTEGER PRIMARY KEY alias, so a
4611    // serial-0 there reads the rowid rather than NULL.
4612    decode_record_inner(payload, enc, Some(rowid))
4613}
4614
4615/// Decode an index-b-tree record payload (roadmap §1.4). Unlike a table record it
4616/// has NO `INTEGER PRIMARY KEY` alias — every column is stored literally, so a
4617/// serial-0 first column is a genuine NULL key, never a rowid.
4618fn decode_index_payload(payload: &[u8], enc: TextEncoding) -> Result<Vec<Value>, Error> {
4619    decode_record_inner(payload, enc, None)
4620}
4621
4622/// Decode a SQLite record payload (header + serial array + body) into its column
4623/// values. `rowid_alias` supplies the rowid for a table record's column-0
4624/// `INTEGER PRIMARY KEY` alias (serial 0 → the rowid); `None` (index records)
4625/// leaves a serial-0 column as NULL.
4626fn decode_record_inner(
4627    payload: &[u8],
4628    enc: TextEncoding,
4629    rowid_alias: Option<i64>,
4630) -> Result<Vec<Value>, Error> {
4631    let (header_len, n) = read_varint(payload, 0)?;
4632    let header_len = header_len as usize;
4633    if header_len > payload.len() {
4634        return Err(Error::TruncatedCell);
4635    }
4636    // Pass 1: read serial types from the record header.
4637    let mut serials = Vec::new();
4638    let mut hpos = n;
4639    while hpos < header_len {
4640        let (s, used) = read_varint(payload, hpos)?;
4641        serials.push(s);
4642        hpos += used;
4643    }
4644    // Pass 2: read the body, one value per serial type.
4645    let mut values = Vec::with_capacity(serials.len());
4646    let mut bpos = header_len;
4647    for (idx, &serial) in serials.iter().enumerate() {
4648        let (val, size) = decode_value(payload, bpos, serial, enc)?;
4649        let val = match (idx, serial, rowid_alias) {
4650            // INTEGER PRIMARY KEY alias: NULL in column 0 reads the rowid.
4651            (0, 0, Some(rowid)) => Value::Integer(rowid),
4652            _ => val,
4653        };
4654        values.push(val);
4655        bpos += size;
4656    }
4657    Ok(values)
4658}
4659
4660/// Decode a single value of the given serial type at `off`. Returns the value
4661/// and the number of body bytes it consumed.
4662fn decode_value(
4663    buf: &[u8],
4664    off: usize,
4665    serial: i64,
4666    enc: TextEncoding,
4667) -> Result<(Value, usize), Error> {
4668    Ok(match serial {
4669        // 0 = NULL; 10/11 are reserved for internal use and surfaced as NULL.
4670        0 | 10 | 11 => (Value::Null, 0),
4671        1 => (
4672            Value::Integer(i64::from(read_be_u64(buf, off, 1)? as i8)),
4673            1,
4674        ),
4675        2 => (
4676            Value::Integer(i64::from(read_be_u64(buf, off, 2)? as i16)),
4677            2,
4678        ),
4679        3 => (Value::Integer(sign_extend(read_be_u64(buf, off, 3)?, 3)), 3),
4680        4 => (
4681            Value::Integer(i64::from(read_be_u64(buf, off, 4)? as i32)),
4682            4,
4683        ),
4684        5 => (Value::Integer(sign_extend(read_be_u64(buf, off, 6)?, 6)), 6),
4685        6 => (Value::Integer(read_be_u64(buf, off, 8)? as i64), 8),
4686        7 => {
4687            let bits = read_be_u64(buf, off, 8)?;
4688            (Value::Real(f64::from_bits(bits)), 8)
4689        }
4690        8 => (Value::Integer(0), 0),
4691        9 => (Value::Integer(1), 0),
4692        n if n >= 12 && n % 2 == 0 => {
4693            let len = ((n - 12) / 2) as usize;
4694            let bytes = buf.get(off..off + len).ok_or(Error::TruncatedCell)?;
4695            (Value::Blob(bytes.to_vec()), len)
4696        }
4697        n => {
4698            // odd, >= 13: text, decoded per the database's text encoding
4699            // (UTF-8 / UTF-16LE / UTF-16BE). Lossy so a corrupt byte can't panic.
4700            let len = ((n - 13) / 2) as usize;
4701            let bytes = buf.get(off..off + len).ok_or(Error::TruncatedCell)?;
4702            (Value::Text(enc.decode(bytes)), len)
4703        }
4704    })
4705}
4706
4707/// Read `width` (1..=8) big-endian bytes into a raw u64 (no sign extension).
4708fn read_be_u64(buf: &[u8], off: usize, width: usize) -> Result<u64, Error> {
4709    let bytes = buf.get(off..off + width).ok_or(Error::TruncatedCell)?;
4710    let mut acc: u64 = 0;
4711    for &b in bytes {
4712        acc = (acc << 8) | u64::from(b);
4713    }
4714    Ok(acc)
4715}
4716
4717/// Sign-extend a `width`-byte (3 or 6) value held in the low bits of `raw`.
4718fn sign_extend(raw: u64, width: usize) -> i64 {
4719    let bits = width * 8;
4720    let shift = 64 - bits;
4721    ((raw as i64) << shift) >> shift
4722}
4723
4724/// Read a `SQLite` varint (1..=9 bytes) at `off`. Returns value + bytes consumed.
4725fn read_varint(buf: &[u8], off: usize) -> Result<(i64, usize), Error> {
4726    let mut result: u64 = 0;
4727    for i in 0..8 {
4728        let b = *buf.get(off + i).ok_or(Error::TruncatedCell)?;
4729        result = (result << 7) | u64::from(b & 0x7f);
4730        if b & 0x80 == 0 {
4731            return Ok((result as i64, i + 1));
4732        }
4733    }
4734    // 9th byte contributes all 8 bits.
4735    let b = *buf.get(off + 8).ok_or(Error::TruncatedCell)?;
4736    result = (result << 8) | u64::from(b);
4737    Ok((result as i64, 9))
4738}
4739
4740/// Bounds-checked big-endian u16; out-of-range yields 0 (never panics).
4741fn be_u16(buf: &[u8], off: usize) -> u16 {
4742    let mut b = [0u8; 2];
4743    if let Some(s) = buf.get(off..off + 2) {
4744        b.copy_from_slice(s);
4745    }
4746    u16::from_be_bytes(b)
4747}
4748
4749/// Byte width of the minimal `SQLite` varint encoding of a non-negative `value`
4750/// (task #73, used to re-derive a clobbered record's `header_len`). Mirrors the
4751/// 7-bit big-endian grouping of [`enc_varint_into`]; a value needing more than 8
4752/// groups uses the 9-byte form. Negative inputs (illegal serial types) are
4753/// treated as a single byte and rejected upstream by `serial_body_len`.
4754fn varint_len(value: i64) -> usize {
4755    if value < 0 {
4756        return 1; // cov:unreachable: callers pass only non-negative serials/lengths
4757    }
4758    enc_varint_into(value as usize).len()
4759}
4760
4761/// Minimal `SQLite` varint encoding of a non-negative `value` (task #73). 7-bit
4762/// big-endian groups, high bit set on every group but the last (file-format §2).
4763pub(crate) fn enc_varint_into(value: usize) -> Vec<u8> {
4764    if value == 0 {
4765        return vec![0];
4766    }
4767    let mut groups = Vec::new();
4768    let mut n = value as u64;
4769    while n > 0 {
4770        groups.push((n & 0x7f) as u8);
4771        n >>= 7;
4772    }
4773    groups.reverse();
4774    let last = groups.len() - 1;
4775    for (i, g) in groups.iter_mut().enumerate() {
4776        if i != last {
4777            *g |= 0x80;
4778        }
4779    }
4780    groups
4781}
4782
4783/// The 8-byte rollback-journal segment magic (`pager.c` `aJournalMagic`).
4784const JOURNAL_MAGIC: [u8; 8] = [0xd9, 0xd5, 0x05, 0xf9, 0x20, 0xa1, 0x63, 0xd7];
4785
4786/// Hard cap on page records walked in one journal segment, to bound work on a
4787/// crafted/garbage journal whose stride scan would otherwise run the file length.
4788const MAX_JOURNAL_RECORDS: usize = 1_000_000;
4789
4790/// Sector-size candidates probed when reconstructing a zeroed (PERSIST) journal
4791/// header. Real VFS sector sizes exceed 512, so 512 is a candidate, not an
4792/// assumption; the page size is also tried (file-format §"Rollback Journal").
4793const SECTOR_CANDIDATES: [u32; 3] = [512, 4096, 0]; // 0 = "use page_size"
4794
4795/// Parsed (or reconstructed) rollback-journal header (design §5).
4796///
4797/// `Valid` is a header whose magic is intact (Tier A — hot journal / crash
4798/// residue): every parameter, including the checksum `nonce`, is authoritative.
4799/// `ReconstructedZeroed` is the PERSIST post-commit case (Tier B): the first
4800/// sector was zeroed on commit, so the page size comes from the main database
4801/// and the sector size from candidate scoring — the nonce is gone, so page
4802/// checksums cannot be verified.
4803#[derive(Debug, Clone, Copy, PartialEq, Eq)]
4804pub enum JournalHeader {
4805    /// Tier A: header magic present; all fields trusted (`pager.c` offsets).
4806    Valid {
4807        /// Page records declared in this segment (`0xFFFFFFFF`/`0` ⇒ walk to EOF).
4808        n_rec: u32,
4809        /// Database page count at transaction start (`dbOrigSize`).
4810        mx_page: u32,
4811        /// Checksum initializer (`cksumInit`), offset 12.
4812        nonce: u32,
4813        /// VFS sector size the header is padded to.
4814        sector_size: u32,
4815        /// Database page size at transaction start.
4816        page_size: u32,
4817    },
4818    /// Tier B: header zeroed (PERSIST post-commit); parameters reconstructed.
4819    ReconstructedZeroed {
4820        /// Page size taken from the main database header (authoritative).
4821        page_size: u32,
4822        /// Sector size selected by candidate scoring (record offset stride).
4823        sector_size: u32,
4824    },
4825}
4826
4827/// One pre-transaction page image recovered from a rollback journal (design §5).
4828#[derive(Debug, Clone, PartialEq, Eq)]
4829pub struct JournalPageImage {
4830    /// 1-based database page number this image restores.
4831    pub pgno: u32,
4832    /// 0-based segment index this record came from.
4833    pub segment: usize,
4834    /// The original page content (`page_size` bytes).
4835    pub bytes: Vec<u8>,
4836    /// `Some(true/false)` in Tier A (nonce known) — whether the stored checksum
4837    /// matched; `None` in Tier B (nonce zeroed, unverifiable).
4838    pub checksum_valid: Option<bool>,
4839}
4840
4841/// A parsed rollback journal: its header tier plus the ordered, first-wins
4842/// page images (design §3/§5). The temporal inverse of the WAL overlay —
4843/// these images are the database as it was BEFORE the last transaction.
4844#[derive(Debug, Clone, PartialEq, Eq)]
4845pub struct RollbackJournal {
4846    header: JournalHeader,
4847    images: Vec<JournalPageImage>,
4848    /// Page numbers that appeared more than once (first occurrence kept), each
4849    /// listed once in first-seen order. Empty for a well-formed journal.
4850    duplicate_pgnos: Vec<u32>,
4851}
4852
4853/// The journal page checksum (`pager.c` `pager_cksum`): `nonce` plus every-200th
4854/// byte from the tail, starting at `page_size - 200` and stepping down by 200
4855/// while the index is positive, using wrapping u32 arithmetic. It detects torn
4856/// page writes; it is not a cryptographic integrity guarantee.
4857fn journal_cksum(nonce: u32, page: &[u8]) -> u32 {
4858    let mut sum = nonce;
4859    let mut x = page.len() as i64 - 200;
4860    while x > 0 {
4861        // x is in (0, page.len()) by the loop bound, so indexing is in-range.
4862        if let Some(&b) = page.get(x as usize) {
4863            sum = sum.wrapping_add(u32::from(b));
4864        }
4865        x -= 200;
4866    }
4867    sum
4868}
4869
4870/// Walk page records of `page_size` bytes from `start`, with the checksum
4871/// `nonce` (`None` ⇒ Tier B, unverifiable), stopping at EOF or after `limit`
4872/// records. Returns the images in file order; a partial trailing record is
4873/// dropped (truncation tolerance). Bounded by [`MAX_JOURNAL_RECORDS`].
4874fn walk_journal_records(
4875    bytes: &[u8],
4876    start: usize,
4877    page_size: usize,
4878    nonce: Option<u32>,
4879    segment: usize,
4880    limit: usize,
4881) -> Vec<JournalPageImage> {
4882    let stride = 4usize.saturating_add(page_size).saturating_add(4);
4883    let mut out = Vec::new();
4884    let mut off = start;
4885    let cap = limit.min(MAX_JOURNAL_RECORDS);
4886    while out.len() < cap {
4887        let Some(rec) = bytes.get(off..off.saturating_add(stride)) else {
4888            break; // EOF or partial trailing record: stop (truncation tolerant).
4889        };
4890        let pgno = u32::from_be_bytes([rec[0], rec[1], rec[2], rec[3]]);
4891        if pgno == 0 {
4892            break; // page 0 is not a valid record; treat as end-of-segment.
4893        }
4894        let page = &rec[4..4 + page_size];
4895        let stored = u32::from_be_bytes([
4896            rec[4 + page_size],
4897            rec[5 + page_size],
4898            rec[6 + page_size],
4899            rec[7 + page_size],
4900        ]);
4901        let checksum_valid = nonce.map(|n| journal_cksum(n, page) == stored);
4902        out.push(JournalPageImage {
4903            pgno,
4904            segment,
4905            bytes: page.to_vec(),
4906            checksum_valid,
4907        });
4908        off = off.saturating_add(stride);
4909    }
4910    out
4911}
4912
4913/// Score a candidate record walk for the Tier-B sector reconstruction: more
4914/// records and all page numbers within `1..=page_bound` rank higher; a record
4915/// count of zero scores zero so an off-stride candidate never wins.
4916fn score_journal_candidate(images: &[JournalPageImage], page_bound: u32) -> usize {
4917    if images.is_empty() {
4918        return 0;
4919    }
4920    let in_range = images
4921        .iter()
4922        .filter(|i| i.pgno >= 1 && i.pgno <= page_bound)
4923        .count();
4924    // All-in-range walks are strongly preferred; weight the in-range fraction so
4925    // a candidate that mostly decodes to impossible page numbers loses to one
4926    // that decodes cleanly even with fewer records.
4927    if in_range == images.len() {
4928        1000 + images.len()
4929    } else {
4930        in_range
4931    }
4932}
4933
4934impl RollbackJournal {
4935    /// LOWER-LEVEL, UNAUTHENTICATED parse (design §5): interpret `bytes` as a
4936    /// rollback journal given an externally-supplied `page_size`. Does NOT bind
4937    /// the journal to a particular database — prefer [`Database::rollback_prior`],
4938    /// which supplies the authoritative page size from the main db.
4939    ///
4940    /// Tier A (magic present) trusts the header and verifies each checksum. Tier B
4941    /// (magic absent — PERSIST post-commit) reconstructs the sector size by
4942    /// candidate scoring and walks records (checksums unverifiable). Robust: a
4943    /// malformed/truncated journal yields fewer images, never a panic; a page size
4944    /// that is not a power of two in `[512, 65536]` is a typed
4945    /// [`Error::BadJournalPageSize`] carrying the offending value.
4946    pub fn parse(bytes: &[u8], page_size: u32) -> Result<Self, Error> {
4947        if !(512..=65536).contains(&page_size) || !page_size.is_power_of_two() {
4948            return Err(Error::BadJournalPageSize(page_size));
4949        }
4950        let ps = page_size as usize;
4951        let page_bound = u32::try_from(bytes.len() / ps.max(1)).unwrap_or(u32::MAX);
4952
4953        let header_valid = bytes.len() >= 28 && bytes.starts_with(&JOURNAL_MAGIC);
4954        if header_valid {
4955            // Tier A: trust the header.
4956            let n_rec = be_u32(bytes, 8);
4957            let nonce = be_u32(bytes, 12);
4958            let mx_page = be_u32(bytes, 16);
4959            let sector_size = be_u32(bytes, 20);
4960            let hdr_page_size = be_u32(bytes, 24);
4961            // nRec ∈ {0, 0xFFFFFFFF} ⇒ walk to EOF; else exactly n_rec records.
4962            let limit = if n_rec == 0 || n_rec == u32::MAX {
4963                MAX_JOURNAL_RECORDS
4964            } else {
4965                n_rec as usize
4966            };
4967            let start = sector_size.max(1) as usize;
4968            let imgs = walk_journal_records(bytes, start, ps, Some(nonce), 0, limit);
4969            let header = JournalHeader::Valid {
4970                n_rec,
4971                mx_page,
4972                nonce,
4973                sector_size,
4974                // The journal's pages are images of THIS db, so the externally
4975                // supplied page size is authoritative; expose it even if the
4976                // header field disagrees (a tampered/mismatched header field).
4977                page_size: if hdr_page_size == page_size {
4978                    hdr_page_size
4979                } else {
4980                    page_size
4981                },
4982            };
4983            return Ok(Self::from_walk(header, imgs));
4984        }
4985
4986        // Tier B: header zeroed/absent (PERSIST post-commit). Score sector
4987        // candidates and pick the best; checksums are unverifiable (nonce gone).
4988        let mut best: Option<(usize, u32, Vec<JournalPageImage>)> = None;
4989        for cand in SECTOR_CANDIDATES {
4990            let sector = if cand == 0 { page_size } else { cand };
4991            let imgs =
4992                walk_journal_records(bytes, sector as usize, ps, None, 0, MAX_JOURNAL_RECORDS);
4993            let score = score_journal_candidate(&imgs, page_bound);
4994            // `map_or(true, …)` not `is_none_or` to keep the library MSRV at 1.80
4995            // (`Option::is_none_or` stabilised in 1.82); clippy is MSRV-aware.
4996            let better = best.as_ref().map_or(true, |(bs, _, _)| score > *bs);
4997            if better && score > 0 {
4998                best = Some((score, sector, imgs));
4999            }
5000        }
5001        // No candidate decoded a single in-range record (garbage, or a journal too
5002        // short for one record): an empty Tier-B journal, sector size unknown →
5003        // page size. Degrade gracefully rather than erroring.
5004        let (sector_size, imgs) = best
5005            .map(|(_, s, i)| (s, i))
5006            .unwrap_or((page_size, Vec::new()));
5007        let header = JournalHeader::ReconstructedZeroed {
5008            page_size,
5009            sector_size,
5010        };
5011        Ok(Self::from_walk(header, imgs))
5012    }
5013
5014    /// Apply first-wins dedup to a walked record set, recording whether any
5015    /// `pgno` repeated (the duplicate-page anomaly, design §3).
5016    fn from_walk(header: JournalHeader, walked: Vec<JournalPageImage>) -> Self {
5017        let mut seen = std::collections::BTreeSet::new();
5018        let mut images = Vec::with_capacity(walked.len());
5019        let mut duplicate_pgnos: Vec<u32> = Vec::new();
5020        for img in walked {
5021            if seen.insert(img.pgno) {
5022                images.push(img);
5023            } else if !duplicate_pgnos.contains(&img.pgno) {
5024                // Keep the FIRST occurrence as the truest pre-transaction image;
5025                // record WHICH page repeated (once) rather than a bare flag, so the
5026                // anomaly can name the offending page number.
5027                duplicate_pgnos.push(img.pgno);
5028            }
5029        }
5030        Self {
5031            header,
5032            images,
5033            duplicate_pgnos,
5034        }
5035    }
5036
5037    /// The parsed (or reconstructed) header.
5038    #[must_use]
5039    pub fn header(&self) -> &JournalHeader {
5040        &self.header
5041    }
5042
5043    /// The ordered, first-wins pre-transaction page images.
5044    #[must_use]
5045    pub fn page_images(&self) -> &[JournalPageImage] {
5046        &self.images
5047    }
5048
5049    /// Whether a `pgno` appeared more than once across the parsed segments — the
5050    /// spec says a page is journaled at most once, so a repeat is consistent with
5051    /// corruption, a savepoint/super-journal artifact, or tampering (design §3).
5052    #[must_use]
5053    pub fn has_duplicate_pgno(&self) -> bool {
5054        !self.duplicate_pgnos.is_empty()
5055    }
5056
5057    /// The page numbers that appeared more than once (first occurrence kept), each
5058    /// listed once in first-seen order — the offending values behind
5059    /// [`Self::has_duplicate_pgno`]. Empty for a well-formed journal.
5060    #[must_use]
5061    pub fn duplicate_pgnos(&self) -> &[u32] {
5062        &self.duplicate_pgnos
5063    }
5064}
5065
5066/// A read-only, page-addressable image of the database AS IT WAS BEFORE the last
5067/// transaction (design §4/§5). The temporal inverse of [`CommitSnapshot`]:
5068/// `prior[pgno]` is the rollback-journal image where present, else the live main
5069/// page. Diffing this against the current database yields the last transaction's
5070/// deletions (rowid present here, absent now) and modifications (present in both,
5071/// values differ — the journal carries the OLD value).
5072///
5073/// Returned by [`Database::rollback_prior`] as a DISTINCT type, never a
5074/// [`Database`], so prior/deleted rows can never be read as "live"
5075/// (secure-by-design). Shares ONE b-tree/overflow walk with the live and
5076/// commit-snapshot reads via the internal `PageSource` seam.
5077#[derive(Debug, Clone, PartialEq, Eq)]
5078pub struct PriorSnapshot {
5079    /// The pre-transaction page images: journal-where-present overlaid on the main
5080    /// db. Materializes EVERY valid journal page type (interior, leaf, overflow,
5081    /// page 1, freelist trunk, pointer-map) so a prior table can be walked through
5082    /// its interior pages and overflow chains reassembled.
5083    overlaid: std::collections::BTreeMap<u32, Vec<u8>>,
5084    /// Usable bytes per page, parsed from the prior snapshot's OWN page-1 header
5085    /// (so a reserved-space change in the last txn is honored).
5086    usable: u32,
5087    /// The 1-based page count bound (max overlaid page), for cycle/over-range
5088    /// guards in the b-tree / overflow walk.
5089    page_bound: u32,
5090    /// Whether any journal page image's number exceeded the current main-db page
5091    /// count — diagnostic only (the txn grew the db).
5092    grew_db: bool,
5093}
5094
5095impl PageSource for PriorSnapshot {
5096    fn page(&self, page: u32) -> Option<PageBytes<'_>> {
5097        self.overlaid
5098            .get(&page)
5099            .map(|v| PageBytes::Borrowed(v.as_slice()))
5100    }
5101    fn usable(&self) -> usize {
5102        self.usable as usize
5103    }
5104    fn page_bound(&self) -> u32 {
5105        self.page_bound
5106    }
5107    fn encoding(&self) -> TextEncoding {
5108        // Encoding from the prior snapshot's OWN page-1 header (byte 56), so a
5109        // historical read decodes TEXT per the encoding as of the prior state.
5110        self.overlaid
5111            .get(&1)
5112            .map(|p| match be_u32(p, TEXT_ENCODING_OFFSET) {
5113                2 => TextEncoding::Utf16Le,
5114                3 => TextEncoding::Utf16Be,
5115                _ => TextEncoding::Utf8,
5116            })
5117            .unwrap_or_default()
5118    }
5119}
5120
5121impl PriorSnapshot {
5122    /// The user tables AS OF the prior state, parsed from the snapshot's OWN page 1
5123    /// (the prior `sqlite_master`), NOT the live database — so a DROP/CREATE in the
5124    /// last transaction is interpreted against the prior schema. Best-effort and
5125    /// panic-free: an unreadable page-1 schema yields an empty vector.
5126    #[must_use]
5127    pub fn tables(&self) -> Vec<SnapshotTable> {
5128        let Ok(schema) = read_table_via(self, 1, 5) else {
5129            return Vec::new(); // cov:unreachable: the prior snapshot has a readable page 1
5130        };
5131        let mut out = Vec::new();
5132        for row in schema {
5133            let is_table = matches!(row.values.first(), Some(Value::Text(t)) if t == "table");
5134            if !is_table {
5135                continue;
5136            }
5137            let Some(Value::Text(name)) = row.values.get(1) else {
5138                continue; // cov:unreachable: a 'table' schema row has a TEXT name
5139            };
5140            if name.starts_with("sqlite_") {
5141                continue;
5142            }
5143            let Some(Value::Integer(root)) = row.values.get(3) else {
5144                continue; // cov:unreachable: a 'table' schema row has an integer rootpage
5145            };
5146            let Ok(rootpage) = u32::try_from(*root) else {
5147                continue; // cov:unreachable: a real rootpage is a small positive page number
5148            };
5149            let sql = match row.values.get(4) {
5150                Some(Value::Text(s)) => s.as_str(),
5151                _ => "", // cov:unreachable: a 'table' schema row carries its CREATE TABLE sql
5152            };
5153            let columns = attribution::column_names(sql).unwrap_or_default();
5154            out.push(SnapshotTable {
5155                name: name.clone(),
5156                rootpage,
5157                columns,
5158                without_rowid: without_rowid_sql(sql),
5159            });
5160        }
5161        out
5162    }
5163
5164    /// The PRIOR `sqlite_master` as a `name -> CREATE SQL` map for every **user**
5165    /// table, parsed from the snapshot's OWN page 1 — the prior-schema half of the
5166    /// Detector-B sidecar schema-change comparison
5167    /// (`docs/design/drop-recreate-attribution.md`).
5168    ///
5169    /// The counterpart to [`Database::schema_sql`] read against the pre-transaction
5170    /// state the `-journal` preserves, so a DROP/CREATE/ALTER in the last
5171    /// transaction is interpreted against the prior schema. Best-effort and
5172    /// panic-free: an unreadable prior page-1 schema yields an empty map.
5173    #[must_use]
5174    pub fn schema_sql(&self) -> std::collections::BTreeMap<String, String> {
5175        let mut out = std::collections::BTreeMap::new();
5176        let Ok(schema) = read_table_via(self, 1, 5) else {
5177            return out; // cov:unreachable: the prior snapshot has a readable page 1
5178        };
5179        for row in schema {
5180            schema_sql_insert(&mut out, &row.values);
5181        }
5182        out
5183    }
5184
5185    /// Read every row of the table b-tree rooted at `rootpage` AS OF the prior
5186    /// state, in rowid order, resolving overflow chains through the snapshot's OWN
5187    /// pages. The snapshot-scoped counterpart to [`Database::read_table`]: a typed
5188    /// [`Error`] (never a panic) on a cyclic/over-deep b-tree or overflow chain.
5189    pub fn read_table(
5190        &self,
5191        rootpage: u32,
5192        column_count: usize,
5193    ) -> Result<Vec<(i64, Vec<Value>)>, Error> {
5194        let rows = read_table_via(self, rootpage, column_count)?;
5195        Ok(rows.into_iter().map(|r| (r.rowid, r.values)).collect())
5196    }
5197
5198    /// Whether the last transaction GREW the database (a journal page number
5199    /// exceeded the current main-db page count). Pages beyond the prior size are
5200    /// new — their pre-images were not journaled — which bounds what rolls back.
5201    #[must_use]
5202    pub fn grew_db(&self) -> bool {
5203        self.grew_db
5204    }
5205
5206    /// Read the table rooted at `rootpage` AS OF the prior state, returning each
5207    /// row's rowid, values, AND the 1-based LEAF page it was decoded from — the
5208    /// per-row page provenance the forensic diff attaches to a recovered prior
5209    /// row. Shares `decode_leaf_cell` with the standard read; a typed [`Error`]
5210    /// (never a panic) on a cyclic/over-deep b-tree.
5211    pub fn read_table_with_pages(
5212        &self,
5213        rootpage: u32,
5214        column_count: usize,
5215    ) -> Result<Vec<(i64, Vec<Value>, u32)>, Error> {
5216        let mut out = Vec::new();
5217        let mut seen = std::collections::BTreeSet::new();
5218        walk_table_page_with_leaf(self, rootpage, column_count, &mut out, &mut seen)?;
5219        Ok(out)
5220    }
5221}
5222
5223/// Walk a table b-tree like [`walk_table_page`] but record each row's LEAF page,
5224/// for the rollback-journal per-row provenance. Bounded identically (visited-set
5225/// caps recursion depth; a revisited page is silently skipped).
5226fn walk_table_page_with_leaf(
5227    src: &dyn PageSource,
5228    page: u32,
5229    column_count: usize,
5230    out: &mut Vec<(i64, Vec<Value>, u32)>,
5231    seen: &mut std::collections::BTreeSet<u32>,
5232) -> Result<(), Error> {
5233    if seen.len() > MAX_PAGES_PER_WALK {
5234        return Err(Error::TooManyPages);
5235    }
5236    if !seen.insert(page) {
5237        return Ok(());
5238    }
5239    let slice = src.page(page).ok_or(Error::PageOutOfRange(page))?;
5240    let slice = &*slice;
5241    let hdr_off = if page == 1 { SQLITE_HEADER_SIZE } else { 0 };
5242    let page_type = *slice.get(hdr_off).ok_or(Error::TruncatedCell)?;
5243    let cell_count = be_u16(slice, hdr_off + 3) as usize;
5244    match page_type {
5245        0x0d => {
5246            let cell_ptr_array = hdr_off + 8;
5247            for i in 0..cell_count {
5248                let p = cell_ptr_array + i * 2;
5249                let cell_off = be_u16(slice, p) as usize;
5250                let row = decode_leaf_cell(src, slice, cell_off, column_count)?;
5251                out.push((row.rowid, row.values, page));
5252            }
5253            Ok(())
5254        }
5255        0x05 => {
5256            let cell_ptr_array = hdr_off + 12;
5257            for i in 0..cell_count {
5258                let p = cell_ptr_array + i * 2;
5259                let cell_off = be_u16(slice, p) as usize;
5260                let child = be_u32(slice, cell_off);
5261                walk_table_page_with_leaf(src, child, column_count, out, seen)?;
5262            }
5263            let right = be_u32(slice, hdr_off + 8);
5264            walk_table_page_with_leaf(src, right, column_count, out, seen)
5265        }
5266        other => Err(Error::NotATablePage(other)),
5267    }
5268}
5269
5270/// Bounds-checked big-endian u32; out-of-range yields 0 (never panics).
5271fn be_u32(buf: &[u8], off: usize) -> u32 {
5272    let mut b = [0u8; 4];
5273    if let Some(s) = buf.get(off..off + 4) {
5274        b.copy_from_slice(s);
5275    }
5276    u32::from_be_bytes(b)
5277}
5278
5279#[cfg(test)]
5280mod tests {
5281    use super::*;
5282
5283    fn page_rc(byte: u8) -> std::rc::Rc<[u8]> {
5284        std::rc::Rc::from(vec![byte].into_boxed_slice())
5285    }
5286
5287    /// Encode `v` as a 9-byte SQLite varint (round-trips through `read_varint`).
5288    fn varint9(v: u64) -> [u8; 9] {
5289        let mut out = [0u8; 9];
5290        let top56 = v >> 8;
5291        for (i, b) in out.iter_mut().take(8).enumerate() {
5292            *b = (((top56 >> (7 * (7 - i))) & 0x7f) as u8) | 0x80;
5293        }
5294        out[8] = (v & 0xff) as u8;
5295        out
5296    }
5297
5298    #[test]
5299    fn inferred_carve_does_not_overflow_on_huge_serials() {
5300        // A record whose serial array declares column body lengths summing past
5301        // usize::MAX must be REJECTED, never panic (debug) or wrap (release). Real
5302        // free-space bytes (Belkasoft corpus) hit this; here we craft it minimally:
5303        // five maximal (i64::MAX) serials, each a text/blob length ~(i64::MAX-12)/2.
5304        let big = varint9(i64::MAX as u64); // serial_body_len ~4.6e18; five overflow usize
5305        let n_serials = 5usize;
5306        let header_len = 1 + n_serials * 9; // 1-byte header_len varint + 5 serials
5307        let payload_len = header_len; // reach the body-sum loop before any body exists
5308        let mut buf = Vec::new();
5309        buf.push(payload_len as u8); // payload_len varint (small, 1 byte)
5310        buf.push(1u8); // rowid varint = 1 (positive)
5311        buf.push(header_len as u8); // header_len varint (1 byte, < 128)
5312        for _ in 0..n_serials {
5313            buf.extend_from_slice(&big);
5314        }
5315        // Must return None (rejected), and above all must not panic/overflow.
5316        let got = try_carve_cell_at(&buf, 0, None, TextEncoding::Utf8);
5317        assert!(
5318            got.is_none(),
5319            "a body-length-overflowing record must be rejected"
5320        );
5321    }
5322
5323    #[test]
5324    fn page_cache_hits_reorders_and_evicts_past_cap() {
5325        let mut cache = PageCache::new();
5326        // Fill exactly to CAP, then one more → the oldest (key 0) is evicted.
5327        for i in 0..=PageCache::CAP {
5328            cache.put(i, page_rc(i as u8));
5329        }
5330        assert!(cache.get(0).is_none(), "oldest entry evicted once past CAP");
5331        assert!(
5332            cache.get(PageCache::CAP).is_some(),
5333            "the newest entry is retained (get-hit + touch)"
5334        );
5335        // Re-put an existing key → the already-present branch (touch, no growth).
5336        let before = cache.order.len();
5337        cache.put(PageCache::CAP, page_rc(0xff));
5338        assert_eq!(cache.order.len(), before, "re-put must not grow the order");
5339        assert_eq!(cache.get(PageCache::CAP).as_deref(), Some(&[0xff][..]));
5340    }
5341
5342    #[test]
5343    fn varint_single_byte() {
5344        assert_eq!(read_varint(&[0x05], 0).unwrap(), (5, 1));
5345    }
5346
5347    #[test]
5348    fn varint_two_bytes() {
5349        // 0x81 0x00 => (1<<7) = 128
5350        assert_eq!(read_varint(&[0x81, 0x00], 0).unwrap(), (128, 2));
5351    }
5352
5353    #[test]
5354    fn varint_truncated_is_err() {
5355        assert_eq!(read_varint(&[0x81], 0), Err(Error::TruncatedCell));
5356    }
5357
5358    #[test]
5359    fn sign_extend_three_byte_negative() {
5360        // 0xFFFFFF as 3-byte => -1
5361        assert_eq!(sign_extend(0x00FF_FFFF, 3), -1);
5362    }
5363
5364    #[test]
5365    fn decode_value_text_and_blob() {
5366        let (v, n) = decode_value(b"hi", 0, 17, TextEncoding::Utf8).unwrap(); // 17 => text len (17-13)/2 =2
5367        assert_eq!(v, Value::Text("hi".into()));
5368        assert_eq!(n, 2);
5369        let (v, n) = decode_value(&[0xAA, 0xBB], 0, 16, TextEncoding::Utf8).unwrap(); // 16 => blob len 2
5370        assert_eq!(v, Value::Blob(vec![0xAA, 0xBB]));
5371        assert_eq!(n, 2);
5372    }
5373
5374    #[test]
5375    fn decode_value_text_utf16_le_and_be() {
5376        // The TEXT decode path honors the database encoding (file-format §1.3.1):
5377        // the same code points must round-trip from both byte orders. This drives
5378        // `decode_utf16` deterministically, without depending on an external
5379        // `sqlite3`-minted fixture (the integration tests skip when absent).
5380        // Serial 21 => text byte length (21-13)/2 = 4 = two UTF-16 code units.
5381        let le = [b'h', 0x00, b'i', 0x00];
5382        let (v, n) = decode_value(&le, 0, 21, TextEncoding::Utf16Le).unwrap();
5383        assert_eq!(v, Value::Text("hi".into()));
5384        assert_eq!(n, 4);
5385        let be = [0x00, b'h', 0x00, b'i'];
5386        let (v, n) = decode_value(&be, 0, 21, TextEncoding::Utf16Be).unwrap();
5387        assert_eq!(v, Value::Text("hi".into()));
5388        assert_eq!(n, 4);
5389    }
5390
5391    #[test]
5392    fn localstorage_decodes_known_utf16le_bytes() {
5393        // Independent oracle: these UTF-16-LE bytes are derived from the Unicode
5394        // code points and the surrogate-pair formula, NOT from Rust's encoder, so
5395        // a matching round-trip validates the decoder against the documented
5396        // construction (Evidence-Based Rigor tier 2).
5397        //   'A' U+0041      -> 41 00
5398        //   '中' U+4E2D      -> 2D 4E
5399        //   '😀' U+1F600     -> surrogate pair D83D DE00 -> 3D D8 00 DE
5400        let bytes = [0x41, 0x00, 0x2D, 0x4E, 0x3D, 0xD8, 0x00, 0xDE];
5401        let out = decode_localstorage_value(&bytes);
5402        assert_eq!(out.text, "A中😀");
5403        assert!(!out.lossy, "a fully-paired BLOB is not lossy");
5404    }
5405
5406    #[test]
5407    fn localstorage_empty_blob_is_empty_not_lossy() {
5408        let out = decode_localstorage_value(&[]);
5409        assert_eq!(out.text, "");
5410        assert!(!out.lossy);
5411    }
5412
5413    #[test]
5414    fn localstorage_odd_length_blob_is_lossy_not_panic() {
5415        // 'A' (41 00) then a lone trailing byte 42 — half a code unit was cut off.
5416        let out = decode_localstorage_value(&[0x41, 0x00, 0x42]);
5417        assert_eq!(out.text, "A");
5418        assert!(out.lossy, "a trailing half code unit is a lossy truncation");
5419    }
5420
5421    #[test]
5422    fn localstorage_lone_surrogate_is_replacement_and_lossy() {
5423        // High surrogate D83D (LE 3D D8) with no following low surrogate.
5424        let out = decode_localstorage_value(&[0x3D, 0xD8]);
5425        assert_eq!(out.text, "\u{FFFD}");
5426        assert!(out.lossy);
5427    }
5428
5429    #[test]
5430    fn item_table_schema_recognized_and_others_rejected() {
5431        assert!(is_local_storage_item_table("ItemTable"));
5432        assert!(!is_local_storage_item_table("moz_places"));
5433        assert!(!is_local_storage_item_table("itemtable"));
5434        assert!(!is_local_storage_item_table(""));
5435    }
5436
5437    #[test]
5438    fn decode_value_int_literals() {
5439        assert_eq!(
5440            decode_value(&[], 0, 8, TextEncoding::Utf8).unwrap(),
5441            (Value::Integer(0), 0)
5442        );
5443        assert_eq!(
5444            decode_value(&[], 0, 9, TextEncoding::Utf8).unwrap(),
5445            (Value::Integer(1), 0)
5446        );
5447    }
5448
5449    #[test]
5450    fn bad_magic_rejected() {
5451        let mut b = vec![0u8; 100];
5452        b[..16].copy_from_slice(b"NOT SQLITE 3\0\0\0\0");
5453        assert_eq!(parse_header(&b), Err(Error::BadMagic));
5454    }
5455
5456    #[test]
5457    fn too_short_rejected() {
5458        assert_eq!(parse_header(&[0u8; 10]), Err(Error::TooShort));
5459    }
5460
5461    /// The deleted-record carving fixture (see `docs/corpus-catalog.md`).
5462    const DELETED_DB: &[u8] = include_bytes!("../../tests/data/deleted_places.db");
5463    /// A clean DB with one live `moz_places` table and no deletions.
5464    const CLEAN_DB: &[u8] = include_bytes!("../../tests/data/places.db");
5465
5466    #[test]
5467    fn free_regions_is_complement_of_live_extents() {
5468        // Live cells [10,20) and [30,40) within content area [5, 50).
5469        let live = [(10, 20), (30, 40)];
5470        let regions = free_regions(&live, 5, 50);
5471        assert_eq!(regions, vec![(5, 10), (20, 30), (40, 50)]);
5472        // No live cells -> the whole span is free.
5473        assert_eq!(free_regions(&[], 5, 50), vec![(5, 50)]);
5474        // Live cell covering the whole span -> no free region.
5475        assert!(free_regions(&[(0, 100)], 5, 50).is_empty());
5476    }
5477
5478    #[test]
5479    fn live_cell_len_reads_on_page_footprint() {
5480        // Cell: payload_len=3 (varint 0x03), rowid=1 (varint 0x01), 3 payload bytes.
5481        let buf = [0x03, 0x01, 0xAA, 0xBB, 0xCC];
5482        let usable = 4096;
5483        assert_eq!(live_cell_len(&buf, 0, usable), Some(1 + 1 + 3));
5484        // Truncated prefix -> None, never panics.
5485        assert_eq!(live_cell_len(&[0x81], 0, usable), None);
5486    }
5487
5488    #[test]
5489    fn carve_free_regions_recovers_in_page_remnant() {
5490        let db = Database::open(DELETED_DB.to_vec()).unwrap();
5491        // Page 8 is an allocated leaf (live ids 181..=200) whose free gap holds
5492        // deleted-row residue including rowid 237.
5493        let page = db.raw_page(8).unwrap();
5494        let carved = db.carve_free_regions(&page, 6);
5495        assert!(carved.iter().any(|c| c.rowid == 237));
5496        // 0-FP: never a live (id<=200) rowid.
5497        assert!(carved.iter().all(|c| c.rowid > 200));
5498        // A non-leaf page yields nothing.
5499        assert!(db.carve_free_regions(&[0x05u8; 4096], 6).is_empty());
5500        // An empty / too-short slice yields nothing (no panic).
5501        assert!(db.carve_free_regions(&[], 6).is_empty());
5502    }
5503
5504    #[test]
5505    fn carve_leaf_cells_reads_allocated_cells_and_rejects_non_leaf() {
5506        let db = Database::open(DELETED_DB.to_vec()).unwrap();
5507        // Page 8 is an allocated table-leaf (live ids 181..=200); carve_leaf_cells
5508        // decodes every cell the page records as allocated, so the live ids appear
5509        // (unlike carve_free_regions, which excludes them).
5510        let page = db.raw_page(8).unwrap();
5511        let cells = db.carve_leaf_cells(&page);
5512        assert!(
5513            cells.iter().any(|c| c.rowid == 181),
5514            "must read the allocated cells of the leaf"
5515        );
5516        // Page 1 is passed whole (starts with the file magic) → header read at 100.
5517        let _ = db.carve_leaf_cells(&db.raw_page(1).unwrap());
5518        // A non-leaf page (interior 0x05) and an empty/too-short slice yield nothing
5519        // (no panic) — the same defensive arms carve_free_regions guards.
5520        assert!(db.carve_leaf_cells(&[0x05u8; 4096]).is_empty());
5521        assert!(db.carve_leaf_cells(&[]).is_empty());
5522    }
5523
5524    #[test]
5525    fn carve_free_regions_handles_page_one_and_inferred() {
5526        let db = Database::open(DELETED_DB.to_vec()).unwrap();
5527        // Page 1 is passed whole (starts with the file magic) -> the b-tree header
5528        // is read at offset 100, exercising the page-1 branch.
5529        let page1 = db.raw_page(1).unwrap();
5530        let _ = db.carve_free_regions(&page1, 6);
5531        // With column_count_hint = 0, the inferred path runs over the free regions.
5532        let page8 = db.raw_page(8).unwrap();
5533        let inferred = db.carve_free_regions(&page8, 0);
5534        assert!(inferred.iter().any(|c| c.rowid == 237));
5535    }
5536
5537    #[test]
5538    fn live_cell_len_accounts_for_overflow_pointer() {
5539        let usable = 4096usize;
5540        // Non-spilling cell: payload_len small -> footprint = prefix + payload.
5541        // varint 0x03 (payload_len=3), 0x01 (rowid=1), 3 payload bytes.
5542        assert_eq!(live_cell_len(&[0x03, 0x01, 0, 0, 0], 0, usable), Some(5));
5543
5544        // Spilling cell: a payload_len far above the local threshold takes the
5545        // overflow branch -> footprint = prefix + local + 4 (overflow pointer).
5546        // Encode payload_len = 5000 as a 2-byte varint (0xA7 0x08), rowid = 1.
5547        let mut buf = vec![0xA7, 0x08, 0x01];
5548        buf.extend(std::iter::repeat_n(0u8, 5000));
5549        let total = 5000usize;
5550        let local = local_payload_len(total, usable);
5551        assert!(local < total, "this payload must spill");
5552        assert_eq!(live_cell_len(&buf, 0, usable), Some(2 + 1 + local + 4));
5553    }
5554
5555    #[test]
5556    fn carve_cells_inferred_matches_fixed_count() {
5557        let db = Database::open(DELETED_DB.to_vec()).unwrap();
5558        // A freed leaf page body carves the same rows whether the column count is
5559        // fixed at 6 or inferred.
5560        let page = db.raw_page(10).unwrap();
5561        let fixed = db.carve_cells(&page, 6);
5562        let inferred = db.carve_cells_inferred(&page);
5563        assert!(!fixed.is_empty());
5564        let fixed_ids: std::collections::BTreeSet<i64> = fixed.iter().map(|c| c.rowid).collect();
5565        let inf_ids: std::collections::BTreeSet<i64> = inferred.iter().map(|c| c.rowid).collect();
5566        assert!(fixed_ids.is_subset(&inf_ids));
5567    }
5568
5569    #[test]
5570    fn has_user_table_distinguishes_live_and_dropped() {
5571        let live = Database::open(CLEAN_DB.to_vec()).unwrap();
5572        assert!(live.has_user_table());
5573        let with_deletions = Database::open(DELETED_DB.to_vec()).unwrap();
5574        assert!(with_deletions.has_user_table());
5575    }
5576
5577    #[test]
5578    fn live_rowids_collects_live_rows_only() {
5579        let db = Database::open(CLEAN_DB.to_vec()).unwrap();
5580        let ids = db.live_rowids();
5581        // places.db has 5 live rows, rowids 1..=5.
5582        assert_eq!(ids.len(), 5);
5583        assert!(ids.contains(&1) && ids.contains(&5));
5584
5585        // On the deletions fixture, live rowids are 1..=200; none of the deleted
5586        // 201..=400 appear.
5587        let del = Database::open(DELETED_DB.to_vec()).unwrap();
5588        let live = del.live_rowids();
5589        assert!(live.contains(&1) && live.contains(&200));
5590        assert!(!live.contains(&201) && !live.contains(&400));
5591    }
5592
5593    #[test]
5594    fn live_rows_decodes_current_values() {
5595        let db = Database::open(CLEAN_DB.to_vec()).unwrap();
5596        let rows = db.live_rows();
5597        // places.db has 5 live rows keyed by rowid 1..=5, each decoded to values.
5598        assert_eq!(rows.len(), 5);
5599        // Row 1's url column (index 1) is the rust-lang URL (cross-checks that
5600        // values are decoded, not just rowids collected).
5601        let r1 = rows.get(&1).expect("row 1 present");
5602        assert!(
5603            matches!(r1.get(1), Some(Value::Text(t)) if t.contains("rust-lang")),
5604            "row 1 values must be decoded: {r1:?}"
5605        );
5606        // The value map and the rowid set agree on which rows are live.
5607        let ids = db.live_rowids();
5608        assert_eq!(
5609            rows.keys().copied().collect::<Vec<_>>(),
5610            ids.into_iter().collect::<Vec<_>>()
5611        );
5612
5613        // The deletions fixture's table b-tree has an INTERIOR root page (0x05),
5614        // so this exercises the interior-walk branch of collect_rows and confirms
5615        // values are decoded for all 200 live rows.
5616        let del = Database::open(DELETED_DB.to_vec()).unwrap();
5617        let del_rows = del.live_rows();
5618        assert_eq!(del_rows.len(), 200);
5619        let r1 = del_rows.get(&1).expect("live row 1");
5620        assert!(
5621            matches!(r1.get(1), Some(Value::Text(t)) if t.contains("site-1.example")),
5622            "interior-walked live row 1 must decode its url: {r1:?}"
5623        );
5624    }
5625
5626    #[test]
5627    fn live_table_rows_dumps_each_user_table_in_rowid_order() {
5628        let db = Database::open(CLEAN_DB.to_vec()).unwrap();
5629        let dumps = db.live_table_rows();
5630        // places.db has exactly one user table (moz_places); sqlite_* excluded.
5631        assert_eq!(dumps.len(), 1, "one user-table dump expected: {dumps:?}");
5632        let t = &dumps[0];
5633        assert_eq!(t.name, "moz_places");
5634        // Real column names come from the CREATE TABLE, not generic c0..cN.
5635        assert!(
5636            t.column_names.iter().any(|c| c == "url"),
5637            "real column names expected: {:?}",
5638            t.column_names
5639        );
5640        // The rowids must be the live set, in ascending order.
5641        let rowids: Vec<i64> = t.rows.iter().map(|r| r.rowid).collect();
5642        assert_eq!(rowids, vec![1, 2, 3, 4, 5], "rowid order: {rowids:?}");
5643        // The url cell of row 1 decodes (cross-check values are real).
5644        assert!(
5645            matches!(t.rows[0].values.get(1), Some(Value::Text(s)) if s.contains("rust-lang")),
5646            "row 1 url must decode: {:?}",
5647            t.rows[0].values
5648        );
5649    }
5650
5651    #[test]
5652    fn live_table_rows_excludes_internal_tables_and_handles_interior_btree() {
5653        // The deletions fixture has an INTERIOR root page; all 200 live rows dump
5654        // in ascending rowid order, and no sqlite_* table appears.
5655        let db = Database::open(DELETED_DB.to_vec()).unwrap();
5656        let dumps = db.live_table_rows();
5657        assert!(
5658            dumps.iter().all(|t| !t.name.starts_with("sqlite_")),
5659            "internal tables excluded: {:?}",
5660            dumps.iter().map(|t| &t.name).collect::<Vec<_>>()
5661        );
5662        let places = dumps
5663            .iter()
5664            .find(|t| t.name == "moz_places")
5665            .expect("moz_places dump");
5666        assert_eq!(places.rows.len(), 200, "all live rows dumped");
5667        let ids: Vec<i64> = places.rows.iter().map(|r| r.rowid).collect();
5668        assert!(
5669            ids.windows(2).all(|w| w[0] < w[1]),
5670            "rows in ascending rowid order"
5671        );
5672        assert_eq!(*ids.first().unwrap(), 1);
5673        assert_eq!(*ids.last().unwrap(), 200);
5674    }
5675
5676    #[test]
5677    fn live_table_rows_falls_back_to_generic_columns_on_unparseable_schema() {
5678        // Robustness: a damaged CREATE TABLE whose column list cannot be parsed
5679        // must dump the table with generic c0..cN columns (never a fabricated
5680        // real header), while its rows still read. Mint a valid db, then blank out
5681        // the `( ... )` column list in the stored schema SQL in place (same byte
5682        // length), so column_defs yields None for that table.
5683        use crate::rebuild::{build_recovered_db_tables, RecoveredTable as RT};
5684        let seed = vec![RT {
5685            name: "people".to_string(),
5686            columns: vec!["id".to_string(), "name".to_string()],
5687            rows: vec![vec![Value::Integer(1), Value::Text("alice".into())]],
5688        }];
5689        let mut bytes = build_recovered_db_tables(&seed);
5690
5691        // Find the stored `CREATE TABLE "people" (...)` text and overwrite from the
5692        // first '(' through the matching ')' with spaces, leaving `CREATE TABLE
5693        // "people"` (no column list) — unparseable to column_defs.
5694        let needle = b"CREATE TABLE \"people\"";
5695        let start = bytes
5696            .windows(needle.len())
5697            .position(|w| w == needle)
5698            .expect("schema SQL present");
5699        let open = bytes[start..]
5700            .iter()
5701            .position(|&b| b == b'(')
5702            .map(|p| start + p)
5703            .expect("column list open paren");
5704        let close = bytes[open..]
5705            .iter()
5706            .position(|&b| b == b')')
5707            .map(|p| open + p)
5708            .expect("column list close paren");
5709        for b in &mut bytes[open..=close] {
5710            *b = b' ';
5711        }
5712
5713        let db = Database::open(bytes).expect("corrupted-schema db still opens");
5714        let dumps = db.live_table_rows();
5715        let people = dumps
5716            .iter()
5717            .find(|t| t.name == "people")
5718            .expect("people dump present");
5719        // Generic columns sized to the row width (2), never the real id/name.
5720        assert_eq!(
5721            people.column_names,
5722            vec!["c0".to_string(), "c1".to_string()]
5723        );
5724        // The row still decoded despite the schema damage.
5725        assert_eq!(people.rows.len(), 1);
5726        assert_eq!(people.rows[0].values.first(), Some(&Value::Integer(1)));
5727    }
5728
5729    /// Real-corpus freeblock reconstruction: 0C-01 page 2 has six freeblock-head
5730    /// cells the forward parser cannot reach; reconstruction recovers them
5731    /// (including the destroyed-rowid `id` column) from the surviving serial tail.
5732    const NEMETZ_0C_01: &[u8] = include_bytes!("../../tests/data/nemetz/0C/0C-01.db");
5733
5734    #[test]
5735    fn reconstruct_freeblock_records_recovers_clobbered_rows() {
5736        let db = Database::open(NEMETZ_0C_01.to_vec()).unwrap();
5737        let page = db.raw_page(2).unwrap();
5738        let recovered = db.reconstruct_freeblock_records(&page);
5739        // Row 20005 is a freeblock-head cell only reconstruction can recover.
5740        assert!(recovered.iter().any(|c| c.values
5741            == vec![
5742                Value::Integer(20005),
5743                Value::Integer(3_780_322_152),
5744                Value::Integer(3_909_007_646),
5745                Value::Integer(120_462_986),
5746                Value::Integer(1_290_558_629),
5747            ]));
5748        assert!(recovered
5749            .iter()
5750            .all(|c| c.rowid == 0 && c.confidence <= 0.5));
5751    }
5752
5753    /// Real-corpus span-walking reconstruction (task #66): 0D-07 page 3 coalesces
5754    /// three deleted cells into a single freeblock `[0xf79,0xfe0)` —
5755    /// `Luca|Schumacher` (the head), then `Kurt|Schubert`, then `Georg|Schulz`,
5756    /// each prefixed by a stale `00 00 00 NN` freeblock header that clobbers its
5757    /// leading four bytes. A single-shot head reconstruction recovers only the
5758    /// first; walking the template across the whole span recovers all three.
5759    const NEMETZ_0D_07: &[u8] = include_bytes!("../../tests/data/nemetz/0D/0D-07.db");
5760
5761    #[test]
5762    fn reconstruct_freeblock_records_walks_coalesced_cells() {
5763        let db = Database::open(NEMETZ_0D_07.to_vec()).unwrap();
5764        let page = db.raw_page(3).unwrap();
5765        let recovered = db.reconstruct_freeblock_records(&page);
5766        let has = |name: &str, surname: &str| {
5767            recovered.iter().any(|c| {
5768                matches!(c.values.get(1), Some(Value::Text(t)) if t == name)
5769                    && matches!(c.values.get(2), Some(Value::Text(t)) if t == surname)
5770            })
5771        };
5772        // The span-head cell a single-shot reconstruction already reached.
5773        assert!(has("Luca", "Schumacher"), "head cell must be recovered");
5774        // The two trailing cells deeper inside the same freeblock — only a
5775        // span-walk reaches these.
5776        assert!(
5777            has("Kurt", "Schubert"),
5778            "second coalesced cell must be recovered"
5779        );
5780        assert!(
5781            has("Georg", "Schulz"),
5782            "third coalesced cell must be recovered"
5783        );
5784        // Every reconstruction carries a destroyed rowid and low confidence.
5785        assert!(recovered
5786            .iter()
5787            .all(|c| c.rowid == 0 && c.confidence <= 0.5));
5788    }
5789
5790    /// Helper: a real opened DB to call the page-slice methods against crafted
5791    /// page byte slices (the methods take `page_bytes` explicitly).
5792    fn opened() -> Database {
5793        Database::open(NEMETZ_0C_01.to_vec()).unwrap()
5794    }
5795
5796    /// A leaf page advertising a freeblock chain but whose cells do not parse
5797    /// yields no template, so reconstruction returns empty (covers the
5798    /// `freeblock_template` rejection arms and the final `None`).
5799    #[test]
5800    fn reconstruct_freeblock_records_without_template_is_empty() {
5801        let db = opened();
5802        let mut page = vec![0u8; 256];
5803        page[0] = 0x0d; // table-leaf
5804        page[1] = 0x00;
5805        page[2] = 0x40; // first freeblock at offset 64
5806        page[3] = 0x00;
5807        page[4] = 0x01; // cell_count = 1
5808                        // The single cell pointer (offset 8) points at 0 -> cell_off == 0 -> skipped,
5809                        // so no template can be derived.
5810        page[8] = 0x00;
5811        page[9] = 0x00;
5812        // A freeblock at 64: next=0, size=8 (in-bounds), but no template anyway.
5813        page[64] = 0x00;
5814        page[65] = 0x00;
5815        page[66] = 0x00;
5816        page[67] = 0x08;
5817        assert!(db.reconstruct_freeblock_records(&page).is_empty());
5818    }
5819
5820    /// A cyclic freeblock `next` chain terminates (covers the cycle-break guard)
5821    /// and a freeblock whose size runs past the page is skipped — all without a
5822    /// panic.
5823    #[test]
5824    fn reconstruct_freeblock_records_breaks_cyclic_chain() {
5825        let db = opened();
5826        // Build a page WITH a usable template by copying 0C-01 page 2's header +
5827        // first live cell, then point the freeblock chain at itself.
5828        let src = db.raw_page(2).unwrap().to_vec();
5829        let mut page = src.clone();
5830        // Repoint first-freeblock to a self-cycle at offset 100: next -> 100.
5831        page[1] = 0x00;
5832        page[2] = 100;
5833        page[100] = 0x00;
5834        page[101] = 100; // next = 100 (points to itself)
5835        page[102] = 0xff;
5836        page[103] = 0xff; // size huge -> runs past page -> skipped
5837                          // Must not panic and must terminate.
5838        let _ = db.reconstruct_freeblock_records(&page);
5839    }
5840
5841    // ---- Tier-2 fragment salvage (task #72) --------------------------------
5842
5843    #[test]
5844    fn is_distinctive_classifies_every_storage_class() {
5845        // TEXT >= 4 UTF-8 bytes and REAL are distinctive; everything else is not.
5846        assert!(is_distinctive(&Value::Text("Anja".into())));
5847        assert!(is_distinctive(&Value::Text("\u{00e4}\u{00f6}".into()))); // 4 UTF-8 bytes
5848        assert!(is_distinctive(&Value::Real(3.5)));
5849        assert!(!is_distinctive(&Value::Text("abc".into()))); // 3 bytes
5850        assert!(!is_distinctive(&Value::Text(String::new())));
5851        assert!(!is_distinctive(&Value::Text("ab\u{fffd}x".into()))); // replacement char
5852        assert!(!is_distinctive(&Value::Integer(20004)));
5853        assert!(!is_distinctive(&Value::Null));
5854        assert!(!is_distinctive(&Value::Blob(vec![1, 2, 3, 4, 5])));
5855    }
5856
5857    /// Build a synthetic 256-byte table-leaf (0x0d) page for the fragment tests.
5858    ///
5859    /// Schema implied by the template live cell: 3 columns
5860    /// `(c0: 1-byte int, c1: TEXT-4, c2: TEXT-4)` → serials `[1, 21, 21]`,
5861    /// `header_len = 4`. The live cell (the freeblock template source) is placed
5862    /// at `live_off`. A single freeblock spanning `[fb, fb + fb_size)` holds the
5863    /// freed-cell payload `freed`, whose leading 4 bytes are the stale freeblock
5864    /// header (`next`, `size`) — exactly what freeblock conversion clobbers.
5865    fn synth_frag_page(live_off: usize, fb: usize, fb_size: usize, freed: &[u8]) -> Vec<u8> {
5866        let mut page = vec![0u8; 256];
5867        page[0] = 0x0d; // table-leaf
5868        page[1] = (fb >> 8) as u8;
5869        page[2] = (fb & 0xff) as u8;
5870        page[3] = 0x00;
5871        page[4] = 0x01; // cell_count = 1
5872        page[5] = (live_off >> 8) as u8;
5873        page[6] = (live_off & 0xff) as u8; // cellContentArea = live_off
5874        page[8] = (live_off >> 8) as u8;
5875        page[9] = (live_off & 0xff) as u8; // cell pointer -> live_off
5876
5877        // Live template cell: payload_len=13, rowid=5, header_len=4, serials
5878        // [int1, text4, text4], body 1+4+4.
5879        let live = [
5880            13u8, 5u8, 0x04, 0x01, 0x15, 0x15, 0x09, b'L', b'i', b'v', b'e', b'R', b'o', b'w', b'!',
5881        ];
5882        page[live_off..live_off + live.len()].copy_from_slice(&live);
5883
5884        // Lay the freed-cell bytes first, then stamp the stale freeblock header
5885        // (next=0, size=fb_size) over its first 4 bytes — exactly what freeblock
5886        // conversion does (the header clobbers the freed cell's leading 4 bytes).
5887        page[fb..fb + freed.len()].copy_from_slice(freed);
5888        page[fb] = 0x00;
5889        page[fb + 1] = 0x00;
5890        page[fb + 2] = (fb_size >> 8) as u8;
5891        page[fb + 3] = (fb_size & 0xff) as u8;
5892        page
5893    }
5894
5895    /// (a) Truncated tail: the freed cell's body overruns the freeblock span, so
5896    /// full reconstruction fails — salvage emits the decodable column prefix
5897    /// (incl. a distinctive TEXT cell) with correct `missing`/confidence, while
5898    /// `reconstruct_freeblock_records` recovers nothing from that anchor.
5899    #[test]
5900    fn fragment_salvage_truncated_tail() {
5901        let db = opened();
5902        // surviving serials [21,21] at fb+4,fb+5; body c0(1)+c1(4)+c2(4) at fb+6.
5903        // A full record needs fb+15. Span size 12 ends at fb+12: c0,c1 fit, c2
5904        // overruns → salvage keeps [c0, c1].
5905        let mut freed = vec![0u8; 16];
5906        freed[4] = 0x15;
5907        freed[5] = 0x15;
5908        freed[6] = 0x07;
5909        freed[7..11].copy_from_slice(b"Anja");
5910        freed[11..15].copy_from_slice(b"Frnk");
5911        let page = synth_frag_page(96, 64, 12, &freed);
5912
5913        let frags = db.reconstruct_freeblock_fragments(&page);
5914        assert_eq!(frags.len(), 1, "exactly one fragment salvaged");
5915        let f = &frags[0];
5916        assert_eq!(f.offset, 64);
5917        assert_eq!(
5918            f.surviving,
5919            vec![(0, Value::Integer(7)), (1, Value::Text("Anja".into()))]
5920        );
5921        assert_eq!(f.missing, 1, "c2 did not decode");
5922        assert!((f.confidence - 0.2).abs() < f32::EPSILON);
5923        let cells = db.reconstruct_freeblock_records(&page);
5924        // The page's only freeblock anchor is the truncated one at offset 64, and
5925        // full reconstruction recovers nothing from it — so the full-record set is
5926        // empty. Asserting emptiness is the precise, deterministic intent.
5927        assert!(
5928            cells.is_empty(),
5929            "the truncated anchor yields no full record, got {}",
5930            cells.len()
5931        );
5932    }
5933
5934    /// (b) A surviving column whose body cannot fit ends the prefix early —
5935    /// salvage keeps the columns decoded before the failure.
5936    #[test]
5937    fn fragment_salvage_partial_tail() {
5938        let db = opened();
5939        let mut freed = vec![0u8; 16];
5940        freed[4] = 0x15;
5941        freed[5] = 0x15;
5942        freed[6] = 0x07;
5943        freed[7..11].copy_from_slice(b"Lena");
5944        let page = synth_frag_page(96, 64, 11, &freed); // c1 fits, c2 overruns
5945        let frags = db.reconstruct_freeblock_fragments(&page);
5946        assert_eq!(frags.len(), 1);
5947        assert_eq!(
5948            frags[0].surviving,
5949            vec![(0, Value::Integer(7)), (1, Value::Text("Lena".into()))]
5950        );
5951    }
5952
5953    /// (c) A fully reconstructable freeblock yields NO fragment (mutual exclusion).
5954    #[test]
5955    fn fragment_salvage_full_record_yields_no_fragment() {
5956        let db = opened();
5957        let mut freed = vec![0u8; 16];
5958        freed[4] = 0x15;
5959        freed[5] = 0x15;
5960        freed[6] = 0x07;
5961        freed[7..11].copy_from_slice(b"Whol");
5962        freed[11..15].copy_from_slice(b"Erow");
5963        let page = synth_frag_page(96, 64, 15, &freed);
5964        let cells = db.reconstruct_freeblock_records(&page);
5965        assert!(
5966            cells.iter().any(|c| c.offset == 64),
5967            "full record recovered"
5968        );
5969        assert!(
5970            db.reconstruct_freeblock_fragments(&page).is_empty(),
5971            "no fragment when the full record is recoverable"
5972        );
5973    }
5974
5975    /// (d) Salvage yielding only non-distinctive (INTEGER) cells emits NO fragment.
5976    #[test]
5977    fn fragment_salvage_integer_only_is_rejected() {
5978        let db = opened();
5979        let mut freed = vec![0u8; 12];
5980        freed[4] = 0x01; // surviving 1-byte int
5981        freed[5] = 0x01; // surviving 1-byte int
5982        freed[6] = 0x07;
5983        freed[7] = 0x08;
5984        let page = synth_frag_page(96, 64, 8, &freed); // c2 overruns; only ints decode
5985        assert!(
5986            db.reconstruct_freeblock_fragments(&page).is_empty(),
5987            "integer-only prefix is not distinctive — no fragment"
5988        );
5989    }
5990
5991    /// (e) Fragment salvage does NOT extend the span walk: a failed head stops
5992    /// the walk, emitting at most one fragment, never sliding forward.
5993    #[test]
5994    fn fragment_salvage_does_not_extend_walk() {
5995        let db = opened();
5996        let mut freed = vec![0u8; 16];
5997        freed[4] = 0x15;
5998        freed[5] = 0x15;
5999        freed[6] = 0x07;
6000        freed[7..11].copy_from_slice(b"Stop");
6001        freed[11..15].copy_from_slice(b"Here");
6002        let page = synth_frag_page(96, 64, 12, &freed);
6003        assert_eq!(db.reconstruct_freeblock_fragments(&page).len(), 1);
6004    }
6005
6006    /// (Step 2) Real-artifact validation: 0D-01 page 2 salvages the genuine
6007    /// partial deleted row for id 20004 — `Text("Anja")`/`Text("Frank")` survive
6008    /// in a freeblock whose full-row reconstruction fails. Full pass unchanged.
6009    const NEMETZ_0D_01: &[u8] = include_bytes!("../../tests/data/nemetz/0D/0D-01.db");
6010
6011    #[test]
6012    fn fragment_salvage_recovers_anja_on_0d01() {
6013        let db = Database::open(NEMETZ_0D_01.to_vec()).unwrap();
6014        let page = db.raw_page(2).unwrap();
6015        let frags = db.reconstruct_freeblock_fragments(&page);
6016        let f = frags
6017            .iter()
6018            .find(|f| {
6019                f.surviving
6020                    .iter()
6021                    .any(|(_, v)| matches!(v, Value::Text(t) if t == "Anja"))
6022            })
6023            .expect("0D-01 page 2 must salvage the Anja fragment");
6024        assert!(f
6025            .surviving
6026            .iter()
6027            .any(|(_, v)| matches!(v, Value::Text(t) if t == "Frank")));
6028        assert!((f.confidence - 0.2).abs() < f32::EPSILON);
6029        let cells = db.reconstruct_freeblock_records(&page);
6030        assert!(cells.iter().all(|c| !c
6031            .values
6032            .iter()
6033            .any(|v| matches!(v, Value::Text(t) if t == "Anja"))));
6034    }
6035
6036    // ---- task #73: chain-aware overflow recovery — spilled-cell recognition ----
6037
6038    /// Encode a SQLite varint (minimal big-endian 7-bit groups).
6039    fn enc_varint(mut n: u64) -> Vec<u8> {
6040        if n == 0 {
6041            return vec![0];
6042        }
6043        let mut groups = Vec::new();
6044        while n > 0 {
6045            groups.push((n & 0x7f) as u8);
6046            n >>= 7;
6047        }
6048        groups.reverse();
6049        let last = groups.len() - 1;
6050        for (i, g) in groups.iter_mut().enumerate() {
6051            if i != last {
6052                *g |= 0x80;
6053            }
6054        }
6055        groups
6056    }
6057
6058    /// Build the **local prefix** bytes of a freed spilled table-leaf cell:
6059    /// `payload_len varint, rowid varint, record header, local payload bytes,
6060    /// 4-byte big-endian first-overflow pointer`. Returns `(bytes, P, local,
6061    /// serials)`. The record is `(id INTEGER, name TEXT, code TEXT)` with `code`
6062    /// large enough to force a spill past `usable - 35`.
6063    fn synth_spilled_prefix(
6064        rowid: i64,
6065        id: i64,
6066        name: &str,
6067        code_len: usize,
6068        usable: usize,
6069        first_overflow: u32,
6070    ) -> (Vec<u8>, usize, usize, Vec<i64>) {
6071        let id_serial = 1i64; // 1-byte integer
6072        let name_serial = 13 + 2 * name.len() as i64; // TEXT
6073        let code_serial = 13 + 2 * code_len as i64; // TEXT
6074        let serials = vec![id_serial, name_serial, code_serial];
6075        let mut serial_bytes = Vec::new();
6076        for &s in &serials {
6077            serial_bytes.extend(enc_varint(s as u64));
6078        }
6079        // header_len varint counts itself — solve the fixed point.
6080        let mut header_len = serial_bytes.len() + 1;
6081        while enc_varint(header_len as u64).len() + serial_bytes.len() != header_len {
6082            header_len += 1;
6083        }
6084        let mut header = enc_varint(header_len as u64);
6085        header.extend(&serial_bytes);
6086        let body_len = 1 + name.len() + code_len;
6087        let payload_len = header.len() + body_len;
6088        let local = local_payload_len(payload_len, usable);
6089
6090        // Full payload = header ++ id-body ++ name-body ++ code-body.
6091        let mut payload = header.clone();
6092        payload.push(id as u8); // 1-byte id
6093        payload.extend(name.as_bytes());
6094        payload.extend(std::iter::repeat_n(b'C', code_len));
6095        assert_eq!(payload.len(), payload_len);
6096
6097        // Cell = prefix varints ++ local payload prefix ++ 4-byte overflow ptr.
6098        let mut cell = enc_varint(payload_len as u64);
6099        cell.extend(enc_varint(rowid as u64));
6100        cell.extend(&payload[..local]);
6101        cell.extend(first_overflow.to_be_bytes());
6102        (cell, payload_len, local, serials)
6103    }
6104
6105    #[test]
6106    fn spilled_recognizer_reads_intact_prefix() {
6107        let usable = 4096usize;
6108        let (cell, p, local, serials) = synth_spilled_prefix(20012, 42, "Ella", 4200, usable, 13);
6109        assert!(p > usable - 35, "this record must spill");
6110        // Place the cell inside a larger scanned slice at a nonzero offset.
6111        let off = 50usize;
6112        let mut buf = vec![0u8; off];
6113        buf.extend(&cell);
6114        let sc = try_carve_spilled_cell_at(&buf, off, usable, Some(3))
6115            .expect("must recognize the intact-prefix spilled cell");
6116        assert_eq!(sc.payload_len, p);
6117        assert_eq!(sc.local_len, local);
6118        assert_eq!(sc.rowid, 20012);
6119        assert_eq!(sc.first_overflow, 13);
6120        assert_eq!(sc.serials, serials);
6121        assert_eq!(sc.offset, off);
6122    }
6123
6124    #[test]
6125    fn spilled_recognizer_abstains_for_in_page_payload() {
6126        let usable = 4096usize;
6127        // A small (in-page) payload: the existing carve path owns it.
6128        // header (3 serials) + body for a tiny code -> P <= usable-35.
6129        let (cell, p, _local, _s) = synth_spilled_prefix(7, 1, "Bob", 10, usable, 9);
6130        assert!(p <= usable - 35, "this record must NOT spill");
6131        assert!(try_carve_spilled_cell_at(&cell, 0, usable, Some(3)).is_none());
6132    }
6133
6134    #[test]
6135    fn spilled_recognizer_abstains_on_truncated_pointer() {
6136        let usable = 4096usize;
6137        let (cell, _p, _local, _s) = synth_spilled_prefix(20012, 42, "Ella", 4200, usable, 13);
6138        // Drop the final 2 bytes so the 4-byte overflow pointer is out of bounds.
6139        let truncated = &cell[..cell.len() - 2];
6140        assert!(try_carve_spilled_cell_at(truncated, 0, usable, Some(3)).is_none());
6141    }
6142
6143    #[test]
6144    fn spilled_recognizer_abstains_on_column_mismatch() {
6145        let usable = 4096usize;
6146        let (cell, _p, _local, _s) = synth_spilled_prefix(20012, 42, "Ella", 4200, usable, 13);
6147        // Expect 5 columns but the record has 3.
6148        assert!(try_carve_spilled_cell_at(&cell, 0, usable, Some(5)).is_none());
6149        // Inferred (None) still recognizes it.
6150        assert!(try_carve_spilled_cell_at(&cell, 0, usable, None).is_some());
6151    }
6152
6153    #[test]
6154    fn spilled_recognizer_abstains_on_nonpositive_rowid() {
6155        let usable = 4096usize;
6156        let (cell, _p, _local, _s) = synth_spilled_prefix(0, 42, "Ella", 4000, usable, 13);
6157        assert!(try_carve_spilled_cell_at(&cell, 0, usable, Some(3)).is_none());
6158    }
6159
6160    // ---- task #73: freed overflow-chain walk + freelist leaf/trunk split ----
6161
6162    /// Build a minimal multi-page `SQLite` DB image with `page_count` pages of
6163    /// `page_size` bytes. Page 1 carries a valid 100-byte header (so
6164    /// `Database::open` succeeds) with the given freelist trunk pointer and count
6165    /// at offsets 32/36. All pages are zero-filled; the caller writes overflow /
6166    /// trunk content afterwards. Returns the byte vector.
6167    fn synth_db(page_size: usize, page_count: usize, trunk: u32, fl_count: u32) -> Vec<u8> {
6168        let mut b = vec![0u8; page_size * page_count];
6169        b[..16].copy_from_slice(SQLITE_MAGIC);
6170        b[16..18].copy_from_slice(&(page_size as u16).to_be_bytes());
6171        b[18] = 1; // file format write version
6172        b[19] = 1; // file format read version
6173        b[20] = 0; // reserved space
6174        b[21] = 64;
6175        b[22] = 32;
6176        b[23] = 32;
6177        b[32..36].copy_from_slice(&trunk.to_be_bytes());
6178        b[36..40].copy_from_slice(&fl_count.to_be_bytes());
6179        // A minimal table-leaf page-1 body (type 0x0d, 0 cells) so header parsing
6180        // and page-count helpers behave.
6181        b[100] = 0x0d;
6182        b
6183    }
6184
6185    /// Write a freelist trunk page at `page` listing `leaves` and chaining to
6186    /// `next_trunk` (0 = end).
6187    fn write_trunk(b: &mut [u8], page_size: usize, page: u32, next_trunk: u32, leaves: &[u32]) {
6188        let base = (page as usize - 1) * page_size;
6189        b[base..base + 4].copy_from_slice(&next_trunk.to_be_bytes());
6190        b[base + 4..base + 8].copy_from_slice(&(leaves.len() as u32).to_be_bytes());
6191        for (i, &lf) in leaves.iter().enumerate() {
6192            b[base + 8 + i * 4..base + 12 + i * 4].copy_from_slice(&lf.to_be_bytes());
6193        }
6194    }
6195
6196    /// Write an overflow page at `page`: 4-byte big-endian `next` then `content`.
6197    fn write_overflow(b: &mut [u8], page_size: usize, page: u32, next: u32, content: &[u8]) {
6198        let base = (page as usize - 1) * page_size;
6199        b[base..base + 4].copy_from_slice(&next.to_be_bytes());
6200        b[base + 4..base + 4 + content.len()].copy_from_slice(content);
6201    }
6202
6203    #[test]
6204    fn freelist_split_separates_leaves_and_trunks() {
6205        let ps = 512usize;
6206        // Pages: 1 header, 2 trunk, leaves 3,4,5.
6207        let mut b = synth_db(ps, 6, 2, 4);
6208        write_trunk(&mut b, ps, 2, 0, &[3, 4, 5]);
6209        let db = Database::open(b).unwrap();
6210        let (leaves, trunks) = db.freelist_pages_split().unwrap();
6211        assert_eq!(leaves, [3u32, 4, 5].into_iter().collect());
6212        assert_eq!(trunks, [2u32].into_iter().collect());
6213        // The legacy combined accessor still returns leaves ++ trunk.
6214        let all: std::collections::BTreeSet<u32> =
6215            db.freelist_pages().unwrap().into_iter().collect();
6216        assert_eq!(all, [2u32, 3, 4, 5].into_iter().collect());
6217    }
6218
6219    #[test]
6220    fn freed_chain_assembles_single_leaf_page() {
6221        let ps = 512usize;
6222        let usable = ps; // reserved 0
6223        let mut b = synth_db(ps, 6, 2, 4);
6224        write_trunk(&mut b, ps, 2, 0, &[3, 4, 5]);
6225        // Chain content on leaf page 3: a single page holds `remaining` bytes.
6226        let remaining = 100usize;
6227        let content: Vec<u8> = (0..remaining).map(|i| (i % 251) as u8).collect();
6228        write_overflow(&mut b, ps, 3, 0, &content);
6229        let db = Database::open(b).unwrap();
6230        let (leaves, _trunks) = db.freelist_pages_split().unwrap();
6231        let (bytes, chain) = db
6232            .read_freed_overflow_chain(3, remaining, usable, &leaves)
6233            .expect("intact single-leaf chain must assemble");
6234        assert_eq!(bytes, content);
6235        assert_eq!(chain, vec![3]);
6236    }
6237
6238    #[test]
6239    fn freed_chain_assembles_multi_leaf_pages() {
6240        let ps = 512usize;
6241        let usable = ps;
6242        let per_page = usable - 4;
6243        let mut b = synth_db(ps, 8, 2, 5);
6244        write_trunk(&mut b, ps, 2, 0, &[3, 4, 5, 6]);
6245        // 2-page chain: page 3 -> page 4. remaining spans into page 4.
6246        let remaining = per_page + 50;
6247        let content: Vec<u8> = (0..remaining).map(|i| (i % 251) as u8).collect();
6248        write_overflow(&mut b, ps, 3, 4, &content[..per_page]);
6249        write_overflow(&mut b, ps, 4, 0, &content[per_page..]);
6250        let db = Database::open(b).unwrap();
6251        let (leaves, _t) = db.freelist_pages_split().unwrap();
6252        let (bytes, chain) = db
6253            .read_freed_overflow_chain(3, remaining, usable, &leaves)
6254            .expect("intact 2-leaf chain must assemble");
6255        assert_eq!(bytes, content);
6256        assert_eq!(chain, vec![3, 4]);
6257    }
6258
6259    #[test]
6260    fn freed_chain_breaks_on_non_freelist_page() {
6261        let ps = 512usize;
6262        let usable = ps;
6263        let mut b = synth_db(ps, 6, 2, 2);
6264        write_trunk(&mut b, ps, 2, 0, &[3]); // only page 3 is a leaf
6265        let content = vec![7u8; 100];
6266        // The pointer targets page 4, which is NOT on the freelist.
6267        write_overflow(&mut b, ps, 4, 0, &content);
6268        let db = Database::open(b).unwrap();
6269        let (leaves, _t) = db.freelist_pages_split().unwrap();
6270        assert!(db
6271            .read_freed_overflow_chain(4, 100, usable, &leaves)
6272            .is_err());
6273    }
6274
6275    #[test]
6276    fn freed_chain_breaks_on_trunk_page() {
6277        let ps = 512usize;
6278        let usable = ps;
6279        let mut b = synth_db(ps, 6, 2, 2);
6280        write_trunk(&mut b, ps, 2, 0, &[3]);
6281        let db = Database::open(b).unwrap();
6282        let (leaves, _t) = db.freelist_pages_split().unwrap();
6283        // Page 2 is the trunk — a chain page that is a trunk must break.
6284        assert!(db
6285            .read_freed_overflow_chain(2, 100, usable, &leaves)
6286            .is_err());
6287    }
6288
6289    #[test]
6290    fn freed_chain_breaks_on_cycle() {
6291        let ps = 512usize;
6292        let usable = ps;
6293        let per_page = usable - 4;
6294        let mut b = synth_db(ps, 6, 2, 3);
6295        write_trunk(&mut b, ps, 2, 0, &[3, 4]);
6296        // 3 -> 4 -> 3 cycle; remaining never satisfied.
6297        write_overflow(&mut b, ps, 3, 4, &vec![1u8; per_page]);
6298        write_overflow(&mut b, ps, 4, 3, &vec![2u8; per_page]);
6299        let db = Database::open(b).unwrap();
6300        let (leaves, _t) = db.freelist_pages_split().unwrap();
6301        assert!(db
6302            .read_freed_overflow_chain(3, per_page * 10, usable, &leaves)
6303            .is_err());
6304    }
6305
6306    #[test]
6307    fn freed_chain_breaks_on_premature_zero_pointer() {
6308        let ps = 512usize;
6309        let usable = ps;
6310        let per_page = usable - 4;
6311        let mut b = synth_db(ps, 6, 2, 2);
6312        write_trunk(&mut b, ps, 2, 0, &[3]);
6313        // Page 3 ends the chain (next=0) but `remaining` still wants more bytes.
6314        write_overflow(&mut b, ps, 3, 0, &vec![9u8; per_page]);
6315        let db = Database::open(b).unwrap();
6316        let (leaves, _t) = db.freelist_pages_split().unwrap();
6317        assert!(db
6318            .read_freed_overflow_chain(3, per_page + 10, usable, &leaves)
6319            .is_err());
6320    }
6321
6322    #[test]
6323    fn freed_chain_breaks_on_capacity_overflow() {
6324        let ps = 512usize;
6325        let usable = ps;
6326        let mut b = synth_db(ps, 6, 2, 2);
6327        write_trunk(&mut b, ps, 2, 0, &[3]);
6328        write_overflow(&mut b, ps, 3, 0, &vec![1u8; usable - 4]);
6329        let db = Database::open(b).unwrap();
6330        let (leaves, _t) = db.freelist_pages_split().unwrap();
6331        // remaining far exceeds what one leaf page can deliver — rejected upfront,
6332        // never allocating an attacker-declared payload.
6333        let absurd = (usable - 4) * leaves.len() + 1;
6334        assert!(db
6335            .read_freed_overflow_chain(3, absurd, usable, &leaves)
6336            .is_err());
6337    }
6338
6339    // ---- task #73 step 5: freeblock-clobbered spilled cell (SYNTHETIC ONLY) ----
6340    // Codex ruling #5: there is NO corpus instance for a freeblock-clobbered
6341    // *spilled* cell — this path is validated against a synthetic fixture only
6342    // and is marked unproven-by-corpus in the production code + docs.
6343
6344    /// Build a synthetic 4096-byte-page DB with an allocated table-leaf page 2
6345    /// holding (a) a LIVE template cell of the `(id INTEGER 1-byte, name TEXT,
6346    /// code TEXT)` schema and (b) a freeblock-clobbered SPILLED cell whose 4-byte
6347    /// prefix is overwritten by a stale freeblock header, with its overflow chain
6348    /// on a freed leaf page. Returns the bytes. `break_chain` routes the chain
6349    /// pointer at the freelist trunk instead of a leaf to exercise the rejection.
6350    fn synth_clobbered_spill_db(break_chain: bool) -> Vec<u8> {
6351        let ps = 4096usize;
6352        let usable = ps;
6353        // Pages: 1 header, 2 allocated leaf, 3 trunk, 4 leaf (chain), 5 leaf spare.
6354        let mut b = synth_db(ps, 6, 3, 2);
6355        write_trunk(&mut b, ps, 3, 0, &[4, 5]);
6356
6357        // Record geometry: id=7 (1-byte), name="Zoe", code 4200×'C'.
6358        let name = b"Zoe";
6359        let code_len = 4200usize;
6360        let serials: [i64; 3] = [1, 13 + 2 * name.len() as i64, 13 + 2 * code_len as i64];
6361        let mut serial_bytes = Vec::new();
6362        for &s in &serials {
6363            serial_bytes.extend(enc_varint(s as u64));
6364        }
6365        let mut header_len = serial_bytes.len() + 1;
6366        while enc_varint(header_len as u64).len() + serial_bytes.len() != header_len {
6367            header_len += 1;
6368        }
6369        let mut header = enc_varint(header_len as u64);
6370        header.extend(&serial_bytes);
6371        let mut full_payload = header.clone();
6372        full_payload.push(7u8); // id body
6373        full_payload.extend(name);
6374        full_payload.extend(std::iter::repeat_n(b'C', code_len));
6375        let payload_len = full_payload.len();
6376        let local = local_payload_len(payload_len, usable);
6377        let remaining = payload_len - local;
6378
6379        // --- LIVE template cell at offset 200 on page 2 (a small non-spilling row
6380        //     of the SAME schema so freeblock_template derives the column layout).
6381        let base2 = ps; // page 2 starts at byte 4096
6382        let tmpl_name = b"Al";
6383        let tmpl_code = b"xy";
6384        let tser: [i64; 3] = [
6385            1,
6386            13 + 2 * tmpl_name.len() as i64,
6387            13 + 2 * tmpl_code.len() as i64,
6388        ];
6389        let mut tsb = Vec::new();
6390        for &s in &tser {
6391            tsb.extend(enc_varint(s as u64));
6392        }
6393        let mut thl = tsb.len() + 1;
6394        while enc_varint(thl as u64).len() + tsb.len() != thl {
6395            thl += 1;
6396        }
6397        let mut tpayload = enc_varint(thl as u64);
6398        tpayload.extend(&tsb);
6399        tpayload.push(1u8);
6400        tpayload.extend(tmpl_name);
6401        tpayload.extend(tmpl_code);
6402        let live_off = 200usize;
6403        let mut live_cell = enc_varint(tpayload.len() as u64);
6404        live_cell.extend(enc_varint(1u64)); // rowid 1
6405        live_cell.extend(&tpayload);
6406        b[base2 + live_off..base2 + live_off + live_cell.len()].copy_from_slice(&live_cell);
6407
6408        // Page-2 leaf header (type 0x0d), 1 live cell, freeblock at 0x100, content
6409        // area covering both the live cell and the clobbered spilled cell.
6410        b[base2] = 0x0d;
6411        // first freeblock pointer (offset 1) -> the clobbered spilled cell at 1000.
6412        b[base2 + 1..base2 + 3].copy_from_slice(&1000u16.to_be_bytes());
6413        // cell count (offset 3) = 1
6414        b[base2 + 3..base2 + 5].copy_from_slice(&1u16.to_be_bytes());
6415        // cell content area start (offset 5) — low so both regions are "content".
6416        b[base2 + 5..base2 + 7].copy_from_slice(&100u16.to_be_bytes());
6417        // cell pointer array (1 entry) at offset 8 -> live cell offset.
6418        b[base2 + 8..base2 + 10].copy_from_slice(&(live_off as u16).to_be_bytes());
6419
6420        // --- Clobbered SPILLED cell at offset 1000 on page 2. Lay down the FULL
6421        //     prefix (payload_len varint, rowid varint, header, local payload,
6422        //     overflow ptr), then OVERWRITE the first 4 bytes with a stale
6423        //     freeblock header (next=0x0000, size) to simulate freeblock clobber.
6424        let spill_off = 1000usize;
6425        let mut spill_cell = enc_varint(payload_len as u64);
6426        spill_cell.extend(enc_varint(1u64)); // rowid (will be clobbered)
6427        let prefix_len = spill_cell.len();
6428        spill_cell.extend(&full_payload[..local]);
6429        let chain_first = if break_chain { 3u32 } else { 4u32 };
6430        spill_cell.extend(chain_first.to_be_bytes());
6431        b[base2 + spill_off..base2 + spill_off + spill_cell.len()].copy_from_slice(&spill_cell);
6432        // Clobber the first 4 bytes with a freeblock header: next=0, size=4.
6433        b[base2 + spill_off] = 0;
6434        b[base2 + spill_off + 1] = 0;
6435        b[base2 + spill_off + 2..base2 + spill_off + 4].copy_from_slice(&4u16.to_be_bytes());
6436
6437        // --- The overflow chain content on freed leaf page 4 (next=0).
6438        write_overflow(&mut b, ps, 4, 0, &full_payload[local..local + remaining]);
6439
6440        let _ = prefix_len;
6441        b
6442    }
6443
6444    #[test]
6445    fn clobbered_spilled_cell_reconstructs_with_unknown_rowid() {
6446        let db = Database::open(synth_clobbered_spill_db(false)).unwrap();
6447        let page2 = db.raw_page(2).unwrap();
6448        let recovered = db.carve_overflow_template_records(&page2);
6449        let (cell, chain) = recovered
6450            .iter()
6451            .find(|(c, _)| matches!(c.values.get(1), Some(Value::Text(t)) if t == "Zoe"))
6452            .expect("synthetic clobbered spilled cell must reconstruct");
6453        // rowid destroyed by the freeblock clobber -> surfaced as 0.
6454        assert_eq!(cell.rowid, 0);
6455        // code fully reassembled across the chain.
6456        assert!(matches!(cell.values.get(2), Some(Value::Text(t)) if t.len() == 4200));
6457        assert_eq!(chain, &vec![4u32]);
6458    }
6459
6460    #[test]
6461    fn clobbered_spilled_broken_chain_yields_no_full_row() {
6462        // Chain pointer routed at the freelist TRUNK (page 3) -> rejected.
6463        let db = Database::open(synth_clobbered_spill_db(true)).unwrap();
6464        let page2 = db.raw_page(2).unwrap();
6465        let recovered = db.carve_overflow_template_records(&page2);
6466        // A chain routed through the freelist trunk is rejected outright, so the
6467        // template carve recovers no full row at all (not merely no "Zoe" row).
6468        assert!(
6469            recovered.is_empty(),
6470            "a trunk-routed broken chain must yield no full row, got {} rows",
6471            recovered.len()
6472        );
6473    }
6474
6475    #[test]
6476    fn enc_varint_into_round_trips_zero_and_multibyte() {
6477        // Zero -> single 0 byte (the NULL-serial / empty-header path).
6478        assert_eq!(enc_varint_into(0), vec![0]);
6479        assert_eq!(varint_len(0), 1);
6480        // Multi-byte: 8413 -> 2-byte varint; round-trips via read_varint.
6481        let v = enc_varint_into(8413);
6482        assert_eq!(varint_len(8413), v.len());
6483        assert_eq!(read_varint(&v, 0).unwrap(), (8413, v.len()));
6484        // Negative input (illegal serial) treated as 1 byte (defensive).
6485        assert_eq!(varint_len(-1), 1);
6486    }
6487
6488    /// Build a 4096-byte-page DB with an allocated table-leaf page 2 holding an
6489    /// **intact-prefix** spilled cell in its unallocated gap, with the overflow
6490    /// chain on a freed leaf page (page 4). Mirrors the real 0E geometry so
6491    /// `carve_overflow_records` (and its fragment dual) can be unit-covered without
6492    /// the corpus. `break_chain` routes the pointer at the freelist trunk.
6493    fn synth_gap_spill_db(break_chain: bool, code_len: usize, name: &str) -> Vec<u8> {
6494        let ps = 4096usize;
6495        let usable = ps;
6496        let mut b = synth_db(ps, 6, 3, 2);
6497        write_trunk(&mut b, ps, 3, 0, &[4, 5]);
6498        let base2 = ps;
6499
6500        // Record: (id INTEGER 1-byte, name TEXT, code TEXT) spilled.
6501        let serials: [i64; 3] = [1, 13 + 2 * name.len() as i64, 13 + 2 * code_len as i64];
6502        let mut serial_bytes = Vec::new();
6503        for &s in &serials {
6504            serial_bytes.extend(enc_varint(s as u64));
6505        }
6506        let mut header_len = serial_bytes.len() + 1;
6507        while enc_varint(header_len as u64).len() + serial_bytes.len() != header_len {
6508            header_len += 1;
6509        }
6510        let mut payload = enc_varint(header_len as u64);
6511        payload.extend(&serial_bytes);
6512        payload.push(9u8); // id body
6513        payload.extend(name.as_bytes());
6514        payload.extend(std::iter::repeat_n(b'C', code_len));
6515        let payload_len = payload.len();
6516        let local = local_payload_len(payload_len, usable);
6517        let remaining = payload_len - local;
6518
6519        // Spilled cell at gap offset 1500 on page 2 (intact prefix).
6520        let spill_off = 1500usize;
6521        let mut cell = enc_varint(payload_len as u64);
6522        cell.extend(enc_varint(5u64)); // rowid 5
6523        cell.extend(&payload[..local]);
6524        let first = if break_chain { 3u32 } else { 4u32 };
6525        cell.extend(first.to_be_bytes());
6526        b[base2 + spill_off..base2 + spill_off + cell.len()].copy_from_slice(&cell);
6527
6528        // Page-2 leaf header: 0 live cells, content area at 100 so the gap [8,100..]
6529        // is scanned. No live cells keeps free_regions = the whole content area.
6530        b[base2] = 0x0d;
6531        b[base2 + 1] = 0; // first freeblock = 0
6532        b[base2 + 2] = 0;
6533        b[base2 + 3..base2 + 5].copy_from_slice(&0u16.to_be_bytes()); // 0 cells
6534        b[base2 + 5..base2 + 7].copy_from_slice(&8u16.to_be_bytes()); // cca low
6535
6536        // Chain content on freed leaf page 4.
6537        write_overflow(&mut b, ps, 4, 0, &payload[local..local + remaining]);
6538        b
6539    }
6540
6541    #[test]
6542    fn carve_overflow_records_resolves_gap_spill() {
6543        let db = Database::open(synth_gap_spill_db(false, 4200, "Nora")).unwrap();
6544        let page2 = db.raw_page(2).unwrap();
6545        let recovered = db.carve_overflow_records(&page2);
6546        let (cell, chain) = recovered
6547            .iter()
6548            .find(|(c, _)| matches!(c.values.get(1), Some(Value::Text(t)) if t == "Nora"))
6549            .expect("gap-resident spilled cell must resolve to a full row");
6550        assert_eq!(cell.rowid, 5);
6551        assert!(matches!(cell.values.get(2), Some(Value::Text(t)) if t.len() == 4200));
6552        assert_eq!(chain, &vec![4u32]);
6553        // Graded below the in-page full-row tier (0.9 * factor).
6554        assert!(cell.confidence < 0.72);
6555        // Non-leaf page yields nothing; empty slice yields nothing.
6556        assert!(db.carve_overflow_records(&[0x05u8; 4096]).is_empty());
6557        assert!(db.carve_overflow_records(&[]).is_empty());
6558    }
6559
6560    #[test]
6561    fn carve_overflow_records_rejects_trunk_chain() {
6562        let db = Database::open(synth_gap_spill_db(true, 4200, "Nora")).unwrap();
6563        let page2 = db.raw_page(2).unwrap();
6564        // Chain routed at the trunk -> no full row recovered at all.
6565        let recovered = db.carve_overflow_records(&page2);
6566        assert!(
6567            recovered.is_empty(),
6568            "a trunk-routed chain must yield no full overflow row, got {} rows",
6569            recovered.len()
6570        );
6571    }
6572
6573    #[test]
6574    fn stale_leaf_chain_with_invalid_utf8_is_rejected() {
6575        // NEGATIVE test (the stale-leaf residual): a chain page that IS a freelist
6576        // leaf and assembles to the exact declared length, but whose content is
6577        // unrelated bytes (invalid UTF-8 in the TEXT column). The freelist-leaf
6578        // requirement passes; the strict-UTF-8 extra-signal gate rejects it from
6579        // Tier-1. This documents the design's limit (Codex ruling #2): the leaf
6580        // requirement cannot prove the bytes are the record — only the UTF-8 gate
6581        // catches the cases the lossy decoder would otherwise mask.
6582        let ps = 4096usize;
6583        let usable = ps;
6584        let mut b = synth_db(ps, 6, 3, 2);
6585        write_trunk(&mut b, ps, 3, 0, &[4, 5]);
6586        let base2 = ps;
6587        let name = "Stale";
6588        let code_len = 4200usize;
6589        let serials: [i64; 3] = [1, 13 + 2 * name.len() as i64, 13 + 2 * code_len as i64];
6590        let mut serial_bytes = Vec::new();
6591        for &s in &serials {
6592            serial_bytes.extend(enc_varint(s as u64));
6593        }
6594        let mut header_len = serial_bytes.len() + 1;
6595        while enc_varint(header_len as u64).len() + serial_bytes.len() != header_len {
6596            header_len += 1;
6597        }
6598        let mut payload = enc_varint(header_len as u64);
6599        payload.extend(&serial_bytes);
6600        payload.push(9u8);
6601        payload.extend(name.as_bytes());
6602        payload.extend(std::iter::repeat_n(b'C', code_len));
6603        let payload_len = payload.len();
6604        let local = local_payload_len(payload_len, usable);
6605        let remaining = payload_len - local;
6606
6607        let spill_off = 1500usize;
6608        let mut cell = enc_varint(payload_len as u64);
6609        cell.extend(enc_varint(5u64));
6610        cell.extend(&payload[..local]);
6611        cell.extend(4u32.to_be_bytes());
6612        b[base2 + spill_off..base2 + spill_off + cell.len()].copy_from_slice(&cell);
6613        b[base2] = 0x0d;
6614        b[base2 + 3..base2 + 5].copy_from_slice(&0u16.to_be_bytes());
6615        b[base2 + 5..base2 + 7].copy_from_slice(&8u16.to_be_bytes());
6616
6617        // Stale leaf content: invalid UTF-8 (0xff bytes) where the TEXT body lands.
6618        let stale = vec![0xffu8; remaining];
6619        write_overflow(&mut b, ps, 4, 0, &stale);
6620
6621        let db = Database::open(b).unwrap();
6622        let page2 = db.raw_page(2).unwrap();
6623        // Decodes mechanically (the leaf assembles exactly), but the strict-UTF-8
6624        // gate rejects it -> NOT a Tier-1 full row.
6625        assert!(db.carve_overflow_records(&page2).is_empty());
6626    }
6627
6628    #[test]
6629    fn carve_overflow_fragments_salvages_broken_gap_spill() {
6630        // Broken chain (trunk) -> the local prefix (id + name) salvages as a fragment.
6631        let db = Database::open(synth_gap_spill_db(true, 4200, "Nora")).unwrap();
6632        let page2 = db.raw_page(2).unwrap();
6633        let frags = db.carve_overflow_fragments(&page2);
6634        let f = frags
6635            .iter()
6636            .find(|f| {
6637                f.surviving
6638                    .iter()
6639                    .any(|(_, v)| matches!(v, Value::Text(t) if t == "Nora"))
6640            })
6641            .expect("broken-chain gap spill must salvage a fragment");
6642        // id (col 0) survives locally too.
6643        assert!(f
6644            .surviving
6645            .iter()
6646            .any(|(i, v)| *i == 0 && matches!(v, Value::Integer(9))));
6647        // An intact chain produces NO fragment (it is a full row instead), so the
6648        // fragment set is empty — assert that directly rather than over a vacuous
6649        // per-fragment predicate.
6650        let ok = Database::open(synth_gap_spill_db(false, 4200, "Nora")).unwrap();
6651        let ok_page = ok.raw_page(2).unwrap();
6652        assert!(
6653            ok.carve_overflow_fragments(&ok_page).is_empty(),
6654            "an intact chain yields a full row, not a fragment"
6655        );
6656        // Non-leaf / empty inputs yield nothing.
6657        assert!(db.carve_overflow_fragments(&[0x05u8; 4096]).is_empty());
6658        assert!(db.carve_overflow_fragments(&[]).is_empty());
6659    }
6660
6661    // --- WAL frame checksum (file-format §4.2) -------------------------------
6662
6663    #[test]
6664    fn wal_checksum_known_vector_both_endiannesses() {
6665        // The §4.2 algorithm over a hand-constructed 8-byte input, from a zero
6666        // seed. Input is two 32-bit words x0, x1; the recurrence is
6667        //   s0 += x0 + s1;  s1 += x1 + s0;
6668        // From (s0,s1)=(0,0): s0 = x0; s1 = x1 + x0.
6669        //
6670        // BIG-ENDIAN words (magic 0x377f0683 per the spec): bytes
6671        // [00 00 00 02][00 00 00 03] -> x0=2, x1=3 -> s0=2, s1=5.
6672        let data_be = [0, 0, 0, 2, 0, 0, 0, 3];
6673        assert_eq!(wal_checksum(WalChecksumEndian::Big, 0, 0, &data_be), (2, 5));
6674
6675        // LITTLE-ENDIAN words (magic 0x377f0682): the SAME bytes read LE give
6676        // x0=0x02000000, x1=0x03000000 -> s0=0x02000000,
6677        // s1 = 0x03000000 + 0x02000000 = 0x05000000 (wrapping u32).
6678        assert_eq!(
6679            wal_checksum(WalChecksumEndian::Little, 0, 0, &data_be),
6680            (0x0200_0000, 0x0500_0000)
6681        );
6682
6683        // Seed carries forward: from (s0,s1)=(2,5) over the same BE input ->
6684        // s0 = 2 + (2 + 5) = 9; s1 = 5 + (3 + 9) = 17.
6685        assert_eq!(
6686            wal_checksum(WalChecksumEndian::Big, 2, 5, &data_be),
6687            (9, 17)
6688        );
6689
6690        // Wrapping arithmetic must not panic on overflow (u32 wrap, not i32).
6691        let big = [0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff];
6692        let _ = wal_checksum(WalChecksumEndian::Big, u32::MAX, u32::MAX, &big);
6693    }
6694
6695    #[test]
6696    fn wal_checksum_endian_from_magic_matches_spec() {
6697        // file-format §4.2: 0x377f0683 = BIG-endian words, 0x377f0682 = LITTLE.
6698        assert_eq!(
6699            WalChecksumEndian::from_magic(0x377f_0683),
6700            Some(WalChecksumEndian::Big)
6701        );
6702        assert_eq!(
6703            WalChecksumEndian::from_magic(0x377f_0682),
6704            Some(WalChecksumEndian::Little)
6705        );
6706        assert_eq!(WalChecksumEndian::from_magic(0xdead_beef), None);
6707    }
6708
6709    // --- per-commit schema (CommitSnapshot::tables) -------------------------
6710
6711    /// Wrap a minted main-db image into a `(main, wal)` pair whose WAL commits a
6712    /// full rewrite of every page in ONE commit, with correct §4.2 checksums (so
6713    /// the snapshot is checksum-valid). The snapshot then materializes exactly the
6714    /// minted db, with its real page-1 `sqlite_master` b-tree — the no-sqlite3 way
6715    /// to drive `CommitSnapshot::tables` / snapshot reads against a genuine schema.
6716    fn wrap_db_in_wal(main: &[u8], page_size: u32) -> Vec<u8> {
6717        let ps = page_size as usize;
6718        let n_pages = main.len() / ps;
6719        let endian = WalChecksumEndian::Little; // arbitrary; matches magic below.
6720        let (salt1, salt2) = (0x1234_5678u32, 0x9abc_def0u32);
6721
6722        let mut wal = vec![0u8; 32];
6723        wal[0..4].copy_from_slice(&0x377f_0682u32.to_be_bytes()); // little-endian magic
6724        wal[4..8].copy_from_slice(&3_007_000u32.to_be_bytes());
6725        wal[8..12].copy_from_slice(&page_size.to_be_bytes());
6726        wal[12..16].copy_from_slice(&1u32.to_be_bytes());
6727        wal[16..20].copy_from_slice(&salt1.to_be_bytes());
6728        wal[20..24].copy_from_slice(&salt2.to_be_bytes());
6729        // Header checksum over the first 24 bytes (the seed for the frame chain).
6730        let (mut s0, mut s1) = wal_checksum(endian, 0, 0, &wal[0..24]);
6731        wal[24..28].copy_from_slice(&s0.to_be_bytes());
6732        wal[28..32].copy_from_slice(&s1.to_be_bytes());
6733
6734        for i in 0..n_pages {
6735            let page_no = (i + 1) as u32;
6736            let db_size = if i + 1 == n_pages { n_pages as u32 } else { 0 };
6737            let mut fh = [0u8; 24];
6738            fh[0..4].copy_from_slice(&page_no.to_be_bytes());
6739            fh[4..8].copy_from_slice(&db_size.to_be_bytes());
6740            fh[8..12].copy_from_slice(&salt1.to_be_bytes());
6741            fh[12..16].copy_from_slice(&salt2.to_be_bytes());
6742            let data = &main[i * ps..(i + 1) * ps];
6743            let (n0, n1) = wal_checksum(endian, s0, s1, &fh[0..8]);
6744            let (n0, n1) = wal_checksum(endian, n0, n1, data);
6745            s0 = n0;
6746            s1 = n1;
6747            fh[16..20].copy_from_slice(&s0.to_be_bytes());
6748            fh[20..24].copy_from_slice(&s1.to_be_bytes());
6749            wal.extend_from_slice(&fh);
6750            wal.extend_from_slice(data);
6751        }
6752        wal
6753    }
6754
6755    #[test]
6756    fn snapshot_tables_reads_schema_from_its_own_page_one() {
6757        use crate::rebuild::{build_recovered_db_tables, RecoveredTable as RT};
6758        let seed = vec![RT {
6759            name: "people".to_string(),
6760            columns: vec!["id".to_string(), "name".to_string()],
6761            rows: vec![
6762                vec![Value::Integer(1), Value::Text("alice".into())],
6763                vec![Value::Integer(2), Value::Text("bob".into())],
6764            ],
6765        }];
6766        let main = build_recovered_db_tables(&seed);
6767        let ps = parse_header(&main).unwrap().page_size;
6768        let wal = wrap_db_in_wal(&main, ps);
6769
6770        let db = Database::open_with_wal(main, &wal).unwrap();
6771        let tl = db.wal_timeline().unwrap();
6772        let snap = tl.commit_snapshots().last().unwrap();
6773        assert!(snap.checksum_valid(), "minted WAL must be checksum-valid");
6774
6775        let tables = snap.tables();
6776        let people = tables
6777            .iter()
6778            .find(|t| t.name == "people")
6779            .expect("table 'people' present in snapshot schema");
6780        assert!(people.rootpage >= 2, "rootpage points past page 1");
6781        assert_eq!(people.columns, vec!["id".to_string(), "name".to_string()]);
6782        assert!(!people.without_rowid, "an ordinary rowid table");
6783        // Internal sqlite_* tables are excluded.
6784        assert!(tables.iter().all(|t| !t.name.starts_with("sqlite_")));
6785    }
6786
6787    #[test]
6788    fn snapshot_read_resolves_overflow_through_snapshot_pages_not_live_view() {
6789        // The DEFINING property of the snapshot-scoped read: a spilled (overflow)
6790        // row must decode from the snapshot's OWN pages, even when the live view
6791        // would supply different overflow content. Build a db whose table `t` holds
6792        // one large-blob row (forcing an overflow chain), capture it as the
6793        // snapshot, then CLOBBER the overflow pages in the live main-file image.
6794        // The snapshot read still returns the original blob; a live read sees the
6795        // clobbered bytes — proving the snapshot path does not consult the live view.
6796        use crate::rebuild::{build_recovered_db_tables, RecoveredTable as RT};
6797        let blob: Vec<u8> = (0..9000u32).map(|i| (i % 251) as u8).collect();
6798        let seed = vec![RT {
6799            name: "t".to_string(),
6800            columns: vec!["id".to_string(), "big".to_string()],
6801            rows: vec![vec![Value::Integer(1), Value::Blob(blob.clone())]],
6802        }];
6803        let minted = build_recovered_db_tables(&seed);
6804        let ps = parse_header(&minted).unwrap().page_size;
6805        // The WAL commits the TRUE pages; the snapshot materializes them.
6806        let wal = wrap_db_in_wal(&minted, ps);
6807
6808        // Now clobber the live main image's overflow pages (every page after the
6809        // first two: page 1 schema, page 2 table-leaf, page 3+ overflow) to a
6810        // distinct byte so a live read would mis-decode the blob.
6811        let mut clobbered_main = minted.clone();
6812        for p in clobbered_main.iter_mut().skip(2 * ps as usize) {
6813            *p = 0xEE;
6814        }
6815
6816        let db = Database::open_with_wal(clobbered_main, &wal).unwrap();
6817        let tl = db.wal_timeline().unwrap();
6818        let snap = tl.commit_snapshots().last().unwrap();
6819        let t = snap
6820            .tables()
6821            .into_iter()
6822            .find(|t| t.name == "t")
6823            .expect("table t in snapshot");
6824
6825        let rows = snap.read_table(t.rootpage, t.columns.len()).unwrap();
6826        assert_eq!(rows.len(), 1, "one row at this commit");
6827        let (rowid, values) = &rows[0];
6828        assert_eq!(*rowid, 1);
6829        // The 9000-byte blob reassembles from the SNAPSHOT's overflow pages, intact.
6830        assert_eq!(
6831            values.get(1),
6832            Some(&Value::Blob(blob)),
6833            "overflow blob must reassemble from the snapshot's pages, not the clobbered live view"
6834        );
6835    }
6836
6837    #[test]
6838    fn snapshot_read_walks_interior_btree_in_rowid_order() {
6839        // Many rows force an interior (0x05) table b-tree; the snapshot read must
6840        // descend it and return rows in ascending rowid order — exercising the
6841        // shared walk's interior branch through the snapshot page source.
6842        use crate::rebuild::{build_recovered_db_tables, RecoveredTable as RT};
6843        let rows_seed: Vec<Vec<Value>> = (1..=500i64)
6844            .map(|i| vec![Value::Integer(i), Value::Text(format!("name-{i}"))])
6845            .collect();
6846        let seed = vec![RT {
6847            name: "big".to_string(),
6848            columns: vec!["id".to_string(), "name".to_string()],
6849            rows: rows_seed,
6850        }];
6851        let minted = build_recovered_db_tables(&seed);
6852        let ps = parse_header(&minted).unwrap().page_size;
6853        let wal = wrap_db_in_wal(&minted, ps);
6854
6855        let db = Database::open_with_wal(minted, &wal).unwrap();
6856        let tl = db.wal_timeline().unwrap();
6857        let snap = tl.commit_snapshots().last().unwrap();
6858        let t = snap
6859            .tables()
6860            .into_iter()
6861            .find(|t| t.name == "big")
6862            .expect("table big");
6863        let rows = snap.read_table(t.rootpage, t.columns.len()).unwrap();
6864        assert_eq!(rows.len(), 500, "all rows across the interior b-tree");
6865        let ids: Vec<i64> = rows.iter().map(|(r, _)| *r).collect();
6866        assert!(ids.windows(2).all(|w| w[0] < w[1]), "ascending rowid order");
6867        assert_eq!(*ids.first().unwrap(), 1);
6868        assert_eq!(*ids.last().unwrap(), 500);
6869    }
6870
6871    #[test]
6872    fn without_rowid_sql_detects_the_clause() {
6873        // The WITHOUT ROWID detector keys off the CREATE TABLE tail, tolerant of
6874        // case and whitespace, and does NOT misfire on the literal appearing inside
6875        // a quoted string / column name (file-format §2.4). A WITHOUT ROWID b-tree
6876        // has no rowid key, so this flag gates the snapshot-scoped rowid read.
6877        assert!(without_rowid_sql(
6878            "CREATE TABLE kv(k TEXT PRIMARY KEY, v TEXT) WITHOUT ROWID"
6879        ));
6880        assert!(without_rowid_sql(
6881            "CREATE TABLE kv(k TEXT PRIMARY KEY, v TEXT)  without   rowid"
6882        ));
6883        // Ordinary tables are NOT flagged.
6884        assert!(!without_rowid_sql(
6885            "CREATE TABLE t(id INTEGER PRIMARY KEY, n TEXT)"
6886        ));
6887        // A column literally named with the words, but not the trailing clause, is
6888        // not a false positive.
6889        assert!(!without_rowid_sql(
6890            "CREATE TABLE t(\"without rowid\" TEXT, x INT)"
6891        ));
6892    }
6893
6894    #[test]
6895    fn is_autoincrement_detects_only_the_real_clause() {
6896        // Positive: an ordinary rowid table declaring INTEGER PRIMARY KEY
6897        // AUTOINCREMENT — case-insensitive and whitespace-tolerant.
6898        assert!(is_autoincrement(
6899            "CREATE TABLE students(id INTEGER PRIMARY KEY AUTOINCREMENT, name TEXT)"
6900        ));
6901        assert!(is_autoincrement(
6902            "create table t(  id   integer   primary key   autoincrement )"
6903        ));
6904        // Negative: a plain INTEGER PRIMARY KEY is NOT autoincrement.
6905        assert!(!is_autoincrement(
6906            "CREATE TABLE students(id INTEGER PRIMARY KEY, name TEXT)"
6907        ));
6908        // Negative: a WITHOUT ROWID table cannot be AUTOINCREMENT (no rowid).
6909        assert!(!is_autoincrement(
6910            "CREATE TABLE kv(k INTEGER PRIMARY KEY AUTOINCREMENT, v TEXT) WITHOUT ROWID"
6911        ));
6912        // Negative: a column merely NAMED autoincrement is not the clause.
6913        assert!(!is_autoincrement(
6914            "CREATE TABLE t(\"autoincrement\" INTEGER PRIMARY KEY, x INT)"
6915        ));
6916        // Negative: the keyword inside a quoted string / comment does not qualify.
6917        assert!(!is_autoincrement(
6918            "CREATE TABLE t(id INTEGER PRIMARY KEY, note TEXT DEFAULT 'autoincrement')"
6919        ));
6920        // Negative: AUTOINCREMENT without INTEGER PRIMARY KEY is not a valid clause.
6921        assert!(!is_autoincrement(
6922            "CREATE TABLE t(id INTEGER AUTOINCREMENT, name TEXT)"
6923        ));
6924    }
6925
6926    #[test]
6927    fn sqlite_sequence_reads_present_absent_and_multi() {
6928        // A db with no AUTOINCREMENT table has no sqlite_sequence: empty map
6929        // (NOT seq=0), so callers never invent a high-water mark.
6930        let plain = Database::open(crate::rebuild::build_recovered_db_tables(&[
6931            crate::rebuild::RecoveredTable {
6932                name: "plain".to_string(),
6933                columns: vec!["c0".to_string()],
6934                rows: vec![vec![Value::Integer(1)]],
6935            },
6936        ]))
6937        .expect("minted db opens");
6938        assert!(
6939            plain.sqlite_sequence().is_empty(),
6940            "no AUTOINCREMENT table ⟹ empty sqlite_sequence map"
6941        );
6942
6943        // The b_autoinc fixture maintains sqlite_sequence(students)=5.
6944        let auto =
6945            Database::open(include_bytes!("../../tests/data/drop_recreate/b_autoinc.db").to_vec())
6946                .expect("open b_autoinc.db");
6947        let seq = auto.sqlite_sequence();
6948        assert_eq!(seq.get("students"), Some(&5), "students high-water = 5");
6949
6950        // The upd_autoinc fixture: a single AUTOINCREMENT table t at seq=5.
6951        let upd = Database::open(
6952            include_bytes!("../../tests/data/drop_recreate/upd_autoinc.db").to_vec(),
6953        )
6954        .expect("open upd_autoinc.db");
6955        assert_eq!(upd.sqlite_sequence().get("t"), Some(&5), "t high-water = 5");
6956    }
6957
6958    #[test]
6959    fn schema_sql_reads_current_name_to_create_sql() {
6960        // The live `name -> CREATE SQL` map mirrors live_tables, keyed by name.
6961        let auto =
6962            Database::open(include_bytes!("../../tests/data/drop_recreate/b_autoinc.db").to_vec())
6963                .expect("open b_autoinc.db");
6964        let schema = auto.schema_sql();
6965        let sql = schema.get("students").expect("students present");
6966        assert!(
6967            sql.contains("AUTOINCREMENT"),
6968            "current CREATE SQL carried verbatim: {sql}"
6969        );
6970    }
6971
6972    #[test]
6973    fn prior_snapshot_schema_sql_reads_prior_create_sql() {
6974        // b_journal_altered: the prior (-journal) schema for `students` has NO
6975        // `extra` column, the current schema does → the CREATE SQL texts differ.
6976        let main = include_bytes!("../../tests/data/drop_recreate/b_journal_altered.db").to_vec();
6977        let journal = include_bytes!("../../tests/data/drop_recreate/b_journal_altered.db-journal");
6978        let db = Database::open(main).expect("open b_journal_altered.db");
6979        let prior = db
6980            .rollback_prior(journal)
6981            .expect("rollback_prior parses the PERSIST journal");
6982        let prior_sql = prior.schema_sql();
6983        let prior_students = prior_sql.get("students").expect("prior students present");
6984        assert!(
6985            !prior_students.contains("extra"),
6986            "prior CREATE SQL lacks the ALTER-added column: {prior_students}"
6987        );
6988        let current = db.schema_sql();
6989        assert_ne!(
6990            current.get("students"),
6991            prior_sql.get("students"),
6992            "prior vs current CREATE SQL differ (the ALTER)"
6993        );
6994    }
6995
6996    #[test]
6997    fn prior_snapshot_schema_sql_dml_only_matches_current() {
6998        // b_journal_dml: the last transaction is DML only, so the prior (-journal)
6999        // CREATE SQL for `students` EQUALS the current schema (anti-FP ground truth).
7000        let main = include_bytes!("../../tests/data/drop_recreate/b_journal_dml.db").to_vec();
7001        let journal = include_bytes!("../../tests/data/drop_recreate/b_journal_dml.db-journal");
7002        let db = Database::open(main).expect("open b_journal_dml.db");
7003        let prior = db
7004            .rollback_prior(journal)
7005            .expect("rollback_prior parses the PERSIST journal");
7006        assert_eq!(
7007            db.schema_sql().get("students"),
7008            prior.schema_sql().get("students"),
7009            "DML-only ⟹ prior and current CREATE SQL are identical"
7010        );
7011    }
7012}