sqlite_core/lib.rs
1//! `sqlite-core` — native, read-only, panic-free `SQLite` file-format reader.
2//!
3//! Parses the 100-byte file header (magic + page size), walks table b-trees
4//! (interior + leaf) yielding rows as typed [`Value`]s, reassembles
5//! overflow-page chains for large payloads, walks the freelist
6//! ([`Database::freelist_pages`]), and applies a read-only `-wal` overlay
7//! ([`Database::open_with_wal`]) — all bounds-checked and panic-free on crafted
8//! input. [`Database::carve_cells`] recognizes record-shaped cells in
9//! free/unallocated space for the analyzer's deleted-record recovery. The bespoke
10//! [`WalTimeline`] ([`Database::wal_timeline`]) models a `-wal` as a salt-bounded
11//! segment of materializable [`CommitSnapshot`]s for "carve all snapshots".
12//!
13//! Format constants are consumed from [`forensicnomicon::sqlite`] (the KNOWLEDGE
14//! leaf), including the page-1 header field offsets (reserved-space 20, in-header
15//! DB-size 28, freelist-count 36, text-encoding 56) promoted there in §3.1.
16//! Index-b-tree LEAF reading is a foundation
17//! ([`Database::index_leaf_cells`], roadmap §1.4) — the second substrate for a
18//! table's data and the storage of `WITHOUT ROWID` rows; carving DELETED index
19//! entries and following index-key overflow remain follow-ups.
20//! (UTF-16 text decoding and WAL frame-checksum verification are implemented.)
21
22#![cfg_attr(test, allow(clippy::unwrap_used, clippy::expect_used))]
23
24pub mod attribution;
25pub mod rebuild;
26pub mod row_history;
27pub mod sqlcipher;
28
29// The page-1 header field offsets are consumed from the KNOWLEDGE leaf
30// (forensicnomicon::sqlite ≥ 1.5.0); the previously-local duplicates were promoted
31// there (roadmap §3.1). Aliased to the historical local names so every use site is
32// unchanged and the names read naturally in context.
33use forensicnomicon::sqlite::{
34 SQLITE_DB_SIZE_OFFSET as DB_SIZE_IN_PAGES_OFFSET,
35 SQLITE_FREELIST_COUNT_OFFSET as FREELIST_COUNT_OFFSET, SQLITE_FREELIST_TRUNK_OFFSET,
36 SQLITE_HEADER_SIZE, SQLITE_MAGIC, SQLITE_PAGE_SIZE_OFFSET,
37 SQLITE_RESERVED_SPACE_OFFSET as RESERVED_SPACE_OFFSET,
38 SQLITE_TEXT_ENCODING_OFFSET as TEXT_ENCODING_OFFSET,
39};
40
41/// Errors that can arise while reading a `SQLite` database, all recoverable —
42/// the reader never panics on malformed input.
43#[derive(Debug, Clone, PartialEq, Eq)]
44pub enum Error {
45 /// File is shorter than the 100-byte header.
46 TooShort,
47 /// First 16 bytes are not the `SQLite format 3\0` magic.
48 BadMagic,
49 /// Page-size field is not a power of two in `[512, 65536]`.
50 BadPageSize(u32),
51 /// A page number referenced by the b-tree is out of range for the file.
52 PageOutOfRange(u32),
53 /// A b-tree page had an unexpected type byte where a table page was required.
54 NotATablePage(u8),
55 /// A cell pointer or payload ran past the end of its page.
56 TruncatedCell,
57 /// The b-tree was deeper / wider than the safety cap allows.
58 TooManyPages,
59 /// The freelist trunk chain cycled or exceeded the file's page count.
60 MalformedFreelist,
61 /// An overflow-page chain cycled or exceeded the file's page count.
62 MalformedOverflow,
63 /// A rollback-journal page size was not a power of two in `[512, 65536]`.
64 /// Carries the offending value (Show-the-unrecognized-value).
65 BadJournalPageSize(u32),
66 /// A rollback journal was applied to a database opened WAL-applied, or whose
67 /// page size disagrees with the journal's. WAL and rollback-journal modes are
68 /// mutually exclusive timelines and must not be overlaid.
69 JournalModeConflict,
70 /// The file could not be opened or read (an I/O failure via
71 /// [`Database::open_path`], not a malformed database). Carries the
72 /// [`std::io::ErrorKind`] (show-the-unrecognized-value).
73 Io(std::io::ErrorKind),
74 /// `SQLCipher` decryption failed (wrong key, unsupported cipher parameters, or
75 /// a failed page authentication) via [`Database::open_encrypted`]. Carries
76 /// the underlying [`sqlcipher::DecryptError`] (show-the-unrecognized-value).
77 Decrypt(sqlcipher::DecryptError),
78}
79
80impl From<std::io::Error> for Error {
81 fn from(e: std::io::Error) -> Self {
82 Error::Io(e.kind())
83 }
84}
85
86impl From<sqlcipher::DecryptError> for Error {
87 fn from(e: sqlcipher::DecryptError) -> Self {
88 Error::Decrypt(e)
89 }
90}
91
92/// A freed overflow-page chain could not be followed to a complete, trustworthy
93/// payload (task #73): a chain page that is not a freelist leaf (live / trunk /
94/// unreachable), a cycle, a premature terminator with bytes still owed, an
95/// out-of-range page, or a declared payload exceeding the freelist's capacity.
96/// Carries no detail by design — any break is a uniform "this chain is not
97/// recoverable as a Tier-1 row", and the candidate degrades to a Tier-2 fragment.
98#[derive(Debug, Clone, Copy, PartialEq, Eq)]
99pub struct ChainBreak;
100
101/// A single decoded column value from a table row. Mirrors `SQLite`'s storage
102/// classes.
103#[derive(Debug, Clone, PartialEq)]
104pub enum Value {
105 Null,
106 Integer(i64),
107 Real(f64),
108 Text(String),
109 Blob(Vec<u8>),
110}
111
112/// One table row: its rowid plus decoded column values, in column order.
113#[derive(Debug, Clone, PartialEq)]
114pub struct Row {
115 pub rowid: i64,
116 pub values: Vec<Value>,
117}
118
119/// A live user table dumped for export: its name, the column header to present,
120/// and every live row in rowid order. Produced by [`Database::live_table_rows`].
121///
122/// `column_names` are the table's **real** column names parsed from its
123/// `CREATE TABLE` when available, falling back to generic `c0..c{N-1}` (sized to
124/// the widest row) when the schema parse was low-confidence — so a header is
125/// always present and never a fabricated guess. `rows` preserves b-tree order,
126/// which for an integer-rowid table is ascending rowid order.
127#[derive(Debug, Clone, PartialEq)]
128pub struct LiveTableDump {
129 /// Table name from `sqlite_master.name`.
130 pub name: String,
131 /// Header column names: real names from the schema, or `c0..c{N-1}`.
132 pub column_names: Vec<String>,
133 /// Every live row (rowid + decoded values), in b-tree (rowid) order.
134 pub rows: Vec<Row>,
135}
136
137/// A `WITHOUT ROWID` user table's live rows, produced by
138/// [`Database::without_rowid_table_rows`]. Such a table's data lives entirely in
139/// an index b-tree (there is no rowid), so `rows` holds the decoded index records
140/// in the table's declared column order, in index (primary-key) order.
141#[derive(Debug, Clone, PartialEq)]
142pub struct WithoutRowidTable {
143 /// Table name from `sqlite_master.name`.
144 pub name: String,
145 /// Every live row's decoded column values, in the table's column order.
146 pub rows: Vec<Vec<Value>>,
147}
148
149/// A record-shaped cell recovered from unallocated / free space by
150/// [`Database::carve_cells`]. Carries the decoded row plus enough provenance for
151/// the analyzer to grade it as a "consistent with a deleted row" observation.
152#[derive(Debug, Clone, PartialEq)]
153pub struct CarvedCell {
154 /// Byte offset of the cell within the page slice that was scanned.
155 pub offset: usize,
156 /// Total bytes the candidate cell occupies (cell header + payload), so the
157 /// scanner can skip past a recovered record.
158 pub byte_len: usize,
159 /// Decoded rowid varint.
160 pub rowid: i64,
161 /// Decoded column values, in column order.
162 pub values: Vec<Value>,
163 /// Heuristic confidence in `(0.0, 1.0]` that these bytes are a real record
164 /// rather than a coincidental match.
165 pub confidence: f32,
166}
167
168/// A **partial** deleted record salvaged from a freed-cell reconstruction that
169/// failed full-row validation: the maximal decodable column prefix at a
170/// structural anchor [`Database::reconstruct_freeblock_records`] already trusts.
171///
172/// Deliberately NOT a [`CarvedCell`]: it has no rowid (clobbered) and an
173/// incomplete value set, so the type system keeps it out of the full-row output
174/// — a fragment can never be silently rendered as a recovered row. Emitted only
175/// at an anchor where full reconstruction failed but at least one *distinctive*
176/// cell (TEXT ≥ 4 bytes of valid UTF-8, or REAL) decoded cleanly, so a lone
177/// coincidental integer pattern never anchors a fragment. Graded
178/// `FRAGMENT_CONFIDENCE` — strictly below every full-row class.
179#[derive(Debug, Clone, PartialEq)]
180pub struct CellFragment {
181 /// Byte offset of the failed cell's anchor within the scanned page slice.
182 pub offset: usize,
183 /// Bytes covered by the decoded prefix (anchor to the last decoded body byte).
184 pub byte_len: usize,
185 /// `(column_index, value)` for each column that decoded cleanly, ascending by
186 /// index. Column indexes come from the page's schema template, so they are
187 /// meaningful against the table's column order.
188 pub surviving: Vec<(usize, Value)>,
189 /// Number of the template's columns that did NOT decode (`column_count` minus
190 /// the number of surviving columns).
191 pub missing: usize,
192 /// Always `FRAGMENT_CONFIDENCE` for now; the field is kept so future
193 /// per-fragment grading does not change the public type.
194 pub confidence: f32,
195}
196
197/// A freed table-leaf cell whose declared payload **spills onto an overflow-page
198/// chain** (task #73). Recognized by `try_carve_spilled_cell_at` from the
199/// cell's intact local prefix; the chain itself is resolved separately
200/// ([`Database::read_freed_overflow_chain`]) because that needs whole-database
201/// access. A `SpilledCell` is deliberately NOT a [`CarvedCell`]: until its chain
202/// is walked and validated it cannot masquerade as a recovered row (secure by
203/// design — the type system keeps an unresolved spill out of the full-row output).
204#[derive(Debug, Clone, PartialEq)]
205pub struct SpilledCell {
206 /// Byte offset of the cell within the scanned slice.
207 pub offset: usize,
208 /// On-page footprint of the cell prefix: `n1 + n2 + local_len + 4`.
209 pub byte_len: usize,
210 /// Declared total payload length `P` (header + full body).
211 pub payload_len: usize,
212 /// Decoded rowid varint (intact-prefix anchors); `0` when the prefix was
213 /// clobbered and the rowid is unrecoverable (template path).
214 pub rowid: i64,
215 /// Full serial-type array, decoded from the local record header.
216 pub serials: Vec<i64>,
217 /// Local payload bytes kept on the leaf page (`local_payload_len(P, usable)`).
218 pub local_len: usize,
219 /// Offset, within the scanned slice, at which the local payload begins.
220 pub local_payload_off: usize,
221 /// First overflow-page number (big-endian u32 at `local_payload_off + local_len`).
222 pub first_overflow: u32,
223}
224
225/// Database text encoding (file-format §1.3, header byte 56). Determines how
226/// `TEXT` column bytes are decoded; a fixed property set at database creation.
227#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
228pub enum TextEncoding {
229 /// `1` (and `0`, an unwritten database): UTF-8.
230 #[default]
231 Utf8,
232 /// `2`: UTF-16 little-endian.
233 Utf16Le,
234 /// `3`: UTF-16 big-endian.
235 Utf16Be,
236}
237
238impl TextEncoding {
239 /// Decode a `TEXT` value's raw bytes per this encoding. Lossy so a corrupt
240 /// byte sequence yields U+FFFD rather than a panic or an error.
241 fn decode(self, bytes: &[u8]) -> String {
242 match self {
243 Self::Utf8 => String::from_utf8_lossy(bytes).into_owned(),
244 Self::Utf16Le => Self::decode_utf16(bytes, u16::from_le_bytes),
245 Self::Utf16Be => Self::decode_utf16(bytes, u16::from_be_bytes),
246 }
247 }
248
249 fn decode_utf16(bytes: &[u8], conv: fn([u8; 2]) -> u16) -> String {
250 // The DB-encoding path keeps its lossy-by-default contract: it discards
251 // the flag, so a truncated or corrupt unit still yields U+FFFD as before.
252 // The pairing itself lives in `decode_utf16_units` (DRY — the Local
253 // Storage decode reuses it and keeps the flag).
254 decode_utf16_units(bytes, conv).0
255 }
256}
257
258/// Shared UTF-16 → `String` pairing: pairs 2-byte code units via `conv`, resolves
259/// surrogate pairs, and reports whether the decode was **lossy**. A trailing odd
260/// byte (half a code unit) or an unpaired surrogate emits U+FFFD and sets the
261/// flag; it never panics or errors. Endianness is the caller's via `conv`.
262fn decode_utf16_units(bytes: &[u8], conv: fn([u8; 2]) -> u16) -> (String, bool) {
263 // An odd trailing byte is half a code unit — real data was truncated. It is
264 // dropped by `chunks_exact`; the flag records that a byte was lost.
265 let mut lossy = bytes.len() % 2 != 0;
266 let units = bytes.chunks_exact(2).map(|c| conv([c[0], c[1]]));
267 let mut text = String::new();
268 for unit in char::decode_utf16(units) {
269 if let Ok(c) = unit {
270 text.push(c);
271 } else {
272 lossy = true;
273 text.push(char::REPLACEMENT_CHARACTER);
274 }
275 }
276 (text, lossy)
277}
278
279/// A WebKit/Chrome Local Storage `ItemTable.value` decoded to text, plus whether
280/// the decode was lossy. `lossy` is a struct field, not a side-channel warning,
281/// so a caller cannot render a lossy value as if it were faithfully recovered
282/// (secure by design).
283#[derive(Debug, Clone, PartialEq, Eq, Default)]
284pub struct LocalStorageValue {
285 /// The decoded string; any code unit that could not be decoded is a U+FFFD.
286 pub text: String,
287 /// `true` when at least one input byte/unit could not be decoded cleanly (an
288 /// odd-length BLOB or an unpaired surrogate).
289 pub lossy: bool,
290}
291
292/// Decode a WebKit/Chromium Local Storage `ItemTable.value` BLOB to a `String`.
293///
294/// A `.localstorage` file is a standard `SQLite` database this crate already
295/// reads; the one artifact-specific quirk is that the `value` column is a BLOB
296/// holding the string as raw **UTF-16 little-endian** code units — no BOM, no
297/// type-prefix byte — so a normal dump surfaces it as opaque hex. This turns
298/// such a BLOB back into readable text.
299///
300/// Panic-free and lossy-by-report: an odd-length BLOB (a trailing half code
301/// unit) or an unpaired surrogate yields U+FFFD and sets
302/// [`LocalStorageValue::lossy`] rather than erroring or panicking. An empty BLOB
303/// decodes to the empty string with `lossy == false`.
304#[must_use]
305pub fn decode_localstorage_value(blob: &[u8]) -> LocalStorageValue {
306 let (text, lossy) = decode_utf16_units(blob, u16::from_le_bytes);
307 LocalStorageValue { text, lossy }
308}
309
310/// Recognize the WebKit/Chromium Local Storage `ItemTable(key TEXT, value BLOB)`
311/// table, so a caller knows when [`decode_localstorage_value`] applies to a
312/// dumped table's `value` column.
313///
314/// Keyed on the distinctive table name `ItemTable` — the name WebKit/Chromium
315/// create for Local Storage. The column names are deliberately NOT part of the
316/// test: the real schema declares them with `ON CONFLICT` clauses
317/// (`key TEXT UNIQUE ON CONFLICT REPLACE, value BLOB NOT NULL ON CONFLICT FAIL`)
318/// that a lightweight `CREATE TABLE` parse does not always split cleanly, so a
319/// name match is the robust signal. The row shape (a TEXT key, a BLOB value)
320/// still surfaces positionally in each [`Row`].
321#[must_use]
322pub fn is_local_storage_item_table(table_name: &str) -> bool {
323 table_name == "ItemTable"
324}
325
326/// Parsed 100-byte `SQLite` file header.
327#[derive(Debug, Clone, Copy, PartialEq, Eq)]
328pub struct Header {
329 /// Logical page size in bytes (512..=65536).
330 pub page_size: u32,
331 /// Reserved bytes at the end of each page (usually 0).
332 pub reserved: u8,
333 /// Text encoding for `TEXT` columns (header byte 56).
334 pub text_encoding: TextEncoding,
335}
336
337impl Header {
338 /// Usable bytes per page = `page_size` − reserved (file-format §1.3.4).
339 #[must_use]
340 pub fn usable_size(self) -> u32 {
341 self.page_size.saturating_sub(u32::from(self.reserved))
342 }
343}
344
345/// A read-only view over the raw bytes of a `SQLite` database file.
346///
347/// Holds the whole file in memory — adequate for the spike and for browser
348/// evidence DBs (tens of MB). A `Read + Seek` / mmap backend is a later
349/// refinement and does not change the parsing logic proven here.
350pub struct Database {
351 /// Page byte source: the whole file in memory ([`Database::open`]) or a
352 /// paged, LRU-cached file reader ([`Database::open_path`], roadmap §3.1).
353 source: ByteSource,
354 /// The 100-byte file header, kept resident so fixed-offset header-field reads
355 /// (page count, freelist count/trunk) never touch the byte source.
356 head: Box<[u8]>,
357 header: Header,
358 /// Read-only WAL overlay: newest committed page versions from a `-wal`
359 /// sidecar, applied without checkpointing (never mutates the main file).
360 /// `None` when opened without a WAL.
361 wal: Option<WalOverlay>,
362}
363
364/// A page image handed back by the byte source: a slice borrowed from an
365/// in-memory buffer, or a reference-counted page from the paged LRU cache.
366/// Derefs to `[u8]` so callers treat it as a page slice regardless of origin.
367///
368/// A page-*handle* rather than a `with_page(|bytes| …)` closure because the walk
369/// uses `&dyn PageSource` (a generic closure method would make that trait
370/// non-object-safe) and the recursive b-tree descent cannot hold a pinning
371/// closure across its own recursion. The `Shared` variant keeps a cached page
372/// alive while held, so LRU eviction can never dangle it.
373pub enum PageBytes<'a> {
374 /// Borrowed from an in-memory buffer (the `open` / WAL-overlay path).
375 Borrowed(&'a [u8]),
376 /// Shared out of the paged LRU cache (the `open_path` path).
377 Shared(std::rc::Rc<[u8]>),
378}
379
380impl std::ops::Deref for PageBytes<'_> {
381 type Target = [u8];
382 fn deref(&self) -> &[u8] {
383 match self {
384 PageBytes::Borrowed(s) => s,
385 PageBytes::Shared(r) => r,
386 }
387 }
388}
389
390/// Where a [`Database`]'s page bytes come from.
391enum ByteSource {
392 /// The whole file resident in memory.
393 Mem(Vec<u8>),
394 /// A file read page-by-page through a bounded LRU cache.
395 Paged(Paged),
396}
397
398impl ByteSource {
399 /// Total byte length of the underlying file.
400 fn len(&self) -> usize {
401 match self {
402 ByteSource::Mem(b) => b.len(),
403 ByteSource::Paged(p) => p.len,
404 }
405 }
406
407 /// The 1-based `page`'s bytes, or `None` for page 0 / out of range / an I/O
408 /// error. Bounded and panic-free.
409 fn page(&self, page: u32, page_size: usize) -> Option<PageBytes<'_>> {
410 let start = (page as usize).checked_sub(1)?.checked_mul(page_size)?;
411 let end = start.checked_add(page_size)?;
412 match self {
413 ByteSource::Mem(b) => b.get(start..end).map(PageBytes::Borrowed),
414 ByteSource::Paged(p) if end <= p.len => {
415 p.read_page(start, page_size).map(PageBytes::Shared)
416 }
417 ByteSource::Paged(_) => None,
418 }
419 }
420
421 /// The whole file as one slice when resident in memory; `None` for a paged
422 /// source (which never materializes the whole file). Used only on the
423 /// WAL-overlay path, which is in-memory by construction.
424 fn whole(&self) -> Option<&[u8]> {
425 match self {
426 ByteSource::Mem(b) => Some(b),
427 ByteSource::Paged(_) => None, // cov:unreachable: WAL overlay is in-memory only
428 }
429 }
430}
431
432/// A file read page-by-page through a small LRU cache, so resident memory stays
433/// bounded regardless of file size (roadmap §3.1).
434struct Paged {
435 file: std::cell::RefCell<std::fs::File>,
436 len: usize,
437 cache: std::cell::RefCell<PageCache>,
438}
439
440impl Paged {
441 /// Read `page_size` bytes at `start`, serving from and populating the LRU
442 /// cache. `None` on any I/O error (panic-free).
443 fn read_page(&self, start: usize, page_size: usize) -> Option<std::rc::Rc<[u8]>> {
444 use std::io::{Read, Seek, SeekFrom};
445 if let Some(hit) = self.cache.borrow_mut().get(start) {
446 return Some(hit);
447 }
448 let mut buf = vec![0u8; page_size];
449 {
450 let mut file = self.file.borrow_mut();
451 file.seek(SeekFrom::Start(start as u64)).ok()?;
452 file.read_exact(&mut buf).ok()?;
453 }
454 let rc: std::rc::Rc<[u8]> = std::rc::Rc::from(buf);
455 self.cache.borrow_mut().put(start, std::rc::Rc::clone(&rc));
456 Some(rc)
457 }
458}
459
460/// A tiny bounded LRU of page images keyed by file offset, capping resident
461/// memory to [`PageCache::CAP`] pages so a multi-GB database never loads whole.
462struct PageCache {
463 map: std::collections::HashMap<usize, std::rc::Rc<[u8]>>,
464 order: std::collections::VecDeque<usize>,
465}
466
467impl PageCache {
468 /// Maximum resident pages (`CAP` × `page_size` bytes; 256 pages is about one
469 /// megabyte at a 4-kilobyte page), so a multi-gigabyte database never loads whole.
470 const CAP: usize = 256;
471
472 fn new() -> Self {
473 Self {
474 map: std::collections::HashMap::new(),
475 order: std::collections::VecDeque::new(),
476 }
477 }
478
479 fn get(&mut self, key: usize) -> Option<std::rc::Rc<[u8]>> {
480 let hit = self.map.get(&key).map(std::rc::Rc::clone)?;
481 self.touch(key);
482 Some(hit)
483 }
484
485 fn put(&mut self, key: usize, value: std::rc::Rc<[u8]>) {
486 if self.map.insert(key, value).is_some() {
487 self.touch(key);
488 } else {
489 self.order.push_back(key);
490 if self.order.len() > Self::CAP {
491 if let Some(evicted) = self.order.pop_front() {
492 self.map.remove(&evicted);
493 }
494 }
495 }
496 }
497
498 fn touch(&mut self, key: usize) {
499 if let Some(pos) = self.order.iter().position(|&k| k == key) {
500 self.order.remove(pos);
501 self.order.push_back(key);
502 }
503 }
504}
505
506/// The newest committed version of each WAL page, materialized into owned bytes.
507///
508/// Built once at open; `page_slice` consults it before the main file so a table
509/// walk transparently sees the WAL-applied view. Read-only: building it copies
510/// frame data out of the `-wal` sidecar and never writes back to either file.
511struct WalOverlay {
512 /// page number (1-based) → that page's newest committed contents.
513 pages: std::collections::BTreeMap<u32, Vec<u8>>,
514 /// Every committed frame's page image, in file order, with provenance. Unlike
515 /// `pages` (newest version per page, the consistent view), this keeps EACH
516 /// committed frame so the carver can recover deleted residue that a later
517 /// frame for the same page superseded in `pages` but that still survives in an
518 /// earlier frame's slack — the genuinely-different records an on-disk-only
519 /// carve cannot see.
520 frames: Vec<WalFramePage>,
521 /// The original `-wal` sidecar bytes, retained so [`Database::wal_timeline`]
522 /// can re-parse them into the richer segmented temporal model without the
523 /// caller re-supplying the file. Held read-only; never mutated.
524 raw: Vec<u8>,
525}
526
527/// One committed WAL frame's full page image plus its provenance, exposed by
528/// [`Database::wal_frame_pages`] so the deleted-record carver can scan the
529/// uncheckpointed WAL frames the main file does not yet reflect.
530///
531/// The `(salt1, salt2, frame_index)` triple is the WAL log-sequence identity that
532/// task #55 will formalize: `salt1`/`salt2` pin the checkpoint generation and
533/// `frame_index` the position within it.
534#[derive(Debug, Clone, PartialEq, Eq)]
535pub struct WalFramePage {
536 /// 0-based position of this frame within the `-wal` file (its LSN ordinal).
537 pub frame_index: usize,
538 /// 1-based database page number this frame rewrites.
539 pub page_no: u32,
540 /// WAL header salt-1 (checkpoint generation), shared by every live frame.
541 pub salt1: u32,
542 /// WAL header salt-2 (checkpoint generation), shared by every live frame.
543 pub salt2: u32,
544 /// Whether this is a COMMIT frame (`db_size_after_commit != 0`).
545 pub is_commit: bool,
546 /// The frame's full page image (`page_size` bytes).
547 pub page: Vec<u8>,
548}
549
550/// Hard cap on b-tree pages visited in one table walk, to bound work on a
551/// crafted file with cyclic interior pointers.
552const MAX_PAGES_PER_WALK: usize = 1_000_000;
553
554/// Minimum column count accepted when **inferring** a record's width during
555/// dropped-table carving. A coincidental byte run can look like a self-consistent
556/// 1-column record far too easily; requiring at least two columns (the smallest a
557/// real rowid table with a non-rowid column has) suppresses that false-positive
558/// class without losing real records.
559const MIN_INFERRED_COLUMNS: usize = 2;
560
561/// Confidence multiplier applied to records carved from an allocated page's
562/// in-page free space. Such residue is more often partially overwritten (its
563/// freeblock may have been reused) than whole-page freelist recovery, so it is
564/// graded a notch lower even when it parses cleanly.
565const IN_PAGE_CONFIDENCE_FACTOR: f32 = 0.8;
566
567/// Confidence multiplier applied to a **chain-reassembled overflow** full row
568/// (task #73, [`Database::carve_overflow_records`]). Overflow Tier-1 is NOT part
569/// of the structural 0-false-positive guarantee (Codex ruling #1): a freelist
570/// *leaf* page can be stale — allocated, overwritten, freed, now a leaf holding
571/// unrelated bytes that happen to decode. The freelist-leaf requirement plus the
572/// strict-UTF-8 gate make a clean decode strong evidence, but one indirection
573/// weaker than a contiguous in-page span, so it is graded below the in-page
574/// full-row tier (0.9 × this factor). The residual stale-leaf risk is documented
575/// and the row remains a "consistent with a deleted row" observation, never a
576/// verdict.
577const OVERFLOW_CHAIN_CONFIDENCE_FACTOR: f32 = 0.75;
578
579/// Confidence assigned to a record rebuilt by **freeblock reconstruction**
580/// ([`Database::reconstruct_freeblock_records`]). The cell's first four bytes
581/// (payload-length + rowid varints, the record `header_len`, and the leading
582/// serial type) were destroyed by freeblock conversion, so the record is rebuilt
583/// from its surviving serial-type tail plus a schema-derived header template — a
584/// weaker reconstruction than an intact-header carve, hence graded LOW (a
585/// "consistent with a deleted row" lead the examiner weighs, never a certainty).
586const FREEBLOCK_RECONSTRUCT_CONFIDENCE: f32 = 0.4;
587
588/// Confidence assigned to a Tier-2 [`CellFragment`] — a partial recovery whose
589/// full row could not be reconstructed but at least one distinctive cell survived.
590/// Flat 0.2 = the `MinConfidence::Low` threshold, one notch below freeblock
591/// reconstruction's 0.4 (= Medium): a fragment is the weakest lead in the ladder,
592/// "consistent with a partial deleted row", never a recovered row.
593const FRAGMENT_CONFIDENCE: f32 = 0.2;
594
595/// Upper bound on the number of freeblocks walked on a single page, to cap work
596/// on a crafted file whose freeblock `next` pointers form a long or cyclic chain.
597/// Real pages hold at most a few hundred cells.
598const MAX_FREEBLOCKS_PER_PAGE: usize = 4096;
599
600/// WAL magic, big-endian variant (native byte order in the page checksums; the
601/// little-endian variant `0x377f_0683` differs only in checksum endianness,
602/// which the overlay does not verify). file-format §4.1.
603const WAL_MAGIC_BE: u32 = 0x377f_0682;
604/// WAL magic, little-endian-checksum variant.
605const WAL_MAGIC_LE: u32 = 0x377f_0683;
606
607/// Byte order in which the WAL checksum reads its 32-bit words (file-format
608/// §4.2). NOT the same as the constant names above: per the spec, magic
609/// `0x377f0683` selects **big-endian** words and `0x377f0682` **little-endian**
610/// words. (The legacy `WAL_MAGIC_*` constant names predate this checksum work
611/// and are used only as a "valid magic" set; this enum is the spec-faithful
612/// source of truth for checksum endianness.)
613#[derive(Debug, Clone, Copy, PartialEq, Eq)]
614enum WalChecksumEndian {
615 Big,
616 Little,
617}
618
619impl WalChecksumEndian {
620 /// The checksum word order selected by the WAL header magic (offset 0), or
621 /// `None` for a magic that is neither WAL variant (file-format §4.2).
622 fn from_magic(magic: u32) -> Option<Self> {
623 match magic {
624 0x377f_0683 => Some(Self::Big),
625 0x377f_0682 => Some(Self::Little),
626 _ => None,
627 }
628 }
629
630 /// Read one 32-bit word from `b` (exactly 4 bytes) in this endianness.
631 fn read_word(self, b: [u8; 4]) -> u32 {
632 match self {
633 Self::Big => u32::from_be_bytes(b),
634 Self::Little => u32::from_le_bytes(b),
635 }
636 }
637}
638
639/// Advance the cumulative WAL checksum `(s0, s1)` over `data` (file-format
640/// §4.2). `data` is interpreted as 32-bit words in the given endianness and
641/// consumed 8 bytes (two words) at a time via the Fibonacci-weighted recurrence
642/// `s0 += x[i] + s1; s1 += x[i+1] + s0;`
643/// using wrapping (u32) arithmetic. A trailing partial group (< 8 bytes) is
644/// ignored — the spec defines the checksum only over an even number of words,
645/// and every real WAL input (24-byte header prefix, 8-byte frame-header prefix,
646/// page data) is a multiple of 8 bytes.
647fn wal_checksum(endian: WalChecksumEndian, mut s0: u32, mut s1: u32, data: &[u8]) -> (u32, u32) {
648 let mut chunks = data.chunks_exact(8);
649 for c in &mut chunks {
650 let x0 = endian.read_word([c[0], c[1], c[2], c[3]]);
651 let x1 = endian.read_word([c[4], c[5], c[6], c[7]]);
652 s0 = s0.wrapping_add(x0).wrapping_add(s1);
653 s1 = s1.wrapping_add(x1).wrapping_add(s0);
654 }
655 (s0, s1)
656}
657
658impl Database {
659 /// Parse the file header and validate magic + page size. No WAL overlay.
660 pub fn open(bytes: Vec<u8>) -> Result<Self, Error> {
661 let header = parse_header(&bytes)?;
662 let head = header_prefix(&bytes);
663 Ok(Self {
664 source: ByteSource::Mem(bytes),
665 head,
666 header,
667 wal: None,
668 })
669 }
670
671 /// Decrypt a **`SQLCipher`** database with `key` and open the resulting
672 /// plaintext, detecting the cipher version automatically (see
673 /// [`sqlcipher::decrypt`]). The reader then consumes the decrypted byte
674 /// stream exactly as for a plaintext file — the encryption is transparent
675 /// past this call.
676 ///
677 /// Secure-by-default and read-only: a wrong key or unsupported cipher
678 /// parameters is a loud [`Error::Decrypt`], never a silently-misread
679 /// database; nothing is written back to the evidence file.
680 pub fn open_encrypted(bytes: &[u8], key: &sqlcipher::SqlCipherKey) -> Result<Self, Error> {
681 let decrypted = sqlcipher::decrypt(bytes, key)?;
682 Self::open(decrypted.plaintext)
683 }
684
685 /// Open a database from a filesystem path with a **bounded-memory paged
686 /// read** (roadmap §3.1): pages are streamed on demand through a small LRU
687 /// cache instead of loading the whole file into a `Vec<u8>`, so a multi-GB
688 /// database opens without proportional RAM. Main file only — for the
689 /// WAL-applied view use [`Database::open_with_wal`] (WAL sidecars are small
690 /// and stay in memory).
691 ///
692 /// Read-only and panic-free: an unreadable file or a malformed header is a
693 /// typed [`Error`] ([`Error::Io`] carries the [`std::io::ErrorKind`]); nothing
694 /// is written back.
695 pub fn open_path<P: AsRef<std::path::Path>>(path: P) -> Result<Self, Error> {
696 use std::io::{Read, Seek, SeekFrom};
697 let mut file = std::fs::File::open(path)?;
698 let len = file.metadata()?.len();
699 // Read just the header prefix to parse page size / encoding; the rest of
700 // the file is read page-by-page on demand.
701 let prefix_len = usize::try_from(len)
702 .unwrap_or(usize::MAX)
703 .min(SQLITE_HEADER_SIZE);
704 let mut head = vec![0u8; prefix_len];
705 file.seek(SeekFrom::Start(0))?;
706 file.read_exact(&mut head)?;
707 let header = parse_header(&head)?;
708 let source = ByteSource::Paged(Paged {
709 file: std::cell::RefCell::new(file),
710 len: usize::try_from(len).unwrap_or(usize::MAX),
711 cache: std::cell::RefCell::new(PageCache::new()),
712 });
713 Ok(Self {
714 source,
715 head: head.into(),
716 header,
717 wal: None,
718 })
719 }
720
721 /// Parse the main database plus a `-wal` sidecar, overlaying the newest
722 /// **committed** page versions from the WAL on top of the main file.
723 ///
724 /// This is the forensic-safe alternative to libsqlite checkpointing: neither
725 /// file is mutated. The resulting [`Database`] answers `read_table` with the
726 /// WAL-applied view (use [`Database::open`] for the main-only view). Frames
727 /// past the last commit frame, or whose salt does not match the WAL header,
728 /// are ignored — they are uncommitted / superseded and not part of the
729 /// consistent snapshot.
730 pub fn open_with_wal(bytes: Vec<u8>, wal: &[u8]) -> Result<Self, Error> {
731 let header = parse_header(&bytes)?;
732 let overlay = WalOverlay::parse(wal, header.page_size)?;
733 let head = header_prefix(&bytes);
734 Ok(Self {
735 source: ByteSource::Mem(bytes),
736 head,
737 header,
738 wal: overlay,
739 })
740 }
741
742 /// Materialize the single pre-transaction state from a rollback `-journal`,
743 /// binding it to THIS database (design §5). The journal's page images (the
744 /// bytes BEFORE the last transaction) are overlaid on the live pages, yielding
745 /// a [`PriorSnapshot`] — a DISTINCT read-only view, never a [`Database`], so a
746 /// prior/deleted row can never be read as "live" (secure-by-design).
747 ///
748 /// The main db's page size is authoritative (a PERSIST journal has a zeroed
749 /// header). **Errors with [`Error::JournalModeConflict`]** when `self` was
750 /// opened WAL-applied ([`Database::open_with_wal`]): WAL and rollback-journal
751 /// modes are mutually exclusive timelines and must not be overlaid.
752 ///
753 /// Robust and panic-free: a malformed/truncated journal yields a prior
754 /// snapshot with fewer overlaid pages (degrading toward the live image), never
755 /// a panic; a non-power-of-two page size is a typed
756 /// [`Error::BadJournalPageSize`].
757 pub fn rollback_prior(&self, journal: &[u8]) -> Result<PriorSnapshot, Error> {
758 if self.wal_applied() {
759 return Err(Error::JournalModeConflict);
760 }
761 let page_size = self.header.page_size;
762 let parsed = RollbackJournal::parse(journal, page_size)?;
763
764 // Start from the live main pages, then overlay the journal's prior images.
765 let main_pages = self.file_page_count();
766 let mut overlaid: std::collections::BTreeMap<u32, Vec<u8>> =
767 std::collections::BTreeMap::new();
768 for pgno in 1..=main_pages {
769 if let Some(slice) = self.raw_page(pgno) {
770 overlaid.insert(pgno, slice.to_vec());
771 }
772 }
773 let mut grew_db = false;
774 for img in parsed.page_images() {
775 if img.pgno > main_pages {
776 grew_db = true;
777 }
778 overlaid.insert(img.pgno, img.bytes.clone());
779 }
780
781 // Usable bytes per page from the PRIOR page-1 header (reserved byte @ 20),
782 // so a reserved-space change in the last txn is honored. Fall back to the
783 // live header when page 1 is not in the snapshot.
784 let reserved = overlaid
785 .get(&1)
786 .and_then(|p| p.get(RESERVED_SPACE_OFFSET).copied())
787 .unwrap_or(self.header.reserved);
788 let usable = page_size.saturating_sub(u32::from(reserved));
789 let page_bound = overlaid.keys().copied().next_back().unwrap_or(main_pages);
790
791 Ok(PriorSnapshot {
792 overlaid,
793 usable,
794 page_bound,
795 grew_db,
796 })
797 }
798
799 /// Whether a non-empty WAL overlay is in effect (at least one committed
800 /// frame was applied on top of the main file).
801 #[must_use]
802 pub fn wal_applied(&self) -> bool {
803 self.wal.as_ref().is_some_and(|w| !w.pages.is_empty())
804 }
805
806 /// Every committed `-wal` frame's page image, in file order, with provenance.
807 ///
808 /// Empty when the database was opened without a WAL (or the WAL held no
809 /// committed frames). The carver scans these page images for deleted-cell
810 /// residue that lives ONLY in the uncheckpointed WAL — the genuinely-different
811 /// records the on-disk pages do not hold — tagging each with the
812 /// `(salt1, salt2, frame_index)` log-sequence identity.
813 #[must_use]
814 pub fn wal_frame_pages(&self) -> &[WalFramePage] {
815 self.wal.as_ref().map_or(&[], |w| w.frames.as_slice())
816 }
817
818 /// Build the bespoke, format-exact [`WalTimeline`] for this database's `-wal`
819 /// sidecar, if one was supplied to [`Database::open_with_wal`].
820 ///
821 /// Returns `None` when the database was opened without a WAL, or the WAL held
822 /// no committed frame (no materializable state). The timeline enumerates the
823 /// segment's [`CommitSnapshot`]s — the only materializable database states —
824 /// each addressable by [`CommitId`]; see [`WalTimeline`].
825 ///
826 /// This consults the original `-wal` bytes retained at open time, re-parsing
827 /// them into the richer temporal model (the on-open `WalOverlay` keeps only
828 /// the consistent-view pages; the timeline keeps every segment, snapshot, and
829 /// residue tail). A page-size mismatch or malformed header surfaces as `None`
830 /// here — use [`Database::wal_timeline_from`] when you need the typed
831 /// [`WalValidationError`].
832 #[must_use]
833 pub fn wal_timeline(&self) -> Option<WalTimeline> {
834 let raw = self.wal.as_ref()?.raw.as_slice();
835 WalTimeline::parse(self.source.whole()?, raw, self.header.page_size).ok()
836 }
837
838 /// Parse a main database + `-wal` sidecar directly into a [`WalTimeline`],
839 /// surfacing the typed [`WalValidationError`] when the WAL is malformed.
840 ///
841 /// This is the validation-tier entry point: a page-size mismatch between the DB
842 /// header and the WAL header is a HARD STOP ([`WalValidationError::PageSizeMismatch`]),
843 /// not a silently mis-sliced overlay; a bad magic / unparsable header is
844 /// [`WalValidationError::BadMagic`]. Both are caught at the physical-validation
845 /// tier before any replay.
846 pub fn wal_timeline_from(bytes: &[u8], wal: &[u8]) -> Result<WalTimeline, WalValidationError> {
847 let header = parse_header(bytes).map_err(WalValidationError::Header)?;
848 WalTimeline::parse(bytes, wal, header.page_size)
849 }
850
851 #[must_use]
852 pub fn header(&self) -> Header {
853 self.header
854 }
855
856 /// Number of pages in the database file.
857 ///
858 /// Prefers the in-header DB size (offset 28) when it is a valid, non-zero
859 /// value that is consistent with the file length; otherwise falls back to
860 /// `file_len / page_size`. A mismatch between the two is itself a forensic
861 /// signal (see [`Database::header_page_count`] / [`Database::file_page_count`]).
862 #[must_use]
863 pub fn page_count(&self) -> u32 {
864 let header = self.header_page_count();
865 let file = self.file_page_count();
866 if header != 0 && header == file {
867 header
868 } else {
869 file
870 }
871 }
872
873 /// The page count recorded in the file header (offset 28). May be 0 (legacy
874 /// "size not valid" sentinel) or disagree with the file length after an
875 /// out-of-band truncation/extension.
876 #[must_use]
877 pub fn header_page_count(&self) -> u32 {
878 be_u32(&self.head, DB_SIZE_IN_PAGES_OFFSET)
879 }
880
881 /// The page count implied by the raw file length (`file_len / page_size`).
882 #[must_use]
883 pub fn file_page_count(&self) -> u32 {
884 let ps = self.header.page_size as usize;
885 u32::try_from(self.source.len() / ps).unwrap_or(u32::MAX)
886 }
887
888 /// The freelist page **count** recorded in the file header (offset 36).
889 #[must_use]
890 pub fn freelist_count(&self) -> u32 {
891 be_u32(&self.head, FREELIST_COUNT_OFFSET)
892 }
893
894 /// Walk the freelist trunk/leaf chain and return every free (unallocated)
895 /// page number, in trunk order. Free pages retain the bytes of whatever they
896 /// last held — on a `secure_delete=OFF` database that includes deleted
897 /// records, which the analyzer can carve.
898 ///
899 /// Bounded against crafted cyclic trunk chains: a page already visited, an
900 /// out-of-range page, or a leaf-pointer count larger than a trunk page can
901 /// hold aborts with [`Error::MalformedFreelist`] rather than looping.
902 pub fn freelist_pages(&self) -> Result<Vec<u32>, Error> {
903 let (leaves, trunks) = self.freelist_pages_split()?;
904 // Preserve the historical order: each trunk's leaves, then the trunk.
905 // The split sets are ordered, which is sufficient for every caller (they
906 // treat the result as a set), and keeps a single source of truth.
907 let mut free: Vec<u32> = leaves.into_iter().collect();
908 free.extend(trunks);
909 Ok(free)
910 }
911
912 /// Walk the freelist and return its **leaf** and **trunk** page numbers
913 /// separately (task #73). The distinction is load-bearing for chain-aware
914 /// overflow recovery: a freed page that became a freelist *leaf* keeps its
915 /// former content byte-for-byte, while a *trunk* page has its head
916 /// (next-trunk pointer + leaf count + leaf-number array) written over the
917 /// former content (file-format §"The Freelist"). Only leaves are
918 /// content-preserving, so [`Database::read_freed_overflow_chain`] accepts a
919 /// chain page only when it is a leaf.
920 ///
921 /// Bounded identically to [`Database::freelist_pages`]: a cyclic trunk chain,
922 /// an out-of-range page, or an over-large leaf count aborts with
923 /// [`Error::MalformedFreelist`] rather than looping.
924 pub fn freelist_pages_split(
925 &self,
926 ) -> Result<
927 (
928 std::collections::BTreeSet<u32>,
929 std::collections::BTreeSet<u32>,
930 ),
931 Error,
932 > {
933 let mut leaves = std::collections::BTreeSet::new();
934 let mut trunks = std::collections::BTreeSet::new();
935 let mut trunk = be_u32(&self.head, SQLITE_FREELIST_TRUNK_OFFSET);
936 let total_pages = self.file_page_count();
937 // Each trunk page holds at most (page_size/4 - 2) leaf pointers.
938 let max_leaves = (self.header.page_size as usize / 4).saturating_sub(2);
939 let mut visited = 0usize;
940 let cap = total_pages as usize + 1;
941
942 while trunk != 0 {
943 visited += 1;
944 if visited > cap {
945 return Err(Error::MalformedFreelist);
946 }
947 if trunk > total_pages {
948 return Err(Error::MalformedFreelist);
949 }
950 let slice = self.page_slice(trunk)?;
951 let slice = &*slice;
952 let next = be_u32(slice, 0);
953 let leaf_count = be_u32(slice, 4) as usize;
954 if leaf_count > max_leaves {
955 return Err(Error::MalformedFreelist);
956 }
957 for i in 0..leaf_count {
958 let leaf = be_u32(slice, 8 + i * 4);
959 if leaf == 0 || leaf > total_pages {
960 return Err(Error::MalformedFreelist);
961 }
962 leaves.insert(leaf);
963 }
964 trunks.insert(trunk);
965 trunk = next;
966 }
967 Ok((leaves, trunks))
968 }
969
970 /// Follow a **freed** overflow-page chain starting at `first`, reading raw
971 /// main-file pages only (carving wants on-disk residue, not the WAL view),
972 /// and assemble up to `remaining` content bytes (task #73). The carve-side
973 /// dual of `Database::read_overflow_chain`, with one extra discipline that
974 /// makes it the 0-FP-relevant guard: **every chain page must be a freelist
975 /// leaf** (`freed_leaves`). A page that is not a leaf is live, a trunk, or
976 /// unreachable — following its pointer would risk reading reused or clobbered
977 /// content, so it is a [`ChainBreak`] (Codex ruling #2: the leaf requirement,
978 /// not the UTF-8 gate, is what rejects a destroyed chain).
979 ///
980 /// Returns the assembled content and the ordered list of chain pages on
981 /// success. Robustness (Paranoid Gatekeeper, design §4.2): the anti-bomb cap
982 /// rejects upfront any `remaining` above what the freelist leaves can deliver
983 /// (`(usable - 4) × freed_leaves.len()`), so an attacker-declared huge
984 /// payload dies before any allocation; cycles are caught by a visited set;
985 /// a premature `next == 0` with bytes still wanted, an out-of-range page, or
986 /// page 0 mid-chain all break. Never panics — every read is bounds-checked.
987 pub fn read_freed_overflow_chain(
988 &self,
989 first: u32,
990 remaining: usize,
991 usable: usize,
992 freed_leaves: &std::collections::BTreeSet<u32>,
993 ) -> Result<(Vec<u8>, Vec<u32>), ChainBreak> {
994 let per_page = usable.checked_sub(4).filter(|&p| p > 0).ok_or(ChainBreak)?;
995 // Anti-bomb cap: the chain can deliver at most this many bytes. Reject an
996 // absurd declared payload before allocating (design §4.2).
997 let max_deliverable = per_page.checked_mul(freed_leaves.len()).ok_or(ChainBreak)?;
998 if remaining > max_deliverable {
999 return Err(ChainBreak);
1000 }
1001 let total_pages = self.file_page_count();
1002 let mut content = Vec::with_capacity(remaining);
1003 let mut chain = Vec::new();
1004 let mut visited = std::collections::BTreeSet::new();
1005 let mut page = first;
1006 let mut left = remaining;
1007 while left > 0 {
1008 if page == 0 || page > total_pages {
1009 return Err(ChainBreak);
1010 }
1011 // The load-bearing guard: a chain page must be a freelist LEAF.
1012 if !freed_leaves.contains(&page) {
1013 return Err(ChainBreak);
1014 }
1015 if !visited.insert(page) {
1016 return Err(ChainBreak); // cycle
1017 }
1018 let slice = self.raw_page(page).ok_or(ChainBreak)?;
1019 let slice = &*slice;
1020 let next = be_u32(slice, 0);
1021 let take = left.min(per_page);
1022 let chunk = slice.get(4..4 + take).ok_or(ChainBreak)?;
1023 content.extend_from_slice(chunk);
1024 chain.push(page);
1025 left -= take;
1026 page = next;
1027 }
1028 Ok((content, chain))
1029 }
1030
1031 /// Raw bytes of the 1-based `page` from the **main file only**, ignoring any
1032 /// WAL overlay. Carving wants the on-disk page (where deleted residue lives),
1033 /// not the WAL-applied view. Returns `None` for page 0 or out-of-range pages.
1034 #[must_use]
1035 pub fn raw_page(&self, page: u32) -> Option<PageBytes<'_>> {
1036 if page == 0 {
1037 return None;
1038 }
1039 self.source.page(page, self.header.page_size as usize)
1040 }
1041
1042 /// Scan a slice of page bytes for record-shaped table-leaf cells of exactly
1043 /// `column_count` columns, recovering each as a [`CarvedCell`].
1044 ///
1045 /// This is the carving primitive the forensic analyzer drives over free /
1046 /// unallocated regions: at every byte offset it speculatively parses a
1047 /// `payload_len` varint, a `rowid` varint, and a record header, accepting the
1048 /// candidate only when the serial-type count matches `column_count`, the
1049 /// declared lengths stay within the slice, and every value decodes. Strict
1050 /// validation keeps the false-positive rate low; `confidence` reflects how
1051 /// strongly the bytes are record-shaped. Bounded: each offset does O(record)
1052 /// work and the scan is linear in the slice length.
1053 #[must_use]
1054 pub fn carve_cells(&self, page_bytes: &[u8], column_count: usize) -> Vec<CarvedCell> {
1055 let mut out = Vec::new();
1056 if column_count == 0 {
1057 return out;
1058 }
1059 let mut off = 0usize;
1060 while off < page_bytes.len() {
1061 if let Some(cell) = try_carve_cell_at(
1062 page_bytes,
1063 off,
1064 Some(column_count),
1065 self.header.text_encoding,
1066 ) {
1067 // Skip past this record to avoid re-reporting sub-slices of it.
1068 off += cell.byte_len.max(1);
1069 out.push(cell);
1070 } else {
1071 off += 1;
1072 }
1073 }
1074 out
1075 }
1076
1077 /// Carve record-shaped cells from a page slice **inferring** each record's
1078 /// column count from its own serial-type array, instead of requiring a fixed
1079 /// count. This is what makes **dropped-table / schema-gone** recovery
1080 /// possible: the page's table was `DROP`ped, so `sqlite_master` no longer
1081 /// records a column count, but each record still self-describes its columns.
1082 ///
1083 /// Inferring the count removes one validity check, so the remaining
1084 /// self-consistency checks are kept strict to hold the false-positive rate
1085 /// down: `header_len + body_len == payload_len`, every serial type legal,
1086 /// `rowid > 0`, the payload fully in-bounds, and at least
1087 /// `MIN_INFERRED_COLUMNS` columns. Records carved this way are graded a
1088 /// notch lower in confidence than fixed-count carving.
1089 #[must_use]
1090 pub fn carve_cells_inferred(&self, page_bytes: &[u8]) -> Vec<CarvedCell> {
1091 let mut out = Vec::new();
1092 let mut off = 0usize;
1093 while off < page_bytes.len() {
1094 if let Some(cell) = try_carve_cell_at(page_bytes, off, None, self.header.text_encoding)
1095 {
1096 off += cell.byte_len.max(1);
1097 out.push(cell);
1098 } else {
1099 off += 1;
1100 }
1101 }
1102 out
1103 }
1104
1105 /// Decode **every cell present in a table-leaf page image** (type `0x0D`) by
1106 /// walking its cell-pointer array, inferring each record's column count from
1107 /// its own serial-type array. Unlike [`Database::carve_free_regions`] (which
1108 /// scans only free space and excludes live cells), this returns the cells the
1109 /// page itself records as allocated.
1110 ///
1111 /// This is the primitive WAL-frame recovery needs: a `-wal` frame is a full
1112 /// page snapshot at one point in time, so a cell that is allocated in an
1113 /// EARLIER frame's image but absent from the final WAL-applied view is a row
1114 /// that was deleted later and survives ONLY in that superseded frame. The
1115 /// caller filters the returned cells against the final live view to isolate
1116 /// exactly those genuinely-deleted rows (so a still-live row is never
1117 /// re-surfaced — the filter is the caller's responsibility, mirroring the
1118 /// freeblock-reconstruction discipline).
1119 ///
1120 /// Bounded and panic-free: a malformed cell pointer or record simply yields
1121 /// fewer cells. Non-leaf pages yield nothing.
1122 #[must_use]
1123 pub fn carve_leaf_cells(&self, page_bytes: &[u8]) -> Vec<CarvedCell> {
1124 let hdr_off = if page_bytes.starts_with(SQLITE_MAGIC) {
1125 SQLITE_HEADER_SIZE
1126 } else {
1127 0
1128 };
1129 let Some(&page_type) = page_bytes.get(hdr_off) else {
1130 return Vec::new();
1131 };
1132 if page_type != 0x0d {
1133 return Vec::new(); // only table-leaf pages hold decodable cells here
1134 }
1135 let cell_count = be_u16(page_bytes, hdr_off + 3) as usize;
1136 let cell_ptr_array = hdr_off + 8; // leaf b-tree header is 8 bytes
1137 let mut out = Vec::new();
1138 for i in 0..cell_count {
1139 let cell_off = be_u16(page_bytes, cell_ptr_array + i * 2) as usize;
1140 if cell_off == 0 || cell_off >= page_bytes.len() {
1141 continue; // cov:unreachable: a valid leaf points cells within page
1142 }
1143 if let Some(cell) =
1144 try_carve_cell_at(page_bytes, cell_off, None, self.header.text_encoding)
1145 {
1146 out.push(cell);
1147 }
1148 }
1149 out
1150 }
1151
1152 /// Carve deleted records from the **free (unallocated) regions** of an
1153 /// allocated table-leaf page (type `0x0D`), never re-surfacing a live cell.
1154 ///
1155 /// On an allocated leaf, deleted-cell residue survives in two places: the
1156 /// unallocated gap between the cell-pointer array and the cell-content area,
1157 /// and the slack between/after live cells (a former freeblock whose chain
1158 /// pointer may already be gone). This method computes the exact byte ranges
1159 /// occupied by **live** cells and carves only the complement — so a live
1160 /// (allocated) cell can never be returned as a deleted record. That is the
1161 /// 0-false-positive guarantee, enforced structurally rather than by a filter.
1162 ///
1163 /// `page_bytes` is one whole page. `column_count_hint`, when non-zero, is the
1164 /// table's known column count (matched exactly); pass 0 to infer the count
1165 /// per record (for a page whose schema is gone). Non-leaf pages yield nothing.
1166 #[must_use]
1167 pub fn carve_free_regions(
1168 &self,
1169 page_bytes: &[u8],
1170 column_count_hint: usize,
1171 ) -> Vec<CarvedCell> {
1172 // Page 1 carries the 100-byte file header before its b-tree header; for a
1173 // standalone page slice we assume hdr_off 0 unless it starts with the
1174 // file magic (page 1 passed whole).
1175 let hdr_off = if page_bytes.starts_with(SQLITE_MAGIC) {
1176 SQLITE_HEADER_SIZE
1177 } else {
1178 0
1179 };
1180 let Some(&page_type) = page_bytes.get(hdr_off) else {
1181 return Vec::new();
1182 };
1183 if page_type != 0x0d {
1184 return Vec::new(); // only table-leaf pages have carvable cell residue
1185 }
1186 // Carve each maximal free region (complement of the live cell extents),
1187 // within the cell-content area only — so no allocated cell is ever
1188 // re-surfaced (the 0-false-positive guarantee, enforced structurally).
1189 let mut out = Vec::new();
1190 let regions = self.free_regions_of_leaf(page_bytes, hdr_off);
1191 for (lo, hi) in regions {
1192 let Some(region) = page_bytes.get(lo..hi) else {
1193 continue; // cov:unreachable: free_regions yields in-bounds spans
1194 };
1195 let cells = if column_count_hint == 0 {
1196 self.carve_cells_inferred(region)
1197 } else {
1198 self.carve_cells(region, column_count_hint)
1199 };
1200 for mut cell in cells {
1201 // Translate the offset from region-local to page-local, and grade
1202 // in-page recovery a notch lower (residue here is more often
1203 // partially overwritten than freed-page recovery).
1204 cell.offset += lo;
1205 cell.confidence *= IN_PAGE_CONFIDENCE_FACTOR;
1206 out.push(cell);
1207 }
1208 }
1209 out
1210 }
1211
1212 /// Recover **spilled** deleted records on a table-leaf page whose payload
1213 /// continued onto a freed overflow-page chain (task #73). Scans the page's
1214 /// free regions (the complement of the live cells — same discipline as
1215 /// [`Database::carve_free_regions`], so a live cell is never re-surfaced) for
1216 /// a [`SpilledCell`], then resolves each chain through freelist **leaf** pages
1217 /// only and assembles the full payload.
1218 ///
1219 /// A resolved record is returned only when ALL hold (design §5):
1220 /// 1. the chain is intact through freelist leaves (Codex ruling #2: the leaf
1221 /// requirement is the load-bearing 0-FP guard — a trunk/live/off-freelist
1222 /// chain page is rejected);
1223 /// 2. the assembled bytes total exactly the declared `P` and decode cleanly;
1224 /// 3. **strict UTF-8 on chain-resident TEXT** — an EXTRA reject signal, not a
1225 /// correctness proof (Codex ruling #2: a clobbered chain can still be valid
1226 /// UTF-8, so this cannot prove integrity; it only catches the cases where
1227 /// the lossy decoder would otherwise mask an overwrite as `U+FFFD`).
1228 ///
1229 /// Each returned tuple is `(cell, chain)` where `chain` is the ordered list of
1230 /// overflow pages the bytes came from (for provenance). Confidence is graded
1231 /// BELOW the in-page full-row tier (Codex ruling #1: overflow Tier-1 is a
1232 /// graded recovery, NOT part of the structural 0-FP guarantee — a freelist
1233 /// leaf can be stale, holding unrelated bytes that happen to decode). Bounded
1234 /// and panic-free; a malformed page or chain simply yields fewer records.
1235 #[must_use]
1236 pub fn carve_overflow_records(&self, page_bytes: &[u8]) -> Vec<(CarvedCell, Vec<u32>)> {
1237 let hdr_off = if page_bytes.starts_with(SQLITE_MAGIC) {
1238 SQLITE_HEADER_SIZE
1239 } else {
1240 0
1241 };
1242 let Some(&page_type) = page_bytes.get(hdr_off) else {
1243 return Vec::new();
1244 };
1245 if page_type != 0x0d {
1246 return Vec::new(); // only table-leaf pages carry spilled-cell residue
1247 }
1248 let Ok((freed_leaves, _trunks)) = self.freelist_pages_split() else {
1249 return Vec::new();
1250 };
1251 let usable = self.header.usable_size() as usize;
1252
1253 let mut out = Vec::new();
1254 let regions = self.free_regions_of_leaf(page_bytes, hdr_off);
1255 for (lo, hi) in regions {
1256 let Some(region) = page_bytes.get(lo..hi) else {
1257 continue; // cov:unreachable: free_regions yields in-bounds spans
1258 };
1259 // Scan every offset for a spilled cell (recognizer abstains on in-page
1260 // payloads, so the two carve classes never overlap).
1261 let mut off = 0usize;
1262 while off < region.len() {
1263 let Some(sc) = try_carve_spilled_cell_at(region, off, usable, None) else {
1264 off += 1;
1265 continue;
1266 };
1267 if let Some((mut cell, chain)) =
1268 self.resolve_spilled(region, &sc, usable, &freed_leaves)
1269 {
1270 // Translate the region-local offset to page-local.
1271 cell.offset = lo + sc.offset;
1272 out.push((cell, chain));
1273 off += sc.byte_len.max(1);
1274 } else {
1275 off += 1;
1276 }
1277 }
1278 }
1279 out
1280 }
1281
1282 /// Resolve a recognized [`SpilledCell`] to a full [`CarvedCell`] by walking
1283 /// its freed overflow chain and decoding the assembled payload, applying the
1284 /// strict-UTF-8 chain gate. Returns `Some((cell, chain))` on a fully-validated
1285 /// recovery, `None` on any chain break or gate failure (the candidate then
1286 /// degrades to a Tier-2 fragment elsewhere).
1287 fn resolve_spilled(
1288 &self,
1289 region: &[u8],
1290 sc: &SpilledCell,
1291 usable: usize,
1292 freed_leaves: &std::collections::BTreeSet<u32>,
1293 ) -> Option<(CarvedCell, Vec<u32>)> {
1294 let remaining = sc.payload_len.checked_sub(sc.local_len)?;
1295 let local_payload =
1296 region.get(sc.local_payload_off..sc.local_payload_off + sc.local_len)?;
1297 let (chain_content, chain) = self
1298 .read_freed_overflow_chain(sc.first_overflow, remaining, usable, freed_leaves)
1299 .ok()?;
1300 let mut payload = Vec::with_capacity(sc.payload_len);
1301 payload.extend_from_slice(local_payload);
1302 payload.extend_from_slice(&chain_content);
1303 if payload.len() != sc.payload_len {
1304 return None; // cov:unreachable: chain delivers exactly `remaining` bytes
1305 }
1306
1307 let values = decode_record(
1308 &payload,
1309 sc.serials.len(),
1310 sc.rowid,
1311 self.header.text_encoding,
1312 )
1313 .ok()?;
1314 if values.len() != sc.serials.len() {
1315 return None; // cov:unreachable: decode_record yields one value per serial
1316 }
1317 // Strict-UTF-8 gate on chain-resident TEXT (extra reject signal): the
1318 // lossy decoder turns a clobbered byte into U+FFFD, so any replacement
1319 // char in a decoded TEXT value means the chain-supplied bytes did not
1320 // decode cleanly — reject. NOT a proof of integrity (a stale leaf can hold
1321 // valid UTF-8); the freelist-leaf requirement is the load-bearing guard.
1322 let any_replacement = values.iter().any(|v| match v {
1323 Value::Text(t) => t.contains('\u{FFFD}'),
1324 _ => false,
1325 });
1326 if any_replacement {
1327 return None;
1328 }
1329 // Require at least one distinctive column so a coincidental decode of stale
1330 // bytes does not anchor a full row (the same identity bar as fragments).
1331 if !values.iter().any(is_distinctive) {
1332 return None; // cov:unreachable: the spilled corpus rows carry distinctive TEXT
1333 }
1334
1335 let cell = CarvedCell {
1336 offset: sc.offset,
1337 byte_len: sc.byte_len,
1338 rowid: sc.rowid,
1339 values,
1340 // Graded below the in-page full-row tier (0.9): an overflow chain adds
1341 // one indirection of stale-leaf exposure (Codex ruling #1).
1342 confidence: 0.9 * OVERFLOW_CHAIN_CONFIDENCE_FACTOR,
1343 };
1344 Some((cell, chain))
1345 }
1346
1347 /// Reconstruct **freeblock-clobbered spilled** cells (task #73, design §2.2 /
1348 /// Codex ruling #5). When a freed cell whose payload spilled is also
1349 /// freeblock-clobbered, its declared `P` is destroyed but **re-derivable** from
1350 /// the surviving structure: `P = header_len + Σ serial_body_len` over the full
1351 /// (template + surviving) serial array. When that `P` exceeds `usable - 35` the
1352 /// record is spilled by construction, so we read the 4-byte first-overflow
1353 /// pointer that follows the local payload and resolve the chain through
1354 /// freelist leaves, exactly as the intact-prefix path does — but with
1355 /// `rowid = 0` (the prefix's rowid varint was clobbered, never invented).
1356 ///
1357 /// UNPROVEN-BY-CORPUS (Codex ruling #5): no real Nemetz `0E` cell is *both*
1358 /// freeblock-clobbered *and* spilled — every measured spilled cell kept an
1359 /// intact prefix in the unallocated gap. This path is therefore validated
1360 /// against a **synthetic** fixture only; it is the general solution the
1361 /// no-special-case rule requires (it applies the same spill formula to the
1362 /// clobbered class), but its real-data behavior is not yet observed.
1363 ///
1364 /// Returns `(cell, chain)` per fully-resolved record. Bounded and panic-free.
1365 #[must_use]
1366 pub fn carve_overflow_template_records(
1367 &self,
1368 page_bytes: &[u8],
1369 ) -> Vec<(CarvedCell, Vec<u32>)> {
1370 let hdr_off = if page_bytes.starts_with(SQLITE_MAGIC) {
1371 SQLITE_HEADER_SIZE
1372 } else {
1373 0
1374 };
1375 if page_bytes.get(hdr_off) != Some(&0x0d) {
1376 return Vec::new();
1377 }
1378 let Some(template) = freeblock_template(page_bytes, hdr_off, self.header.text_encoding)
1379 else {
1380 return Vec::new();
1381 };
1382 let Ok((freed_leaves, _trunks)) = self.freelist_pages_split() else {
1383 return Vec::new();
1384 };
1385 let usable = self.header.usable_size() as usize;
1386
1387 let mut out = Vec::new();
1388 // Walk the freeblock chain; at each freeblock head, try a clobbered-spill
1389 // reconstruction (the chain pass reaches the clobbered prefix the
1390 // intact-prefix recognizer cannot read).
1391 let first_freeblock = be_u16(page_bytes, hdr_off + 1) as usize;
1392 let mut fb = first_freeblock;
1393 let mut walked = 0usize;
1394 let mut visited = std::collections::BTreeSet::new();
1395 while fb != 0 && walked < MAX_FREEBLOCKS_PER_PAGE {
1396 walked += 1;
1397 if !visited.insert(fb) {
1398 break; // cyclic next pointer
1399 }
1400 let next = be_u16(page_bytes, fb) as usize;
1401 if let Some((cell, chain)) =
1402 template.reconstruct_spilled(self, page_bytes, fb, usable, &freed_leaves)
1403 {
1404 out.push((cell, chain));
1405 }
1406 fb = next;
1407 }
1408 out
1409 }
1410
1411 /// Tier-2 salvage for **spilled** cells whose overflow chain is broken (task
1412 /// #73, Codex ruling #4): when [`Database::carve_overflow_records`] rejects a
1413 /// recognized spilled cell because its chain failed (a trunk-clobbered or
1414 /// reused chain page), the cell's intact LOCAL prefix still holds the columns
1415 /// whose bodies fit entirely on the leaf page. Those are salvaged as a
1416 /// [`CellFragment`] — the same Tier-2 surface freeblock reconstruction uses.
1417 ///
1418 /// Only columns whose body lies wholly within the local payload are kept; the
1419 /// chain-resident columns are lost (untrusted by definition — the chain that
1420 /// would supply them is the thing that failed). A fragment is emitted only
1421 /// when the salvaged prefix carries ≥ 1 distinctive cell (TEXT ≥ 4 bytes of
1422 /// valid UTF-8, or REAL — the §3.1 gate), so a lone integer prefix never
1423 /// anchors one. Bounded and panic-free.
1424 #[must_use]
1425 pub fn carve_overflow_fragments(&self, page_bytes: &[u8]) -> Vec<CellFragment> {
1426 let hdr_off = if page_bytes.starts_with(SQLITE_MAGIC) {
1427 SQLITE_HEADER_SIZE
1428 } else {
1429 0
1430 };
1431 let Some(&page_type) = page_bytes.get(hdr_off) else {
1432 return Vec::new();
1433 };
1434 if page_type != 0x0d {
1435 return Vec::new();
1436 }
1437 let Ok((freed_leaves, _trunks)) = self.freelist_pages_split() else {
1438 return Vec::new();
1439 };
1440 let usable = self.header.usable_size() as usize;
1441
1442 let mut out = Vec::new();
1443 let regions = self.free_regions_of_leaf(page_bytes, hdr_off);
1444 for (lo, hi) in regions {
1445 let Some(region) = page_bytes.get(lo..hi) else {
1446 continue; // cov:unreachable: free_regions yields in-bounds spans
1447 };
1448 let mut off = 0usize;
1449 while off < region.len() {
1450 let Some(sc) = try_carve_spilled_cell_at(region, off, usable, None) else {
1451 off += 1;
1452 continue;
1453 };
1454 // Only broken chains degrade to a fragment — an intact chain is a
1455 // Tier-1 row (handled by carve_overflow_records), never both.
1456 let remaining = sc.payload_len.saturating_sub(sc.local_len);
1457 let chain_ok = self
1458 .read_freed_overflow_chain(sc.first_overflow, remaining, usable, &freed_leaves)
1459 .is_ok();
1460 if !chain_ok {
1461 if let Some(mut frag) =
1462 salvage_local_prefix(region, &sc, self.header.text_encoding)
1463 {
1464 frag.offset += lo;
1465 out.push(frag);
1466 }
1467 }
1468 off += sc.byte_len.max(1);
1469 }
1470 }
1471 out
1472 }
1473
1474 /// Reconstruct deleted records from the **freeblock chain** of an allocated
1475 /// table-leaf page (type `0x0d`) — the records a forward parse cannot recover
1476 /// because their first four bytes were destroyed by freeblock conversion.
1477 ///
1478 /// When SQLite frees an in-page cell it converts it into a **freeblock**
1479 /// (file-format §1.6): the cell's first two bytes become the next-freeblock
1480 /// offset and the next two the freeblock size, **overwriting the cell's
1481 /// payload-length + rowid varints, the record `header_len` varint, and the
1482 /// leading serial type(s)**. The record's surviving serial-type tail and its
1483 /// whole value body remain intact *after* those four bytes.
1484 ///
1485 /// This method rebuilds each freed cell from that surviving tail plus a
1486 /// **schema template** derived from a LIVE cell on the same page (the table's
1487 /// column count, header length, and the serial types of the leading columns
1488 /// that fall inside the clobbered prefix). The destroyed rowid is surfaced as
1489 /// unknown (`0`) — never invented — and the record is graded LOW.
1490 ///
1491 /// Precision discipline (task #56): a candidate is emitted only when its body
1492 /// decodes cleanly with every serial type legal AND the record fits within
1493 /// the freeblock's `[offset, offset + size)` bounds. Implausible or
1494 /// out-of-bounds candidates are rejected, so reconstruction does not
1495 /// manufacture phantom rows. (The forensic layer additionally drops any
1496 /// reconstruction whose values match a live row, so a live row is never
1497 /// re-surfaced.)
1498 ///
1499 /// Bounded and panic-free: every freeblock pointer, size, and serial length
1500 /// is range-checked against the page before use, and the chain walk is capped
1501 /// at `MAX_FREEBLOCKS_PER_PAGE` to defeat a crafted cyclic `next` chain.
1502 /// Non-leaf pages, pages with no freeblock chain, and pages with no usable
1503 /// schema template yield an empty result.
1504 #[must_use]
1505 pub fn reconstruct_freeblock_records(&self, page_bytes: &[u8]) -> Vec<CarvedCell> {
1506 // Tier-1 cells are the `.0` of the shared two-tier walker, so the full-row
1507 // output and the fragment output ([`Database::reconstruct_freeblock_fragments`])
1508 // can never diverge. The walk (freeblock-chain pass + unallocated-gap pass)
1509 // and its precision discipline live in [`reconstruct_freeblock_inner`].
1510 let _ = self;
1511 reconstruct_freeblock_inner(page_bytes, self.header.text_encoding).0
1512 }
1513
1514 /// Tier-2 partial salvage: the [`CellFragment`]s abandoned by
1515 /// [`Database::reconstruct_freeblock_records`] on this page.
1516 ///
1517 /// At every anchor where full reconstruction failed — an illegal serial in
1518 /// the surviving tail, a tail that overruns the span, or a body that does not
1519 /// fit — the columns that DID decode cleanly before the failure are salvaged
1520 /// as the maximal decodable prefix. A fragment is emitted only when that
1521 /// prefix contains at least one *distinctive* cell (TEXT ≥ 4 bytes of valid
1522 /// UTF-8, or REAL): a lone surviving integer pattern is coincidence-prone and
1523 /// never anchors a fragment.
1524 ///
1525 /// Mutually exclusive with the full reconstructions of
1526 /// [`Database::reconstruct_freeblock_records`] **by construction**: an anchor
1527 /// yields a cell or a fragment, never both. Inherits the same anchor
1528 /// discipline — no sliding scan, no strings-style hunt — so Tier-2 carries
1529 /// Tier-1's precision architecture. Bounded and panic-free identically.
1530 #[must_use]
1531 pub fn reconstruct_freeblock_fragments(&self, page_bytes: &[u8]) -> Vec<CellFragment> {
1532 let _ = self;
1533 reconstruct_freeblock_inner(page_bytes, self.header.text_encoding).1
1534 }
1535
1536 /// Parse the LIVE cells of an index-b-tree **leaf** page (type `0x0a`) into
1537 /// their decoded key records (roadmap §1.4 foundation). A regular index on a
1538 /// rowid table stores each entry as `(indexed columns…, rowid)`; a
1539 /// `WITHOUT ROWID` table stores its whole row here (the row IS the key). This
1540 /// is the structural read every later index-carve / `WITHOUT ROWID` recovery
1541 /// builds on — the second substrate for a table's data, where key columns
1542 /// survive even when the table-leaf residue is gone.
1543 ///
1544 /// Reads live cells only (via the cell-pointer array); returns empty for any
1545 /// non-index-leaf page, so a table page is never mis-read. Bounded and
1546 /// panic-free — every read is bounds-checked; a cell whose payload does not
1547 /// decode is skipped rather than panicking.
1548 ///
1549 /// SCOPE (foundation): decodes the LOCAL payload only. An index key large
1550 /// enough to spill onto an overflow-page chain is decoded up to its on-page
1551 /// bytes (the leading key columns still resolve); full overflow following, and
1552 /// carving DELETED index entries from index-page freeblocks, are follow-ups.
1553 #[must_use]
1554 pub fn index_leaf_cells(&self, page_bytes: &[u8]) -> Vec<Vec<Value>> {
1555 let hdr_off = if page_bytes.starts_with(SQLITE_MAGIC) {
1556 SQLITE_HEADER_SIZE
1557 } else {
1558 0
1559 };
1560 if page_bytes.get(hdr_off) != Some(&0x0a) {
1561 return Vec::new(); // only index-b-tree leaf pages carry index cells
1562 }
1563 let cell_count = be_u16(page_bytes, hdr_off + 3) as usize;
1564 let cell_ptr_array = hdr_off + 8; // an index-leaf header is 8 bytes
1565 let mut out = Vec::with_capacity(cell_count);
1566 for i in 0..cell_count {
1567 let ptr_off = cell_ptr_array + i * 2;
1568 if ptr_off + 1 >= page_bytes.len() {
1569 break;
1570 }
1571 let cell_off = be_u16(page_bytes, ptr_off) as usize;
1572 if cell_off == 0 || cell_off >= page_bytes.len() {
1573 continue;
1574 }
1575 // An index-leaf cell is [payload-length varint][payload][overflow?].
1576 if let Some(values) = self.index_record_at(page_bytes, cell_off) {
1577 out.push(values);
1578 }
1579 }
1580 out
1581 }
1582
1583 /// Decode the index record whose `[payload-length varint][payload]` begins at
1584 /// `off` within `page_bytes`, or `None` if it does not decode. Shared by the
1585 /// leaf read ([`index_leaf_cells`](Self::index_leaf_cells)) and the interior
1586 /// walk (whose cells also carry a key record, after the 4-byte child pointer).
1587 /// Decodes the LOCAL payload only — a key spilled to an overflow chain is
1588 /// decoded up to its on-page bytes (the leading key columns still resolve).
1589 fn index_record_at(&self, page_bytes: &[u8], off: usize) -> Option<Vec<Value>> {
1590 let (payload_len, n) = read_varint(page_bytes, off).ok()?;
1591 let payload_start = off + n;
1592 let payload_len = usize::try_from(payload_len).ok()?;
1593 let end = payload_start
1594 .saturating_add(payload_len)
1595 .min(page_bytes.len());
1596 let payload = page_bytes.get(payload_start..end)?;
1597 decode_index_payload(payload, self.header.text_encoding).ok()
1598 }
1599
1600 /// The live rows of every `WITHOUT ROWID` user table (roadmap §1.4).
1601 ///
1602 /// A `WITHOUT ROWID` table stores its whole row in an **index b-tree** — there
1603 /// is no separate table b-tree and no rowid — so the ordinary
1604 /// [`read_table`](Self::read_table) reader (which walks table pages 0x0d/0x05)
1605 /// is blind to it. This resolves each such table from `sqlite_master`, walks
1606 /// its index b-tree (interior 0x02 → leaf 0x0a), and returns its live rows,
1607 /// keyed by table name. Ordinary rowid tables are not returned.
1608 ///
1609 /// Bounded and panic-free: a malformed/cyclic b-tree stops the walk (visited
1610 /// set + page cap) rather than looping; an unreadable schema yields an empty
1611 /// result. Rows are the decoded index records, in the table's column order.
1612 #[must_use]
1613 pub fn without_rowid_table_rows(&self) -> Vec<WithoutRowidTable> {
1614 let Ok(schema) = self.read_table(1, 5) else {
1615 return Vec::new(); // cov:unreachable: a validly-opened DB has a readable page-1 schema
1616 };
1617 let mut out = Vec::new();
1618 for row in schema {
1619 // sqlite_master row: (type, name, tbl_name, rootpage, sql).
1620 let is_table = matches!(row.values.first(), Some(Value::Text(t)) if t == "table");
1621 if !is_table {
1622 continue;
1623 }
1624 let Some(Value::Text(name)) = row.values.get(1) else {
1625 continue; // cov:unreachable: a 'table' schema row has a TEXT name
1626 };
1627 if name.starts_with("sqlite_") {
1628 continue;
1629 }
1630 let sql = match row.values.get(4) {
1631 Some(Value::Text(s)) => s.as_str(),
1632 _ => "", // cov:unreachable: a 'table' schema row carries its CREATE TABLE sql
1633 };
1634 if !without_rowid_sql(sql) {
1635 continue; // ordinary rowid table — read_table handles those
1636 }
1637 let Some(Value::Integer(root)) = row.values.get(3) else {
1638 continue; // cov:unreachable: a 'table' schema row has an integer rootpage
1639 };
1640 let Ok(root) = u32::try_from(*root) else {
1641 continue; // cov:unreachable: a real rootpage is a small positive page number
1642 };
1643 let mut rows = Vec::new();
1644 let mut seen = std::collections::BTreeSet::new();
1645 self.collect_index_rows(root, &mut rows, &mut seen);
1646 out.push(WithoutRowidTable {
1647 name: name.clone(),
1648 rows,
1649 });
1650 }
1651 out
1652 }
1653
1654 /// Walk the index b-tree rooted at `page`, appending every leaf cell's decoded
1655 /// record to `rows`. Interior pages (0x02) recurse through their child pointers
1656 /// and rightmost child; leaf pages (0x0a) yield their cells via
1657 /// [`index_leaf_cells`](Self::index_leaf_cells). Bounded identically to
1658 /// [`collect_rows`](Self::collect_rows): a page is visited at most once and the
1659 /// walk is capped, so a crafted cyclic/oversized tree cannot loop.
1660 fn collect_index_rows(
1661 &self,
1662 page: u32,
1663 rows: &mut Vec<Vec<Value>>,
1664 seen: &mut std::collections::BTreeSet<u32>,
1665 ) {
1666 if page == 0 || seen.len() > MAX_PAGES_PER_WALK || !seen.insert(page) {
1667 return;
1668 }
1669 let Ok(slice) = self.page_slice(page) else {
1670 return; // cov:unreachable: schema rootpages and their children are in range
1671 };
1672 let slice = &*slice;
1673 let hdr_off = if page == 1 { SQLITE_HEADER_SIZE } else { 0 };
1674 let Some(&page_type) = slice.get(hdr_off) else {
1675 return; // cov:unreachable: a full page slice always has its header byte
1676 };
1677 match page_type {
1678 0x0a => rows.extend(self.index_leaf_cells(slice)),
1679 0x02 => {
1680 let cell_count = be_u16(slice, hdr_off + 3) as usize;
1681 let cell_ptr_array = hdr_off + 12; // an index-interior header is 12 bytes
1682 for i in 0..cell_count {
1683 let cell_off = be_u16(slice, cell_ptr_array + i * 2) as usize;
1684 // An interior cell is [4-byte left-child page][key record]. In
1685 // an INDEX b-tree the key IS a real entry (a WITHOUT ROWID row),
1686 // so decode it too — not just the child pointer, unlike a table
1687 // b-tree where interior cells are pure navigation.
1688 let child = be_u32(slice, cell_off);
1689 self.collect_index_rows(child, rows, seen);
1690 if let Some(values) = self.index_record_at(slice, cell_off + 4) {
1691 rows.push(values);
1692 }
1693 }
1694 let right = be_u32(slice, hdr_off + 8);
1695 self.collect_index_rows(right, rows, seen);
1696 }
1697 _ => {} // cov:unreachable: a WITHOUT ROWID b-tree page is index leaf (0x0a) or interior (0x02)
1698 }
1699 }
1700
1701 /// The maximal FREE (unallocated) byte ranges of a table-leaf page — the
1702 /// complement of its live cells within the cell-content area. Shared by
1703 /// [`Database::carve_free_regions`] and
1704 /// [`Database::reconstruct_freeblock_records`] so both scan exactly the same
1705 /// ranges and never touch a live cell. Returns empty for a non-leaf page.
1706 fn free_regions_of_leaf(&self, page_bytes: &[u8], hdr_off: usize) -> Vec<(usize, usize)> {
1707 if page_bytes.get(hdr_off) != Some(&0x0d) {
1708 return Vec::new(); // cov:unreachable: callers gate on page_type == 0x0d
1709 }
1710 let cell_count = be_u16(page_bytes, hdr_off + 3) as usize;
1711 let cell_ptr_array = hdr_off + 8; // leaf header is 8 bytes
1712 let usable = self.header.usable_size() as usize;
1713 let mut live: Vec<(usize, usize)> = Vec::with_capacity(cell_count);
1714 for i in 0..cell_count {
1715 let cell_off = be_u16(page_bytes, cell_ptr_array + i * 2) as usize;
1716 if cell_off == 0 || cell_off >= page_bytes.len() {
1717 continue; // cov:unreachable: a valid leaf points cells within page
1718 }
1719 if let Some(len) = live_cell_len(page_bytes, cell_off, usable) {
1720 live.push((cell_off, cell_off.saturating_add(len)));
1721 }
1722 }
1723 live.sort_unstable_by_key(|&(s, _)| s);
1724 let content_lo = cell_ptr_array + cell_count * 2;
1725 free_regions(&live, content_lo, page_bytes.len())
1726 }
1727
1728 /// Whether `sqlite_master` (the schema table rooted at page 1) lists at least
1729 /// one **user** table — i.e. a `type='table'` row whose name is not an
1730 /// internal `sqlite_*` table. A database where every table was `DROP`ped (or
1731 /// that never had one) returns `false`; the forensic carver uses this to label
1732 /// freed content as dropped-table residue. Errors (unreadable schema) are
1733 /// treated as "no user table" so the carver degrades safely.
1734 #[must_use]
1735 pub fn has_user_table(&self) -> bool {
1736 // sqlite_master is a 5-column table: (type, name, tbl_name, rootpage, sql).
1737 let Ok(rows) = self.read_table(1, 5) else {
1738 return false; // cov:unreachable: a validly-opened DB has a readable page-1 schema
1739 };
1740 rows.iter().any(|row| {
1741 let is_table = matches!(row.values.first(), Some(Value::Text(t)) if t == "table");
1742 let user = matches!(
1743 row.values.get(1),
1744 Some(Value::Text(n)) if !n.starts_with("sqlite_")
1745 );
1746 is_table && user
1747 })
1748 }
1749
1750 /// Collect the rowids of every **currently-live** row across all user table
1751 /// b-trees (the roots listed in `sqlite_master`). The forensic carver uses
1752 /// this to drop any carved "deleted" record whose rowid is in fact still live
1753 /// — a stale copy of a live row can linger in free space after a b-tree
1754 /// rebalance moved the row to another page, and reporting it as deleted would
1755 /// be a false positive. Rowid collection ignores the column count (the rowid
1756 /// is in the cell prefix), so it works even when a schema row is malformed.
1757 ///
1758 /// Bounded and panic-free: unreadable schema or a malformed b-tree yields a
1759 /// partial (possibly empty) set rather than an error.
1760 #[must_use]
1761 pub fn live_rowids(&self) -> std::collections::BTreeSet<i64> {
1762 let mut ids = std::collections::BTreeSet::new();
1763 let Ok(schema) = self.read_table(1, 5) else {
1764 return ids; // cov:unreachable: a validly-opened DB has a readable page-1 schema
1765 };
1766 for row in schema {
1767 // sqlite_master row: (type, name, tbl_name, rootpage, sql).
1768 let is_table = matches!(row.values.first(), Some(Value::Text(t)) if t == "table");
1769 if !is_table {
1770 continue; // cov:unreachable: the test fixtures' schemas hold only table rows
1771 }
1772 let Some(Value::Integer(root)) = row.values.get(3) else {
1773 continue; // cov:unreachable: a 'table' schema row always has an integer rootpage
1774 };
1775 let Ok(root) = u32::try_from(*root) else {
1776 continue; // cov:unreachable: a real rootpage is a small positive page number
1777 };
1778 let mut seen = std::collections::BTreeSet::new();
1779 self.collect_rowids(root, &mut ids, &mut seen);
1780 }
1781 ids
1782 }
1783
1784 /// Collect every **currently-live** row's decoded column values, keyed by
1785 /// rowid, across all user table b-trees. This is the value-aware companion to
1786 /// [`Database::live_rowids`]: the forensic carver uses it to tell a stale
1787 /// rebalance copy (same rowid AND same values → drop) from a deleted prior
1788 /// version (same rowid but DIFFERENT values → recover, e.g. an edited message
1789 /// or a changed amount).
1790 ///
1791 /// Column values are decoded by inferring the column count from each live
1792 /// cell's own serial-type array (the same self-describing record format the
1793 /// carver uses), so no schema column count is required. Best-effort,
1794 /// bounded, and panic-free: a malformed b-tree yields a partial map.
1795 #[must_use]
1796 pub fn live_rows(&self) -> std::collections::BTreeMap<i64, Vec<Value>> {
1797 let mut rows = std::collections::BTreeMap::new();
1798 let Ok(schema) = self.read_table(1, 5) else {
1799 return rows; // cov:unreachable: a validly-opened DB has a readable page-1 schema
1800 };
1801 for row in schema {
1802 let is_table = matches!(row.values.first(), Some(Value::Text(t)) if t == "table");
1803 if !is_table {
1804 continue; // cov:unreachable: the test fixtures' schemas hold only table rows
1805 }
1806 let Some(Value::Integer(root)) = row.values.get(3) else {
1807 continue; // cov:unreachable: a 'table' schema row always has an integer rootpage
1808 };
1809 let Ok(root) = u32::try_from(*root) else {
1810 continue; // cov:unreachable: a real rootpage is a small positive page number
1811 };
1812 let mut seen = std::collections::BTreeSet::new();
1813 self.collect_rows(root, &mut rows, &mut seen);
1814 }
1815 rows
1816 }
1817
1818 /// Decode every **currently-live** `sqlite_master` row (the schema table
1819 /// rooted at page 1) into its column values: `(type, name, tbl_name,
1820 /// rootpage, sql)`. This is the schema-table companion to
1821 /// [`Database::live_rows`], which collects only USER-table b-trees and so
1822 /// never sees the schema rows themselves.
1823 ///
1824 /// The forensic carver folds these into the same value-based live set it uses
1825 /// to drop stale copies of live user rows: a record carved from a materialized
1826 /// page 1 whose values equal a CURRENT schema row is the live schema entry
1827 /// re-surfaced (drop it), whereas a genuinely-deleted PRIOR schema version has
1828 /// different values (e.g. an old `CREATE TABLE`) and is still recovered.
1829 ///
1830 /// Best-effort, bounded, and panic-free: an unreadable schema yields an empty
1831 /// vector rather than an error.
1832 #[must_use]
1833 pub fn live_schema_rows(&self) -> Vec<Vec<Value>> {
1834 match self.read_table(1, 5) {
1835 Ok(rows) => rows.into_iter().map(|row| row.values).collect(),
1836 Err(_) => Vec::new(), // cov:unreachable: a validly-opened DB has a readable page-1 schema
1837 }
1838 }
1839
1840 /// Every live (schema-present) **user** table, as [`attribution::LiveTable`]:
1841 /// name, rootpage, parsed column names (or `None` when low-confidence), and
1842 /// declared column affinities. Internal `sqlite_*` tables are excluded.
1843 ///
1844 /// The forensic attribution step uses this to know each table's real column
1845 /// names (Tier-1) and its shape signature (Tier-2). Best-effort, bounded,
1846 /// panic-free: an unreadable schema yields an empty vector.
1847 #[must_use]
1848 pub fn live_tables(&self) -> Vec<attribution::LiveTable> {
1849 let mut tables = Vec::new();
1850 let Ok(schema) = self.read_table(1, 5) else {
1851 return tables; // cov:unreachable: a validly-opened DB has a readable page-1 schema
1852 };
1853 for row in schema {
1854 // sqlite_master row: (type, name, tbl_name, rootpage, sql).
1855 let is_table = matches!(row.values.first(), Some(Value::Text(t)) if t == "table");
1856 if !is_table {
1857 continue;
1858 }
1859 let Some(Value::Text(name)) = row.values.get(1) else {
1860 continue; // cov:unreachable: a 'table' schema row always has a TEXT name
1861 };
1862 if name.starts_with("sqlite_") {
1863 continue;
1864 }
1865 let Some(Value::Integer(root)) = row.values.get(3) else {
1866 continue; // cov:unreachable: a 'table' schema row always has an integer rootpage
1867 };
1868 let Ok(rootpage) = u32::try_from(*root) else {
1869 continue; // cov:unreachable: a real rootpage is a small positive page number
1870 };
1871 // The CREATE TABLE statement (column 5). A non-TEXT/absent sql is
1872 // possible on a damaged schema — degrade to no parsed columns.
1873 let sql = match row.values.get(4) {
1874 Some(Value::Text(s)) => s.as_str(),
1875 _ => "", // cov:unreachable: a 'table' schema row carries its CREATE TABLE sql
1876 };
1877 let defs = attribution::column_defs(sql);
1878 let affinities = defs.as_ref().map_or_else(Vec::new, |d| {
1879 d.iter()
1880 .map(|(_, ty)| attribution::column_affinity(ty))
1881 .collect()
1882 });
1883 // Only trust parsed names; if parsing failed, the caller uses c0..cN.
1884 let column_names = defs.map(|d| d.into_iter().map(|(n, _)| n).collect());
1885 tables.push(attribution::LiveTable {
1886 name: name.clone(),
1887 rootpage,
1888 column_names,
1889 affinities,
1890 create_sql: sql.to_string(),
1891 });
1892 }
1893 tables
1894 }
1895
1896 /// The live `sqlite_master` as a `name -> CREATE SQL` map for every **user**
1897 /// table (internal `sqlite_*` tables excluded) — the CURRENT-schema half of
1898 /// the Detector-B sidecar schema-change comparison
1899 /// (`docs/design/drop-recreate-attribution.md`).
1900 ///
1901 /// Reads the same page-1 schema b-tree as [`Self::live_tables`] but keeps the
1902 /// raw CREATE SQL text (not just parsed columns), so a caller can compare the
1903 /// verbatim schema against a sidecar's prior `sqlite_master`. Best-effort,
1904 /// bounded, panic-free: an unreadable schema yields an empty map.
1905 #[must_use]
1906 pub fn schema_sql(&self) -> std::collections::BTreeMap<String, String> {
1907 let mut out = std::collections::BTreeMap::new();
1908 let Ok(schema) = self.read_table(1, 5) else {
1909 return out; // cov:unreachable: a validly-opened DB has a readable page-1 schema
1910 };
1911 for row in schema {
1912 schema_sql_insert(&mut out, &row.values);
1913 }
1914 out
1915 }
1916
1917 /// Per-table, per-rowid VERSION HISTORY reconstructed from this database's WAL
1918 /// temporal model (or just the live view when no `-wal` is present).
1919 ///
1920 /// See [`row_history`] for the full model. Walks each salt epoch's commit
1921 /// snapshots in commit order, then the final live view, and emits — per rowid
1922 /// — the sequence of distinct record values it held (insert / update / delete /
1923 /// reinsert), with evidence-based [`row_history::ViewState`] and NO timestamps.
1924 /// Degrades cleanly to live-only history when [`Database::wal_timeline`] is
1925 /// `None`. `WITHOUT ROWID` tables are recorded with `without_rowid = true` and
1926 /// no versions (they have no rowid to key a history on).
1927 #[must_use]
1928 pub fn row_histories(&self) -> Vec<row_history::TableHistory> {
1929 use row_history::{RowView, VersionOrigin};
1930
1931 // Live tables: name, header columns, live rows, and a WITHOUT ROWID flag
1932 // read from the live schema (a WITHOUT ROWID table has no rowid history).
1933 let live_dumps = self.live_table_rows();
1934 let without_rowid = self.live_without_rowid_map();
1935 // WITHOUT ROWID tables' live rows (index-b-tree read); folded into each
1936 // matching history below (§1.4).
1937 let wr_rows = self.without_rowid_table_rows();
1938
1939 // Per table, build the chronological views: each WAL commit snapshot (in
1940 // epoch order, commit_seq = per-epoch ordinal) then the final live view.
1941 let mut histories = Vec::with_capacity(live_dumps.len());
1942 for dump in live_dumps {
1943 let wr = without_rowid.get(&dump.name).copied().unwrap_or(false);
1944 let mut views: Vec<RowView> = Vec::new();
1945
1946 // Historical views from the WAL timeline, if any.
1947 if let Some(timeline) = self.wal_timeline() {
1948 // commit_seq is monotonic WITHIN a salt epoch only — count per
1949 // segment, never one global sequence spanning a salt reset.
1950 let mut seq_in_segment: std::collections::BTreeMap<WalSegmentId, u32> =
1951 std::collections::BTreeMap::new();
1952 for snapshot in timeline.commit_snapshots() {
1953 let seg = snapshot.id().segment;
1954 let seq = seq_in_segment.entry(seg).or_insert(0);
1955 let commit_seq = *seq;
1956 *seq += 1;
1957
1958 // Resolve THIS table from the snapshot's OWN schema (a rootpage
1959 // can be reused by a different table across commits).
1960 let snap_tables = snapshot.tables();
1961 let Some(st) = snap_tables.iter().find(|t| t.name == dump.name) else {
1962 continue; // table did not exist at this commit
1963 };
1964 if st.without_rowid {
1965 continue; // no rowid history for a WITHOUT ROWID table
1966 }
1967 // schema_known: the snapshot's CREATE TABLE parsed to columns.
1968 let schema_known = !st.columns.is_empty();
1969 let rows = match snapshot.read_table(st.rootpage, st.columns.len()) {
1970 Ok(rows) => rows.into_iter().collect(),
1971 // An unreadable historical b-tree contributes no rows but
1972 // must not abort the whole history.
1973 Err(_) => std::collections::BTreeMap::new(),
1974 };
1975 views.push(RowView {
1976 commit_seq: Some(commit_seq),
1977 is_final: false,
1978 checksum_valid: snapshot.checksum_valid(),
1979 schema_known,
1980 origin: VersionOrigin::Commit(snapshot.id()),
1981 rows,
1982 });
1983 }
1984 }
1985
1986 // The final live view (current on-disk ⊕ WAL state).
1987 let live_rows: std::collections::BTreeMap<i64, Vec<Value>> = dump
1988 .rows
1989 .iter()
1990 .map(|r| (r.rowid, r.values.clone()))
1991 .collect();
1992 views.push(RowView {
1993 commit_seq: None,
1994 is_final: true,
1995 checksum_valid: true,
1996 schema_known: true,
1997 origin: VersionOrigin::Live,
1998 rows: live_rows,
1999 });
2000
2001 let mut history = row_history::table_history(dump.name, dump.column_names, wr, &views);
2002 // A WITHOUT ROWID table has no rowid version history, but its live rows
2003 // live in the index b-tree (§1.4) — read them so the carve output shows
2004 // the table's data, not just a "not version-tracked" note.
2005 if wr {
2006 if let Some(t) = wr_rows.iter().find(|t| t.name == history.table) {
2007 history.without_rowid_rows.clone_from(&t.rows);
2008 }
2009 }
2010 histories.push(history);
2011 }
2012 histories
2013 }
2014
2015 /// Map each live user table's name to whether it is a `WITHOUT ROWID` table,
2016 /// read from the live `sqlite_master` schema. Best-effort and panic-free.
2017 fn live_without_rowid_map(&self) -> std::collections::BTreeMap<String, bool> {
2018 let mut map = std::collections::BTreeMap::new();
2019 let Ok(schema) = self.read_table(1, 5) else {
2020 return map; // cov:unreachable: a validly-opened DB has a readable page-1 schema
2021 };
2022 for row in schema {
2023 let is_table = matches!(row.values.first(), Some(Value::Text(t)) if t == "table");
2024 if !is_table {
2025 continue;
2026 }
2027 let Some(Value::Text(name)) = row.values.get(1) else {
2028 continue; // cov:unreachable: a 'table' schema row has a TEXT name
2029 };
2030 if name.starts_with("sqlite_") {
2031 continue;
2032 }
2033 let sql = match row.values.get(4) {
2034 Some(Value::Text(s)) => s.as_str(),
2035 _ => "", // cov:unreachable: a 'table' schema row carries its CREATE TABLE sql
2036 };
2037 map.insert(name.clone(), without_rowid_sql(sql));
2038 }
2039 map
2040 }
2041
2042 /// The `sqlite_sequence` table `SQLite` maintains for `AUTOINCREMENT` tables,
2043 /// as `name → seq` — `seq` being the highest rowid ever assigned to that table
2044 /// (its monotonic INSERT high-water mark).
2045 ///
2046 /// `sqlite_sequence` exists **only** once at least one `AUTOINCREMENT` table
2047 /// has been created; a database with none returns an **empty** map (never a
2048 /// fabricated `seq = 0`), so a caller can distinguish "no high-water mark" from
2049 /// "high-water mark of 0". Best-effort, bounded, panic-free: an unreadable
2050 /// `sqlite_sequence` b-tree, or a malformed row, is omitted rather than
2051 /// erroring. Note `sqlite_sequence` is a mutable user table — `seq` tracks the
2052 /// INSERT high-water mark, not live rowid assignment — so this is a forensic
2053 /// HINT input, not proof of any row's provenance.
2054 #[must_use]
2055 pub fn sqlite_sequence(&self) -> std::collections::BTreeMap<String, i64> {
2056 let mut map = std::collections::BTreeMap::new();
2057 let Ok(schema) = self.read_table(1, 5) else {
2058 return map; // cov:unreachable: a validly-opened DB has a readable page-1 schema
2059 };
2060 // Locate the sqlite_sequence table's rootpage from the schema.
2061 let mut rootpage: Option<u32> = None;
2062 for row in &schema {
2063 let is_table = matches!(row.values.first(), Some(Value::Text(t)) if t == "table");
2064 if !is_table {
2065 continue;
2066 }
2067 if !matches!(row.values.get(1), Some(Value::Text(n)) if n == "sqlite_sequence") {
2068 continue;
2069 }
2070 if let Some(Value::Integer(root)) = row.values.get(3) {
2071 rootpage = u32::try_from(*root).ok();
2072 }
2073 break;
2074 }
2075 let Some(root) = rootpage else {
2076 return map; // no AUTOINCREMENT table ⟹ no sqlite_sequence ⟹ empty
2077 };
2078 let Ok(rows) = self.read_table(root, 2) else {
2079 return map; // cov:unreachable: a present sqlite_sequence has a readable b-tree
2080 };
2081 for row in rows {
2082 // sqlite_sequence row: (name TEXT, seq INTEGER). A malformed row (wrong
2083 // types) is skipped — never a fabricated entry.
2084 let (Some(Value::Text(name)), Some(Value::Integer(seq))) =
2085 (row.values.first(), row.values.get(1))
2086 else {
2087 continue;
2088 };
2089 map.insert(name.clone(), *seq);
2090 }
2091 map
2092 }
2093
2094 /// Dump every live user table for export: name, header columns, and all live
2095 /// rows in rowid order. The base layer the combined live + recovered workbook
2096 /// is built over.
2097 ///
2098 /// For each [`Database::live_tables`] entry, the b-tree is read via
2099 /// [`Database::read_table`] (so rows arrive in ascending-rowid b-tree order).
2100 /// The header is the table's **real** column names when the schema parse was
2101 /// confident, otherwise generic `c0..c{N-1}` sized to the widest row — a
2102 /// header is always present and never a fabricated name. Best-effort and
2103 /// panic-free: a table whose b-tree is unreadable contributes an empty row set
2104 /// rather than erroring.
2105 #[must_use]
2106 pub fn live_table_rows(&self) -> Vec<LiveTableDump> {
2107 self.live_tables()
2108 .into_iter()
2109 .map(|table| {
2110 // `read_table`'s column_count drives only the INTEGER PRIMARY KEY
2111 // rowid-alias rule; use the declared arity when known, else 0
2112 // (no alias substitution) so a low-confidence schema still dumps.
2113 let declared = table.column_names.as_ref().map_or(0, Vec::len);
2114 let rows = self
2115 .read_table(table.rootpage, declared)
2116 .unwrap_or_default();
2117 let widest = rows.iter().map(|r| r.values.len()).max().unwrap_or(0);
2118 let column_names = match table.column_names {
2119 // Confident schema parse: use the table's real column names.
2120 // Live rows legitimately omit trailing NULLs, so `widest` may
2121 // be < declared — the real header still governs (a recovered
2122 // row pads/truncates to it).
2123 Some(names) => names,
2124 // Low-confidence parse (malformed/unparseable CREATE TABLE):
2125 // generic header sized to the widest row, never a fabricated
2126 // real name. This is the schema-damage robustness guard.
2127 None => (0..widest).map(|i| format!("c{i}")).collect(),
2128 };
2129 LiveTableDump {
2130 name: table.name,
2131 column_names,
2132 rows,
2133 }
2134 })
2135 .collect()
2136 }
2137
2138 /// A map from each **allocated** page that belongs to a live table's b-tree
2139 /// to that table's name. Built by walking every live table's b-tree page set
2140 /// from its rootpage (interior + leaf pages). A page carved as Tier-1
2141 /// in-page residue resolves to its owning table through this map.
2142 ///
2143 /// Best-effort and bounded, mirroring `live_rowids`'s b-tree walk: a
2144 /// malformed b-tree contributes fewer entries rather than erroring.
2145 #[must_use]
2146 pub fn page_to_table_map(&self) -> std::collections::BTreeMap<u32, String> {
2147 let mut map = std::collections::BTreeMap::new();
2148 for table in self.live_tables() {
2149 let mut pages = std::collections::BTreeSet::new();
2150 let mut visited = 0usize;
2151 self.collect_pages(table.rootpage, &mut pages, &mut visited);
2152 for page in pages {
2153 map.insert(page, table.name.clone());
2154 }
2155 }
2156 map
2157 }
2158
2159 /// Walk the table b-tree rooted at `page`, inserting every page it visits
2160 /// (interior + leaf) into `pages`. Best-effort and bounded, mirroring
2161 /// `collect_rowids`.
2162 fn collect_pages(
2163 &self,
2164 page: u32,
2165 pages: &mut std::collections::BTreeSet<u32>,
2166 visited: &mut usize,
2167 ) {
2168 *visited += 1;
2169 if *visited > MAX_PAGES_PER_WALK {
2170 return; // cov:unreachable: test b-trees are far below the 1M-page cap
2171 }
2172 if page == 0 || !pages.insert(page) {
2173 return; // page 0 sentinel, or already visited (cycle guard)
2174 }
2175 let Ok(slice) = self.page_slice(page) else {
2176 return; // cov:unreachable: schema rootpages and their children are in range
2177 };
2178 let slice = &*slice; // PageBytes -> &[u8]; body below is source-agnostic
2179 let hdr_off = if page == 1 { SQLITE_HEADER_SIZE } else { 0 };
2180 let Some(&page_type) = slice.get(hdr_off) else {
2181 return; // cov:unreachable: a full page slice always has its header byte
2182 };
2183 if page_type != 0x05 {
2184 return; // leaf (0x0d) or non-interior: no children to descend
2185 }
2186 let cell_count = be_u16(slice, hdr_off + 3) as usize;
2187 let cell_ptr_array = hdr_off + 12;
2188 for i in 0..cell_count {
2189 let cell_off = be_u16(slice, cell_ptr_array + i * 2) as usize;
2190 let child = be_u32(slice, cell_off);
2191 self.collect_pages(child, pages, visited);
2192 }
2193 let right = be_u32(slice, hdr_off + 8);
2194 self.collect_pages(right, pages, visited);
2195 }
2196
2197 /// Walk the table b-tree rooted at `page`, decoding every live leaf cell's
2198 /// values (column count inferred per cell) into `rows` keyed by rowid.
2199 /// Best-effort and bounded, mirroring [`Database::collect_rowids`].
2200 fn collect_rows(
2201 &self,
2202 page: u32,
2203 rows: &mut std::collections::BTreeMap<i64, Vec<Value>>,
2204 seen: &mut std::collections::BTreeSet<u32>,
2205 ) {
2206 // Visit each page at most once. A manipulated interior left-child or
2207 // right-most pointer (anti-forensic corpus category 12) can point back
2208 // into an already-visited page, and a counter-only guard would still
2209 // recurse a million frames deep before stopping — a stack overflow. The
2210 // visited-set bounds recursion DEPTH to the number of distinct pages,
2211 // mirroring `collect_pages`'s cycle guard.
2212 if page == 0 || seen.len() > MAX_PAGES_PER_WALK || !seen.insert(page) {
2213 return;
2214 }
2215 let Ok(slice) = self.page_slice(page) else {
2216 return; // cov:unreachable: schema rootpages and their children are in range
2217 };
2218 let slice = &*slice; // PageBytes -> &[u8]; body below is source-agnostic
2219 let hdr_off = if page == 1 { SQLITE_HEADER_SIZE } else { 0 };
2220 let Some(&page_type) = slice.get(hdr_off) else {
2221 return; // cov:unreachable: a full page slice always has its header byte
2222 };
2223 let cell_count = be_u16(slice, hdr_off + 3) as usize;
2224 match page_type {
2225 0x0d => {
2226 let cell_ptr_array = hdr_off + 8;
2227 for i in 0..cell_count {
2228 let cell_off = be_u16(slice, cell_ptr_array + i * 2) as usize;
2229 // Decode the live cell with an inferred column count; on any
2230 // parse hiccup (e.g. a table narrower than MIN_INFERRED_COLUMNS),
2231 // fall back to the rowid alone (empty values) so the row is
2232 // still known to be live.
2233 if let Some(cell) =
2234 try_carve_cell_at(slice, cell_off, None, self.header.text_encoding)
2235 {
2236 rows.insert(cell.rowid, cell.values);
2237 } else if let Some(rowid) = live_cell_rowid(slice, cell_off) {
2238 rows.entry(rowid).or_default(); // cov:unreachable: a >=2-col live cell always decodes above
2239 }
2240 }
2241 }
2242 0x05 => {
2243 let cell_ptr_array = hdr_off + 12;
2244 for i in 0..cell_count {
2245 let cell_off = be_u16(slice, cell_ptr_array + i * 2) as usize;
2246 let child = be_u32(slice, cell_off);
2247 self.collect_rows(child, rows, seen);
2248 }
2249 let right = be_u32(slice, hdr_off + 8);
2250 self.collect_rows(right, rows, seen);
2251 }
2252 _ => {} // cov:unreachable: a table b-tree root/child is leaf (0x0d) or interior (0x05)
2253 }
2254 }
2255
2256 /// Walk the table b-tree rooted at `page`, inserting every live leaf cell's
2257 /// rowid into `ids`. Best-effort and bounded: a malformed/cyclic structure
2258 /// stops the walk rather than erroring or looping.
2259 fn collect_rowids(
2260 &self,
2261 page: u32,
2262 ids: &mut std::collections::BTreeSet<i64>,
2263 seen: &mut std::collections::BTreeSet<u32>,
2264 ) {
2265 // Visit each page at most once (see `collect_rows` for the rationale): a
2266 // manipulated child pointer that revisits a page must not recurse
2267 // unboundedly. The visited-set bounds recursion depth to distinct pages.
2268 if page == 0 || seen.len() > MAX_PAGES_PER_WALK || !seen.insert(page) {
2269 return;
2270 }
2271 let Ok(slice) = self.page_slice(page) else {
2272 return; // cov:unreachable: schema rootpages and their children are in range
2273 };
2274 let slice = &*slice; // PageBytes -> &[u8]; body below is source-agnostic
2275 let hdr_off = if page == 1 { SQLITE_HEADER_SIZE } else { 0 };
2276 let Some(&page_type) = slice.get(hdr_off) else {
2277 return; // cov:unreachable: a full page slice always has its header byte
2278 };
2279 let cell_count = be_u16(slice, hdr_off + 3) as usize;
2280 match page_type {
2281 0x0d => {
2282 let cell_ptr_array = hdr_off + 8;
2283 for i in 0..cell_count {
2284 let cell_off = be_u16(slice, cell_ptr_array + i * 2) as usize;
2285 if let Some(rowid) = live_cell_rowid(slice, cell_off) {
2286 ids.insert(rowid);
2287 }
2288 }
2289 }
2290 0x05 => {
2291 let cell_ptr_array = hdr_off + 12;
2292 for i in 0..cell_count {
2293 let cell_off = be_u16(slice, cell_ptr_array + i * 2) as usize;
2294 let child = be_u32(slice, cell_off);
2295 self.collect_rowids(child, ids, seen);
2296 }
2297 let right = be_u32(slice, hdr_off + 8);
2298 self.collect_rowids(right, ids, seen);
2299 }
2300 _ => {} // cov:unreachable: a table b-tree root/child is leaf (0x0d) or interior (0x05)
2301 }
2302 }
2303
2304 /// Walk a single table b-tree rooted at `root_page` (1-based) and collect
2305 /// every leaf row as typed values. `column_count` is the table's declared
2306 /// column count, used to apply the `INTEGER PRIMARY KEY` rowid-alias rule.
2307 ///
2308 /// Shares ONE b-tree/overflow walk with the snapshot-scoped read
2309 /// ([`CommitSnapshot::read_table`]) via an internal page-source abstraction, so
2310 /// the live and historical paths can never diverge.
2311 pub fn read_table(&self, root_page: u32, column_count: usize) -> Result<Vec<Row>, Error> {
2312 read_table_via(self, root_page, column_count)
2313 }
2314
2315 /// Bytes of the 1-based `page` number, or `PageOutOfRange`.
2316 ///
2317 /// When a WAL overlay is in effect and holds a committed version of this
2318 /// page, the overlaid bytes are returned in preference to the main file —
2319 /// this is what makes a table walk see the WAL-applied view. The main file
2320 /// is never mutated.
2321 fn page_slice(&self, page: u32) -> Result<PageBytes<'_>, Error> {
2322 if page == 0 {
2323 return Err(Error::PageOutOfRange(0));
2324 }
2325 if let Some(wal) = &self.wal {
2326 if let Some(overlaid) = wal.pages.get(&page) {
2327 return Ok(PageBytes::Borrowed(overlaid.as_slice()));
2328 }
2329 }
2330 self.source
2331 .page(page, self.header.page_size as usize)
2332 .ok_or(Error::PageOutOfRange(page))
2333 }
2334}
2335
2336/// A source of page images for the shared b-tree / overflow walk — the seam that
2337/// lets the live [`Database`] (main file ⊕ WAL overlay) and a historical
2338/// [`CommitSnapshot`] (materialized commit pages) share ONE table-read
2339/// implementation instead of forking parallel copies.
2340///
2341/// All page numbers are 1-based. Implementations resolve page 1 with the
2342/// 100-byte file header in place (so the walk reads the b-tree header at offset
2343/// `SQLITE_HEADER_SIZE` for page 1, 0 otherwise).
2344trait PageSource {
2345 /// The 1-based `page`'s full image, or `None` for page 0 / out of range.
2346 fn page(&self, page: u32) -> Option<PageBytes<'_>>;
2347 /// Usable bytes per page (`page_size` − reserved-space), for the overflow and
2348 /// local-payload computations.
2349 fn usable(&self) -> usize;
2350 /// The highest valid 1-based page number (the cycle/over-range bound).
2351 fn page_bound(&self) -> u32;
2352 /// The database text encoding, for decoding TEXT values.
2353 fn encoding(&self) -> TextEncoding;
2354}
2355
2356impl PageSource for Database {
2357 fn page(&self, page: u32) -> Option<PageBytes<'_>> {
2358 self.page_slice(page).ok()
2359 }
2360 fn usable(&self) -> usize {
2361 self.header.usable_size() as usize
2362 }
2363 fn page_bound(&self) -> u32 {
2364 self.file_page_count()
2365 }
2366 fn encoding(&self) -> TextEncoding {
2367 self.header.text_encoding
2368 }
2369}
2370
2371impl PageSource for CommitSnapshot {
2372 fn page(&self, page: u32) -> Option<PageBytes<'_>> {
2373 self.overlaid
2374 .get(&page)
2375 .map(|v| PageBytes::Borrowed(v.as_slice()))
2376 }
2377 fn usable(&self) -> usize {
2378 self.usable as usize
2379 }
2380 fn page_bound(&self) -> u32 {
2381 // The committed page count at this snapshot — the cycle/over-range bound
2382 // for an overflow walk over the snapshot's materialized pages.
2383 self.id.db_size_after_commit
2384 }
2385 fn encoding(&self) -> TextEncoding {
2386 // Text encoding from the snapshot's OWN page-1 header (byte 56), so a
2387 // historical read decodes TEXT per the encoding as of this commit.
2388 self.overlaid
2389 .get(&1)
2390 .map(|p| match be_u32(p, TEXT_ENCODING_OFFSET) {
2391 2 => TextEncoding::Utf16Le,
2392 3 => TextEncoding::Utf16Be,
2393 _ => TextEncoding::Utf8,
2394 })
2395 .unwrap_or_default()
2396 }
2397}
2398
2399/// Walk a single table b-tree rooted at `root_page` over any [`PageSource`],
2400/// collecting every leaf row as typed values. The one implementation shared by
2401/// the live and snapshot-scoped reads.
2402/// Insert a `sqlite_master` row's `name -> CREATE SQL` into `out` when the row is
2403/// a **user** table (`type='table'`, name not `sqlite_*`). Shared by
2404/// [`Database::schema_sql`] and [`PriorSnapshot::schema_sql`] so the live and
2405/// prior reads classify schema rows identically. A row that is not a user-table
2406/// row (an index/view/trigger, an internal table, or a malformed row) is skipped.
2407fn schema_sql_insert(out: &mut std::collections::BTreeMap<String, String>, values: &[Value]) {
2408 // sqlite_master row: (type, name, tbl_name, rootpage, sql).
2409 let is_table = matches!(values.first(), Some(Value::Text(t)) if t == "table");
2410 if !is_table {
2411 return;
2412 }
2413 let Some(Value::Text(name)) = values.get(1) else {
2414 return; // cov:unreachable: a 'table' schema row has a TEXT name
2415 };
2416 if name.starts_with("sqlite_") {
2417 return;
2418 }
2419 let sql = match values.get(4) {
2420 Some(Value::Text(s)) => s.clone(),
2421 _ => String::new(), // cov:unreachable: a 'table' schema row carries its CREATE TABLE sql
2422 };
2423 out.insert(name.clone(), sql);
2424}
2425
2426fn read_table_via(
2427 src: &dyn PageSource,
2428 root_page: u32,
2429 column_count: usize,
2430) -> Result<Vec<Row>, Error> {
2431 let mut rows = Vec::new();
2432 let mut seen = std::collections::BTreeSet::new();
2433 walk_table_page(src, root_page, column_count, &mut rows, &mut seen)?;
2434 Ok(rows)
2435}
2436
2437fn walk_table_page(
2438 src: &dyn PageSource,
2439 page: u32,
2440 column_count: usize,
2441 rows: &mut Vec<Row>,
2442 seen: &mut std::collections::BTreeSet<u32>,
2443) -> Result<(), Error> {
2444 // Visit each page at most once. A manipulated interior child pointer
2445 // (anti-forensic corpus category 12) can revisit an already-walked page; a
2446 // counter-only guard still recurses up to the cap deep before stopping,
2447 // overflowing the stack. The visited-set bounds recursion DEPTH to the
2448 // number of distinct pages. A revisited page is silently skipped (Ok) so a
2449 // crafted cycle yields the partial-but-valid rows already collected rather
2450 // than an error.
2451 if seen.len() > MAX_PAGES_PER_WALK {
2452 return Err(Error::TooManyPages);
2453 }
2454 if !seen.insert(page) {
2455 return Ok(());
2456 }
2457 let slice = src.page(page).ok_or(Error::PageOutOfRange(page))?;
2458 let slice = &*slice;
2459
2460 // Page 1 carries the 100-byte file header before its b-tree header.
2461 let hdr_off = if page == 1 { SQLITE_HEADER_SIZE } else { 0 };
2462
2463 let page_type = *slice.get(hdr_off).ok_or(Error::TruncatedCell)?;
2464 let cell_count = be_u16(slice, hdr_off + 3) as usize;
2465
2466 match page_type {
2467 0x0d => read_leaf_cells(src, slice, hdr_off, cell_count, column_count, rows),
2468 0x05 => {
2469 // Interior table page: 12-byte header; cell = 4-byte child ptr +
2470 // varint key. Recurse into every child plus the right-most ptr.
2471 let cell_ptr_array = hdr_off + 12;
2472 for i in 0..cell_count {
2473 let p = cell_ptr_array + i * 2;
2474 let cell_off = be_u16(slice, p) as usize;
2475 let child = be_u32(slice, cell_off);
2476 walk_table_page(src, child, column_count, rows, seen)?;
2477 }
2478 let right = be_u32(slice, hdr_off + 8);
2479 walk_table_page(src, right, column_count, rows, seen)
2480 }
2481 other => Err(Error::NotATablePage(other)),
2482 }
2483}
2484
2485fn read_leaf_cells(
2486 src: &dyn PageSource,
2487 slice: &[u8],
2488 hdr_off: usize,
2489 cell_count: usize,
2490 column_count: usize,
2491 rows: &mut Vec<Row>,
2492) -> Result<(), Error> {
2493 let cell_ptr_array = hdr_off + 8; // leaf b-tree header is 8 bytes
2494 for i in 0..cell_count {
2495 let p = cell_ptr_array + i * 2;
2496 let cell_off = be_u16(slice, p) as usize;
2497 let row = decode_leaf_cell(src, slice, cell_off, column_count)?;
2498 rows.push(row);
2499 }
2500 Ok(())
2501}
2502
2503/// Decode one table-leaf cell at `off` into a [`Row`], reassembling the payload
2504/// from its overflow-page chain (resolved through the SAME [`PageSource`]) when
2505/// it spills past the leaf page.
2506fn decode_leaf_cell(
2507 src: &dyn PageSource,
2508 slice: &[u8],
2509 off: usize,
2510 column_count: usize,
2511) -> Result<Row, Error> {
2512 let (payload_len, n1) = read_varint(slice, off)?;
2513 let (rowid, n2) = read_varint(slice, off + n1)?;
2514 let payload_start = off + n1 + n2;
2515 let total = usize::try_from(payload_len).map_err(|_| Error::TruncatedCell)?;
2516
2517 let usable = src.usable();
2518 let local = local_payload_len(total, usable);
2519
2520 let payload = if local >= total {
2521 // Whole payload is on the leaf page (no spill).
2522 slice
2523 .get(payload_start..payload_start + total)
2524 .ok_or(Error::TruncatedCell)?
2525 .to_vec()
2526 } else {
2527 // Spilled: `local` bytes on the leaf, then a 4-byte overflow page
2528 // pointer, then the remainder follows the overflow chain.
2529 let head = slice
2530 .get(payload_start..payload_start + local)
2531 .ok_or(Error::TruncatedCell)?;
2532 let first_overflow = be_u32(slice, payload_start + local);
2533 // Cap the pre-allocation against the untrusted payload length: a payload
2534 // cannot exceed the bytes the file can physically supply — the `local`
2535 // bytes on the leaf plus the content bytes of every page reachable
2536 // through the overflow chain (`per_page * page_bound`). A crafted cell
2537 // that declares a multi-exabyte `payload_len` would otherwise reach
2538 // `Vec::with_capacity(total)` and abort the process with an allocation
2539 // bomb. This is the same over-range condition `read_overflow_chain`
2540 // rejects, pulled ahead of the allocation.
2541 let per_page = usable.saturating_sub(4);
2542 let max_overflow = per_page.saturating_mul(src.page_bound() as usize);
2543 let max_payload = local.saturating_add(max_overflow);
2544 if total > max_payload {
2545 return Err(Error::MalformedOverflow);
2546 }
2547 let mut buf = Vec::with_capacity(total);
2548 buf.extend_from_slice(head);
2549 read_overflow_chain(src, first_overflow, total - local, &mut buf)?;
2550 buf
2551 };
2552
2553 let values = decode_record(&payload, column_count, rowid, src.encoding())?;
2554 Ok(Row { rowid, values })
2555}
2556
2557/// Follow an overflow-page chain starting at `first` (1-based page number) over
2558/// a [`PageSource`], appending up to `remaining` payload bytes to `buf`. Each
2559/// overflow page is a 4-byte big-endian "next page" pointer (0 ends the chain)
2560/// followed by up to `usable - 4` content bytes.
2561///
2562/// Bounded against cyclic/over-long chains via [`Error::MalformedOverflow`].
2563fn read_overflow_chain(
2564 src: &dyn PageSource,
2565 first: u32,
2566 mut remaining: usize,
2567 buf: &mut Vec<u8>,
2568) -> Result<(), Error> {
2569 let usable = src.usable();
2570 let per_page = usable.saturating_sub(4);
2571 if per_page == 0 {
2572 return Err(Error::MalformedOverflow);
2573 }
2574 let total_pages = src.page_bound();
2575 let cap = total_pages as usize + 1;
2576
2577 let mut page = first;
2578 let mut visited = 0usize;
2579 while remaining > 0 {
2580 if page == 0 || page > total_pages {
2581 return Err(Error::MalformedOverflow);
2582 }
2583 visited += 1;
2584 if visited > cap {
2585 return Err(Error::MalformedOverflow);
2586 }
2587 let slice = src.page(page).ok_or(Error::PageOutOfRange(page))?;
2588 let slice = &*slice;
2589 let next = be_u32(slice, 0);
2590 let take = remaining.min(per_page);
2591 let chunk = slice.get(4..4 + take).ok_or(Error::TruncatedCell)?;
2592 buf.extend_from_slice(chunk);
2593 remaining -= take;
2594 page = next;
2595 }
2596 Ok(())
2597}
2598
2599/// Number of payload bytes stored locally on a table-leaf page for a record of
2600/// `total` bytes, given the page's `usable` size (file-format §1.6 overflow
2601/// rule). When the return value equals `total`, the record does not spill.
2602pub(crate) fn local_payload_len(total: usize, usable: usize) -> usize {
2603 let max_local = usable - 35; // X: largest payload kept entirely local
2604 if total <= max_local {
2605 return total;
2606 }
2607 let min_local = (usable - 12) * 32 / 255 - 23; // M
2608 let k = min_local + (total - min_local) % (usable - 4);
2609 if k <= max_local {
2610 k
2611 } else {
2612 min_local
2613 }
2614}
2615
2616impl WalOverlay {
2617 /// Parse a `-wal` sidecar into the newest committed page versions.
2618 ///
2619 /// Returns `Ok(None)` when `wal` is absent of a usable header / has no
2620 /// frames (a no-op overlay). Iterates frames in file order, accumulating the
2621 /// page data of each frame whose salt matches the WAL header; on reaching a
2622 /// COMMIT frame (`db_size_after_commit != 0`) the accumulated pages are
2623 /// promoted into the committed snapshot. Frames after the last commit are
2624 /// uncommitted and dropped. Bounds-checked and breadth-capped against a
2625 /// crafted WAL (a frame whose declared page data runs past the file ends the
2626 /// scan rather than panicking).
2627 fn parse(wal: &[u8], page_size: u32) -> Result<Option<Self>, Error> {
2628 use forensicnomicon::sqlite::{SQLITE_WAL_FRAME_HEADER_SIZE, SQLITE_WAL_HEADER_SIZE};
2629
2630 // No header → no overlay (treat a too-short WAL as empty, not an error:
2631 // a missing/zero-length sidecar is normal and must not fail the open).
2632 let Some(hdr) = wal.get(..SQLITE_WAL_HEADER_SIZE) else {
2633 return Ok(None);
2634 };
2635 let magic = be_u32(hdr, 0);
2636 if magic != WAL_MAGIC_BE && magic != WAL_MAGIC_LE {
2637 return Ok(None);
2638 }
2639 // The WAL records its own page size (offset 8); trust the DB header's
2640 // page size but require agreement to avoid mis-slicing frames.
2641 let wal_page_size = be_u32(hdr, 8);
2642 if wal_page_size != page_size {
2643 return Ok(None);
2644 }
2645 // WAL header layout (file-format §4.1): salt-1 at offset 16, salt-2 at
2646 // offset 20 (the two checksum words follow at 24 and 28).
2647 let salt1 = be_u32(hdr, 16);
2648 let salt2 = be_u32(hdr, 20);
2649
2650 let ps = page_size as usize;
2651 let frame_stride = SQLITE_WAL_FRAME_HEADER_SIZE + ps;
2652
2653 let mut committed: std::collections::BTreeMap<u32, Vec<u8>> =
2654 std::collections::BTreeMap::new();
2655 let mut pending: std::collections::BTreeMap<u32, Vec<u8>> =
2656 std::collections::BTreeMap::new();
2657 // Every committed frame's page image (file order), and the pending frames
2658 // not yet promoted by a COMMIT. Mirrors the page promotion above so
2659 // uncommitted trailing frames are dropped from BOTH the view and the carve.
2660 let mut frames: Vec<WalFramePage> = Vec::new();
2661 let mut pending_frames: Vec<WalFramePage> = Vec::new();
2662
2663 let mut off = SQLITE_WAL_HEADER_SIZE;
2664 // One frame per page in the file is the natural breadth cap; allow a
2665 // generous multiple for repeated rewrites, but keep it bounded.
2666 let max_frames = wal.len() / frame_stride + 1;
2667 let mut frame_no = 0usize;
2668
2669 while let Some(frame) = wal.get(off..off + frame_stride) {
2670 frame_no += 1;
2671 if frame_no > max_frames {
2672 break; // cov:unreachable: the slice walk already bounds frame_no
2673 }
2674 let page_no = be_u32(frame, 0);
2675 let db_size = be_u32(frame, 4);
2676 let fsalt1 = be_u32(frame, 8);
2677 let fsalt2 = be_u32(frame, 12);
2678 // A frame from a different checkpoint generation (salt mismatch) is
2679 // stale residue, not part of this WAL's live content — stop here.
2680 if fsalt1 != salt1 || fsalt2 != salt2 {
2681 break;
2682 }
2683 if page_no == 0 {
2684 break; // malformed frame; stop rather than mis-index
2685 }
2686 let data = frame
2687 .get(SQLITE_WAL_FRAME_HEADER_SIZE..)
2688 .ok_or(Error::TruncatedCell)?;
2689 pending.insert(page_no, data.to_vec());
2690 let is_commit = db_size != 0;
2691 pending_frames.push(WalFramePage {
2692 frame_index: frame_no - 1, // 0-based file order
2693 page_no,
2694 salt1,
2695 salt2,
2696 is_commit,
2697 page: data.to_vec(),
2698 });
2699
2700 if is_commit {
2701 // COMMIT frame: promote everything pending into the snapshot AND
2702 // into the committed frame list (keeping every frame, not just the
2703 // newest version of each page).
2704 for (p, d) in std::mem::take(&mut pending) {
2705 committed.insert(p, d);
2706 }
2707 frames.append(&mut pending_frames);
2708 }
2709 off += frame_stride;
2710 }
2711
2712 if committed.is_empty() {
2713 Ok(None)
2714 } else {
2715 Ok(Some(WalOverlay {
2716 pages: committed,
2717 frames,
2718 raw: wal.to_vec(),
2719 }))
2720 }
2721 }
2722}
2723
2724// ===========================================================================
2725// Bespoke, format-exact WAL temporal model (task #55)
2726// ===========================================================================
2727//
2728// A `-wal` sidecar is NOT an open-ended event log. It is a BOUNDED SEGMENT under a
2729// single salt epoch: every live frame shares the WAL header's (salt1, salt2). A
2730// checkpoint reset renumbers frames and rolls the salts — a DISCONTINUITY, not a
2731// continuation. The only materializable database states are the COMMIT snapshots:
2732// the replay of all valid frames up to a commit frame. A frame BETWEEN commits is
2733// not independently materializable, so it is never surfaced as a snapshot. Tails
2734// past the last commit, or after a salt reset, are WAL residue — forensic leads,
2735// never committed history.
2736//
2737// This model is self-contained in sqlite-core. The future state-history-forensic
2738// [H] adapter attaches at the seam exposed here (WalLsn + CohortTopology +
2739// `checksums_are_tamper_evident`), but sqlite-core does NOT depend on it.
2740
2741/// Cap on the number of salt segments and frames the timeline parser will walk on a
2742/// crafted `-wal`, bounding work against an attacker-supplied file. A real WAL holds
2743/// one segment with at most a few frames per database page.
2744const MAX_WAL_SEGMENTS: usize = 1024;
2745
2746/// Identity of one salt epoch within a `-wal` file: its 0-based segment ordinal.
2747/// A fresh segment begins at file start and after every checkpoint salt reset.
2748#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)]
2749pub struct WalSegmentId(pub usize);
2750
2751/// One salt epoch of a `-wal` file — a single bounded segment.
2752///
2753/// A `-wal` is a bounded segment, not an open-ended log: every live frame here shares
2754/// `(salt1, salt2)`. A checkpoint reset (salt change + frame renumber) starts a NEW
2755/// `WalSegment`; it is a discontinuity, never another epoch of the same segment.
2756#[derive(Debug, Clone, PartialEq, Eq)]
2757pub struct WalSegment {
2758 /// This segment's ordinal within the WAL (0 = the segment at file start).
2759 pub id: WalSegmentId,
2760 /// WAL salt-1 (checkpoint generation), shared by every frame in the segment.
2761 pub salt1: u32,
2762 /// WAL salt-2 (checkpoint generation), shared by every frame in the segment.
2763 pub salt2: u32,
2764 /// Page size declared by the segment's frames (bytes).
2765 pub page_size: u32,
2766 /// Number of frames belonging to this segment.
2767 pub frame_count: usize,
2768 /// The checkpoint sequence number recorded in the WAL header (offset 12). For a
2769 /// segment discovered after a reset within the same file this is the header's
2770 /// value; per-segment sequence is otherwise not separately recorded.
2771 pub checkpoint_seq: u32,
2772}
2773
2774/// Address of a materializable database state: the replay of all valid frames up to
2775/// a COMMIT frame. `CommitId = (segment, commit_frame_index, db_size_after_commit)`.
2776#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)]
2777pub struct CommitId {
2778 /// The salt segment this commit belongs to.
2779 pub segment: WalSegmentId,
2780 /// 0-based file-order index of the COMMIT frame within the segment.
2781 pub commit_frame_index: usize,
2782 /// `db_size_after_commit` recorded in the COMMIT frame header — the database's
2783 /// page count once this commit is materialized.
2784 pub db_size_after_commit: u32,
2785}
2786
2787/// The salt-qualified log-sequence identity of a WAL position — the seam the future
2788/// `state-history-forensic` `[H]` adapter maps onto `LsnKind::SqliteWal`.
2789///
2790/// A bare `frame_index` is meaningless across checkpoint resets (frames renumber), so
2791/// ordering is ALWAYS qualified by `(salt1, salt2)`. The adapter must reconstruct
2792/// `LsnKind::SqliteWal { salt1, salt2, frame_index }` from exactly this triple — never
2793/// from a bare index.
2794#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)]
2795pub struct WalLsn {
2796 /// Salt-1 of the owning segment (checkpoint generation).
2797 pub salt1: u32,
2798 /// Salt-2 of the owning segment (checkpoint generation).
2799 pub salt2: u32,
2800 /// 0-based frame index within that segment.
2801 pub frame_index: usize,
2802}
2803
2804/// Topology of the temporal cohort the WAL exposes — the shape the `[H]` adapter maps
2805/// to `state-history-forensic::CohortTopology`.
2806#[derive(Debug, Clone, Copy, PartialEq, Eq)]
2807pub enum CohortTopology {
2808 /// A single salt epoch: the commit snapshots form one linearly-ordered chain.
2809 LinearSegment,
2810 /// Multiple salt epochs (checkpoint resets) with no replay continuity between
2811 /// them — each segment is linear internally but the segments are disconnected.
2812 Disconnected,
2813}
2814
2815/// One page's image at a particular [`CommitSnapshot`].
2816#[derive(Debug, Clone, PartialEq, Eq)]
2817pub struct CommittedPageVersion {
2818 /// 1-based database page number.
2819 pub page_no: u32,
2820 /// The page's full image (`page_size` bytes) as of this commit.
2821 pub bytes: Vec<u8>,
2822}
2823
2824/// A materializable database state: the replay of all valid frames up to a COMMIT.
2825///
2826/// This is the ONLY independently-materializable WAL state. `page_version` resolves a
2827/// page to its image as of this commit (the newest frame ≤ this commit that rewrote
2828/// the page, else the acquired base image). A frame between commits is never a
2829/// snapshot.
2830#[derive(Debug, Clone, PartialEq, Eq)]
2831pub struct CommitSnapshot {
2832 id: CommitId,
2833 /// Salt-1 of the owning segment, carried so [`CommitSnapshot::lsn`] is
2834 /// self-contained without a back-reference to the segment.
2835 salt1: u32,
2836 /// Salt-2 of the owning segment.
2837 salt2: u32,
2838 /// The materialized page images at this commit: base image overlaid with every
2839 /// committed frame up to and including this commit (newest version per page),
2840 /// capped to `db_size_after_commit` pages. `page_version` reads from this map.
2841 overlaid: std::collections::BTreeMap<u32, Vec<u8>>,
2842 /// Whether the whole frame chain up to and including this commit's COMMIT frame
2843 /// passed the WAL cumulative checksum (file-format §4.2). `false` marks a commit
2844 /// the salt+commit-marker admission would otherwise accept but whose checksum
2845 /// chain is broken (post-reset residue, tampering, or corruption) — kept, not
2846 /// dropped, so the forensic layer can label it.
2847 checksum_valid: bool,
2848 /// Usable bytes per page (`page_size` − reserved), parsed from the snapshot's
2849 /// OWN page-1 header, so a snapshot-scoped read uses the reserved-space value
2850 /// as of this commit rather than the live database's.
2851 usable: u32,
2852}
2853
2854/// One user table as of a [`CommitSnapshot`] — its schema parsed from the
2855/// snapshot's OWN materialized page 1, NOT from the live database. A rootpage can
2856/// be dropped and reused by a different table across commits, so reading the
2857/// schema from the snapshot is the only correct way to interpret its b-trees.
2858#[derive(Debug, Clone, PartialEq, Eq)]
2859pub struct SnapshotTable {
2860 /// The table's `sqlite_master.name`.
2861 pub name: String,
2862 /// 1-based root page of the table's b-tree as of this commit.
2863 pub rootpage: u32,
2864 /// Parsed column names from the table's `CREATE TABLE`, in declared order.
2865 /// Empty when the schema SQL could not be parsed with confidence.
2866 pub columns: Vec<String>,
2867 /// Whether this is a `WITHOUT ROWID` table (file-format §2.4). Such a table
2868 /// uses an INDEX b-tree with no rowid key, so the rowid-based snapshot read
2869 /// does not apply — flagged so a caller never mis-reads it as a rowid table.
2870 pub without_rowid: bool,
2871}
2872
2873/// Whether a `CREATE TABLE` statement declares a `WITHOUT ROWID` table
2874/// (file-format §2.4). Detection keys off the trailing `WITHOUT ROWID` clause,
2875/// case-insensitively and tolerant of internal whitespace, while ignoring any
2876/// occurrence inside a quoted identifier/string so a column literally named
2877/// "without rowid" is not a false positive.
2878/// A `CREATE TABLE` statement with quoted spans removed and whitespace collapsed,
2879/// uppercased — so a clause search sees only unquoted SQL tokens. Strips
2880/// `'...'` / `"..."` / `` `...` `` / `[...]` spans (the four `SQLite` identifier /
2881/// string quotings) exactly as the clause detectors require, so the keyword
2882/// appearing inside a quoted identifier or string literal can never false-match.
2883fn normalized_unquoted_sql(create_sql: &str) -> String {
2884 let bytes = create_sql.as_bytes();
2885 let mut unquoted = String::with_capacity(create_sql.len());
2886 let mut quote: Option<u8> = None;
2887 for &c in bytes {
2888 match quote {
2889 Some(q) => {
2890 if c == q {
2891 quote = None;
2892 }
2893 }
2894 None => match c {
2895 b'\'' | b'"' | b'`' => quote = Some(c),
2896 b'[' => quote = Some(b']'),
2897 _ => unquoted.push(c as char),
2898 },
2899 }
2900 }
2901 unquoted
2902 .split_whitespace()
2903 .collect::<Vec<_>>()
2904 .join(" ")
2905 .to_ascii_uppercase()
2906}
2907
2908fn without_rowid_sql(create_sql: &str) -> bool {
2909 // Look for the clause as a discrete token sequence, ignoring quoted spans and
2910 // case/whitespace (file-format §2.4).
2911 normalized_unquoted_sql(create_sql).contains("WITHOUT ROWID")
2912}
2913
2914/// Whether `create_sql` declares an ordinary rowid table with an
2915/// `INTEGER PRIMARY KEY AUTOINCREMENT` column — the only form for which `SQLite`
2916/// maintains a monotonic `sqlite_sequence` high-water mark.
2917///
2918/// Per the file format, `AUTOINCREMENT` is valid **only** immediately after
2919/// `INTEGER PRIMARY KEY`, and **never** on a `WITHOUT ROWID` table (which has no
2920/// rowid to auto-increment). So this is true iff the normalized, unquoted CREATE
2921/// text contains the exact token run `INTEGER PRIMARY KEY AUTOINCREMENT` and does
2922/// NOT carry the `WITHOUT ROWID` clause. Quoted identifiers / string literals /
2923/// comments are stripped first (mirroring `without_rowid_sql`), so a column
2924/// merely named `"autoincrement"`, or the keyword inside a string, never matches.
2925///
2926/// This is a HINT input only: a true result means the table has an AUTOINCREMENT
2927/// high-water mark the forensic layer can reconcile against, not that any
2928/// particular row predates the current instance.
2929#[must_use]
2930pub fn is_autoincrement(create_sql: &str) -> bool {
2931 let normalized = normalized_unquoted_sql(create_sql);
2932 normalized.contains("INTEGER PRIMARY KEY AUTOINCREMENT")
2933 && !normalized.contains("WITHOUT ROWID")
2934}
2935
2936impl CommitSnapshot {
2937 /// This snapshot's [`CommitId`].
2938 #[must_use]
2939 pub fn id(&self) -> CommitId {
2940 self.id
2941 }
2942
2943 /// The database page count once this commit is materialized.
2944 #[must_use]
2945 pub fn db_size_after_commit(&self) -> u32 {
2946 self.id.db_size_after_commit
2947 }
2948
2949 /// Whether the WAL frame chain up to and including this commit's COMMIT frame
2950 /// validated against the cumulative WAL checksum (file-format §4.2).
2951 ///
2952 /// `true` is the spec-conformant case: every frame's stored `(checksum1,
2953 /// checksum2)` equalled the running checksum advanced over the frame's first
2954 /// 8 header bytes plus its full page data, seeded from the WAL header
2955 /// checksum. `false` means the chain broke at or before this commit — the
2956 /// salt + commit-marker admission accepted it, but it is residue (post-reset
2957 /// leftover, tampering, or corruption). Such a commit is deliberately KEPT
2958 /// (not dropped) so the forensic layer can mark it; a consumer that wants only
2959 /// trustworthy state filters on this flag.
2960 #[must_use]
2961 pub fn checksum_valid(&self) -> bool {
2962 self.checksum_valid
2963 }
2964
2965 /// The salt-qualified [`WalLsn`] of this commit (the `[H]` adapter seam).
2966 #[must_use]
2967 pub fn lsn(&self) -> WalLsn {
2968 WalLsn {
2969 salt1: self.salt1,
2970 salt2: self.salt2,
2971 frame_index: self.id.commit_frame_index,
2972 }
2973 }
2974
2975 /// The 1-based page numbers this commit materialized (base ∪ committed frames
2976 /// up to this commit, capped to `db_size_after_commit`), ascending.
2977 ///
2978 /// The carve-at-snapshot primitive iterates these to drive the carving
2979 /// primitives over each page image, WITHOUT assuming the pages form a
2980 /// contiguous `1..=db_size` range (a truncating commit or a sparse base image
2981 /// can leave gaps). Every returned page resolves via [`Self::page_version`].
2982 #[must_use]
2983 pub fn page_numbers(&self) -> Vec<u32> {
2984 self.overlaid.keys().copied().collect()
2985 }
2986
2987 /// The image of `page_no` as of this commit, or `None` for a page beyond the
2988 /// committed database size that the WAL never rewrote.
2989 #[must_use]
2990 pub fn page_version(&self, page_no: u32) -> Option<CommittedPageVersion> {
2991 let bytes = self.overlaid.get(&page_no)?.clone();
2992 Some(CommittedPageVersion { page_no, bytes })
2993 }
2994
2995 /// The user tables AS OF this commit, parsed from the snapshot's OWN page 1
2996 /// (the `sqlite_master` b-tree), NOT from the live database.
2997 ///
2998 /// A rootpage can be dropped and reused by a different table across commits,
2999 /// so the schema MUST come from the snapshot itself — reading today's live
3000 /// schema would mis-attribute a historical b-tree. Returns one
3001 /// [`SnapshotTable`] per `type='table'` row whose name is not an internal
3002 /// `sqlite_*` table, carrying its rootpage, parsed column names, and a
3003 /// `WITHOUT ROWID` flag (file-format §2.4). Best-effort and panic-free: an
3004 /// unreadable page-1 schema yields an empty vector.
3005 #[must_use]
3006 pub fn tables(&self) -> Vec<SnapshotTable> {
3007 // sqlite_master is a 5-column table rooted at page 1:
3008 // (type, name, tbl_name, rootpage, sql). Walk it through THIS snapshot's
3009 // pages via the shared b-tree reader.
3010 let Ok(schema) = read_table_via(self, 1, 5) else {
3011 return Vec::new(); // cov:unreachable: a committed snapshot has a readable page 1
3012 };
3013 let mut out = Vec::new();
3014 for row in schema {
3015 let is_table = matches!(row.values.first(), Some(Value::Text(t)) if t == "table");
3016 if !is_table {
3017 continue;
3018 }
3019 let Some(Value::Text(name)) = row.values.get(1) else {
3020 continue; // cov:unreachable: a 'table' schema row has a TEXT name
3021 };
3022 if name.starts_with("sqlite_") {
3023 continue;
3024 }
3025 let Some(Value::Integer(root)) = row.values.get(3) else {
3026 continue; // cov:unreachable: a 'table' schema row has an integer rootpage
3027 };
3028 let Ok(rootpage) = u32::try_from(*root) else {
3029 continue; // cov:unreachable: a real rootpage is a small positive page number
3030 };
3031 let sql = match row.values.get(4) {
3032 Some(Value::Text(s)) => s.as_str(),
3033 _ => "", // cov:unreachable: a 'table' schema row carries its CREATE TABLE sql
3034 };
3035 let columns = attribution::column_names(sql).unwrap_or_default();
3036 out.push(SnapshotTable {
3037 name: name.clone(),
3038 rootpage,
3039 columns,
3040 without_rowid: without_rowid_sql(sql),
3041 });
3042 }
3043 out
3044 }
3045
3046 /// Read every row of the table b-tree rooted at `rootpage` AS OF this commit,
3047 /// resolving overflow chains through the snapshot's OWN materialized pages, in
3048 /// rowid order.
3049 ///
3050 /// This is the snapshot-scoped counterpart to [`Database::read_table`]: it
3051 /// shares the SAME b-tree/overflow walk via an internal page-source
3052 /// abstraction, so a large row
3053 /// decodes with the page content as of this commit (not stale/future content
3054 /// the live view would supply). `column_count` drives only the
3055 /// `INTEGER PRIMARY KEY` rowid-alias rule (pass the table's declared arity,
3056 /// e.g. `SnapshotTable::columns.len()`). Returns `(rowid, values)` per row.
3057 ///
3058 /// Bounded and panic-free on hostile input, exactly as the live path: a
3059 /// cyclic/over-deep b-tree or overflow chain surfaces a typed [`Error`] rather
3060 /// than looping or panicking.
3061 pub fn read_table(
3062 &self,
3063 rootpage: u32,
3064 column_count: usize,
3065 ) -> Result<Vec<(i64, Vec<Value>)>, Error> {
3066 let rows = read_table_via(self, rootpage, column_count)?;
3067 Ok(rows.into_iter().map(|r| (r.rowid, r.values)).collect())
3068 }
3069}
3070
3071/// A page-level delta between two materialized states.
3072#[derive(Debug, Clone, PartialEq, Eq)]
3073pub struct WalDiff {
3074 changed: Vec<u32>,
3075}
3076
3077impl WalDiff {
3078 /// The 1-based page numbers whose bytes differ between the two states, ascending.
3079 #[must_use]
3080 pub fn changed_pages(&self) -> &[u32] {
3081 &self.changed
3082 }
3083}
3084
3085/// A stale WAL tail surfaced for forensics — NOT committed history.
3086///
3087/// Frames past the last COMMIT of a segment, frames after a salt reset that cannot be
3088/// replayed into the current segment, or a header/page-size break: all are residue.
3089/// The examiner weighs them; they are never part of a consistent snapshot.
3090#[derive(Debug, Clone, PartialEq, Eq)]
3091pub struct WalResidue {
3092 /// The segment the residue trails (the segment whose last commit it follows).
3093 pub segment: WalSegmentId,
3094 /// 0-based frame index (within the file) of the first residual frame.
3095 pub first_frame_index: usize,
3096 /// Number of residual frames.
3097 pub frame_count: usize,
3098 /// Why these frames are residue rather than committed history.
3099 pub reason: ResidueReason,
3100}
3101
3102/// Why a WAL tail is [`WalResidue`] (an invalidated-frame candidate), not history.
3103#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3104pub enum ResidueReason {
3105 /// Frames written after the segment's last COMMIT (uncommitted tail).
3106 BeyondLastCommit,
3107 /// Frames whose salt no longer matches the segment header (post-reset residue).
3108 SaltReset,
3109}
3110
3111/// Validation tier a WAL has cleared — strictly increasing assurance.
3112///
3113/// `PhysicalValidation` < `CommitValidation` < `ReplaySafe`. The timeline reports the
3114/// highest tier reached; a page-size mismatch never even produces a timeline (it is a
3115/// hard stop at parse, surfaced as [`WalValidationError::PageSizeMismatch`]).
3116#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)]
3117pub enum MaterializationSafety {
3118 /// Header magic / format / page-size / salts / frame boundaries are well-formed,
3119 /// but no committed snapshot was found (nothing to replay).
3120 PhysicalValidated,
3121 /// A last valid commit and committed frame ranges were established, but the
3122 /// read-only replay overlay was not (or could not be) built.
3123 CommitValidated,
3124 /// A read-only replay overlay to the last commit is available — safe to
3125 /// materialize without mutating either file.
3126 ReplaySafe,
3127}
3128
3129/// A WAL that cannot be admitted to the timeline at all (physical-validation hard
3130/// stops). Distinct from "no committed snapshot", which is a valid empty timeline.
3131#[derive(Debug, Clone, PartialEq, Eq)]
3132pub enum WalValidationError {
3133 /// The `-wal` is shorter than its 32-byte header, or carries the wrong magic.
3134 BadMagic,
3135 /// The WAL header's page size disagrees with the DB header's — a HARD STOP, since
3136 /// every frame would be mis-sliced. `db` and `wal` are the two declared sizes.
3137 PageSizeMismatch { db: u32, wal: u32 },
3138 /// The main database header itself failed to parse.
3139 Header(Error),
3140}
3141
3142/// The bespoke, format-exact temporal model of a `-wal` sidecar.
3143///
3144/// Enumerates the salt segments, the materializable [`CommitSnapshot`]s within them
3145/// (CommitId-addressable), and the [`WalResidue`] tails. Materialize a snapshot's page
3146/// images via [`CommitSnapshot::page_version`]; diff the acquired base against the last
3147/// valid commit via [`WalTimeline::diff_base_to_last_commit`].
3148#[derive(Debug, Clone, PartialEq, Eq)]
3149pub struct WalTimeline {
3150 page_size: u32,
3151 base_pages: std::collections::BTreeMap<u32, Vec<u8>>,
3152 segments: Vec<WalSegment>,
3153 snapshots: Vec<CommitSnapshot>,
3154 residue: Vec<WalResidue>,
3155 safety: MaterializationSafety,
3156}
3157
3158impl WalTimeline {
3159 /// Physical-validation tier: header magic + format check.
3160 ///
3161 /// Parses `bytes` (the acquired main DB) and `wal` (the `-wal` sidecar) into the
3162 /// segmented temporal model. A page-size mismatch between the DB header and the
3163 /// WAL header is a HARD STOP; a bad/short header is [`WalValidationError::BadMagic`].
3164 fn parse(bytes: &[u8], wal: &[u8], page_size: u32) -> Result<Self, WalValidationError> {
3165 use forensicnomicon::sqlite::{SQLITE_WAL_FRAME_HEADER_SIZE, SQLITE_WAL_HEADER_SIZE};
3166
3167 // --- PhysicalValidation: header magic / format / page-size / salts -------
3168 let hdr = wal
3169 .get(..SQLITE_WAL_HEADER_SIZE)
3170 .ok_or(WalValidationError::BadMagic)?;
3171 let magic = be_u32(hdr, 0);
3172 if magic != WAL_MAGIC_BE && magic != WAL_MAGIC_LE {
3173 return Err(WalValidationError::BadMagic);
3174 }
3175 let wal_page_size = be_u32(hdr, 8);
3176 if wal_page_size != page_size {
3177 return Err(WalValidationError::PageSizeMismatch {
3178 db: page_size,
3179 wal: wal_page_size,
3180 });
3181 }
3182 let checkpoint_seq = be_u32(hdr, 12);
3183 let mut salt1 = be_u32(hdr, 16);
3184 let mut salt2 = be_u32(hdr, 20);
3185
3186 // Checksum chain seed (file-format §4.2): the running (s0, s1) starts from
3187 // the WAL header's stored checksum (bytes 24..32, always big-endian),
3188 // which is itself the checksum over the first 24 header bytes. The word
3189 // endianness for advancing over frames comes from the magic. `from_magic`
3190 // cannot return None here — the magic was admitted above.
3191 let endian = WalChecksumEndian::from_magic(magic).unwrap_or(WalChecksumEndian::Big);
3192 let header_s0 = be_u32(hdr, 24);
3193 let header_s1 = be_u32(hdr, 28);
3194 // Per-segment running checksum state and whether the chain is still valid.
3195 let mut run_s0 = header_s0;
3196 let mut run_s1 = header_s1;
3197 let mut chain_valid = true;
3198
3199 let ps = page_size as usize;
3200 let frame_stride = SQLITE_WAL_FRAME_HEADER_SIZE + ps;
3201
3202 // The acquired main DB image: the pre-WAL base for replay within the current
3203 // validated segment (NOT "epoch 0" — just the base each commit overlays onto).
3204 let mut base_pages: std::collections::BTreeMap<u32, Vec<u8>> =
3205 std::collections::BTreeMap::new();
3206 // `chunks_exact` yields only whole pages (infallible by construction — no
3207 // out-of-bounds slice to guard); cap at `u32::MAX` pages so the 1-based page
3208 // number never overflows on a pathologically large image.
3209 for (idx, page) in bytes
3210 .chunks_exact(ps)
3211 .take(u32::MAX as usize - 1)
3212 .enumerate()
3213 {
3214 let pno = idx as u32 + 1; // 1-based page number
3215 base_pages.insert(pno, page.to_vec());
3216 }
3217
3218 let mut segments: Vec<WalSegment> = Vec::new();
3219 let mut snapshots: Vec<CommitSnapshot> = Vec::new();
3220 let mut residue: Vec<WalResidue> = Vec::new();
3221
3222 // Per-segment running state.
3223 let mut seg_ordinal = 0usize;
3224 let mut seg_frame_count = 0usize;
3225 // Cumulative newest-page map across all COMMITTED frames of the segment, so a
3226 // snapshot's `overlaid` is base ∪ committed-up-to-this-commit.
3227 let mut committed_pages: std::collections::BTreeMap<u32, Vec<u8>> = base_pages.clone();
3228 let mut pending: std::collections::BTreeMap<u32, Vec<u8>> =
3229 std::collections::BTreeMap::new();
3230 let mut last_commit_global_frame: Option<usize> = None;
3231 let mut uncommitted_tail_start: Option<usize> = None;
3232
3233 let mut off = SQLITE_WAL_HEADER_SIZE;
3234 let max_frames = wal.len() / frame_stride + 1;
3235 let mut frame_no = 0usize;
3236
3237 while let Some(frame) = wal.get(off..off + frame_stride) {
3238 if frame_no >= max_frames {
3239 break; // cov:unreachable: the slice walk already bounds frame_no
3240 }
3241 let page_no = be_u32(frame, 0);
3242 let db_size = be_u32(frame, 4);
3243 let fsalt1 = be_u32(frame, 8);
3244 let fsalt2 = be_u32(frame, 12);
3245
3246 // A salt change opens a NEW segment (checkpoint reset = discontinuity).
3247 // Anything between the prior segment's last commit and here is residue.
3248 if fsalt1 != salt1 || fsalt2 != salt2 {
3249 if segments.len() >= MAX_WAL_SEGMENTS {
3250 break; // cov:unreachable: real WALs hold far fewer than 1024 salt epochs
3251 }
3252 // Close the current segment, recording its residue tail (if any).
3253 Self::close_segment(
3254 &mut segments,
3255 &mut residue,
3256 WalSegmentId(seg_ordinal),
3257 salt1,
3258 salt2,
3259 page_size,
3260 checkpoint_seq,
3261 seg_frame_count,
3262 uncommitted_tail_start,
3263 );
3264 // Begin the next segment under the new salts. Its base for replay is
3265 // the prior committed view (a checkpoint would have flushed it, but on
3266 // a forensic image we keep what we can replay).
3267 seg_ordinal += 1;
3268 salt1 = fsalt1;
3269 salt2 = fsalt2;
3270 seg_frame_count = 0;
3271 pending.clear();
3272 uncommitted_tail_start = None;
3273 // The post-reset frames replay onto the latest committed view.
3274 // committed_pages carries forward.
3275 // The checksum chain for a post-reset segment threads from a WAL
3276 // header we do NOT hold (the new generation's own 32-byte header
3277 // was overwritten), so its frames cannot be validated against our
3278 // seed. Mark the chain broken for this segment: its commits are
3279 // checksum-residue, surfaced for forensics but not trusted.
3280 chain_valid = false;
3281 }
3282
3283 if page_no == 0 {
3284 break; // malformed frame; stop rather than mis-index
3285 }
3286 let data = match frame.get(SQLITE_WAL_FRAME_HEADER_SIZE..) {
3287 Some(d) => d.to_vec(),
3288 None => break, // cov:unreachable: frame slice is exactly frame_stride
3289 };
3290
3291 // Advance the cumulative checksum over this frame (file-format §4.2):
3292 // the first 8 bytes of the frame header (page-no ++ db-size) followed
3293 // by the full page data — NOT the salt/checksum bytes (frame[8..24]).
3294 // Then compare against the frame's stored checksum (frame[16..24], big-
3295 // endian). A mismatch breaks the chain for the rest of the segment.
3296 // Only advance while the chain is still intact (a post-reset segment is
3297 // pre-marked broken and is not re-seedable from our header).
3298 if chain_valid {
3299 let (n0, n1) = wal_checksum(endian, run_s0, run_s1, &frame[0..8]);
3300 let (n0, n1) = wal_checksum(endian, n0, n1, &data);
3301 run_s0 = n0;
3302 run_s1 = n1;
3303 let stored0 = be_u32(frame, 16);
3304 let stored1 = be_u32(frame, 20);
3305 if stored0 != run_s0 || stored1 != run_s1 {
3306 chain_valid = false;
3307 }
3308 }
3309
3310 let frame_index_in_seg = seg_frame_count;
3311 seg_frame_count += 1;
3312 pending.insert(page_no, data);
3313 let is_commit = db_size != 0;
3314
3315 if is_commit {
3316 for (p, d) in std::mem::take(&mut pending) {
3317 committed_pages.insert(p, d);
3318 }
3319 // Drop base/committed pages beyond the committed size so a snapshot
3320 // reflects the database's page count at that commit. `db_size` is
3321 // non-zero here (that is what makes this a COMMIT frame).
3322 committed_pages.retain(|&p, _| p <= db_size);
3323 let id = CommitId {
3324 segment: WalSegmentId(seg_ordinal),
3325 commit_frame_index: frame_index_in_seg,
3326 db_size_after_commit: db_size,
3327 };
3328 let overlaid = committed_pages.clone();
3329 // Usable bytes per page from the snapshot's OWN page-1 header
3330 // (reserved-space byte at offset 20), so a snapshot-scoped read
3331 // honors the reserved value as of this commit. Page 1 is always
3332 // materialized; a missing/short page-1 image degrades to 0 reserved.
3333 let reserved = overlaid
3334 .get(&1)
3335 .and_then(|p| p.get(RESERVED_SPACE_OFFSET).copied())
3336 .unwrap_or(0);
3337 let usable = page_size.saturating_sub(u32::from(reserved));
3338 snapshots.push(CommitSnapshot {
3339 id,
3340 overlaid,
3341 salt1,
3342 salt2,
3343 checksum_valid: chain_valid,
3344 usable,
3345 });
3346 last_commit_global_frame = Some(frame_no);
3347 uncommitted_tail_start = None;
3348 } else if uncommitted_tail_start.is_none() {
3349 uncommitted_tail_start = Some(frame_index_in_seg);
3350 }
3351
3352 frame_no += 1;
3353 off += frame_stride;
3354 }
3355
3356 // Close the final segment (it may have an uncommitted tail).
3357 Self::close_segment(
3358 &mut segments,
3359 &mut residue,
3360 WalSegmentId(seg_ordinal),
3361 salt1,
3362 salt2,
3363 page_size,
3364 checkpoint_seq,
3365 seg_frame_count,
3366 uncommitted_tail_start,
3367 );
3368
3369 let safety = if snapshots.is_empty() {
3370 MaterializationSafety::PhysicalValidated
3371 } else if last_commit_global_frame.is_some() {
3372 MaterializationSafety::ReplaySafe
3373 } else {
3374 MaterializationSafety::CommitValidated // cov:unreachable: a snapshot implies a commit
3375 };
3376
3377 Ok(Self {
3378 page_size,
3379 base_pages,
3380 segments,
3381 snapshots,
3382 residue,
3383 safety,
3384 })
3385 }
3386
3387 #[allow(clippy::too_many_arguments)]
3388 fn close_segment(
3389 segments: &mut Vec<WalSegment>,
3390 residue: &mut Vec<WalResidue>,
3391 id: WalSegmentId,
3392 salt1: u32,
3393 salt2: u32,
3394 page_size: u32,
3395 checkpoint_seq: u32,
3396 frame_count: usize,
3397 uncommitted_tail_start: Option<usize>,
3398 ) {
3399 if frame_count == 0 {
3400 return;
3401 }
3402 segments.push(WalSegment {
3403 id,
3404 salt1,
3405 salt2,
3406 page_size,
3407 frame_count,
3408 checkpoint_seq,
3409 });
3410 if let Some(start) = uncommitted_tail_start {
3411 residue.push(WalResidue {
3412 segment: id,
3413 first_frame_index: start,
3414 frame_count: frame_count - start,
3415 reason: ResidueReason::BeyondLastCommit,
3416 });
3417 }
3418 }
3419
3420 /// The salt segments of this WAL, in file order (one per salt epoch).
3421 #[must_use]
3422 pub fn segments(&self) -> &[WalSegment] {
3423 &self.segments
3424 }
3425
3426 /// Every materializable [`CommitSnapshot`] across all segments, in commit order.
3427 #[must_use]
3428 pub fn commit_snapshots(&self) -> &[CommitSnapshot] {
3429 &self.snapshots
3430 }
3431
3432 /// The stale WAL tails surfaced for forensics (not committed history).
3433 #[must_use]
3434 pub fn residue(&self) -> &[WalResidue] {
3435 &self.residue
3436 }
3437
3438 /// Resolve a [`CommitId`] back to its [`CommitSnapshot`].
3439 #[must_use]
3440 pub fn snapshot_at(&self, id: CommitId) -> Option<&CommitSnapshot> {
3441 self.snapshots.iter().find(|s| s.id == id)
3442 }
3443
3444 /// The highest validation tier this WAL cleared (see [`MaterializationSafety`]).
3445 #[must_use]
3446 pub fn safety(&self) -> MaterializationSafety {
3447 self.safety
3448 }
3449
3450 /// The temporal-cohort topology — `LinearSegment` for one salt epoch, else
3451 /// `Disconnected` across checkpoint resets. The `[H]` adapter maps this onto
3452 /// `state-history-forensic::CohortTopology`.
3453 #[must_use]
3454 pub fn topology(&self) -> CohortTopology {
3455 if self.segments.len() <= 1 {
3456 CohortTopology::LinearSegment
3457 } else {
3458 CohortTopology::Disconnected
3459 }
3460 }
3461
3462 /// Whether the WAL's integrity checks are tamper-EVIDENT. Always `false`: WAL
3463 /// frame checksums are non-cryptographic (corruption detection, not tamper proof),
3464 /// so the `[H]` adapter must record `tamper_resistance = LOW`.
3465 #[must_use]
3466 pub fn checksums_are_tamper_evident(&self) -> bool {
3467 false
3468 }
3469
3470 /// Diff the acquired base image against the last valid commit snapshot, returning
3471 /// the page numbers whose bytes changed. `None` when there is no committed snapshot.
3472 #[must_use]
3473 pub fn diff_base_to_last_commit(&self) -> Option<WalDiff> {
3474 let last = self.snapshots.last()?;
3475 let mut changed = Vec::new();
3476 let mut pages: std::collections::BTreeSet<u32> = std::collections::BTreeSet::new();
3477 pages.extend(self.base_pages.keys().copied());
3478 pages.extend(last.overlaid.keys().copied());
3479 for p in pages {
3480 let base = self.base_pages.get(&p);
3481 let now = last.overlaid.get(&p);
3482 if base != now {
3483 changed.push(p);
3484 }
3485 }
3486 Some(WalDiff { changed })
3487 }
3488
3489 /// The page size (bytes) common to the base image and the WAL frames.
3490 #[must_use]
3491 pub fn page_size(&self) -> u32 {
3492 self.page_size
3493 }
3494
3495 /// Map this WAL timeline onto the canonical `forensicnomicon::history` cohort
3496 /// vocabulary — the `[H]` adapter (#43 / WS-F).
3497 ///
3498 /// Each materializable [`CommitSnapshot`] becomes one `TemporalState<CommitId>`:
3499 /// - **ordering key** — a salt-qualified `LsnKind::SqliteWalFrame` (`frame_seq` is the
3500 /// COMMIT frame index; `commit_seq` is the 0-based commit ordinal within the salt
3501 /// segment). The `(salt1, salt2)` pair keeps the key meaningful across a checkpoint
3502 /// reset, which renumbers frames and rolls the salts.
3503 /// - **clock + safety** — the canonical SQLite-WAL profile, single-sourced from
3504 /// [`forensicnomicon::history::profiles`], so no consumer re-asserts the four
3505 /// classifications locally.
3506 /// - **handle** — the snapshot's [`CommitId`]; resolve it back via [`Self::snapshot_at`].
3507 ///
3508 /// The topology is uniformly `SubJournalCommits`: every state is a committed
3509 /// transaction, and a checkpoint reset is visible as a salt change *inside* the
3510 /// ordering key — there is no separate "disconnected" topology to special-case. The
3511 /// cohort is `PathStable` (a `-wal` belongs to exactly one database path), so the
3512 /// caller supplies the path identity via `artifact`.
3513 #[must_use]
3514 pub fn to_temporal_cohort(
3515 &self,
3516 artifact: forensicnomicon::history::identity::ArtifactRef,
3517 ) -> forensicnomicon::history::cohort::TemporalCohort<CommitId> {
3518 use forensicnomicon::history::cohort::{TemporalCohort, TemporalState};
3519 use forensicnomicon::history::epoch::{CohortTopology, EpochTag, LsnKind};
3520 use forensicnomicon::history::identity::IdentityDiscipline;
3521 use forensicnomicon::history::profiles;
3522
3523 // One canonical profile drives every state's clock + safety — read from
3524 // forensicnomicon, never re-asserted here, so the fleet cannot drift.
3525 let profile = profiles::SourceTemporalProfile::sqlite_wal();
3526 let mut commit_seq_in_segment: std::collections::HashMap<WalSegmentId, u32> =
3527 std::collections::HashMap::new();
3528
3529 let states = self
3530 .snapshots
3531 .iter()
3532 .map(|snap| {
3533 let id = snap.id();
3534 let lsn = snap.lsn();
3535 let seq = commit_seq_in_segment.entry(id.segment).or_insert(0);
3536 let commit_seq = *seq;
3537 *seq += 1;
3538
3539 // Deterministic and collision-free within a cohort: the
3540 // (salt1, salt2, commit_frame_index, db_size_after_commit) quadruple is
3541 // unique per commit state. Packed big-endian into the leading 16 bytes.
3542 let mut tag = [0u8; 32];
3543 tag[0..4].copy_from_slice(&lsn.salt1.to_be_bytes());
3544 tag[4..8].copy_from_slice(&lsn.salt2.to_be_bytes());
3545 tag[8..12].copy_from_slice(&(id.commit_frame_index as u32).to_be_bytes());
3546 tag[12..16].copy_from_slice(&id.db_size_after_commit.to_be_bytes());
3547
3548 TemporalState {
3549 epoch: EpochTag::from_bytes(tag),
3550 ordering_key: Some(LsnKind::SqliteWalFrame {
3551 salt1: lsn.salt1,
3552 salt2: lsn.salt2,
3553 frame_seq: lsn.frame_index as u32,
3554 commit_seq,
3555 }),
3556 wall_time: None,
3557 clock: profile.clock.clone(),
3558 safety: profile.safety.clone(),
3559 handle: id,
3560 }
3561 })
3562 .collect();
3563
3564 TemporalCohort {
3565 artifact,
3566 discipline: IdentityDiscipline::PathStable,
3567 topology: CohortTopology::SubJournalCommits,
3568 states,
3569 }
3570 }
3571}
3572
3573/// Whether a decoded [`Value`] is **distinctive** enough to anchor a Tier-2
3574/// fragment emission (the §3.1 gate): TEXT of ≥ 4 bytes of valid UTF-8 (no
3575/// replacement char), or a REAL. Bare integers (1–8-byte serial patterns),
3576/// NULL, and BLOBs are NOT distinctive alone — a short integer byte-pattern
3577/// coincides far too often in a 4 `KiB` page to serve as identity, so it can ride
3578/// along inside a fragment but never justify emitting one.
3579fn is_distinctive(value: &Value) -> bool {
3580 match value {
3581 Value::Text(t) => t.len() >= 4 && !t.contains('\u{FFFD}'),
3582 Value::Real(_) => true,
3583 Value::Null | Value::Integer(_) | Value::Blob(_) => false,
3584 }
3585}
3586
3587/// The body byte-width of a serial type (file-format §2.1), or `None` for a
3588/// serial value that cannot legally appear in a record body.
3589fn serial_body_len(serial: i64) -> Option<usize> {
3590 match serial {
3591 0 | 8 | 9 | 10 | 11 => Some(0),
3592 1 => Some(1),
3593 2 => Some(2),
3594 3 => Some(3),
3595 4 => Some(4),
3596 5 => Some(6),
3597 6 | 7 => Some(8),
3598 n if n >= 12 => Some(((n - 12) / 2) as usize),
3599 _ => None, // negative serial: impossible
3600 }
3601}
3602
3603/// Byte length of a **live** table-leaf cell at `off`, for computing the byte
3604/// extent the cell occupies (so [`Database::carve_free_regions`] can exclude it).
3605/// Returns `None` if the cell header does not parse in bounds.
3606///
3607/// Mirrors the live cell layout: payload-length varint, rowid varint, then the
3608/// local payload (capped at the spill threshold) plus a 4-byte overflow pointer
3609/// when the payload spills. We only need the on-page footprint, so for a spilled
3610/// cell that is `local + 4` bytes, not the full reassembled payload.
3611fn live_cell_len(buf: &[u8], off: usize, usable: usize) -> Option<usize> {
3612 let (payload_len, n1) = read_varint(buf, off).ok()?;
3613 let (_rowid, n2) = read_varint(buf, off + n1).ok()?;
3614 let total = usize::try_from(payload_len).ok()?;
3615 let local = local_payload_len(total, usable);
3616 let on_page = if local >= total {
3617 n1 + n2 + total
3618 } else {
3619 n1 + n2 + local + 4 // 4-byte first-overflow-page pointer
3620 };
3621 Some(on_page)
3622}
3623
3624/// The rowid of a table-leaf cell at `off` — its 2nd varint (after the
3625/// payload-length varint). `None` if either varint is out of bounds. Used to
3626/// identify a live row even when its full record cannot be decoded.
3627fn live_cell_rowid(buf: &[u8], off: usize) -> Option<i64> {
3628 let (_payload_len, n1) = read_varint(buf, off).ok()?;
3629 let (rowid, _) = read_varint(buf, off + n1).ok()?;
3630 Some(rowid)
3631}
3632
3633/// Given the sorted byte extents of live cells, return the maximal **free**
3634/// (unallocated) spans within `[lo, hi)` — the complement of the live extents.
3635/// These are the only ranges [`Database::carve_free_regions`] scans, so a live
3636/// cell can never be re-surfaced.
3637fn free_regions(live: &[(usize, usize)], lo: usize, hi: usize) -> Vec<(usize, usize)> {
3638 let mut regions = Vec::new();
3639 // An inverted or empty range (lo >= hi) has no free regions. Guard before the
3640 // `clamp(lo, hi)` calls below, which panic when lo > hi (untrusted-input path).
3641 if lo >= hi {
3642 return regions;
3643 }
3644 let mut cursor = lo;
3645 for &(s, e) in live {
3646 let s = s.clamp(lo, hi);
3647 let e = e.clamp(lo, hi);
3648 if s > cursor {
3649 regions.push((cursor, s));
3650 }
3651 if e > cursor {
3652 cursor = e;
3653 }
3654 }
3655 if cursor < hi {
3656 regions.push((cursor, hi));
3657 }
3658 regions
3659}
3660
3661/// Derive a [`FreeblockTemplate`] from the first live cell on a table-leaf page:
3662/// the record's header length, its serial-type array, and the byte width of the
3663/// cell prefix (payload-length + rowid varints) that the freeblock header
3664/// overwrites. Returns `None` when no live cell parses or the prefix is wider
3665/// than the 4 bytes a freeblock header clobbers (the simple template cannot then
3666/// place the surviving serial tail).
3667/// Shared internal walker producing BOTH recovery tiers in one pass so the cell
3668/// and fragment outputs can never diverge: `(full_cells, fragments)`.
3669/// [`Database::reconstruct_freeblock_records`] takes `.0`,
3670/// [`Database::reconstruct_freeblock_fragments`] takes `.1`. A free function (it
3671/// needs no `Database` state — only the page bytes and the page-derived
3672/// template), keeping the two public entry points a thin projection of one walk.
3673fn reconstruct_freeblock_inner(
3674 page_bytes: &[u8],
3675 enc: TextEncoding,
3676) -> (Vec<CarvedCell>, Vec<CellFragment>) {
3677 let mut cells = Vec::new();
3678 let mut frags = Vec::new();
3679 let hdr_off = if page_bytes.starts_with(SQLITE_MAGIC) {
3680 SQLITE_HEADER_SIZE
3681 } else {
3682 0
3683 };
3684 let Some(&page_type) = page_bytes.get(hdr_off) else {
3685 return (cells, frags);
3686 };
3687 if page_type != 0x0d {
3688 return (cells, frags); // only table-leaf pages have freeblock residue
3689 }
3690 let Some(template) = freeblock_template(page_bytes, hdr_off, enc) else {
3691 return (cells, frags);
3692 };
3693
3694 let first_freeblock = be_u16(page_bytes, hdr_off + 1) as usize;
3695 let mut fb = first_freeblock;
3696 let mut walked = 0usize;
3697 let mut visited = std::collections::BTreeSet::new();
3698 while fb != 0 && walked < MAX_FREEBLOCKS_PER_PAGE {
3699 walked += 1;
3700 if !visited.insert(fb) {
3701 break; // cyclic next pointer
3702 }
3703 let next = be_u16(page_bytes, fb) as usize;
3704 let size = be_u16(page_bytes, fb + 2) as usize;
3705 let Some(fb_end) = fb.checked_add(size) else {
3706 break; // cov:unreachable: usize add of two u16-range values
3707 };
3708 if size >= 4 && fb_end <= page_bytes.len() {
3709 if template.known_lead_serials.is_empty() {
3710 // Empty-lead (2-byte-rowid) page: each freeblock is a single freed
3711 // cell whose serial array fully survives. Reconstruct it ONLY if
3712 // the record tiles the freeblock exactly — the precision gate that
3713 // rejects the misaligned runs a loose walk would manufacture.
3714 cells.extend(template.reconstruct_span_exact(page_bytes, fb, fb_end));
3715 } else {
3716 template
3717 .reconstruct_span_tiered(page_bytes, fb, fb_end, false, &mut cells, &mut frags);
3718 }
3719 }
3720 fb = next;
3721 }
3722
3723 let cell_count = be_u16(page_bytes, hdr_off + 3) as usize;
3724 let cptr_end = hdr_off + 8 + cell_count * 2;
3725 let cca = be_u16(page_bytes, hdr_off + 5) as usize;
3726 // The unallocated-gap pass anchors off a surviving forward cell and a known
3727 // leading serial; it is meaningful only for the (non-empty-lead) span-walk
3728 // templates. Empty-lead pages recover solely through the exact-tile chain pass.
3729 if !template.known_lead_serials.is_empty() && cca > cptr_end && cca <= page_bytes.len() {
3730 for anchor_off in cptr_end..cca {
3731 let Some(anchor) =
3732 try_carve_cell_at(page_bytes, anchor_off, Some(template.column_count), enc)
3733 else {
3734 continue;
3735 };
3736 let has_text = anchor
3737 .values
3738 .iter()
3739 .any(|v| matches!(v, Value::Text(t) if !t.is_empty() && !t.contains('\u{FFFD}')));
3740 if !has_text {
3741 continue;
3742 }
3743 let tail_start = anchor.offset + anchor.byte_len;
3744 template
3745 .reconstruct_span_tiered(page_bytes, tail_start, cca, true, &mut cells, &mut frags);
3746 break; // one anchored run per page — the contiguous freed tail
3747 }
3748 }
3749 (cells, frags)
3750}
3751
3752fn freeblock_template(
3753 page_bytes: &[u8],
3754 hdr_off: usize,
3755 enc: TextEncoding,
3756) -> Option<FreeblockTemplate> {
3757 let cell_count = be_u16(page_bytes, hdr_off + 3) as usize;
3758 let cell_ptr_array = hdr_off + 8;
3759 for i in 0..cell_count {
3760 let cell_off = be_u16(page_bytes, cell_ptr_array + i * 2) as usize;
3761 if cell_off == 0 || cell_off >= page_bytes.len() {
3762 continue;
3763 }
3764 // Prefix: payload-length varint, rowid varint.
3765 let Ok((_payload_len, n1)) = read_varint(page_bytes, cell_off) else {
3766 continue; // cov:unreachable: a live cell-pointer addresses an in-bounds prefix
3767 };
3768 let Ok((_rowid, n2)) = read_varint(page_bytes, cell_off + n1) else {
3769 continue; // cov:unreachable: the rowid varint follows the payload-len varint in-page
3770 };
3771 let prefix_len = n1 + n2;
3772 // The freeblock header overwrites exactly 4 bytes. If the prefix alone is
3773 // wider, no record-header byte is clobbered in a way this simple template
3774 // handles — skip (those tables keep an intact header tail the forward
3775 // carver already reaches).
3776 if prefix_len > 4 {
3777 continue; // cov:unreachable: the corpus tables all encode a <=4-byte cell prefix
3778 }
3779 let payload_start = cell_off + n1 + n2;
3780 let Ok((header_len, hn)) = read_varint(page_bytes, payload_start) else {
3781 continue; // cov:unreachable: a live cell's record header follows its prefix in-page
3782 };
3783 let header_len = usize::try_from(header_len).ok()?;
3784 if header_len < hn {
3785 continue; // cov:unreachable: a live record's header_len covers its own varint
3786 }
3787 // Read the template's serial-type array, recording each serial's byte
3788 // offset within the header so we can split clobbered vs surviving.
3789 let mut serials = Vec::new();
3790 let mut hpos = hn;
3791 let mut ok = true;
3792 while hpos < header_len {
3793 let Ok((s, used)) = read_varint(page_bytes, payload_start + hpos) else {
3794 ok = false; // cov:unreachable: header_len bounds the serial array within the page
3795 break; // cov:unreachable: paired with the read failure above
3796 };
3797 serials.push((s, hpos, used));
3798 hpos += used;
3799 }
3800 if !ok || hpos != header_len || serials.len() < MIN_INFERRED_COLUMNS {
3801 continue; // cov:unreachable: a live cell's header parses cleanly with >= 2 columns
3802 }
3803 return FreeblockTemplate::build(prefix_len, header_len, hn, &serials, enc);
3804 }
3805 None
3806}
3807
3808/// A record-header template derived from a live cell on a table-leaf page, used
3809/// to rebuild freeblock-clobbered records (see
3810/// [`Database::reconstruct_freeblock_records`]).
3811///
3812/// Freeblock conversion overwrites the freed cell's first four bytes — the
3813/// payload-length + rowid varints, the record `header_len`, and the leading
3814/// serial type(s). The surviving serial-type tail and the value body remain. The
3815/// template supplies what was destroyed: the total column count, the serial types
3816/// of the leading (clobbered) columns, and the page offset, relative to the
3817/// freeblock start, at which the surviving serial tail begins.
3818struct FreeblockTemplate {
3819 /// Total number of columns in a record of this table.
3820 column_count: usize,
3821 /// Serial types of the leading columns whose header bytes the freeblock
3822 /// header clobbered (taken from the template; e.g. the fixed-width `id`).
3823 known_lead_serials: Vec<i64>,
3824 /// Offset, relative to the freeblock start, at which the **surviving** serial
3825 /// tail begins (== `prefix_len + first_surviving_serial_header_offset`).
3826 surviving_serials_off: usize,
3827 /// Text encoding of the owning database, so reconstructed text decodes per
3828 /// the header (UTF-8 / UTF-16) rather than assuming UTF-8.
3829 text_encoding: TextEncoding,
3830}
3831
3832impl FreeblockTemplate {
3833 /// Build a template from a parsed live-cell header. `serials` is the list of
3834 /// `(serial_type, header_offset, varint_width)` tuples for every column.
3835 /// Returns `None` when the 4-byte freeblock clobber boundary cannot be
3836 /// resolved to a clean split between leading and surviving serials.
3837 fn build(
3838 prefix_len: usize,
3839 _header_len: usize,
3840 _hn: usize,
3841 serials: &[(i64, usize, usize)],
3842 enc: TextEncoding,
3843 ) -> Option<FreeblockTemplate> {
3844 // Bytes of the record header the 4-byte freeblock header destroys.
3845 let clobbered_header_bytes = 4usize.checked_sub(prefix_len)?;
3846 // The first column whose header bytes survive intact is the first serial
3847 // whose header offset is at or beyond the clobber boundary. Everything
3848 // before it is supplied from the template.
3849 let mut known_lead = Vec::new();
3850 let mut surviving_serials_off = None;
3851 for &(serial, hpos, _used) in serials {
3852 if hpos >= clobbered_header_bytes {
3853 surviving_serials_off = Some(prefix_len + hpos);
3854 break;
3855 }
3856 known_lead.push(serial);
3857 }
3858 // At least one serial must survive to anchor the reconstruction. The
3859 // leading (clobbered) serial list MAY be empty: a 2-byte-or-wider rowid
3860 // varint (rowid >= 128) widens the cell prefix so the 4-byte freeblock
3861 // clobber stops at `header_len`, destroying NO serial type — the whole
3862 // serial array survives. Such pages reconstruct via the exact-tile
3863 // single-cell path (`reconstruct_freeblock_inner` routes on
3864 // `known_lead_serials.is_empty()`), which requires each freed cell to fill
3865 // its freeblock exactly; that precision check keeps the empty-lead case
3866 // phantom-free where a loose span walk would mis-align columns.
3867 let surviving_serials_off = surviving_serials_off?;
3868 Some(FreeblockTemplate {
3869 column_count: serials.len(),
3870 known_lead_serials: known_lead,
3871 surviving_serials_off,
3872 text_encoding: enc,
3873 })
3874 }
3875
3876 /// Reconstruct **every** clobbered cell coalesced into the free span
3877 /// `[lo, hi)` — a chained freeblock or a page's unallocated gap — and append
3878 /// each to `out`.
3879 ///
3880 /// When SQLite frees adjacent cells it coalesces them into one freeblock whose
3881 /// interior still holds the freed cells back-to-back, **each** prefixed by a
3882 /// stale 4-byte freeblock header (`next`/`size`) that clobbers that cell's
3883 /// payload-length + rowid varints and leading serial(s). A single-shot
3884 /// reconstruction at `lo` recovers only the span's first cell; the trailing
3885 /// cells are intact records sitting at the previous record's end. This walks
3886 /// the template across the span: reconstruct at `lo`, advance to that record's
3887 /// end, repeat to `hi`. Every value is derived from the span bounds and the
3888 /// page's own schema template — no per-cell or per-database constant.
3889 ///
3890 /// Each candidate is validated identically to the single-cell case (legal
3891 /// serial types, record fits within `[cell_start, hi)`). The walk is
3892 /// **structural, not a sliding scan**: SQLite coalesces freed cells exactly
3893 /// back-to-back (each freed record's end abuts the next freed cell's clobbered
3894 /// 4-byte prefix), so the next cell begins precisely at the previous record's
3895 /// end. The walk therefore reconstructs at `lo`, advances to that record's
3896 /// end, and repeats — and STOPS the moment a position does not reconstruct
3897 /// cleanly. It never slides forward byte-by-byte hunting for the next cell:
3898 /// that fallback would synthesize a record from any run of bytes that happens
3899 /// to satisfy the legal-serial + fits-in-span checks, manufacturing phantoms
3900 /// in non-cell free space. Anchoring every cell at the prior record's exact
3901 /// end is what keeps the broader span-walk at single-cell precision. Bounded:
3902 /// the walk strictly advances (a record is non-empty) and is capped at
3903 /// [`MAX_FREEBLOCKS_PER_PAGE`] reconstructions per span.
3904 ///
3905 /// Follower precision (the coalesced-freeblock signature): the span's FIRST
3906 /// cell at `lo` is reconstructed unconditionally — `lo` is a real boundary (a
3907 /// freeblock-chain entry, or the gap anchor's first follower). Every SUBSEQUENT
3908 /// follower must carry the structural mark of a freed-and-coalesced cell: its
3909 /// clobbered 4-byte prefix is a stale freeblock header whose 2-byte `next`
3910 /// field is `0x0000` (a terminal/orphaned freeblock — what SQLite leaves when
3911 /// it coalesces freed cells back-to-back). A position whose leading two bytes
3912 /// are non-zero is a byte-shifted remnant, not a coalesced cell, so the run
3913 /// ends there. This is the check that separates a true coalesced tail (0D-06's
3914 /// `00 00 NN NN`-prefixed followers) from a misaligned fragment (0B-02's
3915 /// `24 09 …` remnant), keeping the gap pass phantom-free.
3916 ///
3917 /// `enforce_follower_mark` is `true` for the unallocated-gap pass, where the
3918 /// span is bounded only by `cellContentArea` (not by a page-recorded freeblock
3919 /// size) and so a byte-shifted remnant could otherwise be mistaken for a
3920 /// follower: there EVERY position must carry the `next == 0` mark. It is `false`
3921 /// for the freeblock-chain pass, whose span bounds are the page-recorded
3922 /// `[fb, fb + size)` — a strong boundary that already pins the coalesced run, so
3923 /// the interior followers (whose clobbered bytes are the original record's own
3924 /// varints, not necessarily `00 00 …`) are accepted on the fit-in-span check
3925 /// alone.
3926 ///
3927 /// Tiered walk: it pushes each reconstructed full cell into `cells`, and at the
3928 /// anchor where `reconstruct_one` would `break` it salvages the maximal
3929 /// decodable column prefix into `frags` as a [`CellFragment`] (when the §3.1
3930 /// distinctiveness gate passes) before stopping. Fragment salvage does NOT
3931 /// extend the walk — it stops at exactly the position the full walk does,
3932 /// preserving Tier-1's phantom discipline. Callers that want only the full
3933 /// cells (the Tier-1 [`Database::reconstruct_freeblock_records`]) discard
3934 /// `frags`; both tiers therefore come from one walk and can never diverge.
3935 fn reconstruct_span_tiered(
3936 &self,
3937 page: &[u8],
3938 lo: usize,
3939 hi: usize,
3940 enforce_follower_mark: bool,
3941 cells: &mut Vec<CarvedCell>,
3942 frags: &mut Vec<CellFragment>,
3943 ) {
3944 let mut cell_start = lo;
3945 let mut built = 0usize;
3946 while cell_start < hi && built < MAX_FREEBLOCKS_PER_PAGE {
3947 if enforce_follower_mark && be_u16(page, cell_start) != 0 {
3948 break; // not a coalesced freeblock follower — the contiguous run ends
3949 }
3950 let Some((cell, record_end)) = self.reconstruct_one(page, cell_start, hi) else {
3951 // Full reconstruction failed at this anchor; try to salvage the
3952 // decodable prefix as a fragment, then stop (do not extend the
3953 // walk past the failed anchor).
3954 if let Some(frag) = self.salvage_fragment(page, cell_start, hi) {
3955 frags.push(frag);
3956 }
3957 break;
3958 };
3959 cells.push(cell);
3960 built += 1;
3961 cell_start = record_end;
3962 }
3963 }
3964
3965 /// Salvage the maximal decodable column prefix at `cell_start` (bounded by
3966 /// `span_end`) when full reconstruction failed there. Walks the template +
3967 /// surviving serial array forward, decoding each column's body while it fits
3968 /// in the span; the first illegal serial, out-of-bounds read, or body that
3969 /// overruns the span ends the prefix. Returns a [`CellFragment`] **only** when
3970 /// the salvaged prefix contains at least one distinctive cell (TEXT ≥ 4 bytes
3971 /// of valid UTF-8, or REAL) — the §3.1 emission gate — otherwise `None`.
3972 fn salvage_fragment(
3973 &self,
3974 page: &[u8],
3975 cell_start: usize,
3976 span_end: usize,
3977 ) -> Option<CellFragment> {
3978 let surviving_count = self.column_count - self.known_lead_serials.len();
3979 let tail_start = cell_start.checked_add(self.surviving_serials_off)?;
3980
3981 // Read as many legal surviving serials as decode in-bounds within the span.
3982 // The template's leading serials are always legal (they came from a live
3983 // cell), so the full serial array is `known_lead ++ legal_surviving`.
3984 let mut serials = self.known_lead_serials.clone();
3985 let mut pos = tail_start;
3986 for _ in 0..surviving_count {
3987 let Ok((s, used)) = read_varint(page, pos) else {
3988 break; // cov:unreachable: the surviving serials sit near the cell start, inside the freeblock/gap span the inner walker already bounds to the page; this read mirrors reconstruct_one's bounds guard so a truncated tail ends the prefix rather than panicking
3989 };
3990 if serial_body_len(s).is_none() {
3991 break; // cov:unreachable: serial_body_len is None only for a negative serial, which read_varint yields only from a crafted 9-byte varint; kept as a defence-in-depth guard so a malformed surviving tail ends the prefix rather than mis-decoding
3992 }
3993 let Some(next) = pos.checked_add(used) else {
3994 break; // cov:unreachable: usize add of an in-page varint width
3995 };
3996 if next > span_end {
3997 break; // serial tail overran the span
3998 }
3999 serials.push(s);
4000 pos = next;
4001 }
4002
4003 // Decode column bodies left-to-right, keeping each whose body ends within
4004 // the span. The body begins right after the surviving serial tail.
4005 let body_start = pos;
4006 let mut surviving: Vec<(usize, Value)> = Vec::new();
4007 let mut bpos = body_start;
4008 for (idx, &s) in serials.iter().enumerate() {
4009 let Some(blen) = serial_body_len(s) else {
4010 break; // cov:unreachable: only legal serials were pushed above
4011 };
4012 let Some(body_end) = bpos.checked_add(blen) else {
4013 break; // cov:unreachable: usize add of an in-page body length
4014 };
4015 if body_end > span_end {
4016 break; // this column's body overruns the span — prefix ends here
4017 }
4018 let Some(body) = page.get(bpos..body_end) else {
4019 break; // cov:unreachable: body_end <= span_end <= page.len()
4020 };
4021 let Ok((val, _)) = decode_value(body, 0, s, self.text_encoding) else {
4022 break; // cov:unreachable: serial_body_len-legal serials decode in-bounds
4023 };
4024 surviving.push((idx, val));
4025 bpos = body_end;
4026 }
4027
4028 // Emission gate: at least one distinctive cell (TEXT >= 4 UTF-8 bytes, or
4029 // REAL). A lone integer/NULL/blob prefix is coincidence-prone — no fragment.
4030 if !surviving.iter().any(|(_, v)| is_distinctive(v)) {
4031 return None;
4032 }
4033 let last_body_end = bpos;
4034 Some(CellFragment {
4035 offset: cell_start,
4036 byte_len: last_body_end.saturating_sub(cell_start),
4037 missing: self.column_count - surviving.len(),
4038 surviving,
4039 confidence: FRAGMENT_CONFIDENCE,
4040 })
4041 }
4042
4043 /// Rebuild the single record whose clobbered cell begins at `cell_start`,
4044 /// bounded by the enclosing span end `span_end`: read the surviving serial
4045 /// tail, prepend the template's leading serials, decode the body, and validate
4046 /// the whole record fits within `[cell_start, span_end)`. Returns the carved
4047 /// cell and the record's end offset (the next coalesced cell's start), or
4048 /// `None` on any out-of-bounds or implausible parse.
4049 fn reconstruct_one(
4050 &self,
4051 page: &[u8],
4052 cell_start: usize,
4053 span_end: usize,
4054 ) -> Option<(CarvedCell, usize)> {
4055 let surviving_count = self.column_count - self.known_lead_serials.len();
4056 let tail_start = cell_start.checked_add(self.surviving_serials_off)?;
4057
4058 // Read the surviving serial tail from the freeblock.
4059 let mut serials = self.known_lead_serials.clone();
4060 let mut pos = tail_start;
4061 for _ in 0..surviving_count {
4062 let (s, used) = read_varint(page, pos).ok()?;
4063 // A serial type must be legal; reject the candidate otherwise.
4064 serial_body_len(s)?;
4065 serials.push(s);
4066 pos = pos.checked_add(used)?;
4067 if pos > span_end {
4068 return None;
4069 }
4070 }
4071
4072 // The body begins right after the surviving serial tail. Compute its
4073 // length from the full (template + surviving) serial array.
4074 let mut body_len = 0usize;
4075 for &s in &serials {
4076 body_len = body_len.checked_add(serial_body_len(s)?)?;
4077 }
4078 let body_start = pos;
4079 let record_end = body_start.checked_add(body_len)?;
4080 // The reconstructed record MUST fit within the enclosing span — the core
4081 // precision check that rejects coincidental/garbage reconstructions.
4082 if record_end > span_end {
4083 return None;
4084 }
4085
4086 // Synthesize a record payload (header + body) for the shared decoder so
4087 // values are decoded with the same storage-class fidelity as live rows.
4088 // The rowid is destroyed; pass 0 so a serial-0 column reads as NULL rather
4089 // than a fabricated rowid.
4090 let body = page.get(body_start..record_end)?;
4091 let values = decode_synthetic_record(&serials, body, self.text_encoding)?;
4092 if values.len() != self.column_count {
4093 return None; // cov:unreachable: one value per serial by construction
4094 }
4095
4096 Some((
4097 CarvedCell {
4098 offset: cell_start,
4099 byte_len: record_end - cell_start,
4100 rowid: 0, // destroyed by freeblock conversion — surfaced as unknown
4101 values,
4102 confidence: FREEBLOCK_RECONSTRUCT_CONFIDENCE,
4103 },
4104 record_end,
4105 ))
4106 }
4107
4108 /// Reconstruct ONE freeblock-clobbered empty-leading-serial cell at
4109 /// `cell_start` — a 2-byte-or-wider rowid, so the 4-byte clobber destroyed no
4110 /// serial type and the whole serial array survives at
4111 /// `cell_start + surviving_serials_off`. Returns the carved cell (rowid
4112 /// destroyed → 0) **and the record's end offset**, or `None` on any
4113 /// out-of-bounds parse or a record that overruns `span_end`. Does NOT enforce
4114 /// an exact tile — the span walker [`Self::reconstruct_span_exact`] does.
4115 fn reconstruct_cell_empty_lead(
4116 &self,
4117 page: &[u8],
4118 cell_start: usize,
4119 span_end: usize,
4120 ) -> Option<(CarvedCell, usize)> {
4121 let tail_start = cell_start.checked_add(self.surviving_serials_off)?;
4122 // The whole serial array survives (no clobbered leading serial); read all
4123 // `column_count` serials from the freeblock.
4124 let mut serials = Vec::with_capacity(self.column_count);
4125 let mut pos = tail_start;
4126 for _ in 0..self.column_count {
4127 let (s, used) = read_varint(page, pos).ok()?;
4128 serial_body_len(s)?;
4129 serials.push(s);
4130 pos = pos.checked_add(used)?;
4131 if pos > span_end {
4132 return None;
4133 }
4134 }
4135 let mut body_len = 0usize;
4136 for &s in &serials {
4137 body_len = body_len.checked_add(serial_body_len(s)?)?;
4138 }
4139 let body_start = pos;
4140 let record_end = body_start.checked_add(body_len)?;
4141 if record_end > span_end {
4142 return None;
4143 }
4144 let body = page.get(body_start..record_end)?;
4145 let values = decode_synthetic_record(&serials, body, self.text_encoding)?;
4146 if values.len() != self.column_count {
4147 return None; // cov:unreachable: one value per serial by construction
4148 }
4149 Some((
4150 CarvedCell {
4151 offset: cell_start,
4152 byte_len: record_end - cell_start,
4153 rowid: 0, // destroyed by freeblock conversion — surfaced as unknown
4154 values,
4155 confidence: FREEBLOCK_RECONSTRUCT_CONFIDENCE,
4156 },
4157 record_end,
4158 ))
4159 }
4160
4161 /// Reconstruct every empty-leading-serial cell coalesced into the freeblock
4162 /// `[lo, hi)`, returned ONLY when they tile the freeblock **exactly** (the
4163 /// walk reaches `hi` with no leftover bytes).
4164 ///
4165 /// A single freed cell fills its freeblock exactly; adjacent deletions
4166 /// coalesce into one freeblock whose interior holds the freed cells
4167 /// back-to-back, each clobbered in its first 4 bytes. Walking cell-to-cell and
4168 /// requiring the run to land precisely on `hi` is the precision gate: a
4169 /// misaligned read (a deleted cell whose destroyed rowid width differs from the
4170 /// template's) fails to reach `hi` exactly, so the whole span is rejected
4171 /// rather than emitted as column-shifted phantoms. Bounded by
4172 /// [`MAX_FREEBLOCKS_PER_PAGE`]; a record always advances `cell_start`.
4173 fn reconstruct_span_exact(&self, page: &[u8], lo: usize, hi: usize) -> Vec<CarvedCell> {
4174 let mut cells = Vec::new();
4175 let mut cell_start = lo;
4176 let mut guard = 0usize;
4177 while cell_start < hi && guard < MAX_FREEBLOCKS_PER_PAGE {
4178 guard += 1;
4179 let Some((cell, record_end)) = self.reconstruct_cell_empty_lead(page, cell_start, hi)
4180 else {
4181 return Vec::new(); // a cell did not reconstruct → not a clean tiling
4182 };
4183 if record_end <= cell_start {
4184 return Vec::new(); // cov:unreachable: a non-empty record advances cell_start
4185 }
4186 cells.push(cell);
4187 cell_start = record_end;
4188 }
4189 // Exact tile: leftover bytes (or a walk stopped by the bound) mean a
4190 // misaligned run — emit nothing.
4191 if cell_start == hi {
4192 cells
4193 } else {
4194 Vec::new()
4195 }
4196 }
4197
4198 /// Reconstruct a freeblock-clobbered **spilled** cell at `cell_start` (task
4199 /// #73, design §2.2). A spilled cell always carries a multi-byte
4200 /// `payload_len` varint, so the 4-byte freeblock clobber destroys the
4201 /// `payload_len` + `rowid` varints and the record's `header_len` varint —
4202 /// **but not the serial-type array**, which survives intact immediately after
4203 /// the clobber. We therefore read the full serial array directly from
4204 /// `cell_start + CLOBBER` (using the template only for the column count),
4205 /// re-derive `header_len` and `P = header_len + Σ serial_body_len`, and — when
4206 /// `P > usable - 35` — resolve the spill: `local_payload_len(P, usable)` bytes
4207 /// of payload sit locally (the destroyed header counted within them), the
4208 /// 4-byte first-overflow pointer follows, and the chain is resolved through
4209 /// freelist leaves. Returns `(cell, chain)` with `rowid = 0`, or `None`.
4210 ///
4211 /// UNPROVEN-BY-CORPUS (Codex ruling #5): synthetic-fixture validation only.
4212 /// No real Nemetz cell is both freeblock-clobbered and spilled.
4213 fn reconstruct_spilled(
4214 &self,
4215 db: &Database,
4216 page: &[u8],
4217 cell_start: usize,
4218 usable: usize,
4219 freed_leaves: &std::collections::BTreeSet<u32>,
4220 ) -> Option<(CarvedCell, Vec<u32>)> {
4221 // The freeblock header clobbers exactly 4 bytes. For a spilled cell those
4222 // 4 bytes are payload_len(>=2) + rowid(>=1) + header_len(>=1) varints, so
4223 // the serial array begins right after the clobber.
4224 const CLOBBER: usize = 4;
4225 let serials_start = cell_start.checked_add(CLOBBER)?;
4226 let mut serials = Vec::with_capacity(self.column_count);
4227 let mut pos = serials_start;
4228 for _ in 0..self.column_count {
4229 let (s, used) = read_varint(page, pos).ok()?;
4230 serial_body_len(s)?;
4231 serials.push(s);
4232 pos = pos.checked_add(used)?;
4233 }
4234
4235 // Re-derive the record header bytes that were destroyed: header_len is a
4236 // varint counting itself plus the serial array.
4237 let mut serial_bytes_len = 0usize;
4238 for &s in &serials {
4239 serial_bytes_len += varint_len(s);
4240 }
4241 let mut header_len = serial_bytes_len + 1;
4242 while varint_len(header_len as i64) + serial_bytes_len != header_len {
4243 header_len += 1;
4244 }
4245 // The clobber removed `header_len`'s own varint plus the prefix; verify the
4246 // surviving serial array aligns with the reconstructed header (the bytes
4247 // from serials_start to `pos` are the serial array, length serial_bytes_len).
4248 if pos.checked_sub(serials_start)? != serial_bytes_len {
4249 return None; // cov:unreachable: read_varint widths sum to serial_bytes_len
4250 }
4251 let mut body_len = 0usize;
4252 for &s in &serials {
4253 body_len = body_len.checked_add(serial_body_len(s)?)?;
4254 }
4255 let payload_len = header_len.checked_add(body_len)?;
4256 // Only the spilled class — an in-page payload is the existing template path.
4257 if payload_len <= usable.checked_sub(35)? {
4258 return None;
4259 }
4260 let local_len = local_payload_len(payload_len, usable);
4261
4262 // The body starts right after the surviving serial array. The local payload
4263 // spans `local_len` bytes of (header ++ body); the destroyed header is
4264 // `header_len` of those, so `local_len - header_len` body bytes are present
4265 // locally before the 4-byte first-overflow pointer.
4266 let body_start = pos;
4267 let local_body = local_len.checked_sub(header_len)?;
4268 let local_body_end = body_start.checked_add(local_body)?;
4269 let ptr_off = local_body_end;
4270 let ptr_slice = page.get(ptr_off..ptr_off + 4)?;
4271 let first_overflow =
4272 u32::from_be_bytes([ptr_slice[0], ptr_slice[1], ptr_slice[2], ptr_slice[3]]);
4273 let local_body_bytes = page.get(body_start..local_body_end)?;
4274
4275 let remaining = payload_len - local_len;
4276 let (chain_content, chain) = db
4277 .read_freed_overflow_chain(first_overflow, remaining, usable, freed_leaves)
4278 .ok()?;
4279
4280 // Assemble the full payload: reconstructed header ++ local body ++ chain.
4281 let mut header = enc_varint_into(header_len);
4282 for &s in &serials {
4283 header.extend(enc_varint_into(usize::try_from(s).ok()?));
4284 }
4285 if header.len() != header_len {
4286 return None; // cov:unreachable: header_len was solved to this width
4287 }
4288 let mut payload = Vec::with_capacity(payload_len);
4289 payload.extend_from_slice(&header);
4290 payload.extend_from_slice(local_body_bytes);
4291 payload.extend_from_slice(&chain_content);
4292 if payload.len() != payload_len {
4293 return None; // cov:unreachable: local_body + chain == body_len by construction
4294 }
4295
4296 let values = decode_record(&payload, self.column_count, 0, db.header.text_encoding).ok()?;
4297 if values.len() != self.column_count {
4298 return None; // cov:unreachable: one value per serial
4299 }
4300 let any_replacement = values.iter().any(|v| match v {
4301 Value::Text(t) => t.contains('\u{FFFD}'),
4302 _ => false,
4303 });
4304 if any_replacement {
4305 return None;
4306 }
4307 if !values.iter().any(is_distinctive) {
4308 return None;
4309 }
4310
4311 Some((
4312 CarvedCell {
4313 offset: cell_start,
4314 byte_len: ptr_off + 4 - cell_start,
4315 rowid: 0,
4316 values,
4317 confidence: FREEBLOCK_RECONSTRUCT_CONFIDENCE * OVERFLOW_CHAIN_CONFIDENCE_FACTOR,
4318 },
4319 chain,
4320 ))
4321 }
4322}
4323
4324/// Decode a record body given an explicit serial-type array (the freeblock
4325/// reconstructor supplies the array; the on-disk `header_len` + leading serials
4326/// were destroyed). Mirrors [`decode_record`]'s body pass. Returns `None` on any
4327/// out-of-bounds read so a malformed reconstruction is rejected, never panics.
4328fn decode_synthetic_record(serials: &[i64], body: &[u8], enc: TextEncoding) -> Option<Vec<Value>> {
4329 let mut values = Vec::with_capacity(serials.len());
4330 let mut bpos = 0usize;
4331 for &serial in serials {
4332 let (val, size) = decode_value(body, bpos, serial, enc).ok()?;
4333 values.push(val);
4334 bpos = bpos.checked_add(size)?;
4335 }
4336 Some(values)
4337}
4338
4339/// Attempt to recognize a table-leaf cell at `off` in `buf` as a record.
4340///
4341/// `expected_columns` is `Some(n)` to require exactly `n` columns (fixed-schema
4342/// carving), or `None` to **infer** the column count from the record's own
4343/// serial-type array (dropped-table / schema-gone carving). Returns a
4344/// [`CarvedCell`] only when the bytes are self-consistently record-shaped;
4345/// otherwise `None`. Never panics — every access is bounds-checked.
4346fn try_carve_cell_at(
4347 buf: &[u8],
4348 off: usize,
4349 expected_columns: Option<usize>,
4350 enc: TextEncoding,
4351) -> Option<CarvedCell> {
4352 // Cell prefix: payload_len varint, rowid varint.
4353 let (payload_len, n1) = read_varint(buf, off).ok()?;
4354 let payload_len = usize::try_from(payload_len).ok()?;
4355 if payload_len == 0 {
4356 return None;
4357 }
4358 let (rowid, n2) = read_varint(buf, off + n1).ok()?;
4359 // A negative rowid is legal but vanishingly rare for browser tables; treat a
4360 // non-positive rowid as a non-match to suppress coincidental hits.
4361 if rowid <= 0 {
4362 return None;
4363 }
4364 let payload_start = off + n1 + n2;
4365 let payload = buf.get(payload_start..payload_start + payload_len)?;
4366
4367 // Record header: header_len varint, then one serial type per column.
4368 let (header_len, hn) = read_varint(payload, 0).ok()?;
4369 let header_len = usize::try_from(header_len).ok()?;
4370 if header_len > payload.len() || header_len < hn {
4371 return None;
4372 }
4373 let cap = expected_columns.unwrap_or(0);
4374 let mut serials = Vec::with_capacity(cap);
4375 let mut hpos = hn;
4376 while hpos < header_len {
4377 let (s, used) = read_varint(payload, hpos).ok()?;
4378 serials.push(s);
4379 hpos += used;
4380 }
4381 // The header must consume cleanly, and match the expected column count when
4382 // one was given. When inferring, require a minimum plausible column count to
4383 // suppress coincidental 1-column matches.
4384 if hpos != header_len {
4385 return None;
4386 }
4387 match expected_columns {
4388 Some(n) if serials.len() != n => return None,
4389 None if serials.len() < MIN_INFERRED_COLUMNS => return None,
4390 _ => {}
4391 }
4392 let column_count = serials.len();
4393
4394 // Body length implied by the serial types must equal payload_len - header_len
4395 // — a strong self-consistency check that rejects coincidental matches.
4396 let mut body_len = 0usize;
4397 for &s in &serials {
4398 // Checked: a serial from free-space bytes can declare a body length near
4399 // usize::MAX; summing must reject (None) on overflow, never panic/wrap.
4400 body_len = body_len.checked_add(serial_body_len(s)?)?;
4401 }
4402 if header_len + body_len != payload_len {
4403 return None;
4404 }
4405
4406 // Decode the record (reusing the live decoder for storage-class fidelity).
4407 let values = decode_record(payload, column_count, rowid, enc).ok()?;
4408 if values.len() != column_count {
4409 return None; // cov:unreachable: decode_record yields one value per serial
4410 }
4411
4412 // Confidence: a fully self-consistent record already passed strong checks;
4413 // raise confidence when at least one column is a non-empty, valid-UTF-8 TEXT
4414 // (record-shaped *and* human-meaningful), which coincidental byte runs rarely
4415 // satisfy.
4416 let has_real_text = values.iter().any(|v| match v {
4417 Value::Text(t) => !t.is_empty() && !t.contains('\u{FFFD}'),
4418 _ => false,
4419 });
4420 let confidence = if has_real_text { 0.9 } else { 0.6 };
4421
4422 Some(CarvedCell {
4423 offset: off,
4424 byte_len: (payload_start + payload_len) - off,
4425 rowid,
4426 values,
4427 confidence,
4428 })
4429}
4430
4431/// Recognize a freed **spilled** table-leaf cell at `off` whose payload exceeds
4432/// the in-page threshold (`usable - 35`) and therefore continues on an
4433/// overflow-page chain (task #73). The sibling of [`try_carve_cell_at`] for the
4434/// overflow class: the two partition the candidate space by the spec spill
4435/// threshold, so a cell is recognized by exactly one of them.
4436///
4437/// `expected_columns` is `Some(n)` to require exactly `n` columns, or `None` to
4438/// infer the count (≥ [`MIN_INFERRED_COLUMNS`]). Returns a [`SpilledCell`]
4439/// (recognition only — the chain is resolved later) when the local prefix is
4440/// self-consistent: header fits in the local payload, the serial array consumes
4441/// the header cleanly, `header_len + Σ serial_body_len == P` (length closure
4442/// over the *declared* P), and the local payload plus its 4-byte overflow
4443/// pointer are in-bounds. Never panics — every access is bounds-checked.
4444fn try_carve_spilled_cell_at(
4445 buf: &[u8],
4446 off: usize,
4447 usable: usize,
4448 expected_columns: Option<usize>,
4449) -> Option<SpilledCell> {
4450 let (payload_len, n1) = read_varint(buf, off).ok()?;
4451 let payload_len = usize::try_from(payload_len).ok()?;
4452 // Only the overflow class — in-page payloads belong to `try_carve_cell_at`.
4453 if payload_len <= usable.checked_sub(35)? {
4454 return None;
4455 }
4456 let (rowid, n2) = read_varint(buf, off + n1).ok()?;
4457 if rowid <= 0 {
4458 return None;
4459 }
4460 let payload_start = off + n1 + n2;
4461 let local_len = local_payload_len(payload_len, usable);
4462 // The local payload prefix plus the 4-byte first-overflow pointer must be in
4463 // bounds of the scanned slice.
4464 let prefix = buf.get(payload_start..payload_start + local_len + 4)?;
4465
4466 // The record header must fit entirely within the local prefix — otherwise the
4467 // serial array is not addressable locally and we abstain rather than guess.
4468 let (header_len, hn) = read_varint(prefix, 0).ok()?;
4469 let header_len = usize::try_from(header_len).ok()?;
4470 if header_len > local_len || header_len < hn {
4471 return None;
4472 }
4473 let mut serials = Vec::new();
4474 let mut hpos = hn;
4475 while hpos < header_len {
4476 let (s, used) = read_varint(prefix, hpos).ok()?;
4477 serials.push(s);
4478 hpos += used;
4479 }
4480 if hpos != header_len {
4481 return None;
4482 }
4483 match expected_columns {
4484 Some(n) if serials.len() != n => return None,
4485 None if serials.len() < MIN_INFERRED_COLUMNS => return None,
4486 _ => {}
4487 }
4488
4489 // Length closure over the DECLARED payload: header + body must equal P.
4490 let mut body_len = 0usize;
4491 for &s in &serials {
4492 // Checked: a serial from free-space bytes can declare a body length near
4493 // usize::MAX; summing must reject (None) on overflow, never panic/wrap.
4494 body_len = body_len.checked_add(serial_body_len(s)?)?;
4495 }
4496 if header_len + body_len != payload_len {
4497 return None;
4498 }
4499
4500 let first_overflow = be_u32(prefix, local_len);
4501 Some(SpilledCell {
4502 offset: off,
4503 byte_len: n1 + n2 + local_len + 4,
4504 payload_len,
4505 rowid,
4506 serials,
4507 local_len,
4508 local_payload_off: payload_start,
4509 first_overflow,
4510 })
4511}
4512
4513/// Salvage the columns of a recognized [`SpilledCell`] whose bodies lie wholly
4514/// within the local payload (task #73, Codex ruling #4): the chain-resident
4515/// columns are dropped (the chain that would supply them failed), and the
4516/// surviving local columns become a [`CellFragment`]. Returns `None` unless the
4517/// salvaged prefix carries ≥ 1 distinctive cell (the §3.1 emission gate). The
4518/// returned fragment's `offset` is region-local; the caller translates it.
4519fn salvage_local_prefix(
4520 region: &[u8],
4521 sc: &SpilledCell,
4522 enc: TextEncoding,
4523) -> Option<CellFragment> {
4524 // The body begins right after the local header; decode each column while its
4525 // body ends within the local payload bytes (`local_payload_off + local_len`).
4526 let local_end = sc.local_payload_off.checked_add(sc.local_len)?;
4527 // Recompute the record header length to find where the body starts.
4528 let (header_len, _hn) = read_varint(region, sc.local_payload_off).ok()?;
4529 let header_len = usize::try_from(header_len).ok()?;
4530 let mut bpos = sc.local_payload_off.checked_add(header_len)?;
4531
4532 let mut surviving: Vec<(usize, Value)> = Vec::new();
4533 for (idx, &serial) in sc.serials.iter().enumerate() {
4534 let Some(blen) = serial_body_len(serial) else {
4535 break; // cov:unreachable: recognizer accepted only legal serials
4536 };
4537 let Some(body_end) = bpos.checked_add(blen) else {
4538 break; // cov:unreachable: usize add of an in-page body length
4539 };
4540 if body_end > local_end {
4541 break; // this column's body spills into the chain — local prefix ends
4542 }
4543 let Some(body) = region.get(bpos..body_end) else {
4544 break; // cov:unreachable: body_end <= local_end <= region.len()
4545 };
4546 // Column 0 of a rowid-alias table reads as the rowid when serial 0; here a
4547 // spilled cell's id column is a stored integer, so decode it directly.
4548 let Ok((val, _)) = decode_value(body, 0, serial, enc) else {
4549 break; // cov:unreachable: legal serials decode in-bounds
4550 };
4551 surviving.push((idx, val));
4552 bpos = body_end;
4553 }
4554
4555 if !surviving.iter().any(|(_, v)| is_distinctive(v)) {
4556 return None;
4557 }
4558 Some(CellFragment {
4559 offset: sc.offset,
4560 byte_len: bpos.saturating_sub(sc.local_payload_off),
4561 missing: sc.serials.len() - surviving.len(),
4562 surviving,
4563 confidence: FRAGMENT_CONFIDENCE,
4564 })
4565}
4566
4567/// Parse + validate the 100-byte file header.
4568/// The first up-to-100 bytes (the SQLite header region), kept resident so
4569/// fixed-offset header-field reads never touch the byte source.
4570fn header_prefix(bytes: &[u8]) -> Box<[u8]> {
4571 let n = bytes.len().min(SQLITE_HEADER_SIZE);
4572 bytes[..n].into()
4573}
4574
4575fn parse_header(bytes: &[u8]) -> Result<Header, Error> {
4576 let head = bytes.get(..SQLITE_HEADER_SIZE).ok_or(Error::TooShort)?;
4577 if !head.starts_with(SQLITE_MAGIC) {
4578 return Err(Error::BadMagic);
4579 }
4580 let raw = be_u16(head, SQLITE_PAGE_SIZE_OFFSET);
4581 let page_size: u32 = if raw == 1 { 65536 } else { u32::from(raw) };
4582 let valid = (512..=65536).contains(&page_size) && page_size.is_power_of_two();
4583 if !valid {
4584 return Err(Error::BadPageSize(page_size));
4585 }
4586 let reserved = *head.get(RESERVED_SPACE_OFFSET).ok_or(Error::TooShort)?;
4587 // Header byte 56 (BE u32): 1/0 = UTF-8, 2 = UTF-16LE, 3 = UTF-16BE
4588 // (file-format §1.3.1). Tolerant: an unexpected value degrades to UTF-8
4589 // rather than rejecting the database.
4590 let text_encoding = match be_u32(head, TEXT_ENCODING_OFFSET) {
4591 2 => TextEncoding::Utf16Le,
4592 3 => TextEncoding::Utf16Be,
4593 _ => TextEncoding::Utf8,
4594 };
4595 Ok(Header {
4596 page_size,
4597 reserved,
4598 text_encoding,
4599 })
4600}
4601
4602/// Decode a record (payload) into values. Serial type 0 on the first column of
4603/// a rowid table is the `INTEGER PRIMARY KEY` alias → the cell's rowid.
4604fn decode_record(
4605 payload: &[u8],
4606 _column_count: usize,
4607 rowid: i64,
4608 enc: TextEncoding,
4609) -> Result<Vec<Value>, Error> {
4610 // A table-b-tree record: column 0 is the INTEGER PRIMARY KEY alias, so a
4611 // serial-0 there reads the rowid rather than NULL.
4612 decode_record_inner(payload, enc, Some(rowid))
4613}
4614
4615/// Decode an index-b-tree record payload (roadmap §1.4). Unlike a table record it
4616/// has NO `INTEGER PRIMARY KEY` alias — every column is stored literally, so a
4617/// serial-0 first column is a genuine NULL key, never a rowid.
4618fn decode_index_payload(payload: &[u8], enc: TextEncoding) -> Result<Vec<Value>, Error> {
4619 decode_record_inner(payload, enc, None)
4620}
4621
4622/// Decode a SQLite record payload (header + serial array + body) into its column
4623/// values. `rowid_alias` supplies the rowid for a table record's column-0
4624/// `INTEGER PRIMARY KEY` alias (serial 0 → the rowid); `None` (index records)
4625/// leaves a serial-0 column as NULL.
4626fn decode_record_inner(
4627 payload: &[u8],
4628 enc: TextEncoding,
4629 rowid_alias: Option<i64>,
4630) -> Result<Vec<Value>, Error> {
4631 let (header_len, n) = read_varint(payload, 0)?;
4632 let header_len = header_len as usize;
4633 if header_len > payload.len() {
4634 return Err(Error::TruncatedCell);
4635 }
4636 // Pass 1: read serial types from the record header.
4637 let mut serials = Vec::new();
4638 let mut hpos = n;
4639 while hpos < header_len {
4640 let (s, used) = read_varint(payload, hpos)?;
4641 serials.push(s);
4642 hpos += used;
4643 }
4644 // Pass 2: read the body, one value per serial type.
4645 let mut values = Vec::with_capacity(serials.len());
4646 let mut bpos = header_len;
4647 for (idx, &serial) in serials.iter().enumerate() {
4648 let (val, size) = decode_value(payload, bpos, serial, enc)?;
4649 let val = match (idx, serial, rowid_alias) {
4650 // INTEGER PRIMARY KEY alias: NULL in column 0 reads the rowid.
4651 (0, 0, Some(rowid)) => Value::Integer(rowid),
4652 _ => val,
4653 };
4654 values.push(val);
4655 bpos += size;
4656 }
4657 Ok(values)
4658}
4659
4660/// Decode a single value of the given serial type at `off`. Returns the value
4661/// and the number of body bytes it consumed.
4662fn decode_value(
4663 buf: &[u8],
4664 off: usize,
4665 serial: i64,
4666 enc: TextEncoding,
4667) -> Result<(Value, usize), Error> {
4668 Ok(match serial {
4669 // 0 = NULL; 10/11 are reserved for internal use and surfaced as NULL.
4670 0 | 10 | 11 => (Value::Null, 0),
4671 1 => (
4672 Value::Integer(i64::from(read_be_u64(buf, off, 1)? as i8)),
4673 1,
4674 ),
4675 2 => (
4676 Value::Integer(i64::from(read_be_u64(buf, off, 2)? as i16)),
4677 2,
4678 ),
4679 3 => (Value::Integer(sign_extend(read_be_u64(buf, off, 3)?, 3)), 3),
4680 4 => (
4681 Value::Integer(i64::from(read_be_u64(buf, off, 4)? as i32)),
4682 4,
4683 ),
4684 5 => (Value::Integer(sign_extend(read_be_u64(buf, off, 6)?, 6)), 6),
4685 6 => (Value::Integer(read_be_u64(buf, off, 8)? as i64), 8),
4686 7 => {
4687 let bits = read_be_u64(buf, off, 8)?;
4688 (Value::Real(f64::from_bits(bits)), 8)
4689 }
4690 8 => (Value::Integer(0), 0),
4691 9 => (Value::Integer(1), 0),
4692 n if n >= 12 && n % 2 == 0 => {
4693 let len = ((n - 12) / 2) as usize;
4694 let bytes = buf.get(off..off + len).ok_or(Error::TruncatedCell)?;
4695 (Value::Blob(bytes.to_vec()), len)
4696 }
4697 n => {
4698 // odd, >= 13: text, decoded per the database's text encoding
4699 // (UTF-8 / UTF-16LE / UTF-16BE). Lossy so a corrupt byte can't panic.
4700 let len = ((n - 13) / 2) as usize;
4701 let bytes = buf.get(off..off + len).ok_or(Error::TruncatedCell)?;
4702 (Value::Text(enc.decode(bytes)), len)
4703 }
4704 })
4705}
4706
4707/// Read `width` (1..=8) big-endian bytes into a raw u64 (no sign extension).
4708fn read_be_u64(buf: &[u8], off: usize, width: usize) -> Result<u64, Error> {
4709 let bytes = buf.get(off..off + width).ok_or(Error::TruncatedCell)?;
4710 let mut acc: u64 = 0;
4711 for &b in bytes {
4712 acc = (acc << 8) | u64::from(b);
4713 }
4714 Ok(acc)
4715}
4716
4717/// Sign-extend a `width`-byte (3 or 6) value held in the low bits of `raw`.
4718fn sign_extend(raw: u64, width: usize) -> i64 {
4719 let bits = width * 8;
4720 let shift = 64 - bits;
4721 ((raw as i64) << shift) >> shift
4722}
4723
4724/// Read a `SQLite` varint (1..=9 bytes) at `off`. Returns value + bytes consumed.
4725fn read_varint(buf: &[u8], off: usize) -> Result<(i64, usize), Error> {
4726 let mut result: u64 = 0;
4727 for i in 0..8 {
4728 let b = *buf.get(off + i).ok_or(Error::TruncatedCell)?;
4729 result = (result << 7) | u64::from(b & 0x7f);
4730 if b & 0x80 == 0 {
4731 return Ok((result as i64, i + 1));
4732 }
4733 }
4734 // 9th byte contributes all 8 bits.
4735 let b = *buf.get(off + 8).ok_or(Error::TruncatedCell)?;
4736 result = (result << 8) | u64::from(b);
4737 Ok((result as i64, 9))
4738}
4739
4740/// Bounds-checked big-endian u16; out-of-range yields 0 (never panics).
4741fn be_u16(buf: &[u8], off: usize) -> u16 {
4742 let mut b = [0u8; 2];
4743 if let Some(s) = buf.get(off..off + 2) {
4744 b.copy_from_slice(s);
4745 }
4746 u16::from_be_bytes(b)
4747}
4748
4749/// Byte width of the minimal `SQLite` varint encoding of a non-negative `value`
4750/// (task #73, used to re-derive a clobbered record's `header_len`). Mirrors the
4751/// 7-bit big-endian grouping of [`enc_varint_into`]; a value needing more than 8
4752/// groups uses the 9-byte form. Negative inputs (illegal serial types) are
4753/// treated as a single byte and rejected upstream by `serial_body_len`.
4754fn varint_len(value: i64) -> usize {
4755 if value < 0 {
4756 return 1; // cov:unreachable: callers pass only non-negative serials/lengths
4757 }
4758 enc_varint_into(value as usize).len()
4759}
4760
4761/// Minimal `SQLite` varint encoding of a non-negative `value` (task #73). 7-bit
4762/// big-endian groups, high bit set on every group but the last (file-format §2).
4763pub(crate) fn enc_varint_into(value: usize) -> Vec<u8> {
4764 if value == 0 {
4765 return vec![0];
4766 }
4767 let mut groups = Vec::new();
4768 let mut n = value as u64;
4769 while n > 0 {
4770 groups.push((n & 0x7f) as u8);
4771 n >>= 7;
4772 }
4773 groups.reverse();
4774 let last = groups.len() - 1;
4775 for (i, g) in groups.iter_mut().enumerate() {
4776 if i != last {
4777 *g |= 0x80;
4778 }
4779 }
4780 groups
4781}
4782
4783/// The 8-byte rollback-journal segment magic (`pager.c` `aJournalMagic`).
4784const JOURNAL_MAGIC: [u8; 8] = [0xd9, 0xd5, 0x05, 0xf9, 0x20, 0xa1, 0x63, 0xd7];
4785
4786/// Hard cap on page records walked in one journal segment, to bound work on a
4787/// crafted/garbage journal whose stride scan would otherwise run the file length.
4788const MAX_JOURNAL_RECORDS: usize = 1_000_000;
4789
4790/// Sector-size candidates probed when reconstructing a zeroed (PERSIST) journal
4791/// header. Real VFS sector sizes exceed 512, so 512 is a candidate, not an
4792/// assumption; the page size is also tried (file-format §"Rollback Journal").
4793const SECTOR_CANDIDATES: [u32; 3] = [512, 4096, 0]; // 0 = "use page_size"
4794
4795/// Parsed (or reconstructed) rollback-journal header (design §5).
4796///
4797/// `Valid` is a header whose magic is intact (Tier A — hot journal / crash
4798/// residue): every parameter, including the checksum `nonce`, is authoritative.
4799/// `ReconstructedZeroed` is the PERSIST post-commit case (Tier B): the first
4800/// sector was zeroed on commit, so the page size comes from the main database
4801/// and the sector size from candidate scoring — the nonce is gone, so page
4802/// checksums cannot be verified.
4803#[derive(Debug, Clone, Copy, PartialEq, Eq)]
4804pub enum JournalHeader {
4805 /// Tier A: header magic present; all fields trusted (`pager.c` offsets).
4806 Valid {
4807 /// Page records declared in this segment (`0xFFFFFFFF`/`0` ⇒ walk to EOF).
4808 n_rec: u32,
4809 /// Database page count at transaction start (`dbOrigSize`).
4810 mx_page: u32,
4811 /// Checksum initializer (`cksumInit`), offset 12.
4812 nonce: u32,
4813 /// VFS sector size the header is padded to.
4814 sector_size: u32,
4815 /// Database page size at transaction start.
4816 page_size: u32,
4817 },
4818 /// Tier B: header zeroed (PERSIST post-commit); parameters reconstructed.
4819 ReconstructedZeroed {
4820 /// Page size taken from the main database header (authoritative).
4821 page_size: u32,
4822 /// Sector size selected by candidate scoring (record offset stride).
4823 sector_size: u32,
4824 },
4825}
4826
4827/// One pre-transaction page image recovered from a rollback journal (design §5).
4828#[derive(Debug, Clone, PartialEq, Eq)]
4829pub struct JournalPageImage {
4830 /// 1-based database page number this image restores.
4831 pub pgno: u32,
4832 /// 0-based segment index this record came from.
4833 pub segment: usize,
4834 /// The original page content (`page_size` bytes).
4835 pub bytes: Vec<u8>,
4836 /// `Some(true/false)` in Tier A (nonce known) — whether the stored checksum
4837 /// matched; `None` in Tier B (nonce zeroed, unverifiable).
4838 pub checksum_valid: Option<bool>,
4839}
4840
4841/// A parsed rollback journal: its header tier plus the ordered, first-wins
4842/// page images (design §3/§5). The temporal inverse of the WAL overlay —
4843/// these images are the database as it was BEFORE the last transaction.
4844#[derive(Debug, Clone, PartialEq, Eq)]
4845pub struct RollbackJournal {
4846 header: JournalHeader,
4847 images: Vec<JournalPageImage>,
4848 /// Page numbers that appeared more than once (first occurrence kept), each
4849 /// listed once in first-seen order. Empty for a well-formed journal.
4850 duplicate_pgnos: Vec<u32>,
4851}
4852
4853/// The journal page checksum (`pager.c` `pager_cksum`): `nonce` plus every-200th
4854/// byte from the tail, starting at `page_size - 200` and stepping down by 200
4855/// while the index is positive, using wrapping u32 arithmetic. It detects torn
4856/// page writes; it is not a cryptographic integrity guarantee.
4857fn journal_cksum(nonce: u32, page: &[u8]) -> u32 {
4858 let mut sum = nonce;
4859 let mut x = page.len() as i64 - 200;
4860 while x > 0 {
4861 // x is in (0, page.len()) by the loop bound, so indexing is in-range.
4862 if let Some(&b) = page.get(x as usize) {
4863 sum = sum.wrapping_add(u32::from(b));
4864 }
4865 x -= 200;
4866 }
4867 sum
4868}
4869
4870/// Walk page records of `page_size` bytes from `start`, with the checksum
4871/// `nonce` (`None` ⇒ Tier B, unverifiable), stopping at EOF or after `limit`
4872/// records. Returns the images in file order; a partial trailing record is
4873/// dropped (truncation tolerance). Bounded by [`MAX_JOURNAL_RECORDS`].
4874fn walk_journal_records(
4875 bytes: &[u8],
4876 start: usize,
4877 page_size: usize,
4878 nonce: Option<u32>,
4879 segment: usize,
4880 limit: usize,
4881) -> Vec<JournalPageImage> {
4882 let stride = 4usize.saturating_add(page_size).saturating_add(4);
4883 let mut out = Vec::new();
4884 let mut off = start;
4885 let cap = limit.min(MAX_JOURNAL_RECORDS);
4886 while out.len() < cap {
4887 let Some(rec) = bytes.get(off..off.saturating_add(stride)) else {
4888 break; // EOF or partial trailing record: stop (truncation tolerant).
4889 };
4890 let pgno = u32::from_be_bytes([rec[0], rec[1], rec[2], rec[3]]);
4891 if pgno == 0 {
4892 break; // page 0 is not a valid record; treat as end-of-segment.
4893 }
4894 let page = &rec[4..4 + page_size];
4895 let stored = u32::from_be_bytes([
4896 rec[4 + page_size],
4897 rec[5 + page_size],
4898 rec[6 + page_size],
4899 rec[7 + page_size],
4900 ]);
4901 let checksum_valid = nonce.map(|n| journal_cksum(n, page) == stored);
4902 out.push(JournalPageImage {
4903 pgno,
4904 segment,
4905 bytes: page.to_vec(),
4906 checksum_valid,
4907 });
4908 off = off.saturating_add(stride);
4909 }
4910 out
4911}
4912
4913/// Score a candidate record walk for the Tier-B sector reconstruction: more
4914/// records and all page numbers within `1..=page_bound` rank higher; a record
4915/// count of zero scores zero so an off-stride candidate never wins.
4916fn score_journal_candidate(images: &[JournalPageImage], page_bound: u32) -> usize {
4917 if images.is_empty() {
4918 return 0;
4919 }
4920 let in_range = images
4921 .iter()
4922 .filter(|i| i.pgno >= 1 && i.pgno <= page_bound)
4923 .count();
4924 // All-in-range walks are strongly preferred; weight the in-range fraction so
4925 // a candidate that mostly decodes to impossible page numbers loses to one
4926 // that decodes cleanly even with fewer records.
4927 if in_range == images.len() {
4928 1000 + images.len()
4929 } else {
4930 in_range
4931 }
4932}
4933
4934impl RollbackJournal {
4935 /// LOWER-LEVEL, UNAUTHENTICATED parse (design §5): interpret `bytes` as a
4936 /// rollback journal given an externally-supplied `page_size`. Does NOT bind
4937 /// the journal to a particular database — prefer [`Database::rollback_prior`],
4938 /// which supplies the authoritative page size from the main db.
4939 ///
4940 /// Tier A (magic present) trusts the header and verifies each checksum. Tier B
4941 /// (magic absent — PERSIST post-commit) reconstructs the sector size by
4942 /// candidate scoring and walks records (checksums unverifiable). Robust: a
4943 /// malformed/truncated journal yields fewer images, never a panic; a page size
4944 /// that is not a power of two in `[512, 65536]` is a typed
4945 /// [`Error::BadJournalPageSize`] carrying the offending value.
4946 pub fn parse(bytes: &[u8], page_size: u32) -> Result<Self, Error> {
4947 if !(512..=65536).contains(&page_size) || !page_size.is_power_of_two() {
4948 return Err(Error::BadJournalPageSize(page_size));
4949 }
4950 let ps = page_size as usize;
4951 let page_bound = u32::try_from(bytes.len() / ps.max(1)).unwrap_or(u32::MAX);
4952
4953 let header_valid = bytes.len() >= 28 && bytes.starts_with(&JOURNAL_MAGIC);
4954 if header_valid {
4955 // Tier A: trust the header.
4956 let n_rec = be_u32(bytes, 8);
4957 let nonce = be_u32(bytes, 12);
4958 let mx_page = be_u32(bytes, 16);
4959 let sector_size = be_u32(bytes, 20);
4960 let hdr_page_size = be_u32(bytes, 24);
4961 // nRec ∈ {0, 0xFFFFFFFF} ⇒ walk to EOF; else exactly n_rec records.
4962 let limit = if n_rec == 0 || n_rec == u32::MAX {
4963 MAX_JOURNAL_RECORDS
4964 } else {
4965 n_rec as usize
4966 };
4967 let start = sector_size.max(1) as usize;
4968 let imgs = walk_journal_records(bytes, start, ps, Some(nonce), 0, limit);
4969 let header = JournalHeader::Valid {
4970 n_rec,
4971 mx_page,
4972 nonce,
4973 sector_size,
4974 // The journal's pages are images of THIS db, so the externally
4975 // supplied page size is authoritative; expose it even if the
4976 // header field disagrees (a tampered/mismatched header field).
4977 page_size: if hdr_page_size == page_size {
4978 hdr_page_size
4979 } else {
4980 page_size
4981 },
4982 };
4983 return Ok(Self::from_walk(header, imgs));
4984 }
4985
4986 // Tier B: header zeroed/absent (PERSIST post-commit). Score sector
4987 // candidates and pick the best; checksums are unverifiable (nonce gone).
4988 let mut best: Option<(usize, u32, Vec<JournalPageImage>)> = None;
4989 for cand in SECTOR_CANDIDATES {
4990 let sector = if cand == 0 { page_size } else { cand };
4991 let imgs =
4992 walk_journal_records(bytes, sector as usize, ps, None, 0, MAX_JOURNAL_RECORDS);
4993 let score = score_journal_candidate(&imgs, page_bound);
4994 // `map_or(true, …)` not `is_none_or` to keep the library MSRV at 1.80
4995 // (`Option::is_none_or` stabilised in 1.82); clippy is MSRV-aware.
4996 let better = best.as_ref().map_or(true, |(bs, _, _)| score > *bs);
4997 if better && score > 0 {
4998 best = Some((score, sector, imgs));
4999 }
5000 }
5001 // No candidate decoded a single in-range record (garbage, or a journal too
5002 // short for one record): an empty Tier-B journal, sector size unknown →
5003 // page size. Degrade gracefully rather than erroring.
5004 let (sector_size, imgs) = best
5005 .map(|(_, s, i)| (s, i))
5006 .unwrap_or((page_size, Vec::new()));
5007 let header = JournalHeader::ReconstructedZeroed {
5008 page_size,
5009 sector_size,
5010 };
5011 Ok(Self::from_walk(header, imgs))
5012 }
5013
5014 /// Apply first-wins dedup to a walked record set, recording whether any
5015 /// `pgno` repeated (the duplicate-page anomaly, design §3).
5016 fn from_walk(header: JournalHeader, walked: Vec<JournalPageImage>) -> Self {
5017 let mut seen = std::collections::BTreeSet::new();
5018 let mut images = Vec::with_capacity(walked.len());
5019 let mut duplicate_pgnos: Vec<u32> = Vec::new();
5020 for img in walked {
5021 if seen.insert(img.pgno) {
5022 images.push(img);
5023 } else if !duplicate_pgnos.contains(&img.pgno) {
5024 // Keep the FIRST occurrence as the truest pre-transaction image;
5025 // record WHICH page repeated (once) rather than a bare flag, so the
5026 // anomaly can name the offending page number.
5027 duplicate_pgnos.push(img.pgno);
5028 }
5029 }
5030 Self {
5031 header,
5032 images,
5033 duplicate_pgnos,
5034 }
5035 }
5036
5037 /// The parsed (or reconstructed) header.
5038 #[must_use]
5039 pub fn header(&self) -> &JournalHeader {
5040 &self.header
5041 }
5042
5043 /// The ordered, first-wins pre-transaction page images.
5044 #[must_use]
5045 pub fn page_images(&self) -> &[JournalPageImage] {
5046 &self.images
5047 }
5048
5049 /// Whether a `pgno` appeared more than once across the parsed segments — the
5050 /// spec says a page is journaled at most once, so a repeat is consistent with
5051 /// corruption, a savepoint/super-journal artifact, or tampering (design §3).
5052 #[must_use]
5053 pub fn has_duplicate_pgno(&self) -> bool {
5054 !self.duplicate_pgnos.is_empty()
5055 }
5056
5057 /// The page numbers that appeared more than once (first occurrence kept), each
5058 /// listed once in first-seen order — the offending values behind
5059 /// [`Self::has_duplicate_pgno`]. Empty for a well-formed journal.
5060 #[must_use]
5061 pub fn duplicate_pgnos(&self) -> &[u32] {
5062 &self.duplicate_pgnos
5063 }
5064}
5065
5066/// A read-only, page-addressable image of the database AS IT WAS BEFORE the last
5067/// transaction (design §4/§5). The temporal inverse of [`CommitSnapshot`]:
5068/// `prior[pgno]` is the rollback-journal image where present, else the live main
5069/// page. Diffing this against the current database yields the last transaction's
5070/// deletions (rowid present here, absent now) and modifications (present in both,
5071/// values differ — the journal carries the OLD value).
5072///
5073/// Returned by [`Database::rollback_prior`] as a DISTINCT type, never a
5074/// [`Database`], so prior/deleted rows can never be read as "live"
5075/// (secure-by-design). Shares ONE b-tree/overflow walk with the live and
5076/// commit-snapshot reads via the internal `PageSource` seam.
5077#[derive(Debug, Clone, PartialEq, Eq)]
5078pub struct PriorSnapshot {
5079 /// The pre-transaction page images: journal-where-present overlaid on the main
5080 /// db. Materializes EVERY valid journal page type (interior, leaf, overflow,
5081 /// page 1, freelist trunk, pointer-map) so a prior table can be walked through
5082 /// its interior pages and overflow chains reassembled.
5083 overlaid: std::collections::BTreeMap<u32, Vec<u8>>,
5084 /// Usable bytes per page, parsed from the prior snapshot's OWN page-1 header
5085 /// (so a reserved-space change in the last txn is honored).
5086 usable: u32,
5087 /// The 1-based page count bound (max overlaid page), for cycle/over-range
5088 /// guards in the b-tree / overflow walk.
5089 page_bound: u32,
5090 /// Whether any journal page image's number exceeded the current main-db page
5091 /// count — diagnostic only (the txn grew the db).
5092 grew_db: bool,
5093}
5094
5095impl PageSource for PriorSnapshot {
5096 fn page(&self, page: u32) -> Option<PageBytes<'_>> {
5097 self.overlaid
5098 .get(&page)
5099 .map(|v| PageBytes::Borrowed(v.as_slice()))
5100 }
5101 fn usable(&self) -> usize {
5102 self.usable as usize
5103 }
5104 fn page_bound(&self) -> u32 {
5105 self.page_bound
5106 }
5107 fn encoding(&self) -> TextEncoding {
5108 // Encoding from the prior snapshot's OWN page-1 header (byte 56), so a
5109 // historical read decodes TEXT per the encoding as of the prior state.
5110 self.overlaid
5111 .get(&1)
5112 .map(|p| match be_u32(p, TEXT_ENCODING_OFFSET) {
5113 2 => TextEncoding::Utf16Le,
5114 3 => TextEncoding::Utf16Be,
5115 _ => TextEncoding::Utf8,
5116 })
5117 .unwrap_or_default()
5118 }
5119}
5120
5121impl PriorSnapshot {
5122 /// The user tables AS OF the prior state, parsed from the snapshot's OWN page 1
5123 /// (the prior `sqlite_master`), NOT the live database — so a DROP/CREATE in the
5124 /// last transaction is interpreted against the prior schema. Best-effort and
5125 /// panic-free: an unreadable page-1 schema yields an empty vector.
5126 #[must_use]
5127 pub fn tables(&self) -> Vec<SnapshotTable> {
5128 let Ok(schema) = read_table_via(self, 1, 5) else {
5129 return Vec::new(); // cov:unreachable: the prior snapshot has a readable page 1
5130 };
5131 let mut out = Vec::new();
5132 for row in schema {
5133 let is_table = matches!(row.values.first(), Some(Value::Text(t)) if t == "table");
5134 if !is_table {
5135 continue;
5136 }
5137 let Some(Value::Text(name)) = row.values.get(1) else {
5138 continue; // cov:unreachable: a 'table' schema row has a TEXT name
5139 };
5140 if name.starts_with("sqlite_") {
5141 continue;
5142 }
5143 let Some(Value::Integer(root)) = row.values.get(3) else {
5144 continue; // cov:unreachable: a 'table' schema row has an integer rootpage
5145 };
5146 let Ok(rootpage) = u32::try_from(*root) else {
5147 continue; // cov:unreachable: a real rootpage is a small positive page number
5148 };
5149 let sql = match row.values.get(4) {
5150 Some(Value::Text(s)) => s.as_str(),
5151 _ => "", // cov:unreachable: a 'table' schema row carries its CREATE TABLE sql
5152 };
5153 let columns = attribution::column_names(sql).unwrap_or_default();
5154 out.push(SnapshotTable {
5155 name: name.clone(),
5156 rootpage,
5157 columns,
5158 without_rowid: without_rowid_sql(sql),
5159 });
5160 }
5161 out
5162 }
5163
5164 /// The PRIOR `sqlite_master` as a `name -> CREATE SQL` map for every **user**
5165 /// table, parsed from the snapshot's OWN page 1 — the prior-schema half of the
5166 /// Detector-B sidecar schema-change comparison
5167 /// (`docs/design/drop-recreate-attribution.md`).
5168 ///
5169 /// The counterpart to [`Database::schema_sql`] read against the pre-transaction
5170 /// state the `-journal` preserves, so a DROP/CREATE/ALTER in the last
5171 /// transaction is interpreted against the prior schema. Best-effort and
5172 /// panic-free: an unreadable prior page-1 schema yields an empty map.
5173 #[must_use]
5174 pub fn schema_sql(&self) -> std::collections::BTreeMap<String, String> {
5175 let mut out = std::collections::BTreeMap::new();
5176 let Ok(schema) = read_table_via(self, 1, 5) else {
5177 return out; // cov:unreachable: the prior snapshot has a readable page 1
5178 };
5179 for row in schema {
5180 schema_sql_insert(&mut out, &row.values);
5181 }
5182 out
5183 }
5184
5185 /// Read every row of the table b-tree rooted at `rootpage` AS OF the prior
5186 /// state, in rowid order, resolving overflow chains through the snapshot's OWN
5187 /// pages. The snapshot-scoped counterpart to [`Database::read_table`]: a typed
5188 /// [`Error`] (never a panic) on a cyclic/over-deep b-tree or overflow chain.
5189 pub fn read_table(
5190 &self,
5191 rootpage: u32,
5192 column_count: usize,
5193 ) -> Result<Vec<(i64, Vec<Value>)>, Error> {
5194 let rows = read_table_via(self, rootpage, column_count)?;
5195 Ok(rows.into_iter().map(|r| (r.rowid, r.values)).collect())
5196 }
5197
5198 /// Whether the last transaction GREW the database (a journal page number
5199 /// exceeded the current main-db page count). Pages beyond the prior size are
5200 /// new — their pre-images were not journaled — which bounds what rolls back.
5201 #[must_use]
5202 pub fn grew_db(&self) -> bool {
5203 self.grew_db
5204 }
5205
5206 /// Read the table rooted at `rootpage` AS OF the prior state, returning each
5207 /// row's rowid, values, AND the 1-based LEAF page it was decoded from — the
5208 /// per-row page provenance the forensic diff attaches to a recovered prior
5209 /// row. Shares `decode_leaf_cell` with the standard read; a typed [`Error`]
5210 /// (never a panic) on a cyclic/over-deep b-tree.
5211 pub fn read_table_with_pages(
5212 &self,
5213 rootpage: u32,
5214 column_count: usize,
5215 ) -> Result<Vec<(i64, Vec<Value>, u32)>, Error> {
5216 let mut out = Vec::new();
5217 let mut seen = std::collections::BTreeSet::new();
5218 walk_table_page_with_leaf(self, rootpage, column_count, &mut out, &mut seen)?;
5219 Ok(out)
5220 }
5221}
5222
5223/// Walk a table b-tree like [`walk_table_page`] but record each row's LEAF page,
5224/// for the rollback-journal per-row provenance. Bounded identically (visited-set
5225/// caps recursion depth; a revisited page is silently skipped).
5226fn walk_table_page_with_leaf(
5227 src: &dyn PageSource,
5228 page: u32,
5229 column_count: usize,
5230 out: &mut Vec<(i64, Vec<Value>, u32)>,
5231 seen: &mut std::collections::BTreeSet<u32>,
5232) -> Result<(), Error> {
5233 if seen.len() > MAX_PAGES_PER_WALK {
5234 return Err(Error::TooManyPages);
5235 }
5236 if !seen.insert(page) {
5237 return Ok(());
5238 }
5239 let slice = src.page(page).ok_or(Error::PageOutOfRange(page))?;
5240 let slice = &*slice;
5241 let hdr_off = if page == 1 { SQLITE_HEADER_SIZE } else { 0 };
5242 let page_type = *slice.get(hdr_off).ok_or(Error::TruncatedCell)?;
5243 let cell_count = be_u16(slice, hdr_off + 3) as usize;
5244 match page_type {
5245 0x0d => {
5246 let cell_ptr_array = hdr_off + 8;
5247 for i in 0..cell_count {
5248 let p = cell_ptr_array + i * 2;
5249 let cell_off = be_u16(slice, p) as usize;
5250 let row = decode_leaf_cell(src, slice, cell_off, column_count)?;
5251 out.push((row.rowid, row.values, page));
5252 }
5253 Ok(())
5254 }
5255 0x05 => {
5256 let cell_ptr_array = hdr_off + 12;
5257 for i in 0..cell_count {
5258 let p = cell_ptr_array + i * 2;
5259 let cell_off = be_u16(slice, p) as usize;
5260 let child = be_u32(slice, cell_off);
5261 walk_table_page_with_leaf(src, child, column_count, out, seen)?;
5262 }
5263 let right = be_u32(slice, hdr_off + 8);
5264 walk_table_page_with_leaf(src, right, column_count, out, seen)
5265 }
5266 other => Err(Error::NotATablePage(other)),
5267 }
5268}
5269
5270/// Bounds-checked big-endian u32; out-of-range yields 0 (never panics).
5271fn be_u32(buf: &[u8], off: usize) -> u32 {
5272 let mut b = [0u8; 4];
5273 if let Some(s) = buf.get(off..off + 4) {
5274 b.copy_from_slice(s);
5275 }
5276 u32::from_be_bytes(b)
5277}
5278
5279#[cfg(test)]
5280mod tests {
5281 use super::*;
5282
5283 fn page_rc(byte: u8) -> std::rc::Rc<[u8]> {
5284 std::rc::Rc::from(vec![byte].into_boxed_slice())
5285 }
5286
5287 /// Encode `v` as a 9-byte SQLite varint (round-trips through `read_varint`).
5288 fn varint9(v: u64) -> [u8; 9] {
5289 let mut out = [0u8; 9];
5290 let top56 = v >> 8;
5291 for (i, b) in out.iter_mut().take(8).enumerate() {
5292 *b = (((top56 >> (7 * (7 - i))) & 0x7f) as u8) | 0x80;
5293 }
5294 out[8] = (v & 0xff) as u8;
5295 out
5296 }
5297
5298 #[test]
5299 fn inferred_carve_does_not_overflow_on_huge_serials() {
5300 // A record whose serial array declares column body lengths summing past
5301 // usize::MAX must be REJECTED, never panic (debug) or wrap (release). Real
5302 // free-space bytes (Belkasoft corpus) hit this; here we craft it minimally:
5303 // five maximal (i64::MAX) serials, each a text/blob length ~(i64::MAX-12)/2.
5304 let big = varint9(i64::MAX as u64); // serial_body_len ~4.6e18; five overflow usize
5305 let n_serials = 5usize;
5306 let header_len = 1 + n_serials * 9; // 1-byte header_len varint + 5 serials
5307 let payload_len = header_len; // reach the body-sum loop before any body exists
5308 let mut buf = Vec::new();
5309 buf.push(payload_len as u8); // payload_len varint (small, 1 byte)
5310 buf.push(1u8); // rowid varint = 1 (positive)
5311 buf.push(header_len as u8); // header_len varint (1 byte, < 128)
5312 for _ in 0..n_serials {
5313 buf.extend_from_slice(&big);
5314 }
5315 // Must return None (rejected), and above all must not panic/overflow.
5316 let got = try_carve_cell_at(&buf, 0, None, TextEncoding::Utf8);
5317 assert!(
5318 got.is_none(),
5319 "a body-length-overflowing record must be rejected"
5320 );
5321 }
5322
5323 #[test]
5324 fn page_cache_hits_reorders_and_evicts_past_cap() {
5325 let mut cache = PageCache::new();
5326 // Fill exactly to CAP, then one more → the oldest (key 0) is evicted.
5327 for i in 0..=PageCache::CAP {
5328 cache.put(i, page_rc(i as u8));
5329 }
5330 assert!(cache.get(0).is_none(), "oldest entry evicted once past CAP");
5331 assert!(
5332 cache.get(PageCache::CAP).is_some(),
5333 "the newest entry is retained (get-hit + touch)"
5334 );
5335 // Re-put an existing key → the already-present branch (touch, no growth).
5336 let before = cache.order.len();
5337 cache.put(PageCache::CAP, page_rc(0xff));
5338 assert_eq!(cache.order.len(), before, "re-put must not grow the order");
5339 assert_eq!(cache.get(PageCache::CAP).as_deref(), Some(&[0xff][..]));
5340 }
5341
5342 #[test]
5343 fn varint_single_byte() {
5344 assert_eq!(read_varint(&[0x05], 0).unwrap(), (5, 1));
5345 }
5346
5347 #[test]
5348 fn varint_two_bytes() {
5349 // 0x81 0x00 => (1<<7) = 128
5350 assert_eq!(read_varint(&[0x81, 0x00], 0).unwrap(), (128, 2));
5351 }
5352
5353 #[test]
5354 fn varint_truncated_is_err() {
5355 assert_eq!(read_varint(&[0x81], 0), Err(Error::TruncatedCell));
5356 }
5357
5358 #[test]
5359 fn sign_extend_three_byte_negative() {
5360 // 0xFFFFFF as 3-byte => -1
5361 assert_eq!(sign_extend(0x00FF_FFFF, 3), -1);
5362 }
5363
5364 #[test]
5365 fn decode_value_text_and_blob() {
5366 let (v, n) = decode_value(b"hi", 0, 17, TextEncoding::Utf8).unwrap(); // 17 => text len (17-13)/2 =2
5367 assert_eq!(v, Value::Text("hi".into()));
5368 assert_eq!(n, 2);
5369 let (v, n) = decode_value(&[0xAA, 0xBB], 0, 16, TextEncoding::Utf8).unwrap(); // 16 => blob len 2
5370 assert_eq!(v, Value::Blob(vec![0xAA, 0xBB]));
5371 assert_eq!(n, 2);
5372 }
5373
5374 #[test]
5375 fn decode_value_text_utf16_le_and_be() {
5376 // The TEXT decode path honors the database encoding (file-format §1.3.1):
5377 // the same code points must round-trip from both byte orders. This drives
5378 // `decode_utf16` deterministically, without depending on an external
5379 // `sqlite3`-minted fixture (the integration tests skip when absent).
5380 // Serial 21 => text byte length (21-13)/2 = 4 = two UTF-16 code units.
5381 let le = [b'h', 0x00, b'i', 0x00];
5382 let (v, n) = decode_value(&le, 0, 21, TextEncoding::Utf16Le).unwrap();
5383 assert_eq!(v, Value::Text("hi".into()));
5384 assert_eq!(n, 4);
5385 let be = [0x00, b'h', 0x00, b'i'];
5386 let (v, n) = decode_value(&be, 0, 21, TextEncoding::Utf16Be).unwrap();
5387 assert_eq!(v, Value::Text("hi".into()));
5388 assert_eq!(n, 4);
5389 }
5390
5391 #[test]
5392 fn localstorage_decodes_known_utf16le_bytes() {
5393 // Independent oracle: these UTF-16-LE bytes are derived from the Unicode
5394 // code points and the surrogate-pair formula, NOT from Rust's encoder, so
5395 // a matching round-trip validates the decoder against the documented
5396 // construction (Evidence-Based Rigor tier 2).
5397 // 'A' U+0041 -> 41 00
5398 // '中' U+4E2D -> 2D 4E
5399 // '😀' U+1F600 -> surrogate pair D83D DE00 -> 3D D8 00 DE
5400 let bytes = [0x41, 0x00, 0x2D, 0x4E, 0x3D, 0xD8, 0x00, 0xDE];
5401 let out = decode_localstorage_value(&bytes);
5402 assert_eq!(out.text, "A中😀");
5403 assert!(!out.lossy, "a fully-paired BLOB is not lossy");
5404 }
5405
5406 #[test]
5407 fn localstorage_empty_blob_is_empty_not_lossy() {
5408 let out = decode_localstorage_value(&[]);
5409 assert_eq!(out.text, "");
5410 assert!(!out.lossy);
5411 }
5412
5413 #[test]
5414 fn localstorage_odd_length_blob_is_lossy_not_panic() {
5415 // 'A' (41 00) then a lone trailing byte 42 — half a code unit was cut off.
5416 let out = decode_localstorage_value(&[0x41, 0x00, 0x42]);
5417 assert_eq!(out.text, "A");
5418 assert!(out.lossy, "a trailing half code unit is a lossy truncation");
5419 }
5420
5421 #[test]
5422 fn localstorage_lone_surrogate_is_replacement_and_lossy() {
5423 // High surrogate D83D (LE 3D D8) with no following low surrogate.
5424 let out = decode_localstorage_value(&[0x3D, 0xD8]);
5425 assert_eq!(out.text, "\u{FFFD}");
5426 assert!(out.lossy);
5427 }
5428
5429 #[test]
5430 fn item_table_schema_recognized_and_others_rejected() {
5431 assert!(is_local_storage_item_table("ItemTable"));
5432 assert!(!is_local_storage_item_table("moz_places"));
5433 assert!(!is_local_storage_item_table("itemtable"));
5434 assert!(!is_local_storage_item_table(""));
5435 }
5436
5437 #[test]
5438 fn decode_value_int_literals() {
5439 assert_eq!(
5440 decode_value(&[], 0, 8, TextEncoding::Utf8).unwrap(),
5441 (Value::Integer(0), 0)
5442 );
5443 assert_eq!(
5444 decode_value(&[], 0, 9, TextEncoding::Utf8).unwrap(),
5445 (Value::Integer(1), 0)
5446 );
5447 }
5448
5449 #[test]
5450 fn bad_magic_rejected() {
5451 let mut b = vec![0u8; 100];
5452 b[..16].copy_from_slice(b"NOT SQLITE 3\0\0\0\0");
5453 assert_eq!(parse_header(&b), Err(Error::BadMagic));
5454 }
5455
5456 #[test]
5457 fn too_short_rejected() {
5458 assert_eq!(parse_header(&[0u8; 10]), Err(Error::TooShort));
5459 }
5460
5461 /// The deleted-record carving fixture (see `docs/corpus-catalog.md`).
5462 const DELETED_DB: &[u8] = include_bytes!("../../tests/data/deleted_places.db");
5463 /// A clean DB with one live `moz_places` table and no deletions.
5464 const CLEAN_DB: &[u8] = include_bytes!("../../tests/data/places.db");
5465
5466 #[test]
5467 fn free_regions_is_complement_of_live_extents() {
5468 // Live cells [10,20) and [30,40) within content area [5, 50).
5469 let live = [(10, 20), (30, 40)];
5470 let regions = free_regions(&live, 5, 50);
5471 assert_eq!(regions, vec![(5, 10), (20, 30), (40, 50)]);
5472 // No live cells -> the whole span is free.
5473 assert_eq!(free_regions(&[], 5, 50), vec![(5, 50)]);
5474 // Live cell covering the whole span -> no free region.
5475 assert!(free_regions(&[(0, 100)], 5, 50).is_empty());
5476 }
5477
5478 #[test]
5479 fn live_cell_len_reads_on_page_footprint() {
5480 // Cell: payload_len=3 (varint 0x03), rowid=1 (varint 0x01), 3 payload bytes.
5481 let buf = [0x03, 0x01, 0xAA, 0xBB, 0xCC];
5482 let usable = 4096;
5483 assert_eq!(live_cell_len(&buf, 0, usable), Some(1 + 1 + 3));
5484 // Truncated prefix -> None, never panics.
5485 assert_eq!(live_cell_len(&[0x81], 0, usable), None);
5486 }
5487
5488 #[test]
5489 fn carve_free_regions_recovers_in_page_remnant() {
5490 let db = Database::open(DELETED_DB.to_vec()).unwrap();
5491 // Page 8 is an allocated leaf (live ids 181..=200) whose free gap holds
5492 // deleted-row residue including rowid 237.
5493 let page = db.raw_page(8).unwrap();
5494 let carved = db.carve_free_regions(&page, 6);
5495 assert!(carved.iter().any(|c| c.rowid == 237));
5496 // 0-FP: never a live (id<=200) rowid.
5497 assert!(carved.iter().all(|c| c.rowid > 200));
5498 // A non-leaf page yields nothing.
5499 assert!(db.carve_free_regions(&[0x05u8; 4096], 6).is_empty());
5500 // An empty / too-short slice yields nothing (no panic).
5501 assert!(db.carve_free_regions(&[], 6).is_empty());
5502 }
5503
5504 #[test]
5505 fn carve_leaf_cells_reads_allocated_cells_and_rejects_non_leaf() {
5506 let db = Database::open(DELETED_DB.to_vec()).unwrap();
5507 // Page 8 is an allocated table-leaf (live ids 181..=200); carve_leaf_cells
5508 // decodes every cell the page records as allocated, so the live ids appear
5509 // (unlike carve_free_regions, which excludes them).
5510 let page = db.raw_page(8).unwrap();
5511 let cells = db.carve_leaf_cells(&page);
5512 assert!(
5513 cells.iter().any(|c| c.rowid == 181),
5514 "must read the allocated cells of the leaf"
5515 );
5516 // Page 1 is passed whole (starts with the file magic) → header read at 100.
5517 let _ = db.carve_leaf_cells(&db.raw_page(1).unwrap());
5518 // A non-leaf page (interior 0x05) and an empty/too-short slice yield nothing
5519 // (no panic) — the same defensive arms carve_free_regions guards.
5520 assert!(db.carve_leaf_cells(&[0x05u8; 4096]).is_empty());
5521 assert!(db.carve_leaf_cells(&[]).is_empty());
5522 }
5523
5524 #[test]
5525 fn carve_free_regions_handles_page_one_and_inferred() {
5526 let db = Database::open(DELETED_DB.to_vec()).unwrap();
5527 // Page 1 is passed whole (starts with the file magic) -> the b-tree header
5528 // is read at offset 100, exercising the page-1 branch.
5529 let page1 = db.raw_page(1).unwrap();
5530 let _ = db.carve_free_regions(&page1, 6);
5531 // With column_count_hint = 0, the inferred path runs over the free regions.
5532 let page8 = db.raw_page(8).unwrap();
5533 let inferred = db.carve_free_regions(&page8, 0);
5534 assert!(inferred.iter().any(|c| c.rowid == 237));
5535 }
5536
5537 #[test]
5538 fn live_cell_len_accounts_for_overflow_pointer() {
5539 let usable = 4096usize;
5540 // Non-spilling cell: payload_len small -> footprint = prefix + payload.
5541 // varint 0x03 (payload_len=3), 0x01 (rowid=1), 3 payload bytes.
5542 assert_eq!(live_cell_len(&[0x03, 0x01, 0, 0, 0], 0, usable), Some(5));
5543
5544 // Spilling cell: a payload_len far above the local threshold takes the
5545 // overflow branch -> footprint = prefix + local + 4 (overflow pointer).
5546 // Encode payload_len = 5000 as a 2-byte varint (0xA7 0x08), rowid = 1.
5547 let mut buf = vec![0xA7, 0x08, 0x01];
5548 buf.extend(std::iter::repeat_n(0u8, 5000));
5549 let total = 5000usize;
5550 let local = local_payload_len(total, usable);
5551 assert!(local < total, "this payload must spill");
5552 assert_eq!(live_cell_len(&buf, 0, usable), Some(2 + 1 + local + 4));
5553 }
5554
5555 #[test]
5556 fn carve_cells_inferred_matches_fixed_count() {
5557 let db = Database::open(DELETED_DB.to_vec()).unwrap();
5558 // A freed leaf page body carves the same rows whether the column count is
5559 // fixed at 6 or inferred.
5560 let page = db.raw_page(10).unwrap();
5561 let fixed = db.carve_cells(&page, 6);
5562 let inferred = db.carve_cells_inferred(&page);
5563 assert!(!fixed.is_empty());
5564 let fixed_ids: std::collections::BTreeSet<i64> = fixed.iter().map(|c| c.rowid).collect();
5565 let inf_ids: std::collections::BTreeSet<i64> = inferred.iter().map(|c| c.rowid).collect();
5566 assert!(fixed_ids.is_subset(&inf_ids));
5567 }
5568
5569 #[test]
5570 fn has_user_table_distinguishes_live_and_dropped() {
5571 let live = Database::open(CLEAN_DB.to_vec()).unwrap();
5572 assert!(live.has_user_table());
5573 let with_deletions = Database::open(DELETED_DB.to_vec()).unwrap();
5574 assert!(with_deletions.has_user_table());
5575 }
5576
5577 #[test]
5578 fn live_rowids_collects_live_rows_only() {
5579 let db = Database::open(CLEAN_DB.to_vec()).unwrap();
5580 let ids = db.live_rowids();
5581 // places.db has 5 live rows, rowids 1..=5.
5582 assert_eq!(ids.len(), 5);
5583 assert!(ids.contains(&1) && ids.contains(&5));
5584
5585 // On the deletions fixture, live rowids are 1..=200; none of the deleted
5586 // 201..=400 appear.
5587 let del = Database::open(DELETED_DB.to_vec()).unwrap();
5588 let live = del.live_rowids();
5589 assert!(live.contains(&1) && live.contains(&200));
5590 assert!(!live.contains(&201) && !live.contains(&400));
5591 }
5592
5593 #[test]
5594 fn live_rows_decodes_current_values() {
5595 let db = Database::open(CLEAN_DB.to_vec()).unwrap();
5596 let rows = db.live_rows();
5597 // places.db has 5 live rows keyed by rowid 1..=5, each decoded to values.
5598 assert_eq!(rows.len(), 5);
5599 // Row 1's url column (index 1) is the rust-lang URL (cross-checks that
5600 // values are decoded, not just rowids collected).
5601 let r1 = rows.get(&1).expect("row 1 present");
5602 assert!(
5603 matches!(r1.get(1), Some(Value::Text(t)) if t.contains("rust-lang")),
5604 "row 1 values must be decoded: {r1:?}"
5605 );
5606 // The value map and the rowid set agree on which rows are live.
5607 let ids = db.live_rowids();
5608 assert_eq!(
5609 rows.keys().copied().collect::<Vec<_>>(),
5610 ids.into_iter().collect::<Vec<_>>()
5611 );
5612
5613 // The deletions fixture's table b-tree has an INTERIOR root page (0x05),
5614 // so this exercises the interior-walk branch of collect_rows and confirms
5615 // values are decoded for all 200 live rows.
5616 let del = Database::open(DELETED_DB.to_vec()).unwrap();
5617 let del_rows = del.live_rows();
5618 assert_eq!(del_rows.len(), 200);
5619 let r1 = del_rows.get(&1).expect("live row 1");
5620 assert!(
5621 matches!(r1.get(1), Some(Value::Text(t)) if t.contains("site-1.example")),
5622 "interior-walked live row 1 must decode its url: {r1:?}"
5623 );
5624 }
5625
5626 #[test]
5627 fn live_table_rows_dumps_each_user_table_in_rowid_order() {
5628 let db = Database::open(CLEAN_DB.to_vec()).unwrap();
5629 let dumps = db.live_table_rows();
5630 // places.db has exactly one user table (moz_places); sqlite_* excluded.
5631 assert_eq!(dumps.len(), 1, "one user-table dump expected: {dumps:?}");
5632 let t = &dumps[0];
5633 assert_eq!(t.name, "moz_places");
5634 // Real column names come from the CREATE TABLE, not generic c0..cN.
5635 assert!(
5636 t.column_names.iter().any(|c| c == "url"),
5637 "real column names expected: {:?}",
5638 t.column_names
5639 );
5640 // The rowids must be the live set, in ascending order.
5641 let rowids: Vec<i64> = t.rows.iter().map(|r| r.rowid).collect();
5642 assert_eq!(rowids, vec![1, 2, 3, 4, 5], "rowid order: {rowids:?}");
5643 // The url cell of row 1 decodes (cross-check values are real).
5644 assert!(
5645 matches!(t.rows[0].values.get(1), Some(Value::Text(s)) if s.contains("rust-lang")),
5646 "row 1 url must decode: {:?}",
5647 t.rows[0].values
5648 );
5649 }
5650
5651 #[test]
5652 fn live_table_rows_excludes_internal_tables_and_handles_interior_btree() {
5653 // The deletions fixture has an INTERIOR root page; all 200 live rows dump
5654 // in ascending rowid order, and no sqlite_* table appears.
5655 let db = Database::open(DELETED_DB.to_vec()).unwrap();
5656 let dumps = db.live_table_rows();
5657 assert!(
5658 dumps.iter().all(|t| !t.name.starts_with("sqlite_")),
5659 "internal tables excluded: {:?}",
5660 dumps.iter().map(|t| &t.name).collect::<Vec<_>>()
5661 );
5662 let places = dumps
5663 .iter()
5664 .find(|t| t.name == "moz_places")
5665 .expect("moz_places dump");
5666 assert_eq!(places.rows.len(), 200, "all live rows dumped");
5667 let ids: Vec<i64> = places.rows.iter().map(|r| r.rowid).collect();
5668 assert!(
5669 ids.windows(2).all(|w| w[0] < w[1]),
5670 "rows in ascending rowid order"
5671 );
5672 assert_eq!(*ids.first().unwrap(), 1);
5673 assert_eq!(*ids.last().unwrap(), 200);
5674 }
5675
5676 #[test]
5677 fn live_table_rows_falls_back_to_generic_columns_on_unparseable_schema() {
5678 // Robustness: a damaged CREATE TABLE whose column list cannot be parsed
5679 // must dump the table with generic c0..cN columns (never a fabricated
5680 // real header), while its rows still read. Mint a valid db, then blank out
5681 // the `( ... )` column list in the stored schema SQL in place (same byte
5682 // length), so column_defs yields None for that table.
5683 use crate::rebuild::{build_recovered_db_tables, RecoveredTable as RT};
5684 let seed = vec![RT {
5685 name: "people".to_string(),
5686 columns: vec!["id".to_string(), "name".to_string()],
5687 rows: vec![vec![Value::Integer(1), Value::Text("alice".into())]],
5688 }];
5689 let mut bytes = build_recovered_db_tables(&seed);
5690
5691 // Find the stored `CREATE TABLE "people" (...)` text and overwrite from the
5692 // first '(' through the matching ')' with spaces, leaving `CREATE TABLE
5693 // "people"` (no column list) — unparseable to column_defs.
5694 let needle = b"CREATE TABLE \"people\"";
5695 let start = bytes
5696 .windows(needle.len())
5697 .position(|w| w == needle)
5698 .expect("schema SQL present");
5699 let open = bytes[start..]
5700 .iter()
5701 .position(|&b| b == b'(')
5702 .map(|p| start + p)
5703 .expect("column list open paren");
5704 let close = bytes[open..]
5705 .iter()
5706 .position(|&b| b == b')')
5707 .map(|p| open + p)
5708 .expect("column list close paren");
5709 for b in &mut bytes[open..=close] {
5710 *b = b' ';
5711 }
5712
5713 let db = Database::open(bytes).expect("corrupted-schema db still opens");
5714 let dumps = db.live_table_rows();
5715 let people = dumps
5716 .iter()
5717 .find(|t| t.name == "people")
5718 .expect("people dump present");
5719 // Generic columns sized to the row width (2), never the real id/name.
5720 assert_eq!(
5721 people.column_names,
5722 vec!["c0".to_string(), "c1".to_string()]
5723 );
5724 // The row still decoded despite the schema damage.
5725 assert_eq!(people.rows.len(), 1);
5726 assert_eq!(people.rows[0].values.first(), Some(&Value::Integer(1)));
5727 }
5728
5729 /// Real-corpus freeblock reconstruction: 0C-01 page 2 has six freeblock-head
5730 /// cells the forward parser cannot reach; reconstruction recovers them
5731 /// (including the destroyed-rowid `id` column) from the surviving serial tail.
5732 const NEMETZ_0C_01: &[u8] = include_bytes!("../../tests/data/nemetz/0C/0C-01.db");
5733
5734 #[test]
5735 fn reconstruct_freeblock_records_recovers_clobbered_rows() {
5736 let db = Database::open(NEMETZ_0C_01.to_vec()).unwrap();
5737 let page = db.raw_page(2).unwrap();
5738 let recovered = db.reconstruct_freeblock_records(&page);
5739 // Row 20005 is a freeblock-head cell only reconstruction can recover.
5740 assert!(recovered.iter().any(|c| c.values
5741 == vec![
5742 Value::Integer(20005),
5743 Value::Integer(3_780_322_152),
5744 Value::Integer(3_909_007_646),
5745 Value::Integer(120_462_986),
5746 Value::Integer(1_290_558_629),
5747 ]));
5748 assert!(recovered
5749 .iter()
5750 .all(|c| c.rowid == 0 && c.confidence <= 0.5));
5751 }
5752
5753 /// Real-corpus span-walking reconstruction (task #66): 0D-07 page 3 coalesces
5754 /// three deleted cells into a single freeblock `[0xf79,0xfe0)` —
5755 /// `Luca|Schumacher` (the head), then `Kurt|Schubert`, then `Georg|Schulz`,
5756 /// each prefixed by a stale `00 00 00 NN` freeblock header that clobbers its
5757 /// leading four bytes. A single-shot head reconstruction recovers only the
5758 /// first; walking the template across the whole span recovers all three.
5759 const NEMETZ_0D_07: &[u8] = include_bytes!("../../tests/data/nemetz/0D/0D-07.db");
5760
5761 #[test]
5762 fn reconstruct_freeblock_records_walks_coalesced_cells() {
5763 let db = Database::open(NEMETZ_0D_07.to_vec()).unwrap();
5764 let page = db.raw_page(3).unwrap();
5765 let recovered = db.reconstruct_freeblock_records(&page);
5766 let has = |name: &str, surname: &str| {
5767 recovered.iter().any(|c| {
5768 matches!(c.values.get(1), Some(Value::Text(t)) if t == name)
5769 && matches!(c.values.get(2), Some(Value::Text(t)) if t == surname)
5770 })
5771 };
5772 // The span-head cell a single-shot reconstruction already reached.
5773 assert!(has("Luca", "Schumacher"), "head cell must be recovered");
5774 // The two trailing cells deeper inside the same freeblock — only a
5775 // span-walk reaches these.
5776 assert!(
5777 has("Kurt", "Schubert"),
5778 "second coalesced cell must be recovered"
5779 );
5780 assert!(
5781 has("Georg", "Schulz"),
5782 "third coalesced cell must be recovered"
5783 );
5784 // Every reconstruction carries a destroyed rowid and low confidence.
5785 assert!(recovered
5786 .iter()
5787 .all(|c| c.rowid == 0 && c.confidence <= 0.5));
5788 }
5789
5790 /// Helper: a real opened DB to call the page-slice methods against crafted
5791 /// page byte slices (the methods take `page_bytes` explicitly).
5792 fn opened() -> Database {
5793 Database::open(NEMETZ_0C_01.to_vec()).unwrap()
5794 }
5795
5796 /// A leaf page advertising a freeblock chain but whose cells do not parse
5797 /// yields no template, so reconstruction returns empty (covers the
5798 /// `freeblock_template` rejection arms and the final `None`).
5799 #[test]
5800 fn reconstruct_freeblock_records_without_template_is_empty() {
5801 let db = opened();
5802 let mut page = vec![0u8; 256];
5803 page[0] = 0x0d; // table-leaf
5804 page[1] = 0x00;
5805 page[2] = 0x40; // first freeblock at offset 64
5806 page[3] = 0x00;
5807 page[4] = 0x01; // cell_count = 1
5808 // The single cell pointer (offset 8) points at 0 -> cell_off == 0 -> skipped,
5809 // so no template can be derived.
5810 page[8] = 0x00;
5811 page[9] = 0x00;
5812 // A freeblock at 64: next=0, size=8 (in-bounds), but no template anyway.
5813 page[64] = 0x00;
5814 page[65] = 0x00;
5815 page[66] = 0x00;
5816 page[67] = 0x08;
5817 assert!(db.reconstruct_freeblock_records(&page).is_empty());
5818 }
5819
5820 /// A cyclic freeblock `next` chain terminates (covers the cycle-break guard)
5821 /// and a freeblock whose size runs past the page is skipped — all without a
5822 /// panic.
5823 #[test]
5824 fn reconstruct_freeblock_records_breaks_cyclic_chain() {
5825 let db = opened();
5826 // Build a page WITH a usable template by copying 0C-01 page 2's header +
5827 // first live cell, then point the freeblock chain at itself.
5828 let src = db.raw_page(2).unwrap().to_vec();
5829 let mut page = src.clone();
5830 // Repoint first-freeblock to a self-cycle at offset 100: next -> 100.
5831 page[1] = 0x00;
5832 page[2] = 100;
5833 page[100] = 0x00;
5834 page[101] = 100; // next = 100 (points to itself)
5835 page[102] = 0xff;
5836 page[103] = 0xff; // size huge -> runs past page -> skipped
5837 // Must not panic and must terminate.
5838 let _ = db.reconstruct_freeblock_records(&page);
5839 }
5840
5841 // ---- Tier-2 fragment salvage (task #72) --------------------------------
5842
5843 #[test]
5844 fn is_distinctive_classifies_every_storage_class() {
5845 // TEXT >= 4 UTF-8 bytes and REAL are distinctive; everything else is not.
5846 assert!(is_distinctive(&Value::Text("Anja".into())));
5847 assert!(is_distinctive(&Value::Text("\u{00e4}\u{00f6}".into()))); // 4 UTF-8 bytes
5848 assert!(is_distinctive(&Value::Real(3.5)));
5849 assert!(!is_distinctive(&Value::Text("abc".into()))); // 3 bytes
5850 assert!(!is_distinctive(&Value::Text(String::new())));
5851 assert!(!is_distinctive(&Value::Text("ab\u{fffd}x".into()))); // replacement char
5852 assert!(!is_distinctive(&Value::Integer(20004)));
5853 assert!(!is_distinctive(&Value::Null));
5854 assert!(!is_distinctive(&Value::Blob(vec![1, 2, 3, 4, 5])));
5855 }
5856
5857 /// Build a synthetic 256-byte table-leaf (0x0d) page for the fragment tests.
5858 ///
5859 /// Schema implied by the template live cell: 3 columns
5860 /// `(c0: 1-byte int, c1: TEXT-4, c2: TEXT-4)` → serials `[1, 21, 21]`,
5861 /// `header_len = 4`. The live cell (the freeblock template source) is placed
5862 /// at `live_off`. A single freeblock spanning `[fb, fb + fb_size)` holds the
5863 /// freed-cell payload `freed`, whose leading 4 bytes are the stale freeblock
5864 /// header (`next`, `size`) — exactly what freeblock conversion clobbers.
5865 fn synth_frag_page(live_off: usize, fb: usize, fb_size: usize, freed: &[u8]) -> Vec<u8> {
5866 let mut page = vec![0u8; 256];
5867 page[0] = 0x0d; // table-leaf
5868 page[1] = (fb >> 8) as u8;
5869 page[2] = (fb & 0xff) as u8;
5870 page[3] = 0x00;
5871 page[4] = 0x01; // cell_count = 1
5872 page[5] = (live_off >> 8) as u8;
5873 page[6] = (live_off & 0xff) as u8; // cellContentArea = live_off
5874 page[8] = (live_off >> 8) as u8;
5875 page[9] = (live_off & 0xff) as u8; // cell pointer -> live_off
5876
5877 // Live template cell: payload_len=13, rowid=5, header_len=4, serials
5878 // [int1, text4, text4], body 1+4+4.
5879 let live = [
5880 13u8, 5u8, 0x04, 0x01, 0x15, 0x15, 0x09, b'L', b'i', b'v', b'e', b'R', b'o', b'w', b'!',
5881 ];
5882 page[live_off..live_off + live.len()].copy_from_slice(&live);
5883
5884 // Lay the freed-cell bytes first, then stamp the stale freeblock header
5885 // (next=0, size=fb_size) over its first 4 bytes — exactly what freeblock
5886 // conversion does (the header clobbers the freed cell's leading 4 bytes).
5887 page[fb..fb + freed.len()].copy_from_slice(freed);
5888 page[fb] = 0x00;
5889 page[fb + 1] = 0x00;
5890 page[fb + 2] = (fb_size >> 8) as u8;
5891 page[fb + 3] = (fb_size & 0xff) as u8;
5892 page
5893 }
5894
5895 /// (a) Truncated tail: the freed cell's body overruns the freeblock span, so
5896 /// full reconstruction fails — salvage emits the decodable column prefix
5897 /// (incl. a distinctive TEXT cell) with correct `missing`/confidence, while
5898 /// `reconstruct_freeblock_records` recovers nothing from that anchor.
5899 #[test]
5900 fn fragment_salvage_truncated_tail() {
5901 let db = opened();
5902 // surviving serials [21,21] at fb+4,fb+5; body c0(1)+c1(4)+c2(4) at fb+6.
5903 // A full record needs fb+15. Span size 12 ends at fb+12: c0,c1 fit, c2
5904 // overruns → salvage keeps [c0, c1].
5905 let mut freed = vec![0u8; 16];
5906 freed[4] = 0x15;
5907 freed[5] = 0x15;
5908 freed[6] = 0x07;
5909 freed[7..11].copy_from_slice(b"Anja");
5910 freed[11..15].copy_from_slice(b"Frnk");
5911 let page = synth_frag_page(96, 64, 12, &freed);
5912
5913 let frags = db.reconstruct_freeblock_fragments(&page);
5914 assert_eq!(frags.len(), 1, "exactly one fragment salvaged");
5915 let f = &frags[0];
5916 assert_eq!(f.offset, 64);
5917 assert_eq!(
5918 f.surviving,
5919 vec![(0, Value::Integer(7)), (1, Value::Text("Anja".into()))]
5920 );
5921 assert_eq!(f.missing, 1, "c2 did not decode");
5922 assert!((f.confidence - 0.2).abs() < f32::EPSILON);
5923 let cells = db.reconstruct_freeblock_records(&page);
5924 // The page's only freeblock anchor is the truncated one at offset 64, and
5925 // full reconstruction recovers nothing from it — so the full-record set is
5926 // empty. Asserting emptiness is the precise, deterministic intent.
5927 assert!(
5928 cells.is_empty(),
5929 "the truncated anchor yields no full record, got {}",
5930 cells.len()
5931 );
5932 }
5933
5934 /// (b) A surviving column whose body cannot fit ends the prefix early —
5935 /// salvage keeps the columns decoded before the failure.
5936 #[test]
5937 fn fragment_salvage_partial_tail() {
5938 let db = opened();
5939 let mut freed = vec![0u8; 16];
5940 freed[4] = 0x15;
5941 freed[5] = 0x15;
5942 freed[6] = 0x07;
5943 freed[7..11].copy_from_slice(b"Lena");
5944 let page = synth_frag_page(96, 64, 11, &freed); // c1 fits, c2 overruns
5945 let frags = db.reconstruct_freeblock_fragments(&page);
5946 assert_eq!(frags.len(), 1);
5947 assert_eq!(
5948 frags[0].surviving,
5949 vec![(0, Value::Integer(7)), (1, Value::Text("Lena".into()))]
5950 );
5951 }
5952
5953 /// (c) A fully reconstructable freeblock yields NO fragment (mutual exclusion).
5954 #[test]
5955 fn fragment_salvage_full_record_yields_no_fragment() {
5956 let db = opened();
5957 let mut freed = vec![0u8; 16];
5958 freed[4] = 0x15;
5959 freed[5] = 0x15;
5960 freed[6] = 0x07;
5961 freed[7..11].copy_from_slice(b"Whol");
5962 freed[11..15].copy_from_slice(b"Erow");
5963 let page = synth_frag_page(96, 64, 15, &freed);
5964 let cells = db.reconstruct_freeblock_records(&page);
5965 assert!(
5966 cells.iter().any(|c| c.offset == 64),
5967 "full record recovered"
5968 );
5969 assert!(
5970 db.reconstruct_freeblock_fragments(&page).is_empty(),
5971 "no fragment when the full record is recoverable"
5972 );
5973 }
5974
5975 /// (d) Salvage yielding only non-distinctive (INTEGER) cells emits NO fragment.
5976 #[test]
5977 fn fragment_salvage_integer_only_is_rejected() {
5978 let db = opened();
5979 let mut freed = vec![0u8; 12];
5980 freed[4] = 0x01; // surviving 1-byte int
5981 freed[5] = 0x01; // surviving 1-byte int
5982 freed[6] = 0x07;
5983 freed[7] = 0x08;
5984 let page = synth_frag_page(96, 64, 8, &freed); // c2 overruns; only ints decode
5985 assert!(
5986 db.reconstruct_freeblock_fragments(&page).is_empty(),
5987 "integer-only prefix is not distinctive — no fragment"
5988 );
5989 }
5990
5991 /// (e) Fragment salvage does NOT extend the span walk: a failed head stops
5992 /// the walk, emitting at most one fragment, never sliding forward.
5993 #[test]
5994 fn fragment_salvage_does_not_extend_walk() {
5995 let db = opened();
5996 let mut freed = vec![0u8; 16];
5997 freed[4] = 0x15;
5998 freed[5] = 0x15;
5999 freed[6] = 0x07;
6000 freed[7..11].copy_from_slice(b"Stop");
6001 freed[11..15].copy_from_slice(b"Here");
6002 let page = synth_frag_page(96, 64, 12, &freed);
6003 assert_eq!(db.reconstruct_freeblock_fragments(&page).len(), 1);
6004 }
6005
6006 /// (Step 2) Real-artifact validation: 0D-01 page 2 salvages the genuine
6007 /// partial deleted row for id 20004 — `Text("Anja")`/`Text("Frank")` survive
6008 /// in a freeblock whose full-row reconstruction fails. Full pass unchanged.
6009 const NEMETZ_0D_01: &[u8] = include_bytes!("../../tests/data/nemetz/0D/0D-01.db");
6010
6011 #[test]
6012 fn fragment_salvage_recovers_anja_on_0d01() {
6013 let db = Database::open(NEMETZ_0D_01.to_vec()).unwrap();
6014 let page = db.raw_page(2).unwrap();
6015 let frags = db.reconstruct_freeblock_fragments(&page);
6016 let f = frags
6017 .iter()
6018 .find(|f| {
6019 f.surviving
6020 .iter()
6021 .any(|(_, v)| matches!(v, Value::Text(t) if t == "Anja"))
6022 })
6023 .expect("0D-01 page 2 must salvage the Anja fragment");
6024 assert!(f
6025 .surviving
6026 .iter()
6027 .any(|(_, v)| matches!(v, Value::Text(t) if t == "Frank")));
6028 assert!((f.confidence - 0.2).abs() < f32::EPSILON);
6029 let cells = db.reconstruct_freeblock_records(&page);
6030 assert!(cells.iter().all(|c| !c
6031 .values
6032 .iter()
6033 .any(|v| matches!(v, Value::Text(t) if t == "Anja"))));
6034 }
6035
6036 // ---- task #73: chain-aware overflow recovery — spilled-cell recognition ----
6037
6038 /// Encode a SQLite varint (minimal big-endian 7-bit groups).
6039 fn enc_varint(mut n: u64) -> Vec<u8> {
6040 if n == 0 {
6041 return vec![0];
6042 }
6043 let mut groups = Vec::new();
6044 while n > 0 {
6045 groups.push((n & 0x7f) as u8);
6046 n >>= 7;
6047 }
6048 groups.reverse();
6049 let last = groups.len() - 1;
6050 for (i, g) in groups.iter_mut().enumerate() {
6051 if i != last {
6052 *g |= 0x80;
6053 }
6054 }
6055 groups
6056 }
6057
6058 /// Build the **local prefix** bytes of a freed spilled table-leaf cell:
6059 /// `payload_len varint, rowid varint, record header, local payload bytes,
6060 /// 4-byte big-endian first-overflow pointer`. Returns `(bytes, P, local,
6061 /// serials)`. The record is `(id INTEGER, name TEXT, code TEXT)` with `code`
6062 /// large enough to force a spill past `usable - 35`.
6063 fn synth_spilled_prefix(
6064 rowid: i64,
6065 id: i64,
6066 name: &str,
6067 code_len: usize,
6068 usable: usize,
6069 first_overflow: u32,
6070 ) -> (Vec<u8>, usize, usize, Vec<i64>) {
6071 let id_serial = 1i64; // 1-byte integer
6072 let name_serial = 13 + 2 * name.len() as i64; // TEXT
6073 let code_serial = 13 + 2 * code_len as i64; // TEXT
6074 let serials = vec![id_serial, name_serial, code_serial];
6075 let mut serial_bytes = Vec::new();
6076 for &s in &serials {
6077 serial_bytes.extend(enc_varint(s as u64));
6078 }
6079 // header_len varint counts itself — solve the fixed point.
6080 let mut header_len = serial_bytes.len() + 1;
6081 while enc_varint(header_len as u64).len() + serial_bytes.len() != header_len {
6082 header_len += 1;
6083 }
6084 let mut header = enc_varint(header_len as u64);
6085 header.extend(&serial_bytes);
6086 let body_len = 1 + name.len() + code_len;
6087 let payload_len = header.len() + body_len;
6088 let local = local_payload_len(payload_len, usable);
6089
6090 // Full payload = header ++ id-body ++ name-body ++ code-body.
6091 let mut payload = header.clone();
6092 payload.push(id as u8); // 1-byte id
6093 payload.extend(name.as_bytes());
6094 payload.extend(std::iter::repeat_n(b'C', code_len));
6095 assert_eq!(payload.len(), payload_len);
6096
6097 // Cell = prefix varints ++ local payload prefix ++ 4-byte overflow ptr.
6098 let mut cell = enc_varint(payload_len as u64);
6099 cell.extend(enc_varint(rowid as u64));
6100 cell.extend(&payload[..local]);
6101 cell.extend(first_overflow.to_be_bytes());
6102 (cell, payload_len, local, serials)
6103 }
6104
6105 #[test]
6106 fn spilled_recognizer_reads_intact_prefix() {
6107 let usable = 4096usize;
6108 let (cell, p, local, serials) = synth_spilled_prefix(20012, 42, "Ella", 4200, usable, 13);
6109 assert!(p > usable - 35, "this record must spill");
6110 // Place the cell inside a larger scanned slice at a nonzero offset.
6111 let off = 50usize;
6112 let mut buf = vec![0u8; off];
6113 buf.extend(&cell);
6114 let sc = try_carve_spilled_cell_at(&buf, off, usable, Some(3))
6115 .expect("must recognize the intact-prefix spilled cell");
6116 assert_eq!(sc.payload_len, p);
6117 assert_eq!(sc.local_len, local);
6118 assert_eq!(sc.rowid, 20012);
6119 assert_eq!(sc.first_overflow, 13);
6120 assert_eq!(sc.serials, serials);
6121 assert_eq!(sc.offset, off);
6122 }
6123
6124 #[test]
6125 fn spilled_recognizer_abstains_for_in_page_payload() {
6126 let usable = 4096usize;
6127 // A small (in-page) payload: the existing carve path owns it.
6128 // header (3 serials) + body for a tiny code -> P <= usable-35.
6129 let (cell, p, _local, _s) = synth_spilled_prefix(7, 1, "Bob", 10, usable, 9);
6130 assert!(p <= usable - 35, "this record must NOT spill");
6131 assert!(try_carve_spilled_cell_at(&cell, 0, usable, Some(3)).is_none());
6132 }
6133
6134 #[test]
6135 fn spilled_recognizer_abstains_on_truncated_pointer() {
6136 let usable = 4096usize;
6137 let (cell, _p, _local, _s) = synth_spilled_prefix(20012, 42, "Ella", 4200, usable, 13);
6138 // Drop the final 2 bytes so the 4-byte overflow pointer is out of bounds.
6139 let truncated = &cell[..cell.len() - 2];
6140 assert!(try_carve_spilled_cell_at(truncated, 0, usable, Some(3)).is_none());
6141 }
6142
6143 #[test]
6144 fn spilled_recognizer_abstains_on_column_mismatch() {
6145 let usable = 4096usize;
6146 let (cell, _p, _local, _s) = synth_spilled_prefix(20012, 42, "Ella", 4200, usable, 13);
6147 // Expect 5 columns but the record has 3.
6148 assert!(try_carve_spilled_cell_at(&cell, 0, usable, Some(5)).is_none());
6149 // Inferred (None) still recognizes it.
6150 assert!(try_carve_spilled_cell_at(&cell, 0, usable, None).is_some());
6151 }
6152
6153 #[test]
6154 fn spilled_recognizer_abstains_on_nonpositive_rowid() {
6155 let usable = 4096usize;
6156 let (cell, _p, _local, _s) = synth_spilled_prefix(0, 42, "Ella", 4000, usable, 13);
6157 assert!(try_carve_spilled_cell_at(&cell, 0, usable, Some(3)).is_none());
6158 }
6159
6160 // ---- task #73: freed overflow-chain walk + freelist leaf/trunk split ----
6161
6162 /// Build a minimal multi-page `SQLite` DB image with `page_count` pages of
6163 /// `page_size` bytes. Page 1 carries a valid 100-byte header (so
6164 /// `Database::open` succeeds) with the given freelist trunk pointer and count
6165 /// at offsets 32/36. All pages are zero-filled; the caller writes overflow /
6166 /// trunk content afterwards. Returns the byte vector.
6167 fn synth_db(page_size: usize, page_count: usize, trunk: u32, fl_count: u32) -> Vec<u8> {
6168 let mut b = vec![0u8; page_size * page_count];
6169 b[..16].copy_from_slice(SQLITE_MAGIC);
6170 b[16..18].copy_from_slice(&(page_size as u16).to_be_bytes());
6171 b[18] = 1; // file format write version
6172 b[19] = 1; // file format read version
6173 b[20] = 0; // reserved space
6174 b[21] = 64;
6175 b[22] = 32;
6176 b[23] = 32;
6177 b[32..36].copy_from_slice(&trunk.to_be_bytes());
6178 b[36..40].copy_from_slice(&fl_count.to_be_bytes());
6179 // A minimal table-leaf page-1 body (type 0x0d, 0 cells) so header parsing
6180 // and page-count helpers behave.
6181 b[100] = 0x0d;
6182 b
6183 }
6184
6185 /// Write a freelist trunk page at `page` listing `leaves` and chaining to
6186 /// `next_trunk` (0 = end).
6187 fn write_trunk(b: &mut [u8], page_size: usize, page: u32, next_trunk: u32, leaves: &[u32]) {
6188 let base = (page as usize - 1) * page_size;
6189 b[base..base + 4].copy_from_slice(&next_trunk.to_be_bytes());
6190 b[base + 4..base + 8].copy_from_slice(&(leaves.len() as u32).to_be_bytes());
6191 for (i, &lf) in leaves.iter().enumerate() {
6192 b[base + 8 + i * 4..base + 12 + i * 4].copy_from_slice(&lf.to_be_bytes());
6193 }
6194 }
6195
6196 /// Write an overflow page at `page`: 4-byte big-endian `next` then `content`.
6197 fn write_overflow(b: &mut [u8], page_size: usize, page: u32, next: u32, content: &[u8]) {
6198 let base = (page as usize - 1) * page_size;
6199 b[base..base + 4].copy_from_slice(&next.to_be_bytes());
6200 b[base + 4..base + 4 + content.len()].copy_from_slice(content);
6201 }
6202
6203 #[test]
6204 fn freelist_split_separates_leaves_and_trunks() {
6205 let ps = 512usize;
6206 // Pages: 1 header, 2 trunk, leaves 3,4,5.
6207 let mut b = synth_db(ps, 6, 2, 4);
6208 write_trunk(&mut b, ps, 2, 0, &[3, 4, 5]);
6209 let db = Database::open(b).unwrap();
6210 let (leaves, trunks) = db.freelist_pages_split().unwrap();
6211 assert_eq!(leaves, [3u32, 4, 5].into_iter().collect());
6212 assert_eq!(trunks, [2u32].into_iter().collect());
6213 // The legacy combined accessor still returns leaves ++ trunk.
6214 let all: std::collections::BTreeSet<u32> =
6215 db.freelist_pages().unwrap().into_iter().collect();
6216 assert_eq!(all, [2u32, 3, 4, 5].into_iter().collect());
6217 }
6218
6219 #[test]
6220 fn freed_chain_assembles_single_leaf_page() {
6221 let ps = 512usize;
6222 let usable = ps; // reserved 0
6223 let mut b = synth_db(ps, 6, 2, 4);
6224 write_trunk(&mut b, ps, 2, 0, &[3, 4, 5]);
6225 // Chain content on leaf page 3: a single page holds `remaining` bytes.
6226 let remaining = 100usize;
6227 let content: Vec<u8> = (0..remaining).map(|i| (i % 251) as u8).collect();
6228 write_overflow(&mut b, ps, 3, 0, &content);
6229 let db = Database::open(b).unwrap();
6230 let (leaves, _trunks) = db.freelist_pages_split().unwrap();
6231 let (bytes, chain) = db
6232 .read_freed_overflow_chain(3, remaining, usable, &leaves)
6233 .expect("intact single-leaf chain must assemble");
6234 assert_eq!(bytes, content);
6235 assert_eq!(chain, vec![3]);
6236 }
6237
6238 #[test]
6239 fn freed_chain_assembles_multi_leaf_pages() {
6240 let ps = 512usize;
6241 let usable = ps;
6242 let per_page = usable - 4;
6243 let mut b = synth_db(ps, 8, 2, 5);
6244 write_trunk(&mut b, ps, 2, 0, &[3, 4, 5, 6]);
6245 // 2-page chain: page 3 -> page 4. remaining spans into page 4.
6246 let remaining = per_page + 50;
6247 let content: Vec<u8> = (0..remaining).map(|i| (i % 251) as u8).collect();
6248 write_overflow(&mut b, ps, 3, 4, &content[..per_page]);
6249 write_overflow(&mut b, ps, 4, 0, &content[per_page..]);
6250 let db = Database::open(b).unwrap();
6251 let (leaves, _t) = db.freelist_pages_split().unwrap();
6252 let (bytes, chain) = db
6253 .read_freed_overflow_chain(3, remaining, usable, &leaves)
6254 .expect("intact 2-leaf chain must assemble");
6255 assert_eq!(bytes, content);
6256 assert_eq!(chain, vec![3, 4]);
6257 }
6258
6259 #[test]
6260 fn freed_chain_breaks_on_non_freelist_page() {
6261 let ps = 512usize;
6262 let usable = ps;
6263 let mut b = synth_db(ps, 6, 2, 2);
6264 write_trunk(&mut b, ps, 2, 0, &[3]); // only page 3 is a leaf
6265 let content = vec![7u8; 100];
6266 // The pointer targets page 4, which is NOT on the freelist.
6267 write_overflow(&mut b, ps, 4, 0, &content);
6268 let db = Database::open(b).unwrap();
6269 let (leaves, _t) = db.freelist_pages_split().unwrap();
6270 assert!(db
6271 .read_freed_overflow_chain(4, 100, usable, &leaves)
6272 .is_err());
6273 }
6274
6275 #[test]
6276 fn freed_chain_breaks_on_trunk_page() {
6277 let ps = 512usize;
6278 let usable = ps;
6279 let mut b = synth_db(ps, 6, 2, 2);
6280 write_trunk(&mut b, ps, 2, 0, &[3]);
6281 let db = Database::open(b).unwrap();
6282 let (leaves, _t) = db.freelist_pages_split().unwrap();
6283 // Page 2 is the trunk — a chain page that is a trunk must break.
6284 assert!(db
6285 .read_freed_overflow_chain(2, 100, usable, &leaves)
6286 .is_err());
6287 }
6288
6289 #[test]
6290 fn freed_chain_breaks_on_cycle() {
6291 let ps = 512usize;
6292 let usable = ps;
6293 let per_page = usable - 4;
6294 let mut b = synth_db(ps, 6, 2, 3);
6295 write_trunk(&mut b, ps, 2, 0, &[3, 4]);
6296 // 3 -> 4 -> 3 cycle; remaining never satisfied.
6297 write_overflow(&mut b, ps, 3, 4, &vec![1u8; per_page]);
6298 write_overflow(&mut b, ps, 4, 3, &vec![2u8; per_page]);
6299 let db = Database::open(b).unwrap();
6300 let (leaves, _t) = db.freelist_pages_split().unwrap();
6301 assert!(db
6302 .read_freed_overflow_chain(3, per_page * 10, usable, &leaves)
6303 .is_err());
6304 }
6305
6306 #[test]
6307 fn freed_chain_breaks_on_premature_zero_pointer() {
6308 let ps = 512usize;
6309 let usable = ps;
6310 let per_page = usable - 4;
6311 let mut b = synth_db(ps, 6, 2, 2);
6312 write_trunk(&mut b, ps, 2, 0, &[3]);
6313 // Page 3 ends the chain (next=0) but `remaining` still wants more bytes.
6314 write_overflow(&mut b, ps, 3, 0, &vec![9u8; per_page]);
6315 let db = Database::open(b).unwrap();
6316 let (leaves, _t) = db.freelist_pages_split().unwrap();
6317 assert!(db
6318 .read_freed_overflow_chain(3, per_page + 10, usable, &leaves)
6319 .is_err());
6320 }
6321
6322 #[test]
6323 fn freed_chain_breaks_on_capacity_overflow() {
6324 let ps = 512usize;
6325 let usable = ps;
6326 let mut b = synth_db(ps, 6, 2, 2);
6327 write_trunk(&mut b, ps, 2, 0, &[3]);
6328 write_overflow(&mut b, ps, 3, 0, &vec![1u8; usable - 4]);
6329 let db = Database::open(b).unwrap();
6330 let (leaves, _t) = db.freelist_pages_split().unwrap();
6331 // remaining far exceeds what one leaf page can deliver — rejected upfront,
6332 // never allocating an attacker-declared payload.
6333 let absurd = (usable - 4) * leaves.len() + 1;
6334 assert!(db
6335 .read_freed_overflow_chain(3, absurd, usable, &leaves)
6336 .is_err());
6337 }
6338
6339 // ---- task #73 step 5: freeblock-clobbered spilled cell (SYNTHETIC ONLY) ----
6340 // Codex ruling #5: there is NO corpus instance for a freeblock-clobbered
6341 // *spilled* cell — this path is validated against a synthetic fixture only
6342 // and is marked unproven-by-corpus in the production code + docs.
6343
6344 /// Build a synthetic 4096-byte-page DB with an allocated table-leaf page 2
6345 /// holding (a) a LIVE template cell of the `(id INTEGER 1-byte, name TEXT,
6346 /// code TEXT)` schema and (b) a freeblock-clobbered SPILLED cell whose 4-byte
6347 /// prefix is overwritten by a stale freeblock header, with its overflow chain
6348 /// on a freed leaf page. Returns the bytes. `break_chain` routes the chain
6349 /// pointer at the freelist trunk instead of a leaf to exercise the rejection.
6350 fn synth_clobbered_spill_db(break_chain: bool) -> Vec<u8> {
6351 let ps = 4096usize;
6352 let usable = ps;
6353 // Pages: 1 header, 2 allocated leaf, 3 trunk, 4 leaf (chain), 5 leaf spare.
6354 let mut b = synth_db(ps, 6, 3, 2);
6355 write_trunk(&mut b, ps, 3, 0, &[4, 5]);
6356
6357 // Record geometry: id=7 (1-byte), name="Zoe", code 4200×'C'.
6358 let name = b"Zoe";
6359 let code_len = 4200usize;
6360 let serials: [i64; 3] = [1, 13 + 2 * name.len() as i64, 13 + 2 * code_len as i64];
6361 let mut serial_bytes = Vec::new();
6362 for &s in &serials {
6363 serial_bytes.extend(enc_varint(s as u64));
6364 }
6365 let mut header_len = serial_bytes.len() + 1;
6366 while enc_varint(header_len as u64).len() + serial_bytes.len() != header_len {
6367 header_len += 1;
6368 }
6369 let mut header = enc_varint(header_len as u64);
6370 header.extend(&serial_bytes);
6371 let mut full_payload = header.clone();
6372 full_payload.push(7u8); // id body
6373 full_payload.extend(name);
6374 full_payload.extend(std::iter::repeat_n(b'C', code_len));
6375 let payload_len = full_payload.len();
6376 let local = local_payload_len(payload_len, usable);
6377 let remaining = payload_len - local;
6378
6379 // --- LIVE template cell at offset 200 on page 2 (a small non-spilling row
6380 // of the SAME schema so freeblock_template derives the column layout).
6381 let base2 = ps; // page 2 starts at byte 4096
6382 let tmpl_name = b"Al";
6383 let tmpl_code = b"xy";
6384 let tser: [i64; 3] = [
6385 1,
6386 13 + 2 * tmpl_name.len() as i64,
6387 13 + 2 * tmpl_code.len() as i64,
6388 ];
6389 let mut tsb = Vec::new();
6390 for &s in &tser {
6391 tsb.extend(enc_varint(s as u64));
6392 }
6393 let mut thl = tsb.len() + 1;
6394 while enc_varint(thl as u64).len() + tsb.len() != thl {
6395 thl += 1;
6396 }
6397 let mut tpayload = enc_varint(thl as u64);
6398 tpayload.extend(&tsb);
6399 tpayload.push(1u8);
6400 tpayload.extend(tmpl_name);
6401 tpayload.extend(tmpl_code);
6402 let live_off = 200usize;
6403 let mut live_cell = enc_varint(tpayload.len() as u64);
6404 live_cell.extend(enc_varint(1u64)); // rowid 1
6405 live_cell.extend(&tpayload);
6406 b[base2 + live_off..base2 + live_off + live_cell.len()].copy_from_slice(&live_cell);
6407
6408 // Page-2 leaf header (type 0x0d), 1 live cell, freeblock at 0x100, content
6409 // area covering both the live cell and the clobbered spilled cell.
6410 b[base2] = 0x0d;
6411 // first freeblock pointer (offset 1) -> the clobbered spilled cell at 1000.
6412 b[base2 + 1..base2 + 3].copy_from_slice(&1000u16.to_be_bytes());
6413 // cell count (offset 3) = 1
6414 b[base2 + 3..base2 + 5].copy_from_slice(&1u16.to_be_bytes());
6415 // cell content area start (offset 5) — low so both regions are "content".
6416 b[base2 + 5..base2 + 7].copy_from_slice(&100u16.to_be_bytes());
6417 // cell pointer array (1 entry) at offset 8 -> live cell offset.
6418 b[base2 + 8..base2 + 10].copy_from_slice(&(live_off as u16).to_be_bytes());
6419
6420 // --- Clobbered SPILLED cell at offset 1000 on page 2. Lay down the FULL
6421 // prefix (payload_len varint, rowid varint, header, local payload,
6422 // overflow ptr), then OVERWRITE the first 4 bytes with a stale
6423 // freeblock header (next=0x0000, size) to simulate freeblock clobber.
6424 let spill_off = 1000usize;
6425 let mut spill_cell = enc_varint(payload_len as u64);
6426 spill_cell.extend(enc_varint(1u64)); // rowid (will be clobbered)
6427 let prefix_len = spill_cell.len();
6428 spill_cell.extend(&full_payload[..local]);
6429 let chain_first = if break_chain { 3u32 } else { 4u32 };
6430 spill_cell.extend(chain_first.to_be_bytes());
6431 b[base2 + spill_off..base2 + spill_off + spill_cell.len()].copy_from_slice(&spill_cell);
6432 // Clobber the first 4 bytes with a freeblock header: next=0, size=4.
6433 b[base2 + spill_off] = 0;
6434 b[base2 + spill_off + 1] = 0;
6435 b[base2 + spill_off + 2..base2 + spill_off + 4].copy_from_slice(&4u16.to_be_bytes());
6436
6437 // --- The overflow chain content on freed leaf page 4 (next=0).
6438 write_overflow(&mut b, ps, 4, 0, &full_payload[local..local + remaining]);
6439
6440 let _ = prefix_len;
6441 b
6442 }
6443
6444 #[test]
6445 fn clobbered_spilled_cell_reconstructs_with_unknown_rowid() {
6446 let db = Database::open(synth_clobbered_spill_db(false)).unwrap();
6447 let page2 = db.raw_page(2).unwrap();
6448 let recovered = db.carve_overflow_template_records(&page2);
6449 let (cell, chain) = recovered
6450 .iter()
6451 .find(|(c, _)| matches!(c.values.get(1), Some(Value::Text(t)) if t == "Zoe"))
6452 .expect("synthetic clobbered spilled cell must reconstruct");
6453 // rowid destroyed by the freeblock clobber -> surfaced as 0.
6454 assert_eq!(cell.rowid, 0);
6455 // code fully reassembled across the chain.
6456 assert!(matches!(cell.values.get(2), Some(Value::Text(t)) if t.len() == 4200));
6457 assert_eq!(chain, &vec![4u32]);
6458 }
6459
6460 #[test]
6461 fn clobbered_spilled_broken_chain_yields_no_full_row() {
6462 // Chain pointer routed at the freelist TRUNK (page 3) -> rejected.
6463 let db = Database::open(synth_clobbered_spill_db(true)).unwrap();
6464 let page2 = db.raw_page(2).unwrap();
6465 let recovered = db.carve_overflow_template_records(&page2);
6466 // A chain routed through the freelist trunk is rejected outright, so the
6467 // template carve recovers no full row at all (not merely no "Zoe" row).
6468 assert!(
6469 recovered.is_empty(),
6470 "a trunk-routed broken chain must yield no full row, got {} rows",
6471 recovered.len()
6472 );
6473 }
6474
6475 #[test]
6476 fn enc_varint_into_round_trips_zero_and_multibyte() {
6477 // Zero -> single 0 byte (the NULL-serial / empty-header path).
6478 assert_eq!(enc_varint_into(0), vec![0]);
6479 assert_eq!(varint_len(0), 1);
6480 // Multi-byte: 8413 -> 2-byte varint; round-trips via read_varint.
6481 let v = enc_varint_into(8413);
6482 assert_eq!(varint_len(8413), v.len());
6483 assert_eq!(read_varint(&v, 0).unwrap(), (8413, v.len()));
6484 // Negative input (illegal serial) treated as 1 byte (defensive).
6485 assert_eq!(varint_len(-1), 1);
6486 }
6487
6488 /// Build a 4096-byte-page DB with an allocated table-leaf page 2 holding an
6489 /// **intact-prefix** spilled cell in its unallocated gap, with the overflow
6490 /// chain on a freed leaf page (page 4). Mirrors the real 0E geometry so
6491 /// `carve_overflow_records` (and its fragment dual) can be unit-covered without
6492 /// the corpus. `break_chain` routes the pointer at the freelist trunk.
6493 fn synth_gap_spill_db(break_chain: bool, code_len: usize, name: &str) -> Vec<u8> {
6494 let ps = 4096usize;
6495 let usable = ps;
6496 let mut b = synth_db(ps, 6, 3, 2);
6497 write_trunk(&mut b, ps, 3, 0, &[4, 5]);
6498 let base2 = ps;
6499
6500 // Record: (id INTEGER 1-byte, name TEXT, code TEXT) spilled.
6501 let serials: [i64; 3] = [1, 13 + 2 * name.len() as i64, 13 + 2 * code_len as i64];
6502 let mut serial_bytes = Vec::new();
6503 for &s in &serials {
6504 serial_bytes.extend(enc_varint(s as u64));
6505 }
6506 let mut header_len = serial_bytes.len() + 1;
6507 while enc_varint(header_len as u64).len() + serial_bytes.len() != header_len {
6508 header_len += 1;
6509 }
6510 let mut payload = enc_varint(header_len as u64);
6511 payload.extend(&serial_bytes);
6512 payload.push(9u8); // id body
6513 payload.extend(name.as_bytes());
6514 payload.extend(std::iter::repeat_n(b'C', code_len));
6515 let payload_len = payload.len();
6516 let local = local_payload_len(payload_len, usable);
6517 let remaining = payload_len - local;
6518
6519 // Spilled cell at gap offset 1500 on page 2 (intact prefix).
6520 let spill_off = 1500usize;
6521 let mut cell = enc_varint(payload_len as u64);
6522 cell.extend(enc_varint(5u64)); // rowid 5
6523 cell.extend(&payload[..local]);
6524 let first = if break_chain { 3u32 } else { 4u32 };
6525 cell.extend(first.to_be_bytes());
6526 b[base2 + spill_off..base2 + spill_off + cell.len()].copy_from_slice(&cell);
6527
6528 // Page-2 leaf header: 0 live cells, content area at 100 so the gap [8,100..]
6529 // is scanned. No live cells keeps free_regions = the whole content area.
6530 b[base2] = 0x0d;
6531 b[base2 + 1] = 0; // first freeblock = 0
6532 b[base2 + 2] = 0;
6533 b[base2 + 3..base2 + 5].copy_from_slice(&0u16.to_be_bytes()); // 0 cells
6534 b[base2 + 5..base2 + 7].copy_from_slice(&8u16.to_be_bytes()); // cca low
6535
6536 // Chain content on freed leaf page 4.
6537 write_overflow(&mut b, ps, 4, 0, &payload[local..local + remaining]);
6538 b
6539 }
6540
6541 #[test]
6542 fn carve_overflow_records_resolves_gap_spill() {
6543 let db = Database::open(synth_gap_spill_db(false, 4200, "Nora")).unwrap();
6544 let page2 = db.raw_page(2).unwrap();
6545 let recovered = db.carve_overflow_records(&page2);
6546 let (cell, chain) = recovered
6547 .iter()
6548 .find(|(c, _)| matches!(c.values.get(1), Some(Value::Text(t)) if t == "Nora"))
6549 .expect("gap-resident spilled cell must resolve to a full row");
6550 assert_eq!(cell.rowid, 5);
6551 assert!(matches!(cell.values.get(2), Some(Value::Text(t)) if t.len() == 4200));
6552 assert_eq!(chain, &vec![4u32]);
6553 // Graded below the in-page full-row tier (0.9 * factor).
6554 assert!(cell.confidence < 0.72);
6555 // Non-leaf page yields nothing; empty slice yields nothing.
6556 assert!(db.carve_overflow_records(&[0x05u8; 4096]).is_empty());
6557 assert!(db.carve_overflow_records(&[]).is_empty());
6558 }
6559
6560 #[test]
6561 fn carve_overflow_records_rejects_trunk_chain() {
6562 let db = Database::open(synth_gap_spill_db(true, 4200, "Nora")).unwrap();
6563 let page2 = db.raw_page(2).unwrap();
6564 // Chain routed at the trunk -> no full row recovered at all.
6565 let recovered = db.carve_overflow_records(&page2);
6566 assert!(
6567 recovered.is_empty(),
6568 "a trunk-routed chain must yield no full overflow row, got {} rows",
6569 recovered.len()
6570 );
6571 }
6572
6573 #[test]
6574 fn stale_leaf_chain_with_invalid_utf8_is_rejected() {
6575 // NEGATIVE test (the stale-leaf residual): a chain page that IS a freelist
6576 // leaf and assembles to the exact declared length, but whose content is
6577 // unrelated bytes (invalid UTF-8 in the TEXT column). The freelist-leaf
6578 // requirement passes; the strict-UTF-8 extra-signal gate rejects it from
6579 // Tier-1. This documents the design's limit (Codex ruling #2): the leaf
6580 // requirement cannot prove the bytes are the record — only the UTF-8 gate
6581 // catches the cases the lossy decoder would otherwise mask.
6582 let ps = 4096usize;
6583 let usable = ps;
6584 let mut b = synth_db(ps, 6, 3, 2);
6585 write_trunk(&mut b, ps, 3, 0, &[4, 5]);
6586 let base2 = ps;
6587 let name = "Stale";
6588 let code_len = 4200usize;
6589 let serials: [i64; 3] = [1, 13 + 2 * name.len() as i64, 13 + 2 * code_len as i64];
6590 let mut serial_bytes = Vec::new();
6591 for &s in &serials {
6592 serial_bytes.extend(enc_varint(s as u64));
6593 }
6594 let mut header_len = serial_bytes.len() + 1;
6595 while enc_varint(header_len as u64).len() + serial_bytes.len() != header_len {
6596 header_len += 1;
6597 }
6598 let mut payload = enc_varint(header_len as u64);
6599 payload.extend(&serial_bytes);
6600 payload.push(9u8);
6601 payload.extend(name.as_bytes());
6602 payload.extend(std::iter::repeat_n(b'C', code_len));
6603 let payload_len = payload.len();
6604 let local = local_payload_len(payload_len, usable);
6605 let remaining = payload_len - local;
6606
6607 let spill_off = 1500usize;
6608 let mut cell = enc_varint(payload_len as u64);
6609 cell.extend(enc_varint(5u64));
6610 cell.extend(&payload[..local]);
6611 cell.extend(4u32.to_be_bytes());
6612 b[base2 + spill_off..base2 + spill_off + cell.len()].copy_from_slice(&cell);
6613 b[base2] = 0x0d;
6614 b[base2 + 3..base2 + 5].copy_from_slice(&0u16.to_be_bytes());
6615 b[base2 + 5..base2 + 7].copy_from_slice(&8u16.to_be_bytes());
6616
6617 // Stale leaf content: invalid UTF-8 (0xff bytes) where the TEXT body lands.
6618 let stale = vec![0xffu8; remaining];
6619 write_overflow(&mut b, ps, 4, 0, &stale);
6620
6621 let db = Database::open(b).unwrap();
6622 let page2 = db.raw_page(2).unwrap();
6623 // Decodes mechanically (the leaf assembles exactly), but the strict-UTF-8
6624 // gate rejects it -> NOT a Tier-1 full row.
6625 assert!(db.carve_overflow_records(&page2).is_empty());
6626 }
6627
6628 #[test]
6629 fn carve_overflow_fragments_salvages_broken_gap_spill() {
6630 // Broken chain (trunk) -> the local prefix (id + name) salvages as a fragment.
6631 let db = Database::open(synth_gap_spill_db(true, 4200, "Nora")).unwrap();
6632 let page2 = db.raw_page(2).unwrap();
6633 let frags = db.carve_overflow_fragments(&page2);
6634 let f = frags
6635 .iter()
6636 .find(|f| {
6637 f.surviving
6638 .iter()
6639 .any(|(_, v)| matches!(v, Value::Text(t) if t == "Nora"))
6640 })
6641 .expect("broken-chain gap spill must salvage a fragment");
6642 // id (col 0) survives locally too.
6643 assert!(f
6644 .surviving
6645 .iter()
6646 .any(|(i, v)| *i == 0 && matches!(v, Value::Integer(9))));
6647 // An intact chain produces NO fragment (it is a full row instead), so the
6648 // fragment set is empty — assert that directly rather than over a vacuous
6649 // per-fragment predicate.
6650 let ok = Database::open(synth_gap_spill_db(false, 4200, "Nora")).unwrap();
6651 let ok_page = ok.raw_page(2).unwrap();
6652 assert!(
6653 ok.carve_overflow_fragments(&ok_page).is_empty(),
6654 "an intact chain yields a full row, not a fragment"
6655 );
6656 // Non-leaf / empty inputs yield nothing.
6657 assert!(db.carve_overflow_fragments(&[0x05u8; 4096]).is_empty());
6658 assert!(db.carve_overflow_fragments(&[]).is_empty());
6659 }
6660
6661 // --- WAL frame checksum (file-format §4.2) -------------------------------
6662
6663 #[test]
6664 fn wal_checksum_known_vector_both_endiannesses() {
6665 // The §4.2 algorithm over a hand-constructed 8-byte input, from a zero
6666 // seed. Input is two 32-bit words x0, x1; the recurrence is
6667 // s0 += x0 + s1; s1 += x1 + s0;
6668 // From (s0,s1)=(0,0): s0 = x0; s1 = x1 + x0.
6669 //
6670 // BIG-ENDIAN words (magic 0x377f0683 per the spec): bytes
6671 // [00 00 00 02][00 00 00 03] -> x0=2, x1=3 -> s0=2, s1=5.
6672 let data_be = [0, 0, 0, 2, 0, 0, 0, 3];
6673 assert_eq!(wal_checksum(WalChecksumEndian::Big, 0, 0, &data_be), (2, 5));
6674
6675 // LITTLE-ENDIAN words (magic 0x377f0682): the SAME bytes read LE give
6676 // x0=0x02000000, x1=0x03000000 -> s0=0x02000000,
6677 // s1 = 0x03000000 + 0x02000000 = 0x05000000 (wrapping u32).
6678 assert_eq!(
6679 wal_checksum(WalChecksumEndian::Little, 0, 0, &data_be),
6680 (0x0200_0000, 0x0500_0000)
6681 );
6682
6683 // Seed carries forward: from (s0,s1)=(2,5) over the same BE input ->
6684 // s0 = 2 + (2 + 5) = 9; s1 = 5 + (3 + 9) = 17.
6685 assert_eq!(
6686 wal_checksum(WalChecksumEndian::Big, 2, 5, &data_be),
6687 (9, 17)
6688 );
6689
6690 // Wrapping arithmetic must not panic on overflow (u32 wrap, not i32).
6691 let big = [0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff];
6692 let _ = wal_checksum(WalChecksumEndian::Big, u32::MAX, u32::MAX, &big);
6693 }
6694
6695 #[test]
6696 fn wal_checksum_endian_from_magic_matches_spec() {
6697 // file-format §4.2: 0x377f0683 = BIG-endian words, 0x377f0682 = LITTLE.
6698 assert_eq!(
6699 WalChecksumEndian::from_magic(0x377f_0683),
6700 Some(WalChecksumEndian::Big)
6701 );
6702 assert_eq!(
6703 WalChecksumEndian::from_magic(0x377f_0682),
6704 Some(WalChecksumEndian::Little)
6705 );
6706 assert_eq!(WalChecksumEndian::from_magic(0xdead_beef), None);
6707 }
6708
6709 // --- per-commit schema (CommitSnapshot::tables) -------------------------
6710
6711 /// Wrap a minted main-db image into a `(main, wal)` pair whose WAL commits a
6712 /// full rewrite of every page in ONE commit, with correct §4.2 checksums (so
6713 /// the snapshot is checksum-valid). The snapshot then materializes exactly the
6714 /// minted db, with its real page-1 `sqlite_master` b-tree — the no-sqlite3 way
6715 /// to drive `CommitSnapshot::tables` / snapshot reads against a genuine schema.
6716 fn wrap_db_in_wal(main: &[u8], page_size: u32) -> Vec<u8> {
6717 let ps = page_size as usize;
6718 let n_pages = main.len() / ps;
6719 let endian = WalChecksumEndian::Little; // arbitrary; matches magic below.
6720 let (salt1, salt2) = (0x1234_5678u32, 0x9abc_def0u32);
6721
6722 let mut wal = vec![0u8; 32];
6723 wal[0..4].copy_from_slice(&0x377f_0682u32.to_be_bytes()); // little-endian magic
6724 wal[4..8].copy_from_slice(&3_007_000u32.to_be_bytes());
6725 wal[8..12].copy_from_slice(&page_size.to_be_bytes());
6726 wal[12..16].copy_from_slice(&1u32.to_be_bytes());
6727 wal[16..20].copy_from_slice(&salt1.to_be_bytes());
6728 wal[20..24].copy_from_slice(&salt2.to_be_bytes());
6729 // Header checksum over the first 24 bytes (the seed for the frame chain).
6730 let (mut s0, mut s1) = wal_checksum(endian, 0, 0, &wal[0..24]);
6731 wal[24..28].copy_from_slice(&s0.to_be_bytes());
6732 wal[28..32].copy_from_slice(&s1.to_be_bytes());
6733
6734 for i in 0..n_pages {
6735 let page_no = (i + 1) as u32;
6736 let db_size = if i + 1 == n_pages { n_pages as u32 } else { 0 };
6737 let mut fh = [0u8; 24];
6738 fh[0..4].copy_from_slice(&page_no.to_be_bytes());
6739 fh[4..8].copy_from_slice(&db_size.to_be_bytes());
6740 fh[8..12].copy_from_slice(&salt1.to_be_bytes());
6741 fh[12..16].copy_from_slice(&salt2.to_be_bytes());
6742 let data = &main[i * ps..(i + 1) * ps];
6743 let (n0, n1) = wal_checksum(endian, s0, s1, &fh[0..8]);
6744 let (n0, n1) = wal_checksum(endian, n0, n1, data);
6745 s0 = n0;
6746 s1 = n1;
6747 fh[16..20].copy_from_slice(&s0.to_be_bytes());
6748 fh[20..24].copy_from_slice(&s1.to_be_bytes());
6749 wal.extend_from_slice(&fh);
6750 wal.extend_from_slice(data);
6751 }
6752 wal
6753 }
6754
6755 #[test]
6756 fn snapshot_tables_reads_schema_from_its_own_page_one() {
6757 use crate::rebuild::{build_recovered_db_tables, RecoveredTable as RT};
6758 let seed = vec![RT {
6759 name: "people".to_string(),
6760 columns: vec!["id".to_string(), "name".to_string()],
6761 rows: vec![
6762 vec![Value::Integer(1), Value::Text("alice".into())],
6763 vec![Value::Integer(2), Value::Text("bob".into())],
6764 ],
6765 }];
6766 let main = build_recovered_db_tables(&seed);
6767 let ps = parse_header(&main).unwrap().page_size;
6768 let wal = wrap_db_in_wal(&main, ps);
6769
6770 let db = Database::open_with_wal(main, &wal).unwrap();
6771 let tl = db.wal_timeline().unwrap();
6772 let snap = tl.commit_snapshots().last().unwrap();
6773 assert!(snap.checksum_valid(), "minted WAL must be checksum-valid");
6774
6775 let tables = snap.tables();
6776 let people = tables
6777 .iter()
6778 .find(|t| t.name == "people")
6779 .expect("table 'people' present in snapshot schema");
6780 assert!(people.rootpage >= 2, "rootpage points past page 1");
6781 assert_eq!(people.columns, vec!["id".to_string(), "name".to_string()]);
6782 assert!(!people.without_rowid, "an ordinary rowid table");
6783 // Internal sqlite_* tables are excluded.
6784 assert!(tables.iter().all(|t| !t.name.starts_with("sqlite_")));
6785 }
6786
6787 #[test]
6788 fn snapshot_read_resolves_overflow_through_snapshot_pages_not_live_view() {
6789 // The DEFINING property of the snapshot-scoped read: a spilled (overflow)
6790 // row must decode from the snapshot's OWN pages, even when the live view
6791 // would supply different overflow content. Build a db whose table `t` holds
6792 // one large-blob row (forcing an overflow chain), capture it as the
6793 // snapshot, then CLOBBER the overflow pages in the live main-file image.
6794 // The snapshot read still returns the original blob; a live read sees the
6795 // clobbered bytes — proving the snapshot path does not consult the live view.
6796 use crate::rebuild::{build_recovered_db_tables, RecoveredTable as RT};
6797 let blob: Vec<u8> = (0..9000u32).map(|i| (i % 251) as u8).collect();
6798 let seed = vec![RT {
6799 name: "t".to_string(),
6800 columns: vec!["id".to_string(), "big".to_string()],
6801 rows: vec![vec![Value::Integer(1), Value::Blob(blob.clone())]],
6802 }];
6803 let minted = build_recovered_db_tables(&seed);
6804 let ps = parse_header(&minted).unwrap().page_size;
6805 // The WAL commits the TRUE pages; the snapshot materializes them.
6806 let wal = wrap_db_in_wal(&minted, ps);
6807
6808 // Now clobber the live main image's overflow pages (every page after the
6809 // first two: page 1 schema, page 2 table-leaf, page 3+ overflow) to a
6810 // distinct byte so a live read would mis-decode the blob.
6811 let mut clobbered_main = minted.clone();
6812 for p in clobbered_main.iter_mut().skip(2 * ps as usize) {
6813 *p = 0xEE;
6814 }
6815
6816 let db = Database::open_with_wal(clobbered_main, &wal).unwrap();
6817 let tl = db.wal_timeline().unwrap();
6818 let snap = tl.commit_snapshots().last().unwrap();
6819 let t = snap
6820 .tables()
6821 .into_iter()
6822 .find(|t| t.name == "t")
6823 .expect("table t in snapshot");
6824
6825 let rows = snap.read_table(t.rootpage, t.columns.len()).unwrap();
6826 assert_eq!(rows.len(), 1, "one row at this commit");
6827 let (rowid, values) = &rows[0];
6828 assert_eq!(*rowid, 1);
6829 // The 9000-byte blob reassembles from the SNAPSHOT's overflow pages, intact.
6830 assert_eq!(
6831 values.get(1),
6832 Some(&Value::Blob(blob)),
6833 "overflow blob must reassemble from the snapshot's pages, not the clobbered live view"
6834 );
6835 }
6836
6837 #[test]
6838 fn snapshot_read_walks_interior_btree_in_rowid_order() {
6839 // Many rows force an interior (0x05) table b-tree; the snapshot read must
6840 // descend it and return rows in ascending rowid order — exercising the
6841 // shared walk's interior branch through the snapshot page source.
6842 use crate::rebuild::{build_recovered_db_tables, RecoveredTable as RT};
6843 let rows_seed: Vec<Vec<Value>> = (1..=500i64)
6844 .map(|i| vec![Value::Integer(i), Value::Text(format!("name-{i}"))])
6845 .collect();
6846 let seed = vec![RT {
6847 name: "big".to_string(),
6848 columns: vec!["id".to_string(), "name".to_string()],
6849 rows: rows_seed,
6850 }];
6851 let minted = build_recovered_db_tables(&seed);
6852 let ps = parse_header(&minted).unwrap().page_size;
6853 let wal = wrap_db_in_wal(&minted, ps);
6854
6855 let db = Database::open_with_wal(minted, &wal).unwrap();
6856 let tl = db.wal_timeline().unwrap();
6857 let snap = tl.commit_snapshots().last().unwrap();
6858 let t = snap
6859 .tables()
6860 .into_iter()
6861 .find(|t| t.name == "big")
6862 .expect("table big");
6863 let rows = snap.read_table(t.rootpage, t.columns.len()).unwrap();
6864 assert_eq!(rows.len(), 500, "all rows across the interior b-tree");
6865 let ids: Vec<i64> = rows.iter().map(|(r, _)| *r).collect();
6866 assert!(ids.windows(2).all(|w| w[0] < w[1]), "ascending rowid order");
6867 assert_eq!(*ids.first().unwrap(), 1);
6868 assert_eq!(*ids.last().unwrap(), 500);
6869 }
6870
6871 #[test]
6872 fn without_rowid_sql_detects_the_clause() {
6873 // The WITHOUT ROWID detector keys off the CREATE TABLE tail, tolerant of
6874 // case and whitespace, and does NOT misfire on the literal appearing inside
6875 // a quoted string / column name (file-format §2.4). A WITHOUT ROWID b-tree
6876 // has no rowid key, so this flag gates the snapshot-scoped rowid read.
6877 assert!(without_rowid_sql(
6878 "CREATE TABLE kv(k TEXT PRIMARY KEY, v TEXT) WITHOUT ROWID"
6879 ));
6880 assert!(without_rowid_sql(
6881 "CREATE TABLE kv(k TEXT PRIMARY KEY, v TEXT) without rowid"
6882 ));
6883 // Ordinary tables are NOT flagged.
6884 assert!(!without_rowid_sql(
6885 "CREATE TABLE t(id INTEGER PRIMARY KEY, n TEXT)"
6886 ));
6887 // A column literally named with the words, but not the trailing clause, is
6888 // not a false positive.
6889 assert!(!without_rowid_sql(
6890 "CREATE TABLE t(\"without rowid\" TEXT, x INT)"
6891 ));
6892 }
6893
6894 #[test]
6895 fn is_autoincrement_detects_only_the_real_clause() {
6896 // Positive: an ordinary rowid table declaring INTEGER PRIMARY KEY
6897 // AUTOINCREMENT — case-insensitive and whitespace-tolerant.
6898 assert!(is_autoincrement(
6899 "CREATE TABLE students(id INTEGER PRIMARY KEY AUTOINCREMENT, name TEXT)"
6900 ));
6901 assert!(is_autoincrement(
6902 "create table t( id integer primary key autoincrement )"
6903 ));
6904 // Negative: a plain INTEGER PRIMARY KEY is NOT autoincrement.
6905 assert!(!is_autoincrement(
6906 "CREATE TABLE students(id INTEGER PRIMARY KEY, name TEXT)"
6907 ));
6908 // Negative: a WITHOUT ROWID table cannot be AUTOINCREMENT (no rowid).
6909 assert!(!is_autoincrement(
6910 "CREATE TABLE kv(k INTEGER PRIMARY KEY AUTOINCREMENT, v TEXT) WITHOUT ROWID"
6911 ));
6912 // Negative: a column merely NAMED autoincrement is not the clause.
6913 assert!(!is_autoincrement(
6914 "CREATE TABLE t(\"autoincrement\" INTEGER PRIMARY KEY, x INT)"
6915 ));
6916 // Negative: the keyword inside a quoted string / comment does not qualify.
6917 assert!(!is_autoincrement(
6918 "CREATE TABLE t(id INTEGER PRIMARY KEY, note TEXT DEFAULT 'autoincrement')"
6919 ));
6920 // Negative: AUTOINCREMENT without INTEGER PRIMARY KEY is not a valid clause.
6921 assert!(!is_autoincrement(
6922 "CREATE TABLE t(id INTEGER AUTOINCREMENT, name TEXT)"
6923 ));
6924 }
6925
6926 #[test]
6927 fn sqlite_sequence_reads_present_absent_and_multi() {
6928 // A db with no AUTOINCREMENT table has no sqlite_sequence: empty map
6929 // (NOT seq=0), so callers never invent a high-water mark.
6930 let plain = Database::open(crate::rebuild::build_recovered_db_tables(&[
6931 crate::rebuild::RecoveredTable {
6932 name: "plain".to_string(),
6933 columns: vec!["c0".to_string()],
6934 rows: vec![vec![Value::Integer(1)]],
6935 },
6936 ]))
6937 .expect("minted db opens");
6938 assert!(
6939 plain.sqlite_sequence().is_empty(),
6940 "no AUTOINCREMENT table ⟹ empty sqlite_sequence map"
6941 );
6942
6943 // The b_autoinc fixture maintains sqlite_sequence(students)=5.
6944 let auto =
6945 Database::open(include_bytes!("../../tests/data/drop_recreate/b_autoinc.db").to_vec())
6946 .expect("open b_autoinc.db");
6947 let seq = auto.sqlite_sequence();
6948 assert_eq!(seq.get("students"), Some(&5), "students high-water = 5");
6949
6950 // The upd_autoinc fixture: a single AUTOINCREMENT table t at seq=5.
6951 let upd = Database::open(
6952 include_bytes!("../../tests/data/drop_recreate/upd_autoinc.db").to_vec(),
6953 )
6954 .expect("open upd_autoinc.db");
6955 assert_eq!(upd.sqlite_sequence().get("t"), Some(&5), "t high-water = 5");
6956 }
6957
6958 #[test]
6959 fn schema_sql_reads_current_name_to_create_sql() {
6960 // The live `name -> CREATE SQL` map mirrors live_tables, keyed by name.
6961 let auto =
6962 Database::open(include_bytes!("../../tests/data/drop_recreate/b_autoinc.db").to_vec())
6963 .expect("open b_autoinc.db");
6964 let schema = auto.schema_sql();
6965 let sql = schema.get("students").expect("students present");
6966 assert!(
6967 sql.contains("AUTOINCREMENT"),
6968 "current CREATE SQL carried verbatim: {sql}"
6969 );
6970 }
6971
6972 #[test]
6973 fn prior_snapshot_schema_sql_reads_prior_create_sql() {
6974 // b_journal_altered: the prior (-journal) schema for `students` has NO
6975 // `extra` column, the current schema does → the CREATE SQL texts differ.
6976 let main = include_bytes!("../../tests/data/drop_recreate/b_journal_altered.db").to_vec();
6977 let journal = include_bytes!("../../tests/data/drop_recreate/b_journal_altered.db-journal");
6978 let db = Database::open(main).expect("open b_journal_altered.db");
6979 let prior = db
6980 .rollback_prior(journal)
6981 .expect("rollback_prior parses the PERSIST journal");
6982 let prior_sql = prior.schema_sql();
6983 let prior_students = prior_sql.get("students").expect("prior students present");
6984 assert!(
6985 !prior_students.contains("extra"),
6986 "prior CREATE SQL lacks the ALTER-added column: {prior_students}"
6987 );
6988 let current = db.schema_sql();
6989 assert_ne!(
6990 current.get("students"),
6991 prior_sql.get("students"),
6992 "prior vs current CREATE SQL differ (the ALTER)"
6993 );
6994 }
6995
6996 #[test]
6997 fn prior_snapshot_schema_sql_dml_only_matches_current() {
6998 // b_journal_dml: the last transaction is DML only, so the prior (-journal)
6999 // CREATE SQL for `students` EQUALS the current schema (anti-FP ground truth).
7000 let main = include_bytes!("../../tests/data/drop_recreate/b_journal_dml.db").to_vec();
7001 let journal = include_bytes!("../../tests/data/drop_recreate/b_journal_dml.db-journal");
7002 let db = Database::open(main).expect("open b_journal_dml.db");
7003 let prior = db
7004 .rollback_prior(journal)
7005 .expect("rollback_prior parses the PERSIST journal");
7006 assert_eq!(
7007 db.schema_sql().get("students"),
7008 prior.schema_sql().get("students"),
7009 "DML-only ⟹ prior and current CREATE SQL are identical"
7010 );
7011 }
7012}