Skip to main content

kernel/
page.rs

1//! One page format for the whole file.
2//!
3//! Layout, all little-endian:
4//!   0   magic      u32   0x53454B32
5//!   4   version    u16
6//!   6   kind       u16
7//!   8   tree_id    u16
8//!   10  nentries   u16
9//!   12  page_no    u32   its own number — proves this is the page asked for
10//!   16  free_ptr   u16   lowest payload byte in use; payloads grow downward
11//!   18  format     u16   THE DISK-FORMAT STAMP: `FORMAT_VERSION`, 2 on every
12//!                        page this build creates or rewrites. Zero in an e4
13//!                        pre-release file, which is why a value other than 2
14//!                        is refused rather than read. See
15//!                        docs/core/FORMAT_V2.md.
16//!   20  next_leaf  u32   LEAF: right sibling. INTERIOR: leftmost child
17//!                        (child0). 0 = none.
18//!   24  lsn        u64   RESERVED. Always 0 in Phase 1 — every production
19//!                        call site passes `finalise(0)`. The field exists so
20//!                        the format need not change when it is wired up, and
21//!                        for no other reason. Two hazards attach to it, both
22//!                        real: (1) nothing may compare a page's lsn against a
23//!                        WAL record's to skip replay — log sequence numbers
24//!                        restart after a rotation unless the superblock's
25//!                        floor is applied (see Task 4); (2) recovery may not
26//!                        use it to choose between two leaves claiming the
27//!                        same key — a bulk-packed page and an insert-built
28//!                        page both carry 0. A field documented as meaning
29//!                        something it does not is the same defect as a
30//!                        durability label that does not correspond to
31//!                        behaviour.
32//!   32  reserved2  u32
33//!   36  crc32c     u32   over [0..36] ++ [40..PAGE_SIZE]
34//!   40  slot directory: (u16 offset, u16 len) per entry, growing forward
35//!       ... free space ...
36//!       payloads, growing backward from PAGE_SIZE
37//!
38//! SACRIFICE (Law 4): 8 bytes of every page are identity and checksum, and every
39//! read pays a CRC over 4 KiB. Bought: no damaged page is ever decoded and served,
40//! and no page can be silently substituted for another.
41
42use crate::{Error, Result};
43
44pub const PAGE_SIZE: usize = 4096;
45pub const HEADER_LEN: usize = 40;
46const MAGIC: u32 = 0x5345_4B32;
47const VERSION: u16 = 1;
48const SLOT_LEN: usize = 4;
49
50/// The sekejap disk format, stamped into bytes 18-19 of every page.
51///
52/// Defined ONCE, here, because bytes 18-19 of a page are the only place the
53/// number is written; `sekejap_core::FORMAT_VERSION` and
54/// `sekejap::FORMAT_VERSION` are re-exports of this constant, not copies.
55///
56/// It is NOT `meta::FORMAT_VERSION`, which is the logical superblock version
57/// of the inherited kernel `Store` (1 or 2, chosen by the `compact-cells`
58/// cargo feature) and says what the pages of that store CONTAIN. This one is
59/// the envelope: which sekejap disk format the file is. There is no v1 --
60/// every sekejap from 0.17.0 writes and reads 2, an e4 pre-release file
61/// carries 0 here, and a value that is not 2 is refused with the file
62/// unchanged (docs/core/FORMAT_V2.md).
63pub const FORMAT_VERSION: u16 = 2;
64/// Offset of the disk-format stamp inside a page.
65const FORMAT_AT: usize = 18;
66
67/// The disk-format stamp a page image carries, read straight from bytes
68/// 18-19. Public so a test, a repair tool or an inspector can ask what a
69/// page claims WITHOUT opening it -- `PageRef::open` refuses anything but
70/// [`FORMAT_VERSION`], so by then the answer is already known.
71pub fn format_version(b: &[u8]) -> u16 { rd_u16(b, FORMAT_AT) }
72
73/// The largest a leaf record (`BTree`'s key+value encoding, `SLOT_LEN` bytes
74/// of directory overhead included) can be and still fit an empty leaf.
75/// `BTree::insert` enforces exactly this bound; it is public so a caller
76/// that wants to reject an oversized record BEFORE doing anything
77/// consequential with it -- `Store::put` logging it to the WAL, say -- can
78/// check against the identical number instead of a second copy that could
79/// drift (Task 17 re-review, R1: a record the WAL logged and the tree then
80/// refused is exactly how a legitimate frame ended up sitting in the log
81/// with nothing there to apply it, which is not a case any reader should
82/// have to characterise after the fact).
83pub const MAX_RECORD_LEN: usize = PAGE_SIZE - HEADER_LEN - SLOT_LEN;
84
85#[derive(Debug, Clone, Copy, PartialEq, Eq)]
86#[repr(u16)]
87pub enum PageKind { Free = 0, Meta = 1, Leaf = 2, Interior = 3, Overflow = 4 }
88
89impl PageKind {
90    fn from_u16(v: u16) -> Option<Self> {
91        Some(match v {
92            0 => PageKind::Free, 1 => PageKind::Meta, 2 => PageKind::Leaf,
93            3 => PageKind::Interior, 4 => PageKind::Overflow, _ => return None,
94        })
95    }
96}
97
98fn rd_u16(b: &[u8], at: usize) -> u16 { u16::from_le_bytes([b[at], b[at + 1]]) }
99fn rd_u32(b: &[u8], at: usize) -> u32 { u32::from_le_bytes(b[at..at + 4].try_into().unwrap()) }
100fn rd_u64(b: &[u8], at: usize) -> u64 { u64::from_le_bytes(b[at..at + 8].try_into().unwrap()) }
101
102/// Checksum over everything except the checksum field itself.
103pub fn checksum(b: &[u8]) -> u32 {
104    #[cfg(feature = "write-trace")]
105    let trace_started = crate::write_trace::active().then(std::time::Instant::now);
106    let mut h = crc32c::crc32c(&b[0..36]);
107    h = crc32c::crc32c_append(h, &b[40..PAGE_SIZE]);
108    #[cfg(feature = "write-trace")]
109    if let Some(started) = trace_started {
110        crate::write_trace::add(crate::write_trace::Field::PageCrc, started.elapsed());
111        crate::write_trace::page_crc();
112    }
113    h
114}
115
116pub struct PageMut<'a> { b: &'a mut [u8] }
117
118impl<'a> PageMut<'a> {
119    pub fn init(b: &'a mut [u8], kind: PageKind, tree_id: u16, page_no: u32) -> Self {
120        assert_eq!(b.len(), PAGE_SIZE);
121        b.fill(0);
122        b[0..4].copy_from_slice(&MAGIC.to_le_bytes());
123        b[4..6].copy_from_slice(&VERSION.to_le_bytes());
124        b[6..8].copy_from_slice(&(kind as u16).to_le_bytes());
125        b[8..10].copy_from_slice(&tree_id.to_le_bytes());
126        b[12..16].copy_from_slice(&page_no.to_le_bytes());
127        b[16..18].copy_from_slice(&(PAGE_SIZE as u16).to_le_bytes());
128        b[FORMAT_AT..FORMAT_AT + 2].copy_from_slice(&FORMAT_VERSION.to_le_bytes());
129        PageMut { b }
130    }
131
132    /// Reopen an already-initialised buffer for modification.
133    pub fn reopen(b: &'a mut [u8]) -> Self { PageMut { b } }
134
135    fn nentries(&self) -> usize { rd_u16(self.b, 10) as usize }
136    /// Number of entries currently in a page under construction. Public
137    /// because `btree.rs`'s splits, and `pack_tree` in a later task, need to
138    /// know where the next `insert_slot` call will land while building a
139    /// fresh page from a collected `Vec` of records.
140    pub fn nentries_pub(&self) -> usize { self.nentries() }
141    fn free_ptr(&self) -> usize { rd_u16(self.b, 16) as usize }
142
143    pub fn free_space(&self) -> usize {
144        self.free_ptr().saturating_sub(HEADER_LEN + self.nentries() * SLOT_LEN)
145    }
146
147    pub fn set_next_leaf(&mut self, p: u32) { self.b[20..24].copy_from_slice(&p.to_le_bytes()); }
148    /// Same field, named for its meaning on an interior page.
149    pub fn set_child0(&mut self, p: u32) { self.set_next_leaf(p) }
150
151    /// Insert `rec` so that it becomes entry `at`, shifting later slots right.
152    pub fn insert_slot(&mut self, at: usize, rec: &[u8]) -> Result<()> {
153        let n = self.nentries();
154        if at > n { return Err(Error::TooLarge); }
155        if rec.len() + SLOT_LEN > self.free_space() { return Err(Error::TooLarge); }
156
157        let new_ptr = self.free_ptr() - rec.len();
158        self.b[new_ptr..new_ptr + rec.len()].copy_from_slice(rec);
159        self.b[16..18].copy_from_slice(&(new_ptr as u16).to_le_bytes());
160
161        let dir = HEADER_LEN;
162        let from = dir + at * SLOT_LEN;
163        let to = dir + n * SLOT_LEN;
164        self.b.copy_within(from..to, from + SLOT_LEN);
165        self.b[from..from + 2].copy_from_slice(&(new_ptr as u16).to_le_bytes());
166        self.b[from + 2..from + 4].copy_from_slice(&(rec.len() as u16).to_le_bytes());
167        self.b[10..12].copy_from_slice(&((n + 1) as u16).to_le_bytes());
168        Ok(())
169    }
170
171    /// Remove entry `at`. The payload bytes are abandoned in place; they are
172    /// reclaimed only when the page is compacted.
173    ///
174    /// SACRIFICE (Law 4): a page can hold dead payload bytes it cannot reuse
175    /// until compaction. Bought: deletion touches only the slot directory.
176    pub fn remove_slot(&mut self, at: usize) {
177        let n = self.nentries();
178        assert!(at < n);
179        let dir = HEADER_LEN;
180        let from = dir + (at + 1) * SLOT_LEN;
181        let to = dir + n * SLOT_LEN;
182        self.b.copy_within(from..to, from - SLOT_LEN);
183        self.b[10..12].copy_from_slice(&((n - 1) as u16).to_le_bytes());
184    }
185
186    /// Rewrite every live payload packed against the end of the page, in slot
187    /// order, and reset `free_ptr` to reclaim whatever `remove_slot` left
188    /// abandoned. This is the reclamation `remove_slot`'s doc comment
189    /// promises; without it that comment was a cheque the code could not
190    /// cash, and a caller that relied on it (a replace on an already-tight
191    /// leaf) would corrupt the page instead.
192    ///
193    /// Bounded by one page: the snapshot below is at most `PAGE_SIZE` bytes,
194    /// so this is RAM proportional to the page, not the store (Law 1).
195    pub fn compact(&mut self) {
196        let n = self.nentries();
197        // Snapshot every live payload before any byte moves, since the
198        // write-back below overwrites the very region these slices point
199        // into.
200        let payloads: Vec<Vec<u8>> = (0..n)
201            .map(|i| {
202                let base = HEADER_LEN + i * SLOT_LEN;
203                let off = rd_u16(self.b, base) as usize;
204                let len = rd_u16(self.b, base + 2) as usize;
205                self.b[off..off + len].to_vec()
206            })
207            .collect();
208
209        let mut ptr = PAGE_SIZE;
210        for (i, payload) in payloads.iter().enumerate() {
211            ptr -= payload.len();
212            self.b[ptr..ptr + payload.len()].copy_from_slice(payload);
213            let base = HEADER_LEN + i * SLOT_LEN;
214            self.b[base..base + 2].copy_from_slice(&(ptr as u16).to_le_bytes());
215            // The length half of the slot entry is already correct — a
216            // payload's length never changes across a compaction.
217        }
218        self.b[16..18].copy_from_slice(&(ptr as u16).to_le_bytes());
219    }
220
221    /// Close out a mutation: stamp the LSN. The checksum is stamped by
222    /// [`seal`] immediately before a page is WRITTEN -- a page mutated fifty
223    /// times before eviction needs one checksum, not fifty (measured: 12.34
224    /// full-page CRCs per row, 10 of them re-verifying resident pages, 34% of
225    /// the write path). A dirty resident page's checksum field is meaningless;
226    /// nothing reads it, and the only path to disk seals first.
227    pub fn finalise(&mut self, lsn: u64) {
228        self.b[24..32].copy_from_slice(&lsn.to_le_bytes());
229    }
230}
231
232/// Stamp the checksum over current contents. Called by the pool immediately
233/// before write_at, and nowhere else: checksummed going to the medium,
234/// verified coming back, in between it is just memory (DuckDB: block manager;
235/// SQLite: cksumvfs -- both at the I/O boundary, never per pin).
236/// Stamp the publishing generation into the (formerly reserved) lsn field,
237/// then checksum. Called at the ONLY two places bytes leave for disk
238/// (flush_all, eviction), so every on-disk page carries the generation of
239/// the epoch that wrote it -- the ordering signal recovery's duplicate-key
240/// collapse needs once page numbers are recycled (2n). In-memory
241/// `finalise(0)` call sites are untouched: the stamp happens on the way out.
242pub fn seal(b: &mut [u8], gen: u64) {
243    b[24..32].copy_from_slice(&gen.to_le_bytes());
244    // The stamp is written here as well as in `init` because `seal` is the
245    // ONE place a page image leaves for the medium: a page reached through
246    // `reopen` rather than `init` -- a rewrite -- gets the same guarantee
247    // without every mutation site having to remember it.
248    b[FORMAT_AT..FORMAT_AT + 2].copy_from_slice(&FORMAT_VERSION.to_le_bytes());
249    let c = checksum(b);
250    b[36..40].copy_from_slice(&c.to_le_bytes());
251}
252
253#[derive(Debug)]
254pub struct PageRef<'a> { b: &'a [u8] }
255
256impl<'a> PageRef<'a> {
257    /// Verify and open. `want` is the page number the caller asked the pool for.
258    /// Verify and open a page just off the medium (checksum included). Call
259    /// exactly where a page arrives from disk: BufferPool::load, and the raw
260    /// buffers recover.rs reads itself.
261    pub fn open(b: &'a [u8], want: u32) -> Result<Self> {
262        Self::open_inner(b, want, true)
263    }
264
265    /// Open a resident pool page: every structural check, no CRC -- the pool
266    /// verified it at load and nothing has been believed off the medium since.
267    /// NOT a Law 5 relaxation: the law binds the medium boundary. Bounds
268    /// checks stay per-pin (cheap, and they stop wild reads).
269    pub fn open_resident(b: &'a [u8], want: u32) -> Result<Self> {
270        Self::open_inner(b, want, false)
271    }
272
273    fn open_inner(b: &'a [u8], want: u32, verify_crc: bool) -> Result<Self> {
274        let bad = |why| Err(Error::Corrupt { page_no: want, why });
275        if b.len() != PAGE_SIZE { return bad("wrong buffer length"); }
276        if rd_u32(b, 0) != MAGIC { return bad("bad magic"); }
277        if rd_u16(b, 4) != VERSION { return bad("unknown format version"); }
278        if verify_crc && rd_u32(b, 36) != checksum(b) { return bad("checksum mismatch"); }
279        // After the checksum, never before: a page whose bytes are damaged is
280        // damage, not a foreign format. Only an INTACT page gets to claim a
281        // disk-format version, and a claim other than 2 is refused rather
282        // than read (docs/core/FORMAT_V2.md).
283        let stamp = rd_u16(b, FORMAT_AT);
284        if stamp != FORMAT_VERSION { return Err(Error::UnsupportedFormat { found: stamp }); }
285        if rd_u32(b, 12) != want { return bad("page_no mismatch"); }
286        if PageKind::from_u16(rd_u16(b, 6)).is_none() { return bad("unknown page kind"); }
287
288        let n = rd_u16(b, 10) as usize;
289        let dir_end = HEADER_LEN + n * SLOT_LEN;
290        if dir_end > PAGE_SIZE { return bad("nentries exceeds page"); }
291        let free_ptr = rd_u16(b, 16) as usize;
292        if free_ptr < dir_end || free_ptr > PAGE_SIZE {
293            return bad("free pointer is outside the page payload area");
294        }
295        for i in 0..n {
296            let off = rd_u16(b, HEADER_LEN + i * SLOT_LEN) as usize;
297            let len = rd_u16(b, HEADER_LEN + i * SLOT_LEN + 2) as usize;
298            if off < free_ptr || off.checked_add(len).is_none_or(|end| end > PAGE_SIZE) {
299                return bad("slot out of bounds");
300            }
301        }
302        Ok(PageRef { b })
303    }
304
305    /// Open a page whose slot directory was ALREADY validated once during
306    /// this residency (the pool's per-frame `validated` bit). Skips the
307    /// O(entries) slot-bounds loop; keeps the constant-time identity checks
308    /// (magic, version, page_no, kind) because they also guard against a
309    /// caller-side page-number mixup, not just against disk damage. The
310    /// authoritative validation boundary is `load` (D8): content can only
311    /// have changed since via our own `PageMut` writes, which maintain the
312    /// slot invariants by construction.
313    pub fn open_resident_validated(b: &'a [u8], want: u32) -> Result<Self> {
314        let bad = |why| Err(Error::Corrupt { page_no: want, why });
315        if b.len() != PAGE_SIZE { return bad("wrong buffer length"); }
316        if rd_u32(b, 0) != MAGIC { return bad("bad magic"); }
317        if rd_u16(b, 4) != VERSION { return bad("unknown format version"); }
318        let stamp = rd_u16(b, FORMAT_AT);
319        if stamp != FORMAT_VERSION { return Err(Error::UnsupportedFormat { found: stamp }); }
320        if rd_u32(b, 12) != want { return bad("page_no mismatch"); }
321        if PageKind::from_u16(rd_u16(b, 6)).is_none() { return bad("unknown page kind"); }
322        Ok(PageRef { b })
323    }
324
325    pub fn kind(&self) -> PageKind { PageKind::from_u16(rd_u16(self.b, 6)).unwrap() }
326    pub fn tree_id(&self) -> u16 { rd_u16(self.b, 8) }
327    pub fn nentries(&self) -> usize { rd_u16(self.b, 10) as usize }
328    pub fn page_no(&self) -> u32 { rd_u32(self.b, 12) }
329    pub fn next_leaf(&self) -> u32 { rd_u32(self.b, 20) }
330    /// Same field, named for its meaning on an interior page.
331    pub fn child0(&self) -> u32 { self.next_leaf() }
332    pub fn lsn(&self) -> u64 { rd_u64(self.b, 24) }
333
334    /// The real free-space figure — identical to `PageMut::free_space` —
335    /// available without taking a write pin. A caller that instead sums
336    /// live entry lengths gets a number that silently diverges from this
337    /// one the moment anything on the page was ever deleted, because a
338    /// deletion reclaims only its slot-directory entry until `compact` runs.
339    pub fn free_space(&self) -> usize {
340        let free_ptr = rd_u16(self.b, 16) as usize;
341        free_ptr.saturating_sub(HEADER_LEN + self.nentries() * SLOT_LEN)
342    }
343
344    /// Entry `i`'s slot-directory pair, verbatim: (offset, length) into this
345    /// page image. `slot()` is these two numbers already applied; a caller
346    /// that wants to REMEMBER where a record lives -- rather than copy it out
347    /// -- needs the numbers themselves. Valid for the same reason `slot` is:
348    /// `open`/`open_resident` bounds-checked every pair before handing the
349    /// page over.
350    pub fn slot_bounds(&self, i: usize) -> (u16, u16) {
351        let base = HEADER_LEN + i * SLOT_LEN;
352        (rd_u16(self.b, base), rd_u16(self.b, base + 2))
353    }
354
355    pub fn slot(&self, i: usize) -> &'a [u8] {
356        let off = rd_u16(self.b, HEADER_LEN + i * SLOT_LEN) as usize;
357        let len = rd_u16(self.b, HEADER_LEN + i * SLOT_LEN + 2) as usize;
358        &self.b[off..off + len]
359    }
360}
361
362#[cfg(test)]
363mod tests {
364    use super::*;
365
366    fn buf() -> Vec<u8> { vec![0u8; PAGE_SIZE] }
367
368    #[test]
369    fn a_finalised_page_reads_back_its_identity() {
370        let mut b = buf();
371        let mut p = PageMut::init(&mut b, PageKind::Leaf, 7, 12345);
372        p.finalise(99);
373        seal(&mut b, 99); // seal's stamp is authoritative for the on-disk lsn
374        let r = PageRef::open(&b, 12345).expect("verifies");
375        assert_eq!(r.kind(), PageKind::Leaf);
376        assert_eq!(r.tree_id(), 7);
377        assert_eq!(r.page_no(), 12345);
378        assert_eq!(r.lsn(), 99);
379        assert_eq!(r.nentries(), 0);
380    }
381
382    #[test]
383    fn entries_round_trip_in_slot_order() {
384        let mut b = buf();
385        let mut p = PageMut::init(&mut b, PageKind::Leaf, 1, 4);
386        p.insert_slot(0, b"alpha").unwrap();
387        p.insert_slot(1, b"beta").unwrap();
388        p.insert_slot(1, b"between").unwrap(); // inserted between the two
389        p.finalise(1);
390        seal(&mut b, 7);
391
392        let r = PageRef::open(&b, 4).unwrap();
393        assert_eq!(r.nentries(), 3);
394        assert_eq!(r.slot(0), b"alpha");
395        assert_eq!(r.slot(1), b"between");
396        assert_eq!(r.slot(2), b"beta");
397    }
398
399    #[test]
400    fn a_free_pointer_outside_its_page_is_refused() {
401        let mut b = buf();
402        let mut p = PageMut::init(&mut b, PageKind::Leaf, 1, 4);
403        p.finalise(1);
404        b[16..18].copy_from_slice(&((PAGE_SIZE + 1) as u16).to_le_bytes());
405        seal(&mut b, 7);
406
407        assert!(matches!(
408            PageRef::open(&b, 4),
409            Err(Error::Corrupt { page_no: 4, .. })
410        ));
411    }
412
413    #[test]
414    fn a_page_asked_for_by_the_wrong_number_is_refused() {
415        let mut b = buf();
416        let mut p = PageMut::init(&mut b, PageKind::Leaf, 1, 500);
417        p.finalise(1);
418        seal(&mut b, 7);
419        // The page is intact, but it is not the page that was requested.
420        match PageRef::open(&b, 501) {
421            Err(crate::Error::Corrupt { page_no: 501, why }) => assert_eq!(why, "page_no mismatch"),
422            other => panic!("expected identity refusal, got {other:?}"),
423        }
424    }
425
426    #[test]
427    fn a_single_flipped_byte_anywhere_is_refused() {
428        let mut b = buf();
429        let mut p = PageMut::init(&mut b, PageKind::Leaf, 1, 3);
430        p.insert_slot(0, b"payload-bytes").unwrap();
431        p.finalise(1);
432        seal(&mut b, 7);
433        assert!(PageRef::open(&b, 3).is_ok());
434
435        for pos in [0usize, 9, 21, 37, 41, 100, PAGE_SIZE - 1] {
436            let mut damaged = b.clone();
437            damaged[pos] ^= 0xff;
438            assert!(
439                PageRef::open(&damaged, 3).is_err(),
440                "corruption at byte {pos} was not detected"
441            );
442        }
443    }
444
445    #[test]
446    fn a_corrupt_entry_count_cannot_read_past_the_page() {
447        let mut b = buf();
448        let mut p = PageMut::init(&mut b, PageKind::Leaf, 1, 3);
449        p.insert_slot(0, b"x").unwrap();
450        p.finalise(1);
451        seal(&mut b, 7);
452        // Forge a huge nentries and re-checksum, so only the bound check can save us.
453        b[10..12].copy_from_slice(&60_000u16.to_le_bytes());
454        recrc(&mut b);
455        match PageRef::open(&b, 3) {
456            Err(crate::Error::Corrupt { why, .. }) => assert_eq!(why, "nentries exceeds page"),
457            other => panic!("expected bound refusal, got {other:?}"),
458        }
459    }
460
461    #[test]
462    fn a_record_larger_than_the_page_is_refused_not_panicked() {
463        let mut b = buf();
464        let mut p = PageMut::init(&mut b, PageKind::Leaf, 1, 3);
465        assert!(matches!(p.insert_slot(0, &vec![0u8; PAGE_SIZE]), Err(crate::Error::TooLarge)));
466    }
467
468    /// `compact` is the reclamation `remove_slot`'s doc comment promises.
469    /// Pinned directly: remove the middle entry, compact, and confirm both
470    /// that free space grows back by the removed entry's true size (not
471    /// just its slot-directory entry) and that the surviving entries still
472    /// read back correctly in order.
473    #[test]
474    fn compact_reclaims_a_removed_entrys_payload() {
475        let mut b = buf();
476        let mut p = PageMut::init(&mut b, PageKind::Leaf, 1, 5);
477        p.insert_slot(0, b"alpha").unwrap();
478        p.insert_slot(1, b"between-bytes").unwrap();
479        p.insert_slot(2, b"beta").unwrap();
480        let free_before_remove = p.free_space();
481
482        p.remove_slot(1); // drop "between-bytes"; its payload is only abandoned
483        let free_after_remove = p.free_space();
484        // remove_slot reclaims just the 4-byte slot-directory entry — not
485        // the 13 payload bytes of "between-bytes".
486        assert_eq!(free_after_remove, free_before_remove + 4);
487
488        p.compact();
489        let free_after_compact = p.free_space();
490        assert_eq!(
491            free_after_compact,
492            free_after_remove + "between-bytes".len(),
493            "compact must reclaim the abandoned payload, not just the slot"
494        );
495        p.finalise(1);
496        seal(&mut b, 7);
497
498        let r = PageRef::open(&b, 5).unwrap();
499        assert_eq!(r.nentries(), 2);
500        assert_eq!(r.slot(0), b"alpha");
501        assert_eq!(r.slot(1), b"beta");
502    }
503
504    /// `child0`/`set_child0` are the header field the whole interior-page
505    /// convention hangs off: the leftmost child has no separator key and
506    /// lives here instead of in the slot array. Load-bearing, so it gets its
507    /// own direct round-trip rather than only incidental coverage from btree
508    /// tests.
509    #[test]
510    fn child0_round_trips_through_an_interior_page() {
511        let mut b = buf();
512        let mut p = PageMut::init(&mut b, PageKind::Interior, 1, 9);
513        p.set_child0(4242);
514        p.finalise(1);
515        seal(&mut b, 7);
516
517        let r = PageRef::open(&b, 9).unwrap();
518        assert_eq!(r.kind(), PageKind::Interior);
519        assert_eq!(r.child0(), 4242);
520    }
521
522    fn recrc(b: &mut [u8]) {
523        let c = crate::page::checksum(b);
524        b[36..40].copy_from_slice(&c.to_le_bytes());
525    }
526}