Skip to main content

fdu_core/
snapshot.rs

1//! Persisting an index to disk and reading it back.
2//!
3//! # Status: format v5 is a bounded bootstrap format
4//!
5//! This module implements a flat, uncompressed writer and a bounded streaming reader.
6//! It exists so the cache *lifecycle* — semantic-scope invalidation, atomic replacement,
7//! complete-only persistence, resource limits, and corrupt-equals-empty — is nailed down
8//! before the optimized wire format is designed. The loader checks file size, trailer,
9//! and payload checksum before parsing, then checks the header, record count, and path
10//! lengths before allocating record data and rebuilds one entry at a time.
11//!
12//! The target format is different in every other respect: zstd-compressed blocks with an
13//! index block at the tail, so opening costs one small read and directory listings
14//! decompress on demand into a small LRU cache; item references encoded as
15//! `(block << k) | offset` and delta-encoded when they point inside the same block;
16//! sibling groups written contiguously so one directory listing costs one block
17//! decompression; front-coded names; and pre-computed roll-ups stored per directory so a
18//! query never re-aggregates. O(1) open with lazy materialization matters more than raw
19//! decode throughput, because it matches how a UI actually navigates: open now, expand
20//! later.
21//!
22//! Two things here are **not** placeholders and must survive the format change:
23//!
24//! - **A corrupt or unrecognized snapshot is treated as absent, never as an error and
25//!   never as data.** A cache that fails closed costs a rescan; a cache that fails open
26//!   silently lies.
27//! - **Replacement is atomic.** Write a temporary file, then rename over the target, so a
28//!   crash mid-write leaves the previous snapshot intact rather than a half-written one.
29
30use std::ffi::{OsStr, OsString};
31use std::fs::{self, OpenOptions};
32use std::io::{self, BufReader, Read, Seek, SeekFrom, Write};
33use std::path::{Component, Path, PathBuf};
34use std::sync::atomic::{AtomicU64, Ordering};
35
36use crate::engine_contract::{Attrs, EntryKind, Error, Observation, Op, Result, Source};
37use crate::index::{EntryId, Index, IndexHandle};
38use crate::stored_state::{
39    ControlTierIdentity, SNAPSHOT_IDENTITY_BYTES, Serves, SnapshotIdentity, serves_snapshot,
40};
41
42/// Result of loading a snapshot for one requested stored-state identity.
43#[derive(Debug)]
44#[allow(clippy::large_enum_variant)] // One transient load result; avoid boxing the returned index.
45pub enum LoadOutcome {
46    /// The snapshot supplied an index, either exactly or through a lawful projection.
47    Served {
48        /// The index in the requested scope.
49        index: Index,
50        /// The identity the snapshot declares, which is what a plan admits: a projected
51        /// index reports the requested identity, not this one.
52        stored: SnapshotIdentity,
53    },
54    /// The snapshot was valid, but its identity cannot answer this request.
55    Refused {
56        /// The validated identity that did not serve the request.
57        identity: SnapshotIdentity,
58        /// The validated snapshot root, used before explaining a scope mismatch.
59        root: PathBuf,
60    },
61    /// No valid snapshot was present.
62    Absent,
63}
64
65/// Leading magic. Distinguishes an fdu snapshot from any other file that lands here.
66const MAGIC: &[u8; 8] = b"FDUSNAP\x00";
67
68/// Trailing magic. Present only once the whole snapshot has been written, so a truncated
69/// file is detected on load rather than parsed into a plausible-looking partial tree.
70const TRAILER: &[u8; 8] = b"FDUEND\x00\x00";
71
72/// CRC-32C of every byte before the footer. This detects plausible payload corruption
73/// that remains structurally valid; it is an integrity check, not authentication.
74const CHECKSUM_BYTES: usize = std::mem::size_of::<u32>();
75
76/// Reversed Castagnoli polynomial used by CRC-32C.
77const CRC32C_POLYNOMIAL: u32 = 0x82f6_3b78;
78
79/// One table lookup per byte keeps integrity validation from dominating snapshot load.
80const CRC32C_TABLES: [[u32; 256]; 8] = make_crc32c_tables();
81
82/// On-disk format version. Bump on any layout change; old snapshots are then discarded
83/// rather than misread.
84///
85/// 4: the control section also carries both control limits, the budget and the line
86/// limit, and every refused control file, so a snapshot of an index that refused some
87/// reloads with the same coverage rather than claiming every rule applied.
88///
89/// 5: the header records the identity of each tier the snapshot holds, in the
90/// stored-state model's fixed-width encodings: the entry tier's scope, type rules, and
91/// reducer set, and the control tier's observation and limits. Observation and limits are
92/// no longer mixed into the scope as one hash, and the limits moved out of the control
93/// section, so a section is re-admitted under the header's limits and one they would not
94/// admit is refused. The header also records `writing_pass_started_at_ns`, the start of the
95/// pass that last wrote the image.
96///
97/// 6: the entry tier records ignored population and, when narrowed, the governing
98/// control policy. Narrowed scans cannot reuse broader metadata without projection.
99const FORMAT_VERSION: u32 = 6;
100
101/// Version of the rules that decide which bucket an entry's bytes are tallied under.
102///
103/// Separate from [`FORMAT_VERSION`] because the two answer different questions: the
104/// layout can be unchanged while the meaning of what was written moves. A snapshot stores
105/// the bucket an entry was assigned, not the name it was assigned from, so a rule change
106/// cannot be re-derived on load — the entries have to be discarded and re-walked.
107///
108/// 1: initial rules. 2: files with no extension are tallied under `(none)` instead of
109/// being left out of the extension roll-up entirely.
110const CLASSIFICATION_VERSION: u32 = 2;
111
112/// Version of the per-entry facts used to decide whether retained state is still valid.
113///
114/// Version 2 adds Windows change time, volume identity, and file index. A pre-v2 Windows
115/// snapshot stores zeroes in those fields and cannot safely serve cache-only as if those
116/// facts had been observed. This is mixed into the engine fingerprint on every platform
117/// so one build has one store identity.
118const VALIDITY_VERSION: u32 = 2;
119
120/// Snapshot path encoding used by Unix targets.
121#[cfg(unix)]
122const PATH_ENCODING_UNIX_BYTES: u8 = 1;
123
124/// Snapshot path encoding used by Windows targets.
125#[cfg(windows)]
126const PATH_ENCODING_WINDOWS_WIDE: u8 = 2;
127
128/// Portable UTF-8 fallback for targets outside Unix and Windows.
129#[cfg(not(any(unix, windows)))]
130const PATH_ENCODING_UTF8: u8 = 3;
131
132/// Marks the root's absent parent.
133const NO_PARENT: u32 = u32::MAX;
134
135/// Byte offset of `writing_pass_started_at_ns`: after the magic, the format version, the
136/// engine fingerprint, and the path encoding.
137///
138/// Fixed, so a save can read the stamp of the image already at its path without parsing
139/// the rest of it.
140const WRITING_PASS_STARTED_AT_OFFSET: usize = MAGIC.len() + 4 + 8 + 1;
141
142/// Byte offset of the tier identities, which follow the stamp.
143const IDENTITY_OFFSET: usize = WRITING_PASS_STARTED_AT_OFFSET + 8;
144
145/// Byte offset of the root path, which follows the tier identities.
146#[cfg(test)]
147const ROOT_OFFSET: usize = IDENTITY_OFFSET + SNAPSHOT_IDENTITY_BYTES;
148
149/// Smallest possible on-disk record: parent slot, kind, name length, and six 8-byte
150/// attribute fields, with a zero-length name. Used to sanity-check a declared entry
151/// count against the bytes actually present.
152const MIN_RECORD_BYTES: usize = 4 + 1 + 4 + 8 * 6;
153
154/// Binary size unit used by the snapshot resource limits.
155const GIBIBYTE: u64 = 1024 * 1024 * 1024;
156
157/// Largest snapshot image accepted by the bootstrap streaming reader.
158const MAX_SNAPSHOT_BYTES: u64 = 64 * GIBIBYTE;
159
160/// Process-local discriminator for exclusive sibling temporary files.
161static NEXT_TEMP_FILE: AtomicU64 = AtomicU64::new(0);
162
163/// Per-process entropy mixed into temporary file names.
164///
165/// Correctness never depended on this: `O_EXCL` is what guarantees one creator wins,
166/// and the counter above is what keeps two threads of this process from even trying the
167/// same name. Randomness closes two gaps the counter cannot.
168///
169/// A killed writer leaves its temporary behind and nothing reaps it. With a
170/// pid-and-counter name, a later process that recycles that pid restarts its counter at
171/// zero and collides with the corpse on every run — correct, because it retries, but it
172/// accumulates litter and in the pathological case walks the whole retry budget. And
173/// the names are otherwise entirely predictable, which matters because
174/// [`MAX_TEMP_CREATE_ATTEMPTS`] exists specifically to survive a hostile directory: an
175/// attacker who can guess the names can pre-create the whole budget and deny every save.
176///
177/// Seeded from `RandomState`, whose keys the standard library takes from the operating
178/// system, so this needs no dependency and no clock.
179static TEMP_FILE_ENTROPY: std::sync::LazyLock<u64> = std::sync::LazyLock::new(|| {
180    use std::hash::{BuildHasher, Hasher};
181    std::collections::hash_map::RandomState::new().build_hasher().finish()
182});
183
184/// Stale files can survive a killed writer. Try enough unique sequence values to step
185/// over them without turning an attacker-controlled directory into an unbounded loop.
186const MAX_TEMP_CREATE_ATTEMPTS: usize = 1024;
187
188/// How old an abandoned temporary must be before a later writer removes it.
189///
190/// Per-process entropy in the name means a corpse can never collide with a future
191/// writer — which also means no future writer will ever reuse or overwrite it. Unique
192/// names turn an occasional collision into permanent litter, and each corpse is a whole
193/// snapshot image, so something has to collect them.
194///
195/// A day is far longer than any real save and far shorter than "never". The threshold
196/// is what makes this safe without any liveness check: a temporary this old cannot
197/// belong to a writer that is still running, and pid-based liveness tests are both
198/// unportable and wrong under pid reuse.
199///
200/// Shared with the cache lifecycle, which reclaims the same corpses on request rather
201/// than waiting for the next writer, and must answer "old enough to be nobody's" the same
202/// way this does.
203pub(crate) const STALE_TEMP_AGE: std::time::Duration = std::time::Duration::from_secs(24 * 60 * 60);
204
205/// Largest encoded root or entry name accepted from a snapshot.
206const MAX_PATH_BYTES: u32 = 1024 * 1024;
207
208/// Upper bound on records accepted even when a sparse file could physically hold more.
209const MAX_SNAPSHOT_ENTRIES: u64 = 100_000_000;
210
211/// Binary size unit for the control-section ceiling.
212const MEBIBYTE: usize = 1024 * 1024;
213
214/// Largest control table, in retained charge and in total source bytes, a snapshot may
215/// carry.
216///
217/// A parser guard over untrusted lengths, deliberately independent of both control limits:
218/// they are a caller's runtime settings and may be unbounded, while this bounds what a
219/// corrupt or hostile file can make the loader allocate. [`save`] refuses a table above
220/// it, so a snapshot the loader would reject is never written as if it were usable.
221const SNAPSHOT_CONTROL_TABLE_CEILING: usize = 256 * MEBIBYTE;
222
223/// Encoded refusal reasons.
224const REFUSED_FOR_BUDGET: u8 = 1;
225const REFUSED_FOR_LINE_LIMIT: u8 = 2;
226
227/// A fingerprint of everything that would change how the engine interprets a tree.
228///
229/// When this does not match, the whole snapshot is discarded. That is the cheap,
230/// wholesale answer to "the rules changed, so every derived verdict in the cache might
231/// be wrong" — far simpler than trying to work out which entries a rule change affected.
232///
233/// Currently derived from the crate version, the format version, and the version of the
234/// classification rules. When compiled type-recognition rules and reducer registrations
235/// arrive, their hashes belong here too.
236pub fn engine_fingerprint() -> u64 {
237    let mut hash = 0xcbf2_9ce4_8422_2325_u64;
238    let mut mix = |bytes: &[u8]| {
239        for byte in bytes {
240            hash ^= u64::from(*byte);
241            hash = hash.wrapping_mul(0x1000_0000_01b3);
242        }
243    };
244    mix(env!("CARGO_PKG_VERSION").as_bytes());
245    mix(&FORMAT_VERSION.to_le_bytes());
246    mix(&CLASSIFICATION_VERSION.to_le_bytes());
247    mix(&VALIDITY_VERSION.to_le_bytes());
248    hash
249}
250
251/// Write `index` to `path`, replacing any existing snapshot atomically.
252pub fn save(index: &Index, path: &Path) -> Result<()> {
253    if !crate::stored_state::entries_writable(index) {
254        return Err(Error::Snapshot(
255            "refusing to persist an index that is stale, reconciling, or incomplete".into(),
256        ));
257    }
258    // The loader refuses a table whose limits its scope does not claim, so writing one
259    // would publish a snapshot that every later open discards.
260    index.require_control_limits_in_scope(index.control_table().limits())?;
261    let mut buf: Vec<u8> = Vec::new();
262    buf.extend_from_slice(MAGIC);
263    buf.extend_from_slice(&FORMAT_VERSION.to_le_bytes());
264    buf.extend_from_slice(&engine_fingerprint().to_le_bytes());
265    buf.push(path_encoding());
266    debug_assert_eq!(buf.len(), WRITING_PASS_STARTED_AT_OFFSET);
267    buf.extend_from_slice(&index.writing_pass_started_at_ns().to_le_bytes());
268    // Every tier identity shares the prologue's engine fingerprint, which is this build's.
269    buf.extend_from_slice(&index.snapshot_identity().encode());
270
271    put_os_str(&mut buf, index.root_path().as_os_str())?;
272
273    // Pre-order, so a parent's record always precedes its children's and the loader can
274    // rebuild the tree in one forward pass with no fixups.
275    let mut records: Vec<(u32, EntryId)> = Vec::new();
276    let mut stack: Vec<(u32, EntryId)> = vec![(NO_PARENT, EntryId::ROOT)];
277    while let Some((parent_slot, id)) = stack.pop() {
278        let slot = u32::try_from(records.len())
279            .map_err(|_| Error::Snapshot("snapshot exceeds u32 entry capacity".into()))?;
280        records.push((parent_slot, id));
281        let children = index
282            .children_of(id)
283            .ok_or_else(|| Error::Snapshot("stale entry handle while saving".into()))?;
284        for (_, child) in children.rev() {
285            stack.push((slot, child));
286        }
287    }
288
289    let count = u64::try_from(records.len())
290        .map_err(|_| Error::Snapshot("snapshot entry count overflow".into()))?;
291    buf.extend_from_slice(&count.to_le_bytes());
292
293    for (parent_slot, id) in records {
294        buf.extend_from_slice(&parent_slot.to_le_bytes());
295        let kind = index
296            .kind_of(id)
297            .ok_or_else(|| Error::Snapshot("stale entry handle while saving".into()))?;
298        buf.push(kind as u8);
299        let name = index
300            .name_of(id)
301            .ok_or_else(|| Error::Snapshot("stale entry handle while saving".into()))?;
302        put_os_str(&mut buf, name)?;
303        let attrs = index
304            .attrs_of(id)
305            .ok_or_else(|| Error::Snapshot("stale entry handle while saving".into()))?;
306        buf.extend_from_slice(&attrs.size.to_le_bytes());
307        buf.extend_from_slice(&attrs.allocated.to_le_bytes());
308        buf.extend_from_slice(&attrs.mtime_ns.to_le_bytes());
309        buf.extend_from_slice(&attrs.ctime_ns.to_le_bytes());
310        buf.extend_from_slice(&attrs.inode.to_le_bytes());
311        buf.extend_from_slice(&attrs.dev.to_le_bytes());
312    }
313    put_controls(&mut buf, index.control_table())?;
314    publish(path, buf)
315}
316
317/// Seal a snapshot payload and write it to `path`, keeping an image already there that
318/// encodes the same facts.
319///
320/// Two passes over an unchanged tree encode identical images except for each pass's
321/// start, and rewriting for the stamp alone would give back the skip [`write_atomically`]
322/// exists for. A default run never reads its snapshot for a metadata query, so on that path
323/// the rewrite was the whole cost of having a cache: about 70 ms of a 375 ms run over 175k
324/// entries (exp-066), and a 14 MB write and `F_FULLFSYNC` on every run (exp-067). So an image at `path` that differs only in its stamp is kept, stamp
325/// included, and only its mtime moves. The kept stamp is the start of an earlier pass that
326/// wrote exactly these facts, which can understate how recently they were verified and
327/// never overstates it.
328fn publish(path: &Path, mut payload: Vec<u8>) -> Result<()> {
329    if keep_equivalent_image(path, &payload) {
330        return Ok(());
331    }
332    seal(&mut payload);
333    // Nothing at `path` is this image under any stamp, this pass's included, so comparing
334    // again before replacing it could only repeat the answer.
335    replace_atomically(path, &payload)
336}
337
338/// Keep the file at `path` when it is `payload` sealed under any pass's stamp, moving only
339/// its mtime, to the stamp in `payload`.
340///
341/// The mtime is cache-maintenance metadata, not filesystem-observation provenance: a copy
342/// or a touch must not change when the tree was observed. Moving it to the start of the pass
343/// that would have stamped the image lets the equivalent-image fast path retain its useful
344/// monotonic hint without changing the persisted observation stamp. It never moves backwards:
345/// a save of an index loaded under an older stamp leaves a later mtime in place.
346///
347/// One read through one handle, comparing every payload byte before computing the
348/// checksum, so a changed tree stops at its first difference and costs no checksum here,
349/// and an unchanged one costs the one checksum sealing it would have. The checksum is
350/// still verified before the file is kept: a file whose own checksum is wrong reads as
351/// absent, and keeping it would keep every later run cold.
352fn keep_equivalent_image(path: &Path, payload: &[u8]) -> bool {
353    const FOOTER_BYTES: usize = CHECKSUM_BYTES + TRAILER.len();
354    let Ok(mut file) = fs::File::open(path) else { return false };
355    let Ok(metadata) = file.metadata() else { return false };
356    let image_len = payload.len().checked_add(FOOTER_BYTES).and_then(|len| u64::try_from(len).ok());
357    if image_len != Some(metadata.len()) {
358        return false;
359    }
360    let (before_stamp, after_stamp) =
361        (&payload[..WRITING_PASS_STARTED_AT_OFFSET], &payload[IDENTITY_OFFSET..]);
362    let mut stamp = [0u8; 8];
363    if !reads_as(&mut file, before_stamp)
364        || file.read_exact(&mut stamp).is_err()
365        || !reads_as(&mut file, after_stamp)
366    {
367        return false;
368    }
369    let checksum =
370        ![before_stamp, &stamp[..], after_stamp].into_iter().fold(u32::MAX, crc32c_update);
371    let mut footer = [0u8; FOOTER_BYTES];
372    if file.read_exact(&mut footer).is_err()
373        || footer[..CHECKSUM_BYTES] != checksum.to_le_bytes()
374        || footer[CHECKSUM_BYTES..] != *TRAILER
375        // A file that grew under the comparison is not this image.
376        || !matches!(file.read(&mut [0u8; 1]), Ok(0))
377    {
378        return false;
379    }
380    let pass_started = payload
381        .get(WRITING_PASS_STARTED_AT_OFFSET..IDENTITY_OFFSET)
382        .and_then(|stamp| <[u8; 8]>::try_from(stamp).ok())
383        .map(i64::from_le_bytes)
384        .and_then(|nanos| u64::try_from(nanos).ok())
385        .and_then(|nanos| {
386            std::time::UNIX_EPOCH.checked_add(std::time::Duration::from_nanos(nanos))
387        });
388    if let Some(pass_started) = pass_started {
389        if !matches!(metadata.modified(), Ok(modified) if modified >= pass_started) {
390            // Best effort, as for any kept image: a stale mtime understates freshness.
391            let _ = touch(path, pass_started);
392        }
393    }
394    true
395}
396
397/// Append the checksum of every byte so far, then the trailer.
398fn seal(payload: &mut Vec<u8>) {
399    let checksum = crc32c(payload);
400    payload.extend_from_slice(&checksum.to_le_bytes());
401    payload.extend_from_slice(TRAILER);
402}
403
404/// Capture and persist a coherent shared-index image without holding its lock during
405/// serialization or filesystem I/O.
406pub fn save_handle(index: &IndexHandle, path: &Path) -> Result<()> {
407    let snapshot = index.snapshot()?;
408    save(&snapshot, path)
409}
410
411/// Load a snapshot, or return `None` when there is nothing usable at `path`.
412///
413/// Returns `None` — not an error — for a missing file, a foreign file, a version or
414/// engine-fingerprint mismatch, and a truncated or corrupt file. Every one of those
415/// means the same thing to a caller: there is no warm cache, so scan.
416pub fn load(path: &Path) -> Result<Option<Index>> {
417    load_with_size_limit(path, MAX_SNAPSHOT_BYTES)
418}
419
420/// Load a snapshot under the registry that will classify and analyze its entries.
421///
422/// A snapshot made under different rules is a clean cache miss. The snapshot carries
423/// only their derived identity, so the caller must supply the matching registry rather
424/// than allowing the loader to attach unrelated default rules.
425pub fn load_with_types(
426    path: &Path,
427    types: std::sync::Arc<crate::classify::TypeRegistry>,
428) -> Result<Option<Index>> {
429    load_with_types_and_size_limit(path, MAX_SNAPSHOT_BYTES, types, None).map(|outcome| {
430        match outcome {
431            LoadOutcome::Served { index, .. } => Some(index),
432            LoadOutcome::Refused { .. } | LoadOutcome::Absent => None,
433        }
434    })
435}
436
437/// Load a snapshot that can serve `wanted`, projecting observed control state away when
438/// the entry tier is equal and the request does not observe controls.
439pub fn load_serving(
440    path: &Path,
441    types: std::sync::Arc<crate::classify::TypeRegistry>,
442    wanted: SnapshotIdentity,
443) -> Result<LoadOutcome> {
444    load_with_types_and_size_limit(path, MAX_SNAPSHOT_BYTES, types, Some(wanted))
445}
446
447fn load_with_size_limit(path: &Path, max_snapshot_bytes: u64) -> Result<Option<Index>> {
448    load_with_types_and_size_limit(
449        path,
450        max_snapshot_bytes,
451        crate::classify::TypeRegistry::compiled_shared(),
452        None,
453    )
454    .map(|outcome| match outcome {
455        LoadOutcome::Served { index, .. } => Some(index),
456        LoadOutcome::Refused { .. } | LoadOutcome::Absent => None,
457    })
458}
459
460fn load_with_types_and_size_limit(
461    path: &Path,
462    max_snapshot_bytes: u64,
463    types: std::sync::Arc<crate::classify::TypeRegistry>,
464    wanted: Option<SnapshotIdentity>,
465) -> Result<LoadOutcome> {
466    let mut file = match fs::File::open(path) {
467        Ok(file) => file,
468        Err(e) if e.kind() == std::io::ErrorKind::NotFound => return Ok(LoadOutcome::Absent),
469        Err(e) => return Err(Error::io(path, e)),
470    };
471    let file_len = file.metadata().map_err(|e| Error::io(path, e))?.len();
472    let footer_bytes = CHECKSUM_BYTES
473        .checked_add(TRAILER.len())
474        .and_then(|bytes| u64::try_from(bytes).ok())
475        .ok_or_else(|| Error::Snapshot("snapshot footer size overflow".into()))?;
476    if file_len > max_snapshot_bytes || file_len < footer_bytes {
477        return Ok(LoadOutcome::Absent);
478    }
479
480    let footer_offset = i64::try_from(footer_bytes)
481        .map_err(|_| Error::Snapshot("snapshot footer size overflow".into()))?;
482    file.seek(SeekFrom::End(-footer_offset)).map_err(|e| Error::io(path, e))?;
483    let expected_checksum = match read_footer_checksum(&mut file) {
484        Ok(checksum) => checksum,
485        Err(error) if error.kind() == std::io::ErrorKind::UnexpectedEof => {
486            return Ok(LoadOutcome::Absent);
487        }
488        Err(error) => return Err(Error::io(path, error)),
489    };
490    let mut trailer = [0u8; TRAILER.len()];
491    match file.read_exact(&mut trailer) {
492        Ok(()) if &trailer == TRAILER => {}
493        Ok(()) => return Ok(LoadOutcome::Absent),
494        Err(e) if e.kind() == std::io::ErrorKind::UnexpectedEof => return Ok(LoadOutcome::Absent),
495        Err(e) => return Err(Error::io(path, e)),
496    }
497    let payload_len = file_len
498        .checked_sub(footer_bytes)
499        .ok_or_else(|| Error::Snapshot("snapshot length underflow".into()))?;
500    file.seek(SeekFrom::Start(0)).map_err(|e| Error::io(path, e))?;
501
502    // One pass, not two: the checksum accumulates as the parser consumes, instead of
503    // a separate full read of the image before parsing begins. The verdict still
504    // gates the data — the parsed index is returned only after the digest over the
505    // complete payload matches, so nothing is ever served from bytes that failed
506    // their checksum. What changes is the failure mode for structurally-valid
507    // corruption: the parser may do work before the mismatch is known, and the
508    // result is then discarded. Structural corruption is caught by the parser's own
509    // bounds and consistency checks exactly as before, fail-closed either way.
510    let mut reader = Crc32cReader::new(BufReader::new(file.take(payload_len)));
511    let outcome = parse_stream(&mut reader, payload_len, types, wanted);
512    match outcome {
513        Ok(outcome) => {
514            // A successful parse consumed every payload byte (the trailing-byte check
515            // proves it), so the running digest covers the whole image.
516            if reader.finish() == expected_checksum { Ok(outcome) } else { Ok(LoadOutcome::Absent) }
517        }
518        Err(ParseError::Invalid) => Ok(LoadOutcome::Absent),
519        Err(ParseError::Io(source)) => Err(Error::io(path, source)),
520    }
521}
522
523/// A reader that folds CRC-32C over every byte the caller consumes.
524struct Crc32cReader<R> {
525    inner: R,
526    state: u32,
527}
528
529impl<R: Read> Crc32cReader<R> {
530    fn new(inner: R) -> Self {
531        Self { inner, state: u32::MAX }
532    }
533
534    fn finish(&self) -> u32 {
535        !self.state
536    }
537}
538
539impl<R: Read> Read for Crc32cReader<R> {
540    fn read(&mut self, buf: &mut [u8]) -> std::io::Result<usize> {
541        let read = self.inner.read(buf)?;
542        self.state = crc32c_update(self.state, &buf[..read]);
543        Ok(read)
544    }
545}
546
547fn read_footer_checksum(reader: &mut impl Read) -> std::io::Result<u32> {
548    let mut bytes = [0u8; CHECKSUM_BYTES];
549    reader.read_exact(&mut bytes)?;
550    Ok(u32::from_le_bytes(bytes))
551}
552
553pub(crate) fn crc32c(bytes: &[u8]) -> u32 {
554    !crc32c_update(u32::MAX, bytes)
555}
556
557/// Slicing-by-8: fold eight input bytes per step through eight derived tables
558/// instead of one byte through one. Table 0 is the classic byte table, so the
559/// remainder loop and the 8-byte path share one source of truth; the digest is
560/// bit-identical to the byte-at-a-time form, which a test asserts over uneven
561/// lengths alongside the standard check value.
562fn crc32c_update(mut state: u32, bytes: &[u8]) -> u32 {
563    let mut chunks = bytes.chunks_exact(8);
564    for chunk in &mut chunks {
565        let low = (state ^ u32::from_le_bytes(chunk[..4].try_into().expect("chunk holds 8 bytes")))
566            .to_le_bytes();
567        let high =
568            u32::from_le_bytes(chunk[4..].try_into().expect("chunk holds 8 bytes")).to_le_bytes();
569        state = CRC32C_TABLES[7][usize::from(low[0])]
570            ^ CRC32C_TABLES[6][usize::from(low[1])]
571            ^ CRC32C_TABLES[5][usize::from(low[2])]
572            ^ CRC32C_TABLES[4][usize::from(low[3])]
573            ^ CRC32C_TABLES[3][usize::from(high[0])]
574            ^ CRC32C_TABLES[2][usize::from(high[1])]
575            ^ CRC32C_TABLES[1][usize::from(high[2])]
576            ^ CRC32C_TABLES[0][usize::from(high[3])];
577    }
578    for byte in chunks.remainder() {
579        let index = usize::from(state.to_le_bytes()[0] ^ *byte);
580        state = CRC32C_TABLES[0][index] ^ (state >> 8);
581    }
582    state
583}
584
585const fn make_crc32c_tables() -> [[u32; 256]; 8] {
586    let mut tables = [[0u32; 256]; 8];
587    let mut index = 0usize;
588    let mut value = 0u32;
589    while index < 256 {
590        let mut crc = value;
591        let mut bit = 0;
592        while bit < u8::BITS {
593            crc = (crc >> 1) ^ (CRC32C_POLYNOMIAL & 0u32.wrapping_sub(crc & 1));
594            bit += 1;
595        }
596        tables[0][index] = crc;
597        index += 1;
598        value += 1;
599    }
600    // tables[k] advances a byte's contribution k further positions: one more
601    // byte-table step applied to the previous table's value.
602    let mut table = 1usize;
603    while table < tables.len() {
604        let mut index = 0usize;
605        while index < 256 {
606            let previous = tables[table - 1][index];
607            tables[table][index] = tables[0][(previous & 0xFF) as usize] ^ (previous >> 8);
608            index += 1;
609        }
610        table += 1;
611    }
612    tables
613}
614
615/// Read only a snapshot's header, without materializing its index.
616///
617/// Returns `None` for anything this build cannot serve — absent, truncated, foreign,
618/// a different format version, or a mismatched engine fingerprint. Corrupt equals
619/// absent here exactly as it does on the load path: a caller asking what is in the cache
620/// must not be stopped by one unreadable file.
621pub fn read_header(path: &Path) -> Result<Option<crate::cache::SnapshotInfo>> {
622    Ok(match identify(path)? {
623        Some(Identity::Current(info)) => Some(info),
624        Some(Identity::Stale(_) | Identity::Foreign) | None => None,
625    })
626}
627
628/// What a file's leading bytes and trailer say it is.
629#[derive(Debug)]
630pub(crate) enum Identity {
631    /// A snapshot this build reads.
632    Current(crate::cache::SnapshotInfo),
633    /// It begins with the snapshot magic, so fdu wrote it, but this build cannot serve it.
634    Stale(crate::cache::StaleReason),
635    /// It does not begin with the snapshot magic.
636    Foreign,
637}
638
639/// Identify the file at `path` by its contents, or return `None` when nothing is there.
640///
641/// The magic alone decides whether a file is fdu's; the rest of the prologue decides
642/// whether this build can serve it. Every format fdu has written puts the version and the
643/// engine fingerprint at the same offsets after the magic, so a snapshot from another
644/// release is recognised as stale rather than mistaken for a foreign file. That is what
645/// lets the cache lifecycle reclaim the snapshots an upgrade leaves behind.
646///
647/// Opening follows a symbolic link at `path`, so a caller that must not follow one checks
648/// the path's own metadata first.
649pub(crate) fn identify(path: &Path) -> Result<Option<Identity>> {
650    let file = match fs::File::open(path) {
651        Ok(file) => file,
652        Err(error) if error.kind() == std::io::ErrorKind::NotFound => return Ok(None),
653        Err(error) => return Err(Error::io(path, error)),
654    };
655    let trailer_intact = has_intact_trailer(&file).map_err(|error| Error::io(path, error))?;
656    // The trailer check left the cursor at the end; the header lives at the start.
657    let mut file = file;
658    file.seek(SeekFrom::Start(0)).map_err(|error| Error::io(path, error))?;
659    identify_prologue(&mut BufReader::new(file), trailer_intact)
660        .map(Some)
661        .map_err(|error| Error::io(path, error))
662}
663
664/// Classify a file from its prologue. Only an I/O failure is an error; every malformed
665/// byte is an answer.
666fn identify_prologue(reader: &mut impl Read, trailer_intact: bool) -> io::Result<Identity> {
667    use crate::cache::StaleReason;
668
669    match read_array::<_, 8>(reader) {
670        Ok(magic) if magic == *MAGIC => {}
671        Ok(_) | Err(ParseError::Invalid) => return Ok(Identity::Foreign),
672        Err(ParseError::Io(error)) => return Err(error),
673    }
674    let Some(version) = invalid_as_none(read_u32(reader))? else {
675        return Ok(Identity::Stale(StaleReason::Unreadable));
676    };
677    match version.cmp(&FORMAT_VERSION) {
678        std::cmp::Ordering::Less => {
679            return Ok(Identity::Stale(StaleReason::OlderFormat { version }));
680        }
681        std::cmp::Ordering::Greater => {
682            return Ok(Identity::Stale(StaleReason::NewerFormat { version }));
683        }
684        std::cmp::Ordering::Equal => {}
685    }
686    let engine = match invalid_as_none(read_u64(reader))? {
687        Some(fingerprint) if fingerprint == engine_fingerprint() => fingerprint,
688        Some(_) => return Ok(Identity::Stale(StaleReason::OtherEngine)),
689        None => return Ok(Identity::Stale(StaleReason::Unreadable)),
690    };
691    if !trailer_intact {
692        // Truncation removes the tail and leaves the prologue readable, so a header-only
693        // check would call a half-written file current.
694        return Ok(Identity::Stale(StaleReason::Unreadable));
695    }
696    Ok(match invalid_as_none(parse_header_fields(reader, engine))? {
697        Some(header) => Identity::Current(crate::cache::SnapshotInfo {
698            root: header.root,
699            identity: header.identity,
700            entries: header.entries,
701        }),
702        None => Identity::Stale(StaleReason::Unreadable),
703    })
704}
705
706/// Separate a malformed value, which is an answer, from an I/O failure, which is not.
707fn invalid_as_none<T>(result: ParseResult<T>) -> io::Result<Option<T>> {
708    match result {
709        Ok(value) => Ok(Some(value)),
710        Err(ParseError::Invalid) => Ok(None),
711        Err(ParseError::Io(error)) => Err(error),
712    }
713}
714
715/// Whether a file ends with the snapshot trailer.
716///
717/// Cheap enough for a status listing — one seek and eight bytes — and it is exactly what
718/// a truncated write destroys.
719fn has_intact_trailer(file: &fs::File) -> io::Result<bool> {
720    let footer_bytes = CHECKSUM_BYTES + TRAILER.len();
721    let file_len = file.metadata()?.len();
722    if file_len < u64::try_from(footer_bytes).unwrap_or(u64::MAX) {
723        return Ok(false);
724    }
725
726    let mut handle = file;
727    handle.seek(SeekFrom::End(-(i64::try_from(TRAILER.len()).unwrap_or(0))))?;
728    let mut trailer = [0u8; TRAILER.len()];
729    match handle.read_exact(&mut trailer) {
730        Ok(()) => Ok(&trailer == TRAILER),
731        Err(error) if error.kind() == io::ErrorKind::UnexpectedEof => Ok(false),
732        Err(error) => Err(error),
733    }
734}
735
736/// The header fields after a snapshot's prologue.
737struct Header {
738    /// The start of the pass that last wrote the image.
739    ///
740    /// A later pass that verifies the same facts keeps the image and this stamp rather
741    /// than rewriting for the stamp alone, so the value is a lower bound on when the facts
742    /// were last verified: it can predate many verifying passes and never postdates the one
743    /// that wrote them. Until P1.4.4 stamps completed passes, reconciliation does not
744    /// advance it either.
745    writing_pass_started_at_ns: i64,
746    /// The identity of every tier the snapshot holds.
747    identity: SnapshotIdentity,
748    /// The absolute root the snapshot describes.
749    root: PathBuf,
750    /// How many entry records follow.
751    entries: u64,
752}
753
754/// Parse the header fields after the magic, format version, and `engine` fingerprint.
755fn parse_header_fields(reader: &mut impl Read, engine: u64) -> ParseResult<Header> {
756    if read_u8(reader)? != path_encoding() {
757        return Err(ParseError::Invalid);
758    }
759    let writing_pass_started_at_ns = read_i64(reader)?;
760    let identity =
761        SnapshotIdentity::decode(engine, &read_array::<_, SNAPSHOT_IDENTITY_BYTES>(reader)?)
762            .ok_or(ParseError::Invalid)?;
763    let root = PathBuf::from(read_os_string(reader)?);
764    let entries = read_u64(reader)?;
765    if entries == 0 || entries > MAX_SNAPSHOT_ENTRIES {
766        return Err(ParseError::Invalid);
767    }
768    Ok(Header { writing_pass_started_at_ns, identity, root, entries })
769}
770
771/// Parse a bounded payload. Records are applied one at a time so bootstrap paths are not
772/// retained in a second full-tree allocation.
773fn parse_stream(
774    reader: &mut impl Read,
775    payload_len: u64,
776    types: std::sync::Arc<crate::classify::TypeRegistry>,
777    wanted: Option<SnapshotIdentity>,
778) -> ParseResult<LoadOutcome> {
779    if read_array::<_, 8>(reader)? != *MAGIC {
780        return Err(ParseError::Invalid);
781    }
782    let engine = engine_fingerprint();
783    if read_u32(reader)? != FORMAT_VERSION || read_u64(reader)? != engine {
784        return Err(ParseError::Invalid);
785    }
786    let Header { writing_pass_started_at_ns, identity, root: root_path, entries: count } =
787        parse_header_fields(reader, engine)?;
788    let serves = wanted.map_or(Serves::Exact, |wanted| serves_snapshot(identity, wanted));
789    let mut scope = match (serves, wanted) {
790        (Serves::ProjectControlsOff, Some(wanted)) => wanted.scan_scope(),
791        _ => identity.scan_scope(),
792    };
793    if scope.type_rules_fingerprint != types.fingerprint() {
794        if serves != Serves::Refuse {
795            return Err(ParseError::Invalid);
796        }
797        // A foreign registry cannot be attached to a served index. For a refusal,
798        // reconstruct only to validate every record and the checksum, then discard it.
799        // The stored identity below remains unchanged for the caller's diagnostic.
800        //
801        // That costs a full index build for a diagnostic: every record is decoded and
802        // inserted so the checksum can be verified over the bytes it covers, and the
803        // result is dropped. It is paid only under a type-rules mismatch, which a
804        // cache-only request reports and every other policy answers by scanning, so it
805        // buys a truthful refusal (valid but foreign, rather than absent) at the price of
806        // one load on a path that has no answer anyway.
807        scope.type_rules_fingerprint = types.fingerprint();
808    }
809    let minimum_body = count
810        .checked_mul(u64::try_from(MIN_RECORD_BYTES).map_err(|_| ParseError::Invalid)?)
811        .ok_or(ParseError::Invalid)?;
812    if minimum_body > payload_len {
813        return Err(ParseError::Invalid);
814    }
815
816    let mut index = Index::new_with_scope_and_types(&root_path, scope, types);
817    // The facts below were verified by the pass that wrote them, not by this load.
818    index.set_writing_pass_started_at_ns(writing_pass_started_at_ns);
819    // Everything this loader inserts describes the tree as the snapshot found it, not
820    // as this process has seen it. Stamping the entries `Cached` is what lets a
821    // consumer paint them immediately and label them honestly; without it a loaded
822    // index claims to be fresh when nothing has been checked since the file was read.
823    index.set_applying_source(Source::Cached, writing_pass_started_at_ns);
824    // The record count is validated against the bytes actually present above, so it is
825    // safe to size from: reserving here removes the geometric regrowth of a 450k-element
826    // vector without letting a corrupt count drive the allocation.
827    let mut ids: Vec<EntryId> = Vec::with_capacity(usize::try_from(count).unwrap_or(0));
828    for slot in 0..count {
829        let parent_slot = read_u32(reader)?;
830        let kind = EntryKind::from_u8(read_u8(reader)?).ok_or(ParseError::Invalid)?;
831        let name = read_os_string(reader)?;
832        let attrs = Attrs {
833            size: read_u64(reader)?,
834            allocated: read_u64(reader)?,
835            mtime_ns: read_i64(reader)?,
836            ctime_ns: read_i64(reader)?,
837            inode: read_u64(reader)?,
838            dev: read_u64(reader)?,
839        };
840
841        if parent_slot == NO_PARENT {
842            if slot != 0 || kind != EntryKind::Dir || !name.is_empty() {
843                return Err(ParseError::Invalid);
844            }
845            index
846                .apply_baseline(&Observation::new(vec![Op::Upsert {
847                    path: PathBuf::new(),
848                    kind,
849                    attrs,
850                }]))
851                .map_err(|_| ParseError::Invalid)?;
852            ids.push(EntryId::ROOT);
853            continue;
854        }
855
856        let parent = *ids
857            .get(usize::try_from(parent_slot).map_err(|_| ParseError::Invalid)?)
858            .ok_or(ParseError::Invalid)?;
859        if !is_snapshot_name(&name) {
860            return Err(ParseError::Invalid);
861        }
862        // The parent's id is already in hand, so the record is inserted straight beneath
863        // it. Resolving a path to rediscover that parent, and then searching the parent's
864        // children to rediscover the id just created, were both work the format had
865        // already answered. A snapshot naming the same path twice, or parenting an entry
866        // to a non-directory, is corrupt, and `insert_loaded_child` fails closed on both.
867        let id = index.insert_loaded_child(parent, name, kind, attrs).ok_or(ParseError::Invalid)?;
868        ids.push(id);
869    }
870
871    let controls = read_controls(reader, identity.controls)?;
872    if serves == Serves::Exact {
873        index.install_controls(controls).map_err(|_| ParseError::Invalid)?;
874    }
875
876    let mut extra = [0u8; 1];
877    if reader.read(&mut extra).map_err(ParseError::Io)? != 0 {
878        return Err(ParseError::Invalid);
879    }
880    // Once, rather than once per record: the loaded tree is the process baseline, and
881    // nothing it inserted was journalled.
882    index.establish_baseline();
883    // Anything applied after the load is this process checking what the snapshot
884    // claimed, which is a revalidation rather than a first sighting.
885    index.set_applying_source(Source::Revalidated, 0);
886    // The image on disk is this index, so nothing is owed until a pass mutates it.
887    index.set_persistence_owed(false);
888    Ok(if serves == Serves::Refuse {
889        LoadOutcome::Refused { identity, root: root_path }
890    } else {
891        LoadOutcome::Served { index, stored: identity }
892    })
893}
894
895#[derive(Debug)]
896enum ParseError {
897    Invalid,
898    Io(std::io::Error),
899}
900
901type ParseResult<T> = std::result::Result<T, ParseError>;
902
903fn read_array<R: Read, const N: usize>(reader: &mut R) -> ParseResult<[u8; N]> {
904    let mut bytes = [0u8; N];
905    match reader.read_exact(&mut bytes) {
906        Ok(()) => Ok(bytes),
907        Err(error) if error.kind() == std::io::ErrorKind::UnexpectedEof => Err(ParseError::Invalid),
908        Err(error) => Err(ParseError::Io(error)),
909    }
910}
911
912fn read_u8(reader: &mut impl Read) -> ParseResult<u8> {
913    Ok(read_array::<_, 1>(reader)?[0])
914}
915
916fn read_u32(reader: &mut impl Read) -> ParseResult<u32> {
917    Ok(u32::from_le_bytes(read_array(reader)?))
918}
919
920fn read_u64(reader: &mut impl Read) -> ParseResult<u64> {
921    Ok(u64::from_le_bytes(read_array(reader)?))
922}
923
924fn read_i64(reader: &mut impl Read) -> ParseResult<i64> {
925    Ok(i64::from_le_bytes(read_array(reader)?))
926}
927
928fn read_bytes(reader: &mut impl Read) -> ParseResult<Vec<u8>> {
929    let len = read_u32(reader)?;
930    if len > MAX_PATH_BYTES {
931        return Err(ParseError::Invalid);
932    }
933    let mut bytes = vec![0u8; usize::try_from(len).map_err(|_| ParseError::Invalid)?];
934    match reader.read_exact(&mut bytes) {
935        Ok(()) => Ok(bytes),
936        Err(error) if error.kind() == std::io::ErrorKind::UnexpectedEof => Err(ParseError::Invalid),
937        Err(error) => Err(ParseError::Io(error)),
938    }
939}
940
941/// Read the control section, retained sources then refusals, for the control tier the
942/// header records.
943///
944/// Every source is admitted again under the header's limits. The charge does not depend on
945/// admission order, so a table a scan retained always fits; one that does not, a repeated
946/// path, a refusal naming a retained or repeated path, a refusal by a limit the header
947/// records as unbounded, or any source or refusal under a tier that observed nothing, was
948/// not written by [`save`].
949fn read_controls(
950    reader: &mut impl Read,
951    tier: ControlTierIdentity,
952) -> ParseResult<crate::control::ControlTable> {
953    let limits = match tier {
954        ControlTierIdentity::Observed { limits } => limits,
955        // Limits decide nothing when nothing was observed, and the section must be empty.
956        // So a loaded blind table holds the default limits where the scanning index kept
957        // the request's; no reader consults a blind table's limits, and a cache status or
958        // projection that reports them must not show that as a difference.
959        ControlTierIdentity::NotObserved => crate::control::ControlLimits::default(),
960    };
961    let control_count = read_u32(reader)?;
962    if control_count != 0 && !tier.is_observed() {
963        return Err(ParseError::Invalid);
964    }
965    if usize::try_from(control_count)
966        .map_err(|_| ParseError::Invalid)?
967        .saturating_mul(crate::control::CONTROL_SOURCE_OVERHEAD)
968        > SNAPSHOT_CONTROL_TABLE_CEILING
969    {
970        return Err(ParseError::Invalid);
971    }
972    let mut sources = Vec::new();
973    let mut source_bytes = 0_usize;
974    for _ in 0..control_count {
975        let path = PathBuf::from(read_os_string(reader)?);
976        let source = read_control_bytes(reader)?;
977        source_bytes = source_bytes.saturating_add(source.len());
978        if source_bytes > SNAPSHOT_CONTROL_TABLE_CEILING {
979            return Err(ParseError::Invalid);
980        }
981        sources.push((path, source));
982    }
983    let mut controls = crate::control::ControlTable::with_limits(limits);
984    for (path, source) in sources {
985        let admission = controls.upsert(&path, source).map_err(|_| ParseError::Invalid)?;
986        if admission != (crate::control::ControlAdmission::Retained { changed: true }) {
987            return Err(ParseError::Invalid);
988        }
989    }
990    if controls.retained_cost() > SNAPSHOT_CONTROL_TABLE_CEILING {
991        return Err(ParseError::Invalid);
992    }
993    let refused = read_u32(reader)?;
994    if refused != 0 && !tier.is_observed() {
995        return Err(ParseError::Invalid);
996    }
997    for _ in 0..refused {
998        let path = PathBuf::from(read_os_string(reader)?);
999        let reason = match read_u8(reader)? {
1000            REFUSED_FOR_BUDGET => crate::control::ControlRefusalReason::Budget,
1001            REFUSED_FOR_LINE_LIMIT => crate::control::ControlRefusalReason::LineLimit,
1002            _ => return Err(ParseError::Invalid),
1003        };
1004        // An unbounded limit refuses nothing, so no save records a refusal by one.
1005        if !crate::control::is_control_file(&path)
1006            || controls.contains(&path)
1007            || limits.limit_for(reason).is_none()
1008        {
1009            return Err(ParseError::Invalid);
1010        }
1011        controls.record_refusal(&path, reason).map_err(|_| ParseError::Invalid)?;
1012    }
1013    Ok(controls)
1014}
1015
1016fn read_control_bytes(reader: &mut impl Read) -> ParseResult<Vec<u8>> {
1017    let len = read_u32(reader)?;
1018    if usize::try_from(len).map_err(|_| ParseError::Invalid)? > SNAPSHOT_CONTROL_TABLE_CEILING {
1019        return Err(ParseError::Invalid);
1020    }
1021    let mut bytes = vec![0u8; usize::try_from(len).map_err(|_| ParseError::Invalid)?];
1022    match reader.read_exact(&mut bytes) {
1023        Ok(()) => Ok(bytes),
1024        Err(error) if error.kind() == std::io::ErrorKind::UnexpectedEof => Err(ParseError::Invalid),
1025        Err(error) => Err(ParseError::Io(error)),
1026    }
1027}
1028
1029#[cfg(unix)]
1030fn read_os_string(reader: &mut impl Read) -> ParseResult<OsString> {
1031    Ok(os_string_from_bytes(&read_bytes(reader)?))
1032}
1033
1034#[cfg(not(unix))]
1035fn read_os_string(reader: &mut impl Read) -> ParseResult<OsString> {
1036    os_string_from_bytes(&read_bytes(reader)?).ok_or(ParseError::Invalid)
1037}
1038
1039fn put_bytes(buf: &mut Vec<u8>, bytes: &[u8]) -> Result<()> {
1040    let len = u32::try_from(bytes.len())
1041        .map_err(|_| Error::Snapshot("string too long for snapshot".into()))?;
1042    if len > MAX_PATH_BYTES {
1043        return Err(Error::Snapshot("path exceeds snapshot limit".into()));
1044    }
1045    buf.extend_from_slice(&len.to_le_bytes());
1046    buf.extend_from_slice(bytes);
1047    Ok(())
1048}
1049
1050/// Write the control section [`read_controls`] reads: sources, then refusals. The limits
1051/// they were admitted under are the header's control tier identity.
1052fn put_controls(buf: &mut Vec<u8>, controls: &crate::control::ControlTable) -> Result<()> {
1053    if controls.retained_cost() > SNAPSHOT_CONTROL_TABLE_CEILING
1054        || controls.source_bytes() > SNAPSHOT_CONTROL_TABLE_CEILING
1055    {
1056        return Err(Error::Snapshot(format!(
1057            "the control table retains {} bytes of charge, above the {} bytes a snapshot can \
1058             carry; set a control budget below it, or open without a cache",
1059            controls.retained_cost(),
1060            SNAPSHOT_CONTROL_TABLE_CEILING
1061        )));
1062    }
1063    let sources: Vec<_> = controls.sources().collect();
1064    let control_count = u32::try_from(sources.len())
1065        .map_err(|_| Error::Snapshot("control table exceeds u32 capacity".into()))?;
1066    buf.extend_from_slice(&control_count.to_le_bytes());
1067    for (path, source) in sources {
1068        put_os_str(buf, path.as_os_str())?;
1069        let len = u32::try_from(source.len())
1070            .map_err(|_| Error::Snapshot("control source exceeds u32 capacity".into()))?;
1071        buf.extend_from_slice(&len.to_le_bytes());
1072        buf.extend_from_slice(source);
1073    }
1074    let refused = u32::try_from(controls.refused_len())
1075        .map_err(|_| Error::Snapshot("refused controls exceed u32 capacity".into()))?;
1076    buf.extend_from_slice(&refused.to_le_bytes());
1077    for refusal in controls.refusals() {
1078        put_os_str(buf, refusal.path.as_os_str())?;
1079        buf.push(match refusal.reason {
1080            crate::control::ControlRefusalReason::Budget => REFUSED_FOR_BUDGET,
1081            crate::control::ControlRefusalReason::LineLimit => REFUSED_FOR_LINE_LIMIT,
1082        });
1083    }
1084    Ok(())
1085}
1086
1087/// A snapshot entry name is one filesystem component, spelled exactly as stored.
1088///
1089/// `Path::components` treats `a` and `a/` as the same `Normal("a")`. The child map
1090/// distinguishes the raw names, but a `PathBuf`-keyed restore map merges them and
1091/// shrinks the cache-only completeness denominator. The raw name must equal that
1092/// single normal component so a checksummed alias cannot load.
1093fn is_snapshot_name(name: &OsStr) -> bool {
1094    let mut components = Path::new(name).components();
1095    let Some(Component::Normal(component)) = components.next() else { return false };
1096    components.next().is_none() && component == name
1097}
1098
1099#[cfg(unix)]
1100pub(crate) fn path_encoding() -> u8 {
1101    PATH_ENCODING_UNIX_BYTES
1102}
1103
1104#[cfg(windows)]
1105pub(crate) fn path_encoding() -> u8 {
1106    PATH_ENCODING_WINDOWS_WIDE
1107}
1108
1109#[cfg(not(any(unix, windows)))]
1110pub(crate) fn path_encoding() -> u8 {
1111    PATH_ENCODING_UTF8
1112}
1113
1114#[cfg(unix)]
1115pub(crate) fn put_os_str(buf: &mut Vec<u8>, value: &OsStr) -> Result<()> {
1116    use std::os::unix::ffi::OsStrExt;
1117    put_bytes(buf, value.as_bytes())
1118}
1119
1120#[cfg(windows)]
1121pub(crate) fn put_os_str(buf: &mut Vec<u8>, value: &OsStr) -> Result<()> {
1122    use std::os::windows::ffi::OsStrExt;
1123    let mut bytes = Vec::new();
1124    for unit in value.encode_wide() {
1125        bytes.extend_from_slice(&unit.to_le_bytes());
1126    }
1127    put_bytes(buf, &bytes)
1128}
1129
1130#[cfg(not(any(unix, windows)))]
1131pub(crate) fn put_os_str(buf: &mut Vec<u8>, value: &OsStr) -> Result<()> {
1132    let text = value
1133        .to_str()
1134        .ok_or_else(|| Error::Snapshot("path is not valid UTF-8 on this platform".into()))?;
1135    put_bytes(buf, text.as_bytes())
1136}
1137
1138#[cfg(unix)]
1139fn os_string_from_bytes(bytes: &[u8]) -> OsString {
1140    use std::os::unix::ffi::OsStringExt;
1141    OsString::from_vec(bytes.to_vec())
1142}
1143
1144#[cfg(windows)]
1145fn os_string_from_bytes(bytes: &[u8]) -> Option<OsString> {
1146    use std::os::windows::ffi::OsStringExt;
1147    // Stable since 1.87, against an MSRV of 1.85 — see the note in content_cache.rs.
1148    if bytes.len() % std::mem::size_of::<u16>() != 0 {
1149        return None;
1150    }
1151    let units: Vec<u16> = bytes
1152        .chunks_exact(std::mem::size_of::<u16>())
1153        .map(|chunk| u16::from_le_bytes([chunk[0], chunk[1]]))
1154        .collect();
1155    Some(OsString::from_wide(&units))
1156}
1157
1158#[cfg(not(any(unix, windows)))]
1159fn os_string_from_bytes(bytes: &[u8]) -> Option<OsString> {
1160    Some(OsString::from(String::from_utf8(bytes.to_vec()).ok()?))
1161}
1162
1163/// Write to a sibling temporary file, then rename over the target.
1164///
1165/// Rename is atomic within a filesystem, so a reader either sees the whole old snapshot
1166/// or the whole new one. The temporary must be a sibling for that to hold — a rename
1167/// across filesystems is a copy, and copies are not atomic.
1168pub(crate) fn write_atomically(path: &Path, bytes: &[u8]) -> Result<()> {
1169    // Serialization is deterministic, so unchanged input encodes to the bytes already on
1170    // disk, and replacing them is a full write, an `F_FULLFSYNC`, and a rename that leave
1171    // the file exactly as it was. Comparing against the page-cached file costs a few
1172    // milliseconds and cannot go stale, because the bytes are the same bytes. The metadata
1173    // snapshot, whose images also differ in their pass-start stamp, compares in `publish`
1174    // instead, which records what the skip is worth (exp-066, exp-067).
1175    if keep_identical(path, bytes) {
1176        return Ok(());
1177    }
1178    replace_atomically(path, bytes)
1179}
1180
1181/// [`write_atomically`] without first comparing against the file at `path`, for a caller
1182/// that has already compared.
1183fn replace_atomically(path: &Path, bytes: &[u8]) -> Result<()> {
1184    let parent = parent_dir(path);
1185    fs::create_dir_all(parent).map_err(|e| Error::io(parent, e))?;
1186    let (tmp, mut file) = create_temp_file(path, parent)?;
1187    let write_then_sync = file.write_all(bytes).and_then(|()| file.sync_all());
1188    if let Err(e) = write_then_sync {
1189        let _ = fs::remove_file(&tmp);
1190        return Err(Error::io(&tmp, e));
1191    }
1192    drop(file);
1193
1194    if let Err(e) = fs::rename(&tmp, path) {
1195        let _ = fs::remove_file(&tmp);
1196        return Err(Error::io(path, e));
1197    }
1198    reap_stale_temporaries(parent, path, STALE_TEMP_AGE);
1199    Ok(())
1200}
1201
1202/// Leave `path` in place when it already holds exactly `bytes`, moving only its mtime to
1203/// now.
1204///
1205/// The bytes were just produced again, so now is when they were last known to be current,
1206/// and moving the mtime says so without writing the payload. Best effort: a stale mtime
1207/// understates that, which fails safe.
1208fn keep_identical(path: &Path, bytes: &[u8]) -> bool {
1209    if !same_bytes_on_disk(path, bytes) {
1210        return false;
1211    }
1212    let _ = touch(path, std::time::SystemTime::now());
1213    true
1214}
1215
1216/// Whether `path` already holds exactly `bytes`.
1217///
1218/// Streamed in 1 MiB pieces rather than read whole, so deciding not to write a 14 MB
1219/// snapshot does not cost a second 14 MB allocation beside the one being compared. Any
1220/// error -- missing file, short read, a file that grew under the comparison -- reads as
1221/// "different", and the caller writes; the comparison can only ever save work.
1222fn same_bytes_on_disk(path: &Path, bytes: &[u8]) -> bool {
1223    let Ok(mut file) = fs::File::open(path) else { return false };
1224    let Ok(metadata) = file.metadata() else { return false };
1225    if metadata.len() != u64::try_from(bytes.len()).unwrap_or(u64::MAX) {
1226        return false;
1227    }
1228    // A file that holds every byte and then more is not the same file.
1229    reads_as(&mut file, bytes) && matches!(file.read(&mut [0u8; 1]), Ok(0))
1230}
1231
1232/// Whether the next bytes `file` yields are exactly `bytes`.
1233///
1234/// Streamed in 1 MiB pieces, and stopping at the first piece that differs; any read error
1235/// or early end reads as "different".
1236fn reads_as(file: &mut fs::File, bytes: &[u8]) -> bool {
1237    let mut buffer = vec![0u8; bytes.len().min(1 << 20)];
1238    let mut offset = 0usize;
1239    while offset < bytes.len() {
1240        let want = buffer.len().min(bytes.len() - offset);
1241        let read = match file.read(&mut buffer[..want]) {
1242            Ok(0) | Err(_) => return false,
1243            Ok(read) => read,
1244        };
1245        let end = offset + read;
1246        if buffer[..read] != bytes[offset..end] {
1247            return false;
1248        }
1249        offset = end;
1250    }
1251    true
1252}
1253
1254/// Move `path`'s modification time to `when` without touching its contents.
1255fn touch(path: &Path, when: std::time::SystemTime) -> io::Result<()> {
1256    let file = OpenOptions::new().write(true).open(path)?;
1257    file.set_modified(when)
1258}
1259
1260/// Remove long-abandoned temporaries beside `path`.
1261///
1262/// Best effort and deliberately silent: this is housekeeping, and a snapshot that was
1263/// written correctly must not fail because a directory could not be tidied. Anything
1264/// that cannot be read or removed is left for the next writer.
1265fn reap_stale_temporaries(parent: &Path, path: &Path, older_than: std::time::Duration) {
1266    let Some(prefix) = temp_prefix(path) else { return };
1267    let Ok(entries) = fs::read_dir(parent) else { return };
1268    let now = std::time::SystemTime::now();
1269    for entry in entries.flatten() {
1270        let name = entry.file_name();
1271        if !name.as_encoded_bytes().starts_with(prefix.as_encoded_bytes()) {
1272            continue;
1273        }
1274        let stale = entry
1275            .metadata()
1276            .and_then(|meta| meta.modified())
1277            .ok()
1278            .and_then(|modified| now.duration_since(modified).ok())
1279            .is_some_and(|age| age >= older_than);
1280        if stale {
1281            let _ = fs::remove_file(entry.path());
1282        }
1283    }
1284}
1285
1286/// The shared leading portion of every temporary written for `path`.
1287///
1288/// Matching on this rather than on a full parse keeps the reaper from depending on the
1289/// pid, entropy, and sequence encoding: a temporary written by an older build with a
1290/// different suffix layout is still recognised and still collected.
1291fn temp_prefix(path: &Path) -> Option<OsString> {
1292    let mut prefix = OsString::from(".");
1293    prefix.push(path.file_name()?);
1294    prefix.push(".tmp.");
1295    Some(prefix)
1296}
1297
1298/// The directory a target lives in, as a path that can actually be opened.
1299///
1300/// A bare relative target such as `snap.fdu` has an *empty* parent rather than none, and
1301/// the empty path is not the current directory: `create_dir_all` and `rename` tolerate
1302/// it, but `read_dir` does not — which silently disabled the reaper for exactly those
1303/// targets while the write itself kept working.
1304fn parent_dir(path: &Path) -> &Path {
1305    match path.parent() {
1306        Some(parent) if !parent.as_os_str().is_empty() => parent,
1307        _ => Path::new("."),
1308    }
1309}
1310
1311/// The temporary name this process uses for `path` at `sequence`.
1312///
1313/// Split out so a test can plant a collision at a name the writer will actually try.
1314/// Guessing the shape does not work — the entropy is per process — and getting that
1315/// wrong is how the first version of the stale-temporary test came to exercise nothing.
1316fn temp_name(path: &Path, sequence: u64) -> OsString {
1317    let mut name = OsString::from(".");
1318    name.push(path.file_name().unwrap_or_else(|| OsStr::new("snapshot")));
1319    name.push(format!(".tmp.{}.{:016x}.{}", std::process::id(), *TEMP_FILE_ENTROPY, sequence));
1320    name
1321}
1322
1323fn create_temp_file(path: &Path, parent: &Path) -> Result<(PathBuf, fs::File)> {
1324    for _ in 0..MAX_TEMP_CREATE_ATTEMPTS {
1325        let sequence = NEXT_TEMP_FILE.fetch_add(1, Ordering::Relaxed);
1326        let tmp = parent.join(temp_name(path, sequence));
1327        let mut options = OpenOptions::new();
1328        options.write(true).create_new(true);
1329        #[cfg(unix)]
1330        {
1331            use std::os::unix::fs::OpenOptionsExt;
1332            options.mode(0o600);
1333        }
1334        match options.open(&tmp) {
1335            Ok(file) => return Ok((tmp, file)),
1336            Err(error) if error.kind() == std::io::ErrorKind::AlreadyExists => {}
1337            Err(error) => return Err(Error::io(&tmp, error)),
1338        }
1339    }
1340    Err(Error::io(
1341        parent,
1342        std::io::Error::new(
1343            std::io::ErrorKind::AlreadyExists,
1344            "could not reserve a unique snapshot temporary file",
1345        ),
1346    ))
1347}
1348
1349#[cfg(test)]
1350mod tests {
1351    use super::*;
1352    use crate::engine_contract::Observation;
1353    use crate::index::ExtTally;
1354
1355    fn attrs(size: u64, mtime_ns: i64) -> Attrs {
1356        Attrs {
1357            size,
1358            allocated: size.div_ceil(512) * 512,
1359            mtime_ns,
1360            ctime_ns: mtime_ns,
1361            inode: size.wrapping_mul(7).wrapping_add(1),
1362            dev: 3,
1363        }
1364    }
1365
1366    fn sample_index() -> Index {
1367        let mut index = Index::new("/some/root");
1368        index.apply_ok(&Observation::new(vec![
1369            Op::Upsert { path: PathBuf::from("src"), kind: EntryKind::Dir, attrs: attrs(0, 1) },
1370            Op::Upsert {
1371                path: PathBuf::from("src/main.rs"),
1372                kind: EntryKind::File,
1373                attrs: attrs(100, 10),
1374            },
1375            Op::Upsert {
1376                path: PathBuf::from("src/deep"),
1377                kind: EntryKind::Dir,
1378                attrs: attrs(0, 2),
1379            },
1380            Op::Upsert {
1381                path: PathBuf::from("src/deep/nested.rs"),
1382                kind: EntryKind::File,
1383                attrs: attrs(50, 20),
1384            },
1385            Op::Upsert {
1386                path: PathBuf::from("notes.md"),
1387                kind: EntryKind::File,
1388                attrs: attrs(7, 30),
1389            },
1390        ]));
1391        index
1392    }
1393
1394    fn entry_count_offset(bytes: &[u8]) -> usize {
1395        let root_len_at = ROOT_OFFSET;
1396        let root_len = u32::from_le_bytes(
1397            bytes[root_len_at..root_len_at + 4]
1398                .try_into()
1399                .expect("saved snapshot has a root length"),
1400        );
1401        root_len_at + 4 + usize::try_from(root_len).expect("root length fits usize")
1402    }
1403
1404    fn rewrite_checksum(bytes: &mut [u8]) {
1405        let payload_len = bytes.len() - CHECKSUM_BYTES - TRAILER.len();
1406        let checksum = crc32c(&bytes[..payload_len]);
1407        bytes[payload_len..payload_len + CHECKSUM_BYTES].copy_from_slice(&checksum.to_le_bytes());
1408    }
1409
1410    /// The encoded name field and following attributes for every entry record.
1411    fn entry_record_fields(bytes: &[u8]) -> Vec<(std::ops::Range<usize>, std::ops::Range<usize>)> {
1412        let count_at = entry_count_offset(bytes);
1413        let count = u64::from_le_bytes(
1414            bytes[count_at..count_at + 8].try_into().expect("saved snapshot has an entry count"),
1415        );
1416        let mut at = count_at + 8;
1417        let mut fields = Vec::new();
1418        for _ in 0..count {
1419            let name_len_at = at + 5;
1420            let name_len = usize::try_from(u32::from_le_bytes(
1421                bytes[name_len_at..name_len_at + 4]
1422                    .try_into()
1423                    .expect("saved entry has a name length"),
1424            ))
1425            .expect("name length fits usize");
1426            let name_end = name_len_at + 4 + name_len;
1427            fields.push((name_len_at..name_end, name_end..name_end + 6 * 8));
1428            at = name_end + 6 * 8;
1429        }
1430        fields
1431    }
1432
1433    fn replace_entry_name(image: &[u8], record: usize, name: &OsStr) -> Vec<u8> {
1434        let mut encoded = Vec::new();
1435        put_os_str(&mut encoded, name).expect("encode replacement name");
1436        let field = entry_record_fields(image)[record].0.clone();
1437        let mut rewritten = image.to_vec();
1438        rewritten.splice(field, encoded);
1439        rewrite_checksum(&mut rewritten);
1440        rewritten
1441    }
1442
1443    #[test]
1444    fn crc32c_matches_the_standard_check_value() {
1445        assert_eq!(crc32c(b"123456789"), 0xe306_9283);
1446    }
1447
1448    #[test]
1449    fn crc32c_slicing_matches_the_byte_reference_on_uneven_lengths() {
1450        // The 8-byte path and the remainder loop must agree with the classic
1451        // byte-at-a-time recurrence at every alignment, or an old snapshot's digest
1452        // stops verifying. The reference below IS that recurrence, against table 0.
1453        fn reference(bytes: &[u8]) -> u32 {
1454            let mut state = u32::MAX;
1455            for byte in bytes {
1456                let index = usize::from(state.to_le_bytes()[0] ^ *byte);
1457                state = CRC32C_TABLES[0][index] ^ (state >> 8);
1458            }
1459            !state
1460        }
1461        let mut data = Vec::new();
1462        let mut seed = 0x9e37_79b9u32;
1463        for length in [0usize, 1, 7, 8, 9, 15, 16, 63, 64, 65, 1000] {
1464            data.clear();
1465            for _ in 0..length {
1466                seed = seed.wrapping_mul(1_664_525).wrapping_add(1_013_904_223);
1467                data.push(seed.to_le_bytes()[0]);
1468            }
1469            assert_eq!(crc32c(&data), reference(&data), "length {length}");
1470        }
1471    }
1472
1473    #[test]
1474    fn snapshot_names_must_equal_their_single_normal_component() {
1475        assert!(is_snapshot_name(OsStr::new("notes.md")));
1476        assert!(!is_snapshot_name(OsStr::new("notes.md/")));
1477        assert!(!is_snapshot_name(OsStr::new("a/b")));
1478        assert!(!is_snapshot_name(OsStr::new("../bad")));
1479        assert!(!is_snapshot_name(OsStr::new("")));
1480        assert!(!is_snapshot_name(OsStr::new(".")));
1481        assert!(!is_snapshot_name(OsStr::new("..")));
1482        #[cfg(unix)]
1483        {
1484            use std::os::unix::ffi::OsStringExt;
1485            let native = OsString::from_vec(vec![b'n', 0x80]);
1486            assert!(is_snapshot_name(&native), "canonical validation is OsStr identity");
1487            let aliased = OsString::from_vec(vec![b'n', 0x80, b'/']);
1488            assert!(!is_snapshot_name(&aliased));
1489        }
1490    }
1491
1492    #[test]
1493    fn a_valid_forged_rename_loads_and_is_found_by_lookup() {
1494        let dir = tempfile::tempdir().expect("tempdir");
1495        let path = dir.path().join("snapshot.fdu");
1496        let mut index = Index::new("/some/root");
1497        index.apply_ok(&Observation::new(vec![Op::Upsert {
1498            path: PathBuf::from("valid"),
1499            kind: EntryKind::File,
1500            attrs: attrs(1, 1),
1501        }]));
1502        save(&index, &path).expect("save");
1503        let saved = fs::read(&path).expect("read");
1504        let forged = replace_entry_name(&saved, 1, OsStr::new("renamed"));
1505        fs::write(&path, forged).expect("write valid rename");
1506        let restored = load(&path).expect("load").expect("a valid rename is a snapshot");
1507        assert!(restored.lookup(Path::new("renamed")).is_some());
1508        assert!(restored.lookup(Path::new("valid")).is_none());
1509    }
1510
1511    #[test]
1512    fn noncanonical_entry_names_are_rejected_after_integrity_checks() {
1513        let dir = tempfile::tempdir().expect("tempdir");
1514        let path = dir.path().join("snapshot.fdu");
1515        let mut index = Index::new("/some/root");
1516        index.apply_ok(&Observation::new(vec![Op::Upsert {
1517            path: PathBuf::from("valid"),
1518            kind: EntryKind::File,
1519            attrs: attrs(1, 1),
1520        }]));
1521        save(&index, &path).expect("save");
1522        let saved = fs::read(&path).expect("read");
1523
1524        #[cfg(windows)]
1525        let names = ["a/", "a//", "a/.", "./a", r"a\", r"a\\", r"a\."];
1526        #[cfg(not(windows))]
1527        let names = ["a/", "a//", "a/.", "./a"];
1528        for name in names {
1529            let forged = replace_entry_name(&saved, 1, OsStr::new(name));
1530            fs::write(&path, forged).expect("write forged snapshot");
1531            assert!(load(&path).expect("malformed snapshot is a miss").is_none(), "{name:?}");
1532        }
1533    }
1534
1535    #[test]
1536    fn a_checksummed_name_alias_cannot_serve_cache_only_content() {
1537        let dir = tempfile::tempdir().expect("tempdir");
1538        let root = dir.path().join("root");
1539        fs::create_dir(&root).expect("create root");
1540        fs::write(root.join("a"), b"one two\n").expect("write first");
1541        fs::write(root.join("bb"), b"one two\n").expect("write second");
1542        let snapshot_path = dir.path().join("snapshot.fdu");
1543        let config = crate::OpenFixture {
1544            cache_path: Some(snapshot_path.clone()),
1545            policy: crate::CachePolicy::Auto,
1546            analysis: crate::content::AnalysisRequest {
1547                profile: crate::content::AnalysisSet::NONE.with_lines(),
1548                ..crate::content::AnalysisRequest::default()
1549            },
1550            ..crate::OpenFixture::default()
1551        };
1552        crate::open_fixture(&root, &config).expect("seed snapshot and sidecar");
1553
1554        let mut image = fs::read(&snapshot_path).expect("read snapshot");
1555        let fields = entry_record_fields(&image);
1556        assert_eq!(fields.len(), 3, "root and two files");
1557        let first_attrs = image[fields[1].1.clone()].to_vec();
1558        image[fields[2].1.clone()].copy_from_slice(&first_attrs);
1559        let forged = replace_entry_name(&image, 2, OsStr::new("a/"));
1560        fs::write(&snapshot_path, forged).expect("write checksummed alias");
1561
1562        let only = crate::OpenFixture { stale_ok: true, ..config };
1563        assert!(
1564            matches!(crate::open_fixture(&root, &only), Err(Error::Snapshot(_))),
1565            "a malformed snapshot cannot shrink the cache-only completeness denominator"
1566        );
1567    }
1568
1569    #[test]
1570    fn a_loaded_index_reports_cached_provenance_not_fresh() {
1571        // The gap that motivated the provenance model. A snapshot is complete when it
1572        // is written, so a loaded index used to claim `Fresh` — true of when the file
1573        // was made, and exactly backwards for a consumer painting on load, which needs
1574        // to know nothing has been checked since.
1575        let dir = tempfile::tempdir().expect("tempdir");
1576        let path = dir.path().join("snapshot.fdu");
1577        let original = sample_index();
1578        save(&original, &path).expect("save");
1579
1580        let restored = load(&path).expect("load").expect("snapshot present");
1581        let provenance =
1582            restored.provenance(Path::new("src/main.rs")).expect("the loaded entry is present");
1583        assert_eq!(provenance.source, crate::Source::Cached);
1584        assert!(!provenance.is_verified(), "nothing has been stat'd since the load");
1585        assert!(
1586            provenance.observed_at_ns > 0,
1587            "a cached value must say as of when, or a UI cannot label it"
1588        );
1589
1590        // A freshly scanned index is the contrasting case.
1591        assert_eq!(
1592            original.provenance(Path::new("src/main.rs")).expect("present").source,
1593            crate::Source::Scanned
1594        );
1595    }
1596
1597    #[test]
1598    fn a_loaded_root_reports_cached_provenance_not_fresh() {
1599        // The root is the one entry the child-only test above cannot cover, and it is
1600        // the one that matters most: whole-tree totals hang off it, so `fdu ~` reads
1601        // the root's provenance to label its headline number. `apply_upsert` handles
1602        // the root in a separate branch, and that branch used to skip the source stamp
1603        // entirely — leaving a snapshot-loaded root claiming `Scanned` and
1604        // `is_verified()`, the precise silent lie the model exists to prevent.
1605        let dir = tempfile::tempdir().expect("tempdir");
1606        let path = dir.path().join("snapshot.fdu");
1607        save(&sample_index(), &path).expect("save");
1608
1609        let restored = load(&path).expect("load").expect("snapshot present");
1610        let provenance = restored.provenance(Path::new("")).expect("the root is always present");
1611        assert_eq!(provenance.source, crate::Source::Cached, "the root came off disk too");
1612        assert!(!provenance.is_verified(), "nothing has been stat'd since the load");
1613        assert!(
1614            provenance.observed_at_ns > 0,
1615            "a cached total must say as of when, or a UI cannot label it"
1616        );
1617    }
1618
1619    #[test]
1620    fn cached_observation_time_comes_from_the_header_after_touch_or_copy() {
1621        use std::time::{Duration, UNIX_EPOCH};
1622
1623        let dir = tempfile::tempdir().expect("tempdir");
1624        let path = dir.path().join("snapshot.fdu");
1625        let copied = dir.path().join("copied.fdu");
1626        let mut original = sample_index();
1627        original.set_writing_pass_started_at_ns(1_000);
1628        save(&original, &path).expect("save");
1629
1630        // Cache files are routinely kept in place or copied between cache locations. Neither
1631        // operation observes the tree, so neither file mtime may become value provenance.
1632        touch(&path, UNIX_EPOCH + Duration::from_secs(10)).expect("touch cache image");
1633        fs::copy(&path, &copied).expect("copy cache image");
1634        touch(&copied, UNIX_EPOCH + Duration::from_secs(20)).expect("touch copied image");
1635
1636        for cache in [&path, &copied] {
1637            let restored = load(cache).expect("load").expect("present");
1638            for entry in [Path::new(""), Path::new("src/main.rs")] {
1639                assert_eq!(
1640                    restored.provenance(entry).expect("present").observed_at_ns,
1641                    1_000,
1642                    "{} uses its persisted pass start, not its cache-file mtime",
1643                    cache.display()
1644                );
1645            }
1646        }
1647    }
1648
1649    #[test]
1650    fn revalidating_a_loaded_index_promotes_entries_out_of_cached() {
1651        // The failure a reviewer caught on PR #6: entries were stamped only when
1652        // allocated, so a warm sweep could stat every entry and leave them all
1653        // reporting `Cached`. A consumer could then never clear a stale-value
1654        // indicator no matter how much verification ran, which defeats the point.
1655        let dir = tempfile::tempdir().expect("tempdir");
1656        let tree = dir.path().join("tree");
1657        std::fs::create_dir_all(tree.join("sub")).expect("create dirs");
1658        std::fs::write(tree.join("sub/file.txt"), b"contents").expect("write");
1659        let snapshot_path = dir.path().join("snapshot.fdu");
1660
1661        let config = crate::ScanConfig::default();
1662        let (original, report) = crate::scan::scan_into_index(&tree, &config).expect("scan");
1663        assert!(report.is_complete());
1664        save(&original, &snapshot_path).expect("save");
1665
1666        let mut restored = load(&snapshot_path).expect("load").expect("present");
1667        let target = Path::new("sub/file.txt");
1668        assert_eq!(
1669            restored.provenance(target).expect("present").source,
1670            crate::Source::Cached,
1671            "straight off disk, nothing has been checked"
1672        );
1673
1674        // A sweep that finds nothing changed still verified every entry it stat'd.
1675        let reconciled =
1676            crate::scan::reconcile(&mut restored, &config, &mut |_| {}).expect("reconcile");
1677        assert!(reconciled.is_complete());
1678        assert_eq!(reconciled.apply.updated, 0, "the tree did not change");
1679
1680        let provenance = restored.provenance(target).expect("present");
1681        assert_eq!(
1682            provenance.source,
1683            crate::Source::Revalidated,
1684            "an unchanged entry that was freshly stat'd has still been verified"
1685        );
1686        assert!(provenance.is_verified());
1687    }
1688
1689    /// Every entry, not just the root.
1690    ///
1691    /// The loader inserts beneath a known parent instead of replaying an observation, so
1692    /// it no longer shares a code path with the producer that built the original. Root
1693    /// totals cannot catch a roll-up that is wrong halfway down — the errors would have
1694    /// to cancel at the root to hide, but a single misplaced subtree does not touch the
1695    /// root at all. This walks both trees and compares each directory's own roll-up,
1696    /// each entry's kind, attributes and extension tallies, and the shape of the tree.
1697    #[test]
1698    fn round_trip_preserves_every_entry_not_just_the_root() {
1699        let dir = tempfile::tempdir().expect("tempdir");
1700        let tree = dir.path().join("tree");
1701        let snapshot_path = dir.path().join("cache").join("snap.fdu");
1702        // Depth and fan-out both matter: a parent-relative insert that mis-parents an
1703        // entry shows up as a wrong roll-up on an interior directory.
1704        for (relative, contents) in [
1705            ("a.rs", &b"fn main() {}"[..]),
1706            ("deep/one/two/three/leaf.txt", b"leaf"),
1707            ("deep/one/two/sibling.rs", b"sibling"),
1708            ("deep/one/other.md", b"# other"),
1709            ("wide/w1.txt", b"1"),
1710            ("wide/w2.txt", b"22"),
1711            ("wide/w3.rs", b"333"),
1712            ("empty/.keep", b""),
1713        ] {
1714            let path = tree.join(relative);
1715            fs::create_dir_all(path.parent().expect("parent")).expect("create dirs");
1716            fs::write(&path, contents).expect("write");
1717        }
1718
1719        let config = crate::ScanConfig::default();
1720        let (original, report) = crate::scan::scan_into_index(&tree, &config).expect("scan");
1721        assert!(report.is_complete());
1722        save(&original, &snapshot_path).expect("save");
1723        let restored = load(&snapshot_path).expect("load").expect("present");
1724
1725        assert_eq!(restored.len(), original.len(), "entry count");
1726
1727        // Walk the original and demand the same entry, in the same place, with the same
1728        // numbers, in the restored index.
1729        let mut stack = vec![crate::index::EntryId::ROOT];
1730        let mut compared = 0_u64;
1731        while let Some(id) = stack.pop() {
1732            let path = original.path_of(id).expect("original path");
1733            let mirrored = restored.lookup(&path).expect("restored entry at the same path");
1734            assert_eq!(
1735                original.kind_of(id),
1736                restored.kind_of(mirrored),
1737                "kind at {}",
1738                path.display()
1739            );
1740            assert_eq!(
1741                original.attrs_of(id),
1742                restored.attrs_of(mirrored),
1743                "attrs at {}",
1744                path.display()
1745            );
1746            // Roll-ups and children exist only on directories; a file legitimately has
1747            // neither, and demanding them would fail on the tree rather than the loader.
1748            if original.kind_of(id) == Some(crate::EntryKind::Dir) {
1749                let (before, after) = (
1750                    original.rollup_of(id).expect("original rollup"),
1751                    restored.rollup_of(mirrored).expect("restored rollup"),
1752                );
1753                assert_eq!(
1754                    (
1755                        before.files,
1756                        before.dirs,
1757                        before.bytes,
1758                        before.allocated,
1759                        before.newest_mtime_ns
1760                    ),
1761                    (after.files, after.dirs, after.bytes, after.allocated, after.newest_mtime_ns),
1762                    "rollup at {}",
1763                    path.display()
1764                );
1765                assert_eq!(before.by_ext, after.by_ext, "extension tallies at {}", path.display());
1766                let names: Vec<_> = original
1767                    .children_of(id)
1768                    .expect("original children")
1769                    .map(|(name, _)| name.to_os_string())
1770                    .collect();
1771                let mirrored_names: Vec<_> = restored
1772                    .children_of(mirrored)
1773                    .expect("restored children")
1774                    .map(|(name, _)| name.to_os_string())
1775                    .collect();
1776                assert_eq!(names, mirrored_names, "children of {}", path.display());
1777                stack.extend(original.children_of(id).expect("children").map(|(_, child)| child));
1778            }
1779            compared += 1;
1780        }
1781        assert_eq!(compared, original.len(), "every entry was compared");
1782
1783        // The loader must not leave the index looking like it has pending history.
1784        assert_eq!(restored.clock(), crate::Clock::ZERO);
1785        assert!(restored.since(crate::Clock::ZERO).commits.is_empty());
1786    }
1787
1788    #[test]
1789    fn round_trip_preserves_tree_and_rollups() {
1790        let dir = tempfile::tempdir().expect("tempdir");
1791        let path = dir.path().join("cache").join("snap.fdu");
1792        let original = sample_index();
1793
1794        save(&original, &path).expect("save");
1795        let restored = load(&path).expect("load").expect("snapshot present");
1796
1797        assert_eq!(restored.root_path(), Path::new("/some/root"));
1798        assert_eq!(restored.clock(), crate::Clock::ZERO);
1799        assert!(restored.since(crate::Clock::ZERO).commits.is_empty());
1800        assert_eq!(restored.len(), original.len());
1801        // Public roll-ups resolve assignment-ordered extension ids to stable names.
1802        let (restored_total, original_total) = (restored.total(), original.total());
1803        assert_eq!(
1804            (restored_total.files, restored_total.dirs, restored_total.bytes),
1805            (original_total.files, original_total.dirs, original_total.bytes)
1806        );
1807        assert_eq!(restored_total.allocated, original_total.allocated);
1808        assert_eq!(restored_total.newest_mtime_ns, original_total.newest_mtime_ns);
1809        assert_eq!(restored_total.by_ext, original_total.by_ext);
1810        assert!(!original.serving_indexes_enabled());
1811        assert!(!restored.serving_indexes_enabled());
1812        assert_eq!(restored.total().files, 3);
1813        assert_eq!(restored.total().dirs, 2);
1814        assert_eq!(restored.total().bytes, 157);
1815        assert_eq!(
1816            restored.total().by_ext[".rs"],
1817            ExtTally { files: 2, bytes: 150, allocated: 1024 }
1818        );
1819        assert_eq!(
1820            restored.attrs(Path::new("src/deep/nested.rs")),
1821            original.attrs(Path::new("src/deep/nested.rs"))
1822        );
1823    }
1824
1825    #[test]
1826    fn round_trip_preserves_exact_controls_and_fixed_partitions() {
1827        let dir = tempfile::tempdir().expect("tempdir");
1828        let path = dir.path().join("controls.fdu");
1829        let mut original =
1830            Index::new_with_scope("/some/root", crate::test_support::observing_controls());
1831        original.apply_ok(&Observation::new(vec![
1832            Op::Upsert {
1833                path: PathBuf::from(".gitignore"),
1834                kind: EntryKind::File,
1835                attrs: attrs(6, 1),
1836            },
1837            Op::Upsert {
1838                path: PathBuf::from("debug.log"),
1839                kind: EntryKind::File,
1840                attrs: attrs(10, 2),
1841            },
1842            Op::Upsert {
1843                path: PathBuf::from("keep.rs"),
1844                kind: EntryKind::File,
1845                attrs: attrs(20, 3),
1846            },
1847            Op::ControlUpsert { path: PathBuf::from(".gitignore"), source: b"*.log\n".to_vec() },
1848        ]));
1849
1850        save(&original, &path).expect("save");
1851        let restored = load(&path).expect("load").expect("snapshot present");
1852
1853        assert!(
1854            restored
1855                .controls()
1856                .expect("control state observed")
1857                .source_is(Path::new(".gitignore"), b"*.log\n")
1858        );
1859        assert_eq!(restored.controls().expect("control state observed").source_bytes(), 6);
1860        assert_eq!(
1861            restored.is_ignored(Path::new("debug.log")).expect("control state observed"),
1862            Some(true)
1863        );
1864        assert_eq!(
1865            restored.is_ignored(Path::new("keep.rs")).expect("control state observed"),
1866            Some(false)
1867        );
1868        let partitions = restored.partition_total().expect("control state observed");
1869        assert_eq!(partitions, original.partition_total().expect("control state observed"));
1870        assert_eq!(partitions.all.files, 3);
1871        assert_eq!(partitions.unignored.files, 2);
1872    }
1873
1874    #[test]
1875    fn removing_the_last_control_before_save_round_trips_an_empty_table() {
1876        let dir = tempfile::tempdir().expect("tempdir");
1877        let path = dir.path().join("no-controls.fdu");
1878        let mut original =
1879            Index::new_with_scope("/some/root", crate::test_support::observing_controls());
1880        original.apply_ok(&Observation::new(vec![
1881            Op::Upsert {
1882                path: PathBuf::from(".gitignore"),
1883                kind: EntryKind::File,
1884                attrs: attrs(6, 1),
1885            },
1886            Op::Upsert {
1887                path: PathBuf::from("debug.log"),
1888                kind: EntryKind::File,
1889                attrs: attrs(10, 2),
1890            },
1891            Op::ControlUpsert { path: PathBuf::from(".gitignore"), source: b"*.log\n".to_vec() },
1892        ]));
1893        original
1894            .apply_ok(&Observation::new(vec![Op::Remove { path: PathBuf::from(".gitignore") }]));
1895
1896        save(&original, &path).expect("save");
1897        let restored = load(&path).expect("load").expect("snapshot present");
1898
1899        assert!(restored.controls().expect("control state observed").is_empty());
1900        assert_eq!(
1901            restored.is_ignored(Path::new("debug.log")).expect("control state observed"),
1902            Some(false)
1903        );
1904        let partitions = restored.partition_total().expect("control state observed");
1905        assert_eq!(partitions.all, partitions.unignored);
1906    }
1907
1908    #[test]
1909    fn a_control_table_at_its_shared_bound_round_trips() {
1910        let dir = tempfile::tempdir().expect("tempdir");
1911        let path = dir.path().join("bounded-controls.fdu");
1912        let mut original =
1913            Index::new_with_scope("/some/root", crate::test_support::observing_controls());
1914        let source = crate::control::source_at_test_limit();
1915        original.apply_ok(&Observation::new(vec![Op::ControlUpsert {
1916            path: PathBuf::from(".gitignore"),
1917            source: source.clone(),
1918        }]));
1919        assert_eq!(
1920            original.controls().expect("control state observed").retained_cost(),
1921            crate::control::DEFAULT_CONTROL_BUDGET
1922        );
1923
1924        save(&original, &path).expect("save at bound");
1925        let restored = load(&path).expect("load").expect("snapshot present");
1926
1927        assert_eq!(
1928            restored.controls().expect("control state observed").retained_cost(),
1929            original.controls().expect("control state observed").retained_cost()
1930        );
1931        assert!(
1932            restored
1933                .controls()
1934                .expect("control state observed")
1935                .source_is(Path::new(".gitignore"), &source)
1936        );
1937    }
1938
1939    /// A control table must enforce the limits its scope was taken under. A hand-built index
1940    /// whose scope claims other limits is refused at save with a typed error, a snapshot
1941    /// whose control section its header's control tier would not admit fails closed at
1942    /// load, and `Index::new_with_config` builds an index whose table and scope agree.
1943    #[test]
1944    fn control_limits_that_disagree_with_the_scope_are_refused_at_save_and_load() {
1945        let dir = tempfile::tempdir().expect("tempdir");
1946        let path = dir.path().join("limits.fdu");
1947        let lifted = crate::ScanConfig {
1948            control_limits: crate::control::ControlLimits {
1949                budget: None,
1950                ..crate::control::ControlLimits::default()
1951            },
1952            ..crate::ScanConfig::default()
1953        };
1954
1955        let mismatched = Index::new_with_scope("/some/root", lifted.scope());
1956        let error = save(&mismatched, &path).expect_err("the scope claims no budget");
1957        assert!(
1958            matches!(
1959                error,
1960                Error::ControlLimitsOutsideScope { limits }
1961                    if limits == crate::control::ControlLimits::default()
1962            ),
1963            "{error}"
1964        );
1965        assert!(!path.exists(), "nothing is written");
1966
1967        let mut agreeing = Index::new_with_config("/some/root", &lifted);
1968        agreeing.apply_ok(&Observation::new(vec![Op::ControlUpsert {
1969            path: PathBuf::from(".gitignore"),
1970            source: b"*.log\n".to_vec(),
1971        }]));
1972        assert_eq!(agreeing.scope(), lifted.scope());
1973        save(&agreeing, &path).expect("save an index whose table enforces its scope's limits");
1974        let restored = load(&path).expect("load").expect("snapshot present");
1975        assert_eq!(restored.control_coverage(), agreeing.control_coverage());
1976        assert_eq!(restored.snapshot_identity(), lifted.snapshot_identity());
1977
1978        // The limits live only in the header's control tier. A CRC-valid snapshot whose
1979        // header claims a line limit the retained source exceeds, or claims no control
1980        // state beside a retained source, was written by no save, so it fails closed like
1981        // any other corruption.
1982        let saved = fs::read(&path).expect("read snapshot");
1983        let controls_at = IDENTITY_OFFSET + crate::stored_state::ENTRY_TIER_BYTES;
1984        let controls = controls_at..controls_at + crate::stored_state::CONTROL_TIER_BYTES;
1985        let blind = crate::ScanConfig { read_controls: false, ..lifted.clone() };
1986        let forge = |tier: ControlTierIdentity| {
1987            let mut forged = saved.clone();
1988            forged[controls.clone()].copy_from_slice(&tier.encode());
1989            rewrite_checksum(&mut forged);
1990            fs::write(&path, &forged).expect("write forged limits");
1991            (
1992                load(&path).expect("forged equals absent"),
1993                load_serving(&path, blind.types_shared(), blind.snapshot_identity())
1994                    .expect("projected forged equals absent"),
1995            )
1996        };
1997        let tight = crate::control::ControlLimits { line_limit: Some(1), ..lifted.control_limits };
1998        let (exact, projected) = forge(ControlTierIdentity::Observed { limits: tight });
1999        assert!(exact.is_none());
2000        assert!(matches!(projected, LoadOutcome::Absent));
2001        let (exact, projected) = forge(ControlTierIdentity::NotObserved);
2002        assert!(exact.is_none());
2003        assert!(matches!(projected, LoadOutcome::Absent));
2004        // The same splice with the saved tier restores, so the splice is not what failed.
2005        let (exact, projected) = forge(lifted.control_identity());
2006        assert!(exact.is_some());
2007        let LoadOutcome::Served { index: projected, stored } = projected else {
2008            panic!("the valid control payload projects");
2009        };
2010        assert_eq!(serves_snapshot(stored, blind.snapshot_identity()), Serves::ProjectControlsOff);
2011        assert_eq!(projected.snapshot_identity(), blind.snapshot_identity());
2012        assert_eq!(projected.scope(), blind.scope());
2013        assert!(matches!(projected.controls(), Err(Error::ControlStateNotObserved)));
2014    }
2015
2016    /// An index whose control table refused some sources, and its path-ordered tail.
2017    fn index_with_refused_controls() -> Index {
2018        let mut index =
2019            Index::new_with_scope("/some/root", crate::test_support::observing_controls());
2020        let mut line = vec![b'x'; crate::control::DEFAULT_CONTROL_LINE_LIMIT + 1];
2021        line.push(b'\n');
2022        index.apply_ok(&Observation::new(vec![
2023            Op::Upsert {
2024                path: PathBuf::from(".gitignore"),
2025                kind: EntryKind::File,
2026                attrs: attrs(6, 1),
2027            },
2028            Op::Upsert { path: PathBuf::from("a"), kind: EntryKind::Dir, attrs: attrs(0, 1) },
2029            Op::Upsert {
2030                path: PathBuf::from("a/.gitignore"),
2031                kind: EntryKind::File,
2032                attrs: attrs(1, 1),
2033            },
2034            Op::Upsert { path: PathBuf::from("b"), kind: EntryKind::Dir, attrs: attrs(0, 1) },
2035            Op::Upsert {
2036                path: PathBuf::from("b/.gitignore"),
2037                kind: EntryKind::File,
2038                attrs: attrs(1, 1),
2039            },
2040            Op::Upsert {
2041                path: PathBuf::from("b/debug.log"),
2042                kind: EntryKind::File,
2043                attrs: attrs(9, 1),
2044            },
2045            Op::ControlUpsert { path: PathBuf::from(".gitignore"), source: b"*.log\n".to_vec() },
2046            Op::ControlUpsert { path: PathBuf::from("a/.gitignore"), source: line },
2047            Op::ControlUpsert {
2048                path: PathBuf::from("b/.gitignore"),
2049                source: b"y\n".repeat(crate::control::DEFAULT_CONTROL_BUDGET / 2),
2050            },
2051        ]));
2052        index
2053    }
2054
2055    /// A snapshot of an index that refused control files reloads with the same coverage,
2056    /// rather than as an index whose every rule applied.
2057    #[test]
2058    fn a_snapshot_with_refused_controls_reloads_its_coverage_exactly() {
2059        let dir = tempfile::tempdir().expect("tempdir");
2060        let path = dir.path().join("refused.fdu");
2061        let original = index_with_refused_controls();
2062        let crate::control::ControlCoverage::Observed(coverage) = original.control_coverage()
2063        else {
2064            panic!("observed");
2065        };
2066        assert_eq!((coverage.applied, coverage.refused), (1, 2));
2067
2068        save(&original, &path).expect("save a partially covered index");
2069        let restored = load(&path).expect("load").expect("snapshot present");
2070
2071        assert_eq!(restored.control_coverage(), original.control_coverage());
2072        assert_eq!(
2073            restored.is_ignored(Path::new("b/debug.log")).expect("observed"),
2074            original.is_ignored(Path::new("b/debug.log")).expect("observed")
2075        );
2076        assert_eq!(
2077            restored.partition_total().expect("observed"),
2078            original.partition_total().expect("observed")
2079        );
2080    }
2081
2082    /// The refusal records are parsed as strictly as the rest: an unknown reason, and a
2083    /// refusal naming a retained source, are corrupt.
2084    #[test]
2085    fn corrupt_refusal_records_fail_closed() {
2086        let dir = tempfile::tempdir().expect("tempdir");
2087        let path = dir.path().join("refused.fdu");
2088        save(&index_with_refused_controls(), &path).expect("save");
2089        let saved = fs::read(&path).expect("read snapshot");
2090        let footer = CHECKSUM_BYTES + TRAILER.len();
2091        // The last refusal is `b/.gitignore` and its one-byte reason ends the payload.
2092        let reason_at = saved.len() - footer - 1;
2093        assert_eq!(saved[reason_at], REFUSED_FOR_BUDGET);
2094
2095        let mut unknown_reason = saved;
2096        unknown_reason[reason_at] = 9;
2097        rewrite_checksum(&mut unknown_reason);
2098        fs::write(&path, &unknown_reason).expect("write corrupt reason");
2099        assert!(load(&path).expect("corrupt equals absent").is_none());
2100
2101        // A refusal naming a retained source, or one already refused, is corrupt, and so is a
2102        // refusal by a limit the section records as unbounded, which refuses nothing. Built
2103        // with the platform's own path encoding, so the same bytes mean the same on every
2104        // target.
2105        let defaults = crate::control::ControlLimits::default();
2106        let section = |refusals: &[(&str, u8)]| {
2107            let mut section = Vec::new();
2108            section.extend_from_slice(&1_u32.to_le_bytes());
2109            put_os_str(&mut section, OsStr::new(".gitignore")).expect("retained path");
2110            section.extend_from_slice(&6_u32.to_le_bytes());
2111            section.extend_from_slice(b"*.log\n");
2112            let count = u32::try_from(refusals.len()).expect("few refusals");
2113            section.extend_from_slice(&count.to_le_bytes());
2114            for (refusal, reason) in refusals {
2115                put_os_str(&mut section, OsStr::new(refusal)).expect("refused path");
2116                section.push(*reason);
2117            }
2118            section
2119        };
2120        let no_budget = crate::control::ControlLimits { budget: None, ..defaults };
2121        let no_line_limit = crate::control::ControlLimits { line_limit: None, ..defaults };
2122        for (limits, refusals) in [
2123            (defaults, &[("a/.gitignore", REFUSED_FOR_BUDGET)][..]),
2124            (no_budget, &[("a/.gitignore", REFUSED_FOR_LINE_LIMIT)][..]),
2125            (no_line_limit, &[("a/.gitignore", REFUSED_FOR_BUDGET)][..]),
2126        ] {
2127            let tier = ControlTierIdentity::Observed { limits };
2128            let valid = read_controls(&mut section(refusals).as_slice(), tier)
2129                .expect("a well-formed control section");
2130            assert_eq!((valid.len(), valid.refused_len()), (1, 1));
2131        }
2132        for (limits, refusals) in [
2133            (defaults, &[(".gitignore", REFUSED_FOR_BUDGET)][..]),
2134            (
2135                defaults,
2136                &[("a/.gitignore", REFUSED_FOR_BUDGET), ("a/.gitignore", REFUSED_FOR_BUDGET)][..],
2137            ),
2138            (no_budget, &[("a/.gitignore", REFUSED_FOR_BUDGET)][..]),
2139            (no_line_limit, &[("a/.gitignore", REFUSED_FOR_LINE_LIMIT)][..]),
2140        ] {
2141            let tier = ControlTierIdentity::Observed { limits };
2142            assert!(
2143                matches!(
2144                    read_controls(&mut section(refusals).as_slice(), tier),
2145                    Err(ParseError::Invalid)
2146                ),
2147                "{limits:?} {refusals:?}"
2148            );
2149        }
2150        // A tier that observed nothing holds no source and no refusal.
2151        for refusals in [&[][..], &[("a/.gitignore", REFUSED_FOR_BUDGET)][..]] {
2152            assert!(matches!(
2153                read_controls(&mut section(refusals).as_slice(), ControlTierIdentity::NotObserved),
2154                Err(ParseError::Invalid)
2155            ));
2156        }
2157    }
2158
2159    /// A snapshot with no control state loads without walking the tree to reclassify it.
2160    ///
2161    /// Installing the loaded control table re-evaluated every entry's ignored bit, allocating
2162    /// a path per entry, even when the table was empty and no bit could move. That is every
2163    /// warm open of a tree with no `.gitignore`, and it scaled with the tree.
2164    #[test]
2165    fn loading_a_snapshot_without_controls_skips_the_reclassification_walk() {
2166        fn visits_while_loading(path: &Path) -> (u64, Index) {
2167            crate::index::RECLASSIFY_VISITS.with(|visits| visits.set(0));
2168            let restored = load(path).expect("load").expect("snapshot present");
2169            (crate::index::RECLASSIFY_VISITS.with(std::cell::Cell::get), restored)
2170        }
2171
2172        let dir = tempfile::tempdir().expect("tempdir");
2173        let mut original =
2174            Index::new_with_scope("/some/root", crate::test_support::observing_controls());
2175        let mut ops = vec![Op::Upsert {
2176            path: PathBuf::from("src"),
2177            kind: EntryKind::Dir,
2178            attrs: attrs(0, 1),
2179        }];
2180        ops.extend((0..64).map(|sequence| Op::Upsert {
2181            path: PathBuf::from(format!("src/file-{sequence:02}.rs")),
2182            kind: EntryKind::File,
2183            attrs: attrs(sequence + 1, 2),
2184        }));
2185        original.apply_baseline_ok(&Observation::new(ops));
2186        let plain = dir.path().join("plain.fdu");
2187        save(&original, &plain).expect("save");
2188
2189        let (visits, restored) = visits_while_loading(&plain);
2190
2191        assert_eq!(visits, 0, "an empty control table has nothing to reclassify");
2192        assert!(restored.control_table().is_empty());
2193        // The load still answers as the saved index did.
2194        assert_eq!(
2195            restored.is_ignored(Path::new("src/file-00.rs")).expect("control state observed"),
2196            Some(false)
2197        );
2198        assert_eq!(
2199            restored.partition_total().expect("control state observed"),
2200            original.partition_total().expect("control state observed")
2201        );
2202
2203        // The probe sees the walk when there is something to walk for.
2204        original.apply_ok(&Observation::new(vec![Op::ControlUpsert {
2205            path: PathBuf::from("src/.gitignore"),
2206            source: b"file-0*.rs\n".to_vec(),
2207        }]));
2208        let controlled = dir.path().join("controlled.fdu");
2209        save(&original, &controlled).expect("save");
2210
2211        let (visits, restored) = visits_while_loading(&controlled);
2212
2213        assert!(visits > 64, "{visits}");
2214        assert_eq!(
2215            restored.is_ignored(Path::new("src/file-00.rs")).expect("control state observed"),
2216            Some(true)
2217        );
2218        assert_eq!(
2219            restored.is_ignored(Path::new("src/file-10.rs")).expect("control state observed"),
2220            Some(false)
2221        );
2222    }
2223
2224    #[test]
2225    fn round_trip_handles_wide_directory_fanout() {
2226        // Load resolves each record's id from its parent. Doing that by scanning the
2227        // parent's children costs O(width^2) per directory, which is invisible on the
2228        // handful of siblings every other test uses and dominates a real one — a
2229        // node_modules or a Maildir is thousands wide.
2230        const CHILDREN: u64 = 4_096;
2231        let dir = tempfile::tempdir().expect("tempdir");
2232        let path = dir.path().join("wide.fdu");
2233        let mut original = Index::new("/some/root");
2234        let ops = (0..CHILDREN)
2235            .map(|sequence| Op::Upsert {
2236                path: PathBuf::from(format!("child-{sequence:04}.dat")),
2237                kind: EntryKind::File,
2238                attrs: attrs(sequence + 1, i64::try_from(sequence).expect("fanout fits i64")),
2239            })
2240            .collect();
2241        original.apply_baseline_ok(&Observation::new(ops));
2242
2243        save(&original, &path).expect("save wide snapshot");
2244        let restored = load(&path).expect("load wide snapshot").expect("snapshot present");
2245
2246        assert_eq!(restored.total().files, CHILDREN);
2247        assert_eq!(restored.len(), original.len());
2248        // The last child is the one a linear scan reaches last, so it is the one that
2249        // proves the resolution used the parent's map rather than its sibling order.
2250        assert_eq!(
2251            restored.attrs(Path::new("child-4095.dat")),
2252            original.attrs(Path::new("child-4095.dat"))
2253        );
2254    }
2255
2256    #[test]
2257    fn shared_save_captures_before_filesystem_io() {
2258        let dir = tempfile::tempdir().expect("tempdir");
2259        let path = dir.path().join("shared.fdu");
2260        let handle = IndexHandle::new(sample_index());
2261
2262        save_handle(&handle, &path).expect("save shared snapshot");
2263        handle
2264            .apply(&Observation::new(vec![Op::Upsert {
2265                path: PathBuf::from("after.txt"),
2266                kind: EntryKind::File,
2267                attrs: attrs(9, 40),
2268            }]))
2269            .expect("mutate after capture");
2270
2271        let restored = load(&path).expect("load").expect("snapshot present");
2272        assert!(restored.lookup(Path::new("after.txt")).is_none());
2273        assert!(handle.kind(Path::new("after.txt")).expect("query").is_some());
2274    }
2275
2276    #[test]
2277    fn missing_snapshot_is_absent_not_an_error() {
2278        let dir = tempfile::tempdir().expect("tempdir");
2279        let loaded = load(&dir.path().join("nope.fdu")).expect("load must not error");
2280        assert!(loaded.is_none());
2281    }
2282
2283    #[test]
2284    fn configured_size_limit_rejects_before_body_allocation() {
2285        let dir = tempfile::tempdir().expect("tempdir");
2286        let path = dir.path().join("snap.fdu");
2287        save(&sample_index(), &path).expect("save");
2288        let file_len = fs::metadata(&path).expect("metadata").len();
2289
2290        assert!(load_with_size_limit(&path, file_len - 1).expect("load must not error").is_none());
2291    }
2292
2293    #[test]
2294    fn foreign_file_is_treated_as_absent() {
2295        let dir = tempfile::tempdir().expect("tempdir");
2296        let path = dir.path().join("snap.fdu");
2297        fs::write(&path, b"this is not a snapshot at all").expect("write");
2298        assert!(load(&path).expect("load must not error").is_none());
2299    }
2300
2301    #[test]
2302    fn truncated_snapshot_is_treated_as_absent() {
2303        let dir = tempfile::tempdir().expect("tempdir");
2304        let path = dir.path().join("snap.fdu");
2305        save(&sample_index(), &path).expect("save");
2306
2307        let full = fs::read(&path).expect("read");
2308        for cut in [full.len() - 1, full.len() / 2, MAGIC.len() + 2] {
2309            fs::write(&path, &full[..cut]).expect("truncate");
2310            assert!(
2311                load(&path).expect("load must not error").is_none(),
2312                "a snapshot truncated to {cut} bytes must not parse"
2313            );
2314        }
2315    }
2316
2317    #[test]
2318    fn corrupt_body_with_intact_magic_and_trailer_is_rejected() {
2319        let dir = tempfile::tempdir().expect("tempdir");
2320        let path = dir.path().join("snap.fdu");
2321        save(&sample_index(), &path).expect("save");
2322
2323        let mut bytes = fs::read(&path).expect("read");
2324        // Claim far more entries than the body holds.
2325        let count_at = entry_count_offset(&bytes);
2326        bytes[count_at..count_at + 8].copy_from_slice(&u64::MAX.to_le_bytes());
2327        rewrite_checksum(&mut bytes);
2328        fs::write(&path, &bytes).expect("write");
2329
2330        assert!(load(&path).expect("load must not error").is_none());
2331    }
2332
2333    #[test]
2334    fn plausible_attribute_corruption_is_rejected() {
2335        let dir = tempfile::tempdir().expect("tempdir");
2336        let path = dir.path().join("snap.fdu");
2337        save(&sample_index(), &path).expect("save");
2338
2339        let mut bytes = fs::read(&path).expect("read");
2340        let records_at = entry_count_offset(&bytes) + 8;
2341        // The root record has an empty name, so its first attribute starts immediately
2342        // after parent, kind, and encoded-name length. Changing a low size byte keeps the
2343        // image structurally valid and would silently alter the restored state without an
2344        // integrity check.
2345        let root_size_at = records_at + 4 + 1 + 4;
2346        bytes[root_size_at] ^= 1;
2347        fs::write(&path, &bytes).expect("write");
2348
2349        assert!(load(&path).expect("load must not error").is_none());
2350    }
2351
2352    #[test]
2353    fn entry_names_with_path_components_are_rejected() {
2354        let dir = tempfile::tempdir().expect("tempdir");
2355        let path = dir.path().join("snap.fdu");
2356        save(&sample_index(), &path).expect("save");
2357
2358        let mut bytes = fs::read(&path).expect("read");
2359        let count_at = entry_count_offset(&bytes);
2360        let records_at = count_at + 8;
2361        let first_child_name_len_at = records_at + MIN_RECORD_BYTES + 4 + 1;
2362        let old_name_len = u32::from_le_bytes(
2363            bytes[first_child_name_len_at..first_child_name_len_at + 4]
2364                .try_into()
2365                .expect("saved snapshot has a child name length"),
2366        );
2367        let old_name_end = first_child_name_len_at
2368            + 4
2369            + usize::try_from(old_name_len).expect("name length fits usize");
2370        let mut invalid_name = Vec::new();
2371        put_os_str(&mut invalid_name, OsStr::new("../bad")).expect("encode invalid name");
2372        bytes.splice(first_child_name_len_at..old_name_end, invalid_name);
2373        rewrite_checksum(&mut bytes);
2374        fs::write(&path, &bytes).expect("write");
2375
2376        assert!(load(&path).expect("load must not error").is_none());
2377    }
2378
2379    #[test]
2380    fn oversized_declared_path_is_rejected_before_allocation() {
2381        let dir = tempfile::tempdir().expect("tempdir");
2382        let path = dir.path().join("snap.fdu");
2383        save(&sample_index(), &path).expect("save");
2384
2385        let mut bytes = fs::read(&path).expect("read");
2386        let root_len_at = ROOT_OFFSET;
2387        bytes[root_len_at..root_len_at + 4].copy_from_slice(&(MAX_PATH_BYTES + 1).to_le_bytes());
2388        rewrite_checksum(&mut bytes);
2389        fs::write(&path, &bytes).expect("write");
2390
2391        assert!(load(&path).expect("load must not error").is_none());
2392    }
2393
2394    #[test]
2395    fn engine_fingerprint_mismatch_discards_the_snapshot() {
2396        let dir = tempfile::tempdir().expect("tempdir");
2397        let path = dir.path().join("snap.fdu");
2398        save(&sample_index(), &path).expect("save");
2399
2400        let mut bytes = fs::read(&path).expect("read");
2401        let fp_at = MAGIC.len() + 4;
2402        bytes[fp_at] ^= 0xff;
2403        rewrite_checksum(&mut bytes);
2404        fs::write(&path, &bytes).expect("write");
2405
2406        assert!(load(&path).expect("load must not error").is_none());
2407    }
2408
2409    #[test]
2410    fn save_replaces_an_existing_snapshot_and_leaves_no_temp_files() {
2411        let dir = tempfile::tempdir().expect("tempdir");
2412        let path = dir.path().join("snap.fdu");
2413
2414        save(&sample_index(), &path).expect("first save");
2415        let mut smaller = Index::new("/some/root");
2416        smaller.apply_ok(&Observation::new(vec![Op::Upsert {
2417            path: PathBuf::from("only.txt"),
2418            kind: EntryKind::File,
2419            attrs: attrs(1, 1),
2420        }]));
2421        save(&smaller, &path).expect("second save");
2422
2423        let restored = load(&path).expect("load").expect("present");
2424        assert_eq!(restored.total().files, 1);
2425
2426        let leftovers: Vec<_> = fs::read_dir(dir.path())
2427            .expect("read_dir")
2428            .filter_map(std::result::Result::ok)
2429            .map(|e| e.file_name().to_string_lossy().into_owned())
2430            .filter(|name| name != "snap.fdu")
2431            .collect();
2432        assert!(leftovers.is_empty(), "temp files left behind: {leftovers:?}");
2433    }
2434
2435    #[test]
2436    fn an_abandoned_temporary_is_collected_once_it_is_old_enough() {
2437        // Per-process entropy means no future writer will ever generate a corpse's
2438        // name again, so nothing would otherwise reclaim it: unique names turn an
2439        // occasional collision into permanent litter, one whole snapshot image at a
2440        // time. The reaper closes that, and the age threshold is what makes it safe
2441        // without a liveness check — pid-based liveness is unportable and wrong under
2442        // pid reuse.
2443        //
2444        // The threshold is a parameter so both sides are testable without setting
2445        // mtimes, which the standard library cannot do and which is not worth a
2446        // dependency.
2447        let dir = tempfile::tempdir().expect("tempdir");
2448        let path = dir.path().join("snap.fdu");
2449        let prefix = temp_prefix(&path).expect("a named target has a prefix");
2450
2451        let mut corpse = prefix.clone();
2452        corpse.push("111.0123456789abcdef.7");
2453        let unrelated = OsString::from("notes.txt");
2454        fs::write(dir.path().join(&corpse), b"a writer killed long ago").expect("plant");
2455        fs::write(dir.path().join(&unrelated), b"not ours").expect("plant");
2456
2457        // The shipped threshold spares anything that could still be in flight.
2458        reap_stale_temporaries(dir.path(), &path, STALE_TEMP_AGE);
2459        assert!(dir.path().join(&corpse).exists(), "a fresh corpse must be left alone");
2460
2461        // Past the threshold it is collected, and only it.
2462        reap_stale_temporaries(dir.path(), &path, std::time::Duration::ZERO);
2463        assert!(!dir.path().join(&corpse).exists(), "an old corpse must be collected");
2464        assert!(dir.path().join(&unrelated).exists(), "unrelated files are never touched");
2465    }
2466
2467    #[test]
2468    fn the_reaper_only_matches_its_own_targets_temporaries() {
2469        // The prefix carries the target's file name, so two snapshots sharing a
2470        // directory cannot collect each other's work in progress.
2471        let mine = temp_prefix(Path::new("/cache/snap.fdu")).expect("prefix");
2472        let theirs = temp_prefix(Path::new("/cache/other.fdu")).expect("prefix");
2473        assert_eq!(mine, OsString::from(".snap.fdu.tmp."));
2474        assert_ne!(mine, theirs);
2475        assert!(temp_prefix(Path::new("/")).is_none(), "a rootless path has no name");
2476    }
2477
2478    #[test]
2479    fn a_stale_temporary_does_not_block_a_later_write() {
2480        // A killed writer leaves its temporary behind and nothing reaps it until it is
2481        // a day old, so a later write can find a corpse sitting exactly where it wants
2482        // to go. `O_EXCL` plus the retry loop is what makes that safe: the collision is
2483        // detected and stepped over, never shared.
2484        //
2485        // The corpse is planted at names this process will really try, via `temp_name`.
2486        // An earlier version of this test guessed the shape and planted
2487        // `.snap.fdu.tmp.{pid}.0`, which the entropy in the name means the writer can
2488        // never generate — so it collided with nothing and proved nothing, while
2489        // reading as though it had.
2490        let dir = tempfile::tempdir().expect("tempdir");
2491        let path = dir.path().join("snap.fdu");
2492
2493        // Block a window of upcoming sequence values. A window rather than one name
2494        // because the counter is process-global and other tests draw from it too;
2495        // whichever value this write lands on, it starts inside the blocked range.
2496        let first = NEXT_TEMP_FILE.load(Ordering::Relaxed);
2497        let planted: Vec<OsString> =
2498            (first..first + 32).map(|sequence| temp_name(&path, sequence)).collect();
2499        for name in &planted {
2500            fs::write(dir.path().join(name), b"corpse").expect("plant a stale temporary");
2501        }
2502
2503        write_atomically(&path, b"payload").expect("write past the stale temporaries");
2504        assert_eq!(fs::read(&path).expect("read back"), b"payload");
2505
2506        // Every corpse survives: the writer stepped over them rather than reusing or
2507        // removing one, and they are far too young for the reaper.
2508        let mut survivors: Vec<_> = fs::read_dir(dir.path())
2509            .expect("read_dir")
2510            .filter_map(std::result::Result::ok)
2511            .map(|entry| entry.file_name())
2512            .filter(|name| name != "snap.fdu")
2513            .collect();
2514        survivors.sort();
2515        let mut expected = planted;
2516        expected.sort();
2517        assert_eq!(survivors, expected, "the write must step over corpses, not consume them");
2518    }
2519
2520    #[test]
2521    fn an_identical_payload_leaves_the_file_in_place() {
2522        // The default command encodes an unchanged tree to the bytes already on disk.
2523        // Replacing them was a full write, an fsync and a rename for nothing; now the
2524        // file is left alone, which on a filesystem is observable as the same inode.
2525        let dir = tempfile::tempdir().expect("tempdir");
2526        let path = dir.path().join("snap.fdu");
2527        let payload: Vec<u8> =
2528            (0u32..(3 << 20)).map(|i| u8::try_from(i % 251).unwrap_or(0)).collect();
2529
2530        write_atomically(&path, &payload).expect("first write");
2531        let first = fs::metadata(&path).expect("metadata");
2532        let before = std::time::SystemTime::now();
2533        write_atomically(&path, &payload).expect("identical write");
2534        let second = fs::metadata(&path).expect("metadata");
2535
2536        assert_eq!(fs::read(&path).expect("read back"), payload);
2537        assert_same_file(&first, &second);
2538        // The "as of" moved even though the bytes did not: the loader reads the mtime
2539        // as every cached entry's observation time, and the tree was just verified.
2540        assert!(second.modified().expect("mtime") >= before);
2541        // No temporary was created and abandoned on the way to deciding not to write.
2542        let extras = fs::read_dir(dir.path())
2543            .expect("read_dir")
2544            .filter_map(std::result::Result::ok)
2545            .filter(|entry| entry.file_name() != "snap.fdu")
2546            .count();
2547        assert_eq!(extras, 0);
2548    }
2549
2550    #[test]
2551    fn a_different_payload_of_the_same_length_is_written() {
2552        // Length is the cheap pre-check, not the decision: two snapshots of the same
2553        // tree with one mtime changed are the same length and must not be confused.
2554        let dir = tempfile::tempdir().expect("tempdir");
2555        let path = dir.path().join("snap.fdu");
2556        write_atomically(&path, b"payload-a").expect("first write");
2557        write_atomically(&path, b"payload-b").expect("second write");
2558        assert_eq!(fs::read(&path).expect("read back"), b"payload-b");
2559
2560        // And a file that holds the payload as a prefix is not the payload.
2561        fs::write(&path, b"payload-b-and-more").expect("lengthen");
2562        assert!(!same_bytes_on_disk(&path, b"payload-b"));
2563        assert!(!same_bytes_on_disk(&path, b"payload-b-and-mor"));
2564        assert!(same_bytes_on_disk(&path, b"payload-b-and-more"));
2565        assert!(!same_bytes_on_disk(&dir.path().join("absent"), b""));
2566    }
2567
2568    #[cfg(unix)]
2569    fn assert_same_file(first: &fs::Metadata, second: &fs::Metadata) {
2570        use std::os::unix::fs::MetadataExt;
2571        assert_eq!(first.ino(), second.ino(), "the snapshot was replaced, not left in place");
2572    }
2573
2574    #[cfg(not(unix))]
2575    fn assert_same_file(first: &fs::Metadata, second: &fs::Metadata) {
2576        // Creation time survives a skipped write and changes across a rename of a fresh
2577        // temporary, on the platforms that report it.
2578        if let (Ok(a), Ok(b)) = (first.created(), second.created()) {
2579            assert_eq!(a, b, "the snapshot was replaced, not left in place");
2580        }
2581    }
2582
2583    #[test]
2584    fn a_bare_filename_target_resolves_to_an_openable_directory() {
2585        // `Path::parent` returns `Some("")` for a bare name, not `None`, so an
2586        // `unwrap_or(".")` fallback never fires. The write still worked — `rename`
2587        // accepts the empty path — but `read_dir("")` fails, so the reaper returned
2588        // immediately and collected nothing for those targets.
2589        assert_eq!(parent_dir(Path::new("snap.fdu")), Path::new("."));
2590        assert!(fs::read_dir(parent_dir(Path::new("snap.fdu"))).is_ok(), "must be openable");
2591        assert_eq!(parent_dir(Path::new("/cache/snap.fdu")), Path::new("/cache"));
2592        assert_eq!(parent_dir(Path::new("cache/snap.fdu")), Path::new("cache"));
2593    }
2594
2595    #[test]
2596    fn a_temporary_name_is_unique_per_sequence_and_carries_process_entropy() {
2597        // The two components that make a corpse unreachable by a future process, and a
2598        // writer unable to collide with itself.
2599        let path = Path::new("/cache/snap.fdu");
2600        assert_ne!(temp_name(path, 0), temp_name(path, 1), "the counter separates files");
2601        let name = temp_name(path, 0).to_string_lossy().into_owned();
2602        assert!(name.starts_with(".snap.fdu.tmp."), "reaper prefix must match: {name}");
2603        assert!(
2604            name.contains(&format!("{:016x}", *TEMP_FILE_ENTROPY)),
2605            "entropy must be in the name, or a recycled pid regenerates a corpse: {name}"
2606        );
2607    }
2608
2609    #[test]
2610    fn concurrent_atomic_writes_do_not_share_a_temporary_file() {
2611        use std::sync::{Arc, Barrier};
2612
2613        const WRITERS: usize = 8;
2614        const PAYLOAD_BYTES: usize = 1024 * 1024;
2615
2616        let dir = tempfile::tempdir().expect("tempdir");
2617        let path = dir.path().join("snap.fdu");
2618        let barrier = Arc::new(Barrier::new(WRITERS));
2619        let results = std::thread::scope(|scope| {
2620            let handles: Vec<_> = (0..WRITERS)
2621                .map(|writer| {
2622                    let barrier = Arc::clone(&barrier);
2623                    let path = path.clone();
2624                    scope.spawn(move || {
2625                        let byte = u8::try_from(writer + 1).expect("writer id fits");
2626                        let bytes = vec![byte; PAYLOAD_BYTES];
2627                        barrier.wait();
2628                        write_atomically(&path, &bytes)
2629                    })
2630                })
2631                .collect();
2632            handles.into_iter().map(std::thread::ScopedJoinHandle::join).collect::<Vec<_>>()
2633        });
2634
2635        for result in results {
2636            result.expect("writer thread did not panic").expect("concurrent atomic write");
2637        }
2638        let final_bytes = fs::read(&path).expect("read final image");
2639        assert_eq!(final_bytes.len(), PAYLOAD_BYTES);
2640        assert!(final_bytes.iter().all(|byte| *byte == final_bytes[0]));
2641
2642        let leftovers: Vec<_> = fs::read_dir(dir.path())
2643            .expect("read_dir")
2644            .filter_map(std::result::Result::ok)
2645            .map(|entry| entry.file_name())
2646            .filter(|name| name != "snap.fdu")
2647            .collect();
2648        assert!(leftovers.is_empty(), "temp files left behind: {leftovers:?}");
2649    }
2650
2651    #[test]
2652    fn concurrent_snapshot_reader_sees_only_a_complete_old_or_new_image() {
2653        use std::sync::{Arc, Barrier};
2654
2655        let dir = tempfile::tempdir().expect("tempdir");
2656        let path = dir.path().join("snap.fdu");
2657        let mut old_index = Index::new("/some/root");
2658        old_index.apply_ok(&Observation::new(vec![Op::Upsert {
2659            path: PathBuf::from("old.txt"),
2660            kind: EntryKind::File,
2661            attrs: attrs(11, 1),
2662        }]));
2663        let mut new_index = Index::new("/some/root");
2664        new_index.apply_ok(&Observation::new(vec![
2665            Op::Upsert {
2666                path: PathBuf::from("new-a.txt"),
2667                kind: EntryKind::File,
2668                attrs: attrs(20, 2),
2669            },
2670            Op::Upsert {
2671                path: PathBuf::from("new-b.txt"),
2672                kind: EntryKind::File,
2673                attrs: attrs(30, 3),
2674            },
2675        ]));
2676        save(&old_index, &path).expect("save old image");
2677
2678        let start: Arc<Barrier> = Arc::new(Barrier::new(2));
2679        let (done_tx, done_rx) = std::sync::mpsc::sync_channel(1);
2680        std::thread::scope(|scope| {
2681            let writer_start: Arc<Barrier> = Arc::clone(&start);
2682            let writer_path: PathBuf = path.clone();
2683            scope.spawn(move || {
2684                writer_start.wait();
2685                done_tx.send(save(&new_index, &writer_path)).expect("report snapshot write");
2686            });
2687
2688            let reader_start: Arc<Barrier> = Arc::clone(&start);
2689            let reader_path: PathBuf = path.clone();
2690            scope.spawn(move || {
2691                reader_start.wait();
2692                let deadline: std::time::Instant =
2693                    std::time::Instant::now() + std::time::Duration::from_secs(10);
2694                loop {
2695                    assert!(
2696                        std::time::Instant::now() < deadline,
2697                        "snapshot replacement did not finish before the deadline"
2698                    );
2699                    let image: Index =
2700                        load(&reader_path).expect("load during replacement").expect("image");
2701                    match image.total().files {
2702                        1 => {
2703                            assert_eq!(image.total().bytes, 11);
2704                            assert!(image.lookup(Path::new("old.txt")).is_some());
2705                            assert!(image.lookup(Path::new("new-a.txt")).is_none());
2706                        }
2707                        2 => {
2708                            assert_eq!(image.total().bytes, 50);
2709                            assert!(image.lookup(Path::new("old.txt")).is_none());
2710                            assert!(image.lookup(Path::new("new-a.txt")).is_some());
2711                            assert!(image.lookup(Path::new("new-b.txt")).is_some());
2712                        }
2713                        partial => panic!("reader observed partial image with {partial} files"),
2714                    }
2715
2716                    match done_rx.try_recv() {
2717                        Ok(result) => {
2718                            result.expect("replace snapshot");
2719                            break;
2720                        }
2721                        Err(std::sync::mpsc::TryRecvError::Empty) => {}
2722                        Err(std::sync::mpsc::TryRecvError::Disconnected) => {
2723                            panic!("snapshot writer disconnected")
2724                        }
2725                    }
2726                }
2727            });
2728        });
2729
2730        let final_image: Index = load(&path).expect("load final").expect("final image");
2731        assert_eq!(final_image.total().files, 2);
2732        assert_eq!(final_image.total().bytes, 50);
2733    }
2734
2735    #[cfg(unix)]
2736    #[test]
2737    fn saved_snapshot_is_owner_readable_only() {
2738        use std::os::unix::fs::PermissionsExt;
2739
2740        let dir = tempfile::tempdir().expect("tempdir");
2741        let path = dir.path().join("snap.fdu");
2742        save(&sample_index(), &path).expect("save");
2743
2744        let mode = fs::metadata(&path).expect("metadata").permissions().mode() & 0o777;
2745        assert_eq!(mode, 0o600);
2746    }
2747
2748    #[test]
2749    fn empty_index_round_trips() {
2750        let dir = tempfile::tempdir().expect("tempdir");
2751        let path = dir.path().join("snap.fdu");
2752        let empty = Index::new("/some/root");
2753
2754        save(&empty, &path).expect("save");
2755        let restored = load(&path).expect("load").expect("present");
2756        assert!(restored.is_empty());
2757        assert_eq!(restored.total(), empty.total());
2758    }
2759
2760    #[test]
2761    fn partial_index_is_never_persisted() {
2762        let dir = tempfile::tempdir().expect("tempdir");
2763        let path = dir.path().join("partial.fdu");
2764        save(&Index::new("/root"), &path).expect("complete baseline");
2765        let complete_bytes = fs::read(&path).expect("read complete snapshot");
2766
2767        let mut index = Index::new("/root");
2768        index.set_initial_freshness(false);
2769
2770        assert!(matches!(save(&index, &path), Err(Error::Snapshot(_))));
2771        assert_eq!(fs::read(&path).expect("old snapshot remains"), complete_bytes);
2772    }
2773
2774    #[test]
2775    fn semantic_scan_scope_round_trips() {
2776        let dir = tempfile::tempdir().expect("tempdir");
2777        let path = dir.path().join("snap.fdu");
2778        let entries = crate::EntryTierIdentity {
2779            engine: engine_fingerprint(),
2780            scope: crate::EntryScope {
2781                max_depth: Some(7),
2782                follow_symlinks: false,
2783                one_filesystem: true,
2784                hidden_fingerprint: 5,
2785                exclude_special: true,
2786                population: crate::query::IgnoredEntries::Include,
2787                control_fingerprint: 0,
2788            },
2789            type_rules_fingerprint: crate::classify::type_rule_fingerprint(),
2790            reducers_fingerprint: 33,
2791        };
2792        // The default control limits, which the table of an index built with
2793        // `new_with_scope` enforces; any other limits are refused at save.
2794        for controls in [
2795            ControlTierIdentity::Observed { limits: crate::control::ControlLimits::default() },
2796            ControlTierIdentity::NotObserved,
2797        ] {
2798            let identity = SnapshotIdentity { entries, controls };
2799            let index = Index::new_with_scope("/some/root", identity.scan_scope());
2800
2801            save(&index, &path).expect("save");
2802            let restored = load(&path).expect("load").expect("present");
2803            assert_eq!(restored.snapshot_identity(), identity);
2804            assert_eq!(restored.scope(), identity.scan_scope());
2805            let info = read_header(&path).expect("read header").expect("current");
2806            assert_eq!(info.identity, identity);
2807            assert_eq!(info.scope(), identity.scan_scope());
2808        }
2809    }
2810
2811    /// Reading control files decides which entries are ignored, never which entries exist,
2812    /// so the snapshots of one tree with observation on and off record one entry tier, byte
2813    /// for byte, and differ only in the control tier.
2814    #[test]
2815    fn controls_on_and_off_snapshots_share_the_entry_identity() {
2816        let tree = tempfile::tempdir().expect("tree");
2817        fs::write(tree.path().join(".gitignore"), b"*.log\n").expect("write");
2818        fs::write(tree.path().join("debug.log"), b"log").expect("write");
2819        fs::create_dir(tree.path().join("src")).expect("mkdir");
2820        fs::write(tree.path().join("src/main.rs"), b"fn main() {}\n").expect("write");
2821        let dir = tempfile::tempdir().expect("tempdir");
2822        let observed = crate::ScanConfig::default();
2823        let blind = crate::ScanConfig { read_controls: false, ..observed.clone() };
2824
2825        let saved = [(&observed, "on.fdu"), (&blind, "off.fdu")].map(|(config, name)| {
2826            let (index, _) = crate::scan::scan_into_index(tree.path(), config).expect("scan");
2827            let path = dir.path().join(name);
2828            save(&index, &path).expect("save");
2829            let restored = load(&path).expect("load").expect("present");
2830            assert_eq!(restored.snapshot_identity(), config.snapshot_identity());
2831            (restored.snapshot_identity(), fs::read(&path).expect("read"))
2832        });
2833        let [(on, on_bytes), (off, off_bytes)] = saved;
2834
2835        assert_eq!(on.entries, off.entries);
2836        assert_ne!(on.controls, off.controls);
2837        let entry_tier = IDENTITY_OFFSET..IDENTITY_OFFSET + crate::stored_state::ENTRY_TIER_BYTES;
2838        let control_tier = entry_tier.end..ROOT_OFFSET;
2839        assert_eq!(on_bytes[entry_tier.clone()], off_bytes[entry_tier]);
2840        assert_ne!(on_bytes[control_tier.clone()], off_bytes[control_tier]);
2841
2842        let projected = load_serving(
2843            &dir.path().join("on.fdu"),
2844            blind.types_shared(),
2845            blind.snapshot_identity(),
2846        )
2847        .expect("load projection");
2848        let LoadOutcome::Served { index: projected, stored } = projected else {
2849            panic!("an observed snapshot should project to the blind request");
2850        };
2851        assert_eq!(serves_snapshot(stored, blind.snapshot_identity()), Serves::ProjectControlsOff);
2852        let (cold, _) = crate::scan::scan_into_index(tree.path(), &blind).expect("blind scan");
2853        assert_eq!(projected.snapshot_identity(), blind.snapshot_identity());
2854        assert_eq!(projected.scope(), blind.scope());
2855        assert_eq!(projected.len(), cold.len());
2856        assert_eq!(projected.total(), cold.total());
2857        assert!(matches!(projected.controls(), Err(Error::ControlStateNotObserved)));
2858        assert_eq!(
2859            projected.path_state(Path::new("debug.log")),
2860            cold.path_state(Path::new("debug.log"))
2861        );
2862
2863        let mut corrupt = on_bytes;
2864        corrupt[WRITING_PASS_STARTED_AT_OFFSET] ^= 1;
2865        fs::write(dir.path().join("on.fdu"), corrupt).expect("corrupt checksum");
2866        assert!(matches!(
2867            load_serving(
2868                &dir.path().join("on.fdu"),
2869                blind.types_shared(),
2870                blind.snapshot_identity(),
2871            )
2872            .expect("corrupt projection is absent"),
2873            LoadOutcome::Absent
2874        ));
2875    }
2876
2877    /// A format-4 snapshot, laid out as that format wrote it, is fdu's and stale: the
2878    /// prologue kept its offsets, so the cache lifecycle names its version rather than
2879    /// calling it foreign, and loading it is a miss.
2880    #[test]
2881    fn a_v4_snapshot_is_older_format() {
2882        const V4: u32 = 4;
2883        let dir = tempfile::tempdir().expect("tempdir");
2884        let path = dir.path().join("v4.fdu");
2885
2886        let mut image = MAGIC.to_vec();
2887        image.extend_from_slice(&V4.to_le_bytes());
2888        // The v4 engine fingerprint. Recognition reads the version before the engine, so a
2889        // format-4 image is older whatever these eight bytes hold.
2890        image.extend_from_slice(&0x4444_4444_4444_4444_u64.to_le_bytes());
2891        image.push(path_encoding());
2892        // The v4 scope: depth, flags, and the hidden, ignore-rules, type-rules, and reducer
2893        // fingerprints.
2894        image.extend_from_slice(&u64::MAX.to_le_bytes());
2895        image.push(0);
2896        let scope = crate::ScanConfig::default().scope();
2897        for fingerprint in [
2898            scope.hidden_fingerprint,
2899            scope.ignore_rules_fingerprint,
2900            scope.type_rules_fingerprint,
2901            scope.reducers_fingerprint,
2902        ] {
2903            image.extend_from_slice(&fingerprint.to_le_bytes());
2904        }
2905        put_os_str(&mut image, OsStr::new("/some/root")).expect("root");
2906        image.extend_from_slice(&1_u64.to_le_bytes());
2907        image.extend_from_slice(&NO_PARENT.to_le_bytes());
2908        image.push(EntryKind::Dir as u8);
2909        put_os_str(&mut image, OsStr::new("")).expect("root name");
2910        image.extend_from_slice(&[0; 6 * 8]);
2911        // The v4 control section: no sources, both default limits, no refusals.
2912        image.extend_from_slice(&0_u32.to_le_bytes());
2913        for limit in [crate::DEFAULT_CONTROL_BUDGET, crate::DEFAULT_CONTROL_LINE_LIMIT] {
2914            image.push(1);
2915            image.extend_from_slice(&u64::try_from(limit).expect("limit").to_le_bytes());
2916        }
2917        image.extend_from_slice(&0_u32.to_le_bytes());
2918        let checksum = crc32c(&image);
2919        image.extend_from_slice(&checksum.to_le_bytes());
2920        image.extend_from_slice(TRAILER);
2921        fs::write(&path, &image).expect("write v4 image");
2922
2923        assert!(matches!(
2924            identify(&path).expect("identify"),
2925            Some(Identity::Stale(crate::cache::StaleReason::OlderFormat { version: V4 }))
2926        ));
2927        assert!(read_header(&path).expect("read header").is_none());
2928        assert!(load(&path).expect("load").is_none());
2929        assert_eq!(
2930            crate::cache::cache_status(&path).expect("status").state,
2931            crate::cache::CacheState::Stale(crate::cache::StaleReason::OlderFormat { version: V4 })
2932        );
2933    }
2934
2935    /// A later pass over an unchanged tree encodes the same facts under a newer pass start.
2936    /// The image already on disk is kept, its stamp included, rather than rewritten for the
2937    /// stamp alone, and its mtime moves to the later pass's start, never backwards; a
2938    /// changed tree is written with the new pass's stamp.
2939    #[test]
2940    fn a_later_pass_over_the_same_facts_keeps_the_snapshot_in_place() {
2941        use std::time::{Duration, UNIX_EPOCH};
2942
2943        let dir = tempfile::tempdir().expect("tempdir");
2944        let path = dir.path().join("snap.fdu");
2945        let stamp_at = |bytes: &[u8]| {
2946            i64::from_le_bytes(
2947                bytes[WRITING_PASS_STARTED_AT_OFFSET..IDENTITY_OFFSET].try_into().expect("stamp"),
2948            )
2949        };
2950        let mtime = || fs::metadata(&path).expect("metadata").modified().expect("mtime");
2951        let nanos = |time: std::time::SystemTime| {
2952            i64::try_from(time.duration_since(UNIX_EPOCH).expect("after the epoch").as_nanos())
2953                .expect("nanoseconds")
2954        };
2955
2956        let mut first = sample_index();
2957        first.set_writing_pass_started_at_ns(1_000);
2958        save(&first, &path).expect("first save");
2959        let written = fs::metadata(&path).expect("metadata");
2960        assert_eq!(stamp_at(&fs::read(&path).expect("read")), 1_000);
2961        assert_eq!(
2962            load(&path).expect("load").expect("present").writing_pass_started_at_ns(),
2963            1_000
2964        );
2965
2966        // A pass that started after the image was written, on a whole second so every
2967        // filesystem's mtime holds the instant exactly.
2968        let later_started = UNIX_EPOCH
2969            + Duration::from_secs(
2970                mtime().duration_since(UNIX_EPOCH).expect("after the epoch").as_secs() + 1,
2971            );
2972        let mut later = sample_index();
2973        later.set_writing_pass_started_at_ns(nanos(later_started));
2974        save(&later, &path).expect("unchanged save");
2975        let kept = fs::metadata(&path).expect("metadata");
2976        assert_same_file(&written, &kept);
2977        assert_eq!(kept.modified().expect("mtime"), later_started, "the kept image's as-of");
2978        assert_eq!(stamp_at(&fs::read(&path).expect("read")), 1_000);
2979        assert!(load(&path).expect("load").is_some(), "the kept image is still valid");
2980
2981        // An equivalent save under an older stamp, as of an index loaded from an earlier
2982        // image, keeps the later mtime, which was already true of these facts.
2983        save(&first, &path).expect("older unchanged save");
2984        assert_same_file(&written, &fs::metadata(&path).expect("metadata"));
2985        assert_eq!(mtime(), later_started);
2986
2987        // A kept image must still verify: one whose own checksum is wrong reads as absent,
2988        // so it is replaced rather than kept under an older stamp forever.
2989        let mut corrupt = fs::read(&path).expect("read");
2990        let checksum_at = corrupt.len() - TRAILER.len() - CHECKSUM_BYTES;
2991        corrupt[checksum_at] ^= 0xff;
2992        fs::write(&path, &corrupt).expect("corrupt the checksum");
2993        assert!(load(&path).expect("load").is_none());
2994        save(&later, &path).expect("save over a corrupt image");
2995        assert_eq!(stamp_at(&fs::read(&path).expect("read")), nanos(later_started));
2996        assert!(load(&path).expect("load").is_some());
2997
2998        let mut changed = sample_index();
2999        changed.set_writing_pass_started_at_ns(3_000);
3000        changed.apply_ok(&Observation::new(vec![Op::Upsert {
3001            path: PathBuf::from("notes.md"),
3002            kind: EntryKind::File,
3003            attrs: attrs(8, 30),
3004        }]));
3005        save(&changed, &path).expect("changed save");
3006        let rewritten = fs::read(&path).expect("read");
3007        assert_eq!(stamp_at(&rewritten), 3_000);
3008        let restored = load(&path).expect("load").expect("present");
3009        assert_eq!(restored.writing_pass_started_at_ns(), 3_000);
3010        assert_eq!(restored.total(), changed.total());
3011    }
3012
3013    #[cfg(unix)]
3014    #[test]
3015    fn non_utf8_names_round_trip_without_aliasing() {
3016        use std::os::unix::ffi::OsStringExt;
3017
3018        let first = PathBuf::from(OsString::from_vec(vec![b'n', 0x80]));
3019        let second = PathBuf::from(OsString::from_vec(vec![b'n', 0x81]));
3020        let mut root = PathBuf::from("/some");
3021        root.push(OsString::from_vec(vec![b'r', 0x82]));
3022        let mut index = Index::new(&root);
3023        index.apply_baseline_ok(&Observation::new(vec![
3024            Op::Upsert { path: first.clone(), kind: EntryKind::File, attrs: attrs(10, 1) },
3025            Op::Upsert { path: second.clone(), kind: EntryKind::File, attrs: attrs(20, 2) },
3026        ]));
3027
3028        let dir = tempfile::tempdir().expect("tempdir");
3029        let path = dir.path().join("snap.fdu");
3030        save(&index, &path).expect("save");
3031        let restored = load(&path).expect("load").expect("present");
3032
3033        assert_eq!(restored.root_path(), root);
3034        assert_eq!(restored.total().files, 2);
3035        assert_eq!(restored.total().bytes, 30);
3036        assert!(restored.lookup(&first).is_some());
3037        assert!(restored.lookup(&second).is_some());
3038        assert!(!restored.serving_indexes_enabled());
3039    }
3040}