Skip to main content

fdu_core/
snapshot.rs

1//! Persisting an index to disk and reading it back.
2//!
3//! # Status: format v5 is a bounded bootstrap format
4//!
5//! This module implements a flat, uncompressed writer and a bounded streaming reader.
6//! It exists so the cache *lifecycle* — semantic-scope invalidation, atomic replacement,
7//! complete-only persistence, resource limits, and corrupt-equals-empty — is nailed down
8//! before the optimized wire format is designed. The loader checks file size, trailer,
9//! and payload checksum before parsing, then checks the header, record count, and path
10//! lengths before allocating record data and rebuilds one entry at a time.
11//!
12//! The target format is different in every other respect: zstd-compressed blocks with an
13//! index block at the tail, so opening costs one small read and directory listings
14//! decompress on demand into a small LRU cache; item references encoded as
15//! `(block << k) | offset` and delta-encoded when they point inside the same block;
16//! sibling groups written contiguously so one directory listing costs one block
17//! decompression; front-coded names; and pre-computed roll-ups stored per directory so a
18//! query never re-aggregates. O(1) open with lazy materialization matters more than raw
19//! decode throughput, because it matches how a UI actually navigates: open now, expand
20//! later.
21//!
22//! Two things here are **not** placeholders and must survive the format change:
23//!
24//! - **A corrupt or unrecognized snapshot is treated as absent, never as an error and
25//!   never as data.** A cache that fails closed costs a rescan; a cache that fails open
26//!   silently lies.
27//! - **Replacement is atomic.** Write a temporary file, then rename over the target, so a
28//!   crash mid-write leaves the previous snapshot intact rather than a half-written one.
29
30use std::ffi::{OsStr, OsString};
31use std::fs::{self, OpenOptions};
32use std::io::{self, BufReader, Read, Seek, SeekFrom, Write};
33use std::path::{Component, Path, PathBuf};
34use std::sync::atomic::{AtomicU64, Ordering};
35
36use crate::engine_contract::{Attrs, EntryKind, Error, Observation, Op, Result, Source};
37use crate::index::{EntryId, Index, IndexHandle};
38use crate::stored_state::{
39    ControlTierIdentity, SNAPSHOT_IDENTITY_BYTES, Serves, SnapshotIdentity, serves_snapshot,
40};
41
42/// Result of loading a snapshot for one requested stored-state identity.
43#[derive(Debug)]
44#[allow(clippy::large_enum_variant)] // One transient load result; avoid boxing the returned index.
45pub enum LoadOutcome {
46    /// The snapshot supplied an index, either exactly or through a lawful projection.
47    Served {
48        /// The index in the requested scope.
49        index: Index,
50        /// The identity the snapshot declares, which is what a plan admits: a projected
51        /// index reports the requested identity, not this one.
52        stored: SnapshotIdentity,
53    },
54    /// The snapshot was valid, but its identity cannot answer this request.
55    Refused {
56        /// The validated identity that did not serve the request.
57        identity: SnapshotIdentity,
58        /// The validated snapshot root, used before explaining a scope mismatch.
59        root: PathBuf,
60    },
61    /// No valid snapshot was present.
62    Absent,
63}
64
65/// Leading magic. Distinguishes an fdu snapshot from any other file that lands here.
66const MAGIC: &[u8; 8] = b"FDUSNAP\x00";
67
68/// Trailing magic. Present only once the whole snapshot has been written, so a truncated
69/// file is detected on load rather than parsed into a plausible-looking partial tree.
70const TRAILER: &[u8; 8] = b"FDUEND\x00\x00";
71
72/// CRC-32C of every byte before the footer. This detects plausible payload corruption
73/// that remains structurally valid; it is an integrity check, not authentication.
74const CHECKSUM_BYTES: usize = std::mem::size_of::<u32>();
75
76/// Reversed Castagnoli polynomial used by CRC-32C.
77const CRC32C_POLYNOMIAL: u32 = 0x82f6_3b78;
78
79/// One table lookup per byte keeps integrity validation from dominating snapshot load.
80const CRC32C_TABLES: [[u32; 256]; 8] = make_crc32c_tables();
81
82/// On-disk format version. Bump on any layout change; old snapshots are then discarded
83/// rather than misread.
84///
85/// 4: the control section also carries both control limits, the budget and the line
86/// limit, and every refused control file, so a snapshot of an index that refused some
87/// reloads with the same coverage rather than claiming every rule applied.
88///
89/// 5: the header records the identity of each tier the snapshot holds, in the
90/// stored-state model's fixed-width encodings: the entry tier's scope, type rules, and
91/// reducer set, and the control tier's observation and limits. Observation and limits are
92/// no longer mixed into the scope as one hash, and the limits moved out of the control
93/// section, so a section is re-admitted under the header's limits and one they would not
94/// admit is refused. The header also records `writing_pass_started_at_ns`, the start of the
95/// pass that last wrote the image.
96const FORMAT_VERSION: u32 = 5;
97
98/// Version of the rules that decide which bucket an entry's bytes are tallied under.
99///
100/// Separate from [`FORMAT_VERSION`] because the two answer different questions: the
101/// layout can be unchanged while the meaning of what was written moves. A snapshot stores
102/// the bucket an entry was assigned, not the name it was assigned from, so a rule change
103/// cannot be re-derived on load — the entries have to be discarded and re-walked.
104///
105/// 1: initial rules. 2: files with no extension are tallied under `(none)` instead of
106/// being left out of the extension roll-up entirely.
107const CLASSIFICATION_VERSION: u32 = 2;
108
109/// Version of the per-entry facts used to decide whether retained state is still valid.
110///
111/// Version 2 adds Windows change time, volume identity, and file index. A pre-v2 Windows
112/// snapshot stores zeroes in those fields and cannot safely serve cache-only as if those
113/// facts had been observed. This is mixed into the engine fingerprint on every platform
114/// so one build has one store identity.
115const VALIDITY_VERSION: u32 = 2;
116
117/// Snapshot path encoding used by Unix targets.
118#[cfg(unix)]
119const PATH_ENCODING_UNIX_BYTES: u8 = 1;
120
121/// Snapshot path encoding used by Windows targets.
122#[cfg(windows)]
123const PATH_ENCODING_WINDOWS_WIDE: u8 = 2;
124
125/// Portable UTF-8 fallback for targets outside Unix and Windows.
126#[cfg(not(any(unix, windows)))]
127const PATH_ENCODING_UTF8: u8 = 3;
128
129/// Marks the root's absent parent.
130const NO_PARENT: u32 = u32::MAX;
131
132/// Byte offset of `writing_pass_started_at_ns`: after the magic, the format version, the
133/// engine fingerprint, and the path encoding.
134///
135/// Fixed, so a save can read the stamp of the image already at its path without parsing
136/// the rest of it.
137const WRITING_PASS_STARTED_AT_OFFSET: usize = MAGIC.len() + 4 + 8 + 1;
138
139/// Byte offset of the tier identities, which follow the stamp.
140const IDENTITY_OFFSET: usize = WRITING_PASS_STARTED_AT_OFFSET + 8;
141
142/// Byte offset of the root path, which follows the tier identities.
143#[cfg(test)]
144const ROOT_OFFSET: usize = IDENTITY_OFFSET + SNAPSHOT_IDENTITY_BYTES;
145
146/// Smallest possible on-disk record: parent slot, kind, name length, and six 8-byte
147/// attribute fields, with a zero-length name. Used to sanity-check a declared entry
148/// count against the bytes actually present.
149const MIN_RECORD_BYTES: usize = 4 + 1 + 4 + 8 * 6;
150
151/// Binary size unit used by the snapshot resource limits.
152const GIBIBYTE: u64 = 1024 * 1024 * 1024;
153
154/// Largest snapshot image accepted by the bootstrap streaming reader.
155const MAX_SNAPSHOT_BYTES: u64 = 64 * GIBIBYTE;
156
157/// Process-local discriminator for exclusive sibling temporary files.
158static NEXT_TEMP_FILE: AtomicU64 = AtomicU64::new(0);
159
160/// Per-process entropy mixed into temporary file names.
161///
162/// Correctness never depended on this: `O_EXCL` is what guarantees one creator wins,
163/// and the counter above is what keeps two threads of this process from even trying the
164/// same name. Randomness closes two gaps the counter cannot.
165///
166/// A killed writer leaves its temporary behind and nothing reaps it. With a
167/// pid-and-counter name, a later process that recycles that pid restarts its counter at
168/// zero and collides with the corpse on every run — correct, because it retries, but it
169/// accumulates litter and in the pathological case walks the whole retry budget. And
170/// the names are otherwise entirely predictable, which matters because
171/// [`MAX_TEMP_CREATE_ATTEMPTS`] exists specifically to survive a hostile directory: an
172/// attacker who can guess the names can pre-create the whole budget and deny every save.
173///
174/// Seeded from `RandomState`, whose keys the standard library takes from the operating
175/// system, so this needs no dependency and no clock.
176static TEMP_FILE_ENTROPY: std::sync::LazyLock<u64> = std::sync::LazyLock::new(|| {
177    use std::hash::{BuildHasher, Hasher};
178    std::collections::hash_map::RandomState::new().build_hasher().finish()
179});
180
181/// Stale files can survive a killed writer. Try enough unique sequence values to step
182/// over them without turning an attacker-controlled directory into an unbounded loop.
183const MAX_TEMP_CREATE_ATTEMPTS: usize = 1024;
184
185/// How old an abandoned temporary must be before a later writer removes it.
186///
187/// Per-process entropy in the name means a corpse can never collide with a future
188/// writer — which also means no future writer will ever reuse or overwrite it. Unique
189/// names turn an occasional collision into permanent litter, and each corpse is a whole
190/// snapshot image, so something has to collect them.
191///
192/// A day is far longer than any real save and far shorter than "never". The threshold
193/// is what makes this safe without any liveness check: a temporary this old cannot
194/// belong to a writer that is still running, and pid-based liveness tests are both
195/// unportable and wrong under pid reuse.
196///
197/// Shared with the cache lifecycle, which reclaims the same corpses on request rather
198/// than waiting for the next writer, and must answer "old enough to be nobody's" the same
199/// way this does.
200pub(crate) const STALE_TEMP_AGE: std::time::Duration = std::time::Duration::from_secs(24 * 60 * 60);
201
202/// Largest encoded root or entry name accepted from a snapshot.
203const MAX_PATH_BYTES: u32 = 1024 * 1024;
204
205/// Upper bound on records accepted even when a sparse file could physically hold more.
206const MAX_SNAPSHOT_ENTRIES: u64 = 100_000_000;
207
208/// Binary size unit for the control-section ceiling.
209const MEBIBYTE: usize = 1024 * 1024;
210
211/// Largest control table, in retained charge and in total source bytes, a snapshot may
212/// carry.
213///
214/// A parser guard over untrusted lengths, deliberately independent of both control limits:
215/// they are a caller's runtime settings and may be unbounded, while this bounds what a
216/// corrupt or hostile file can make the loader allocate. [`save`] refuses a table above
217/// it, so a snapshot the loader would reject is never written as if it were usable.
218const SNAPSHOT_CONTROL_TABLE_CEILING: usize = 256 * MEBIBYTE;
219
220/// Encoded refusal reasons.
221const REFUSED_FOR_BUDGET: u8 = 1;
222const REFUSED_FOR_LINE_LIMIT: u8 = 2;
223
224/// A fingerprint of everything that would change how the engine interprets a tree.
225///
226/// When this does not match, the whole snapshot is discarded. That is the cheap,
227/// wholesale answer to "the rules changed, so every derived verdict in the cache might
228/// be wrong" — far simpler than trying to work out which entries a rule change affected.
229///
230/// Currently derived from the crate version, the format version, and the version of the
231/// classification rules. When compiled type-recognition rules and reducer registrations
232/// arrive, their hashes belong here too.
233pub fn engine_fingerprint() -> u64 {
234    let mut hash = 0xcbf2_9ce4_8422_2325_u64;
235    let mut mix = |bytes: &[u8]| {
236        for byte in bytes {
237            hash ^= u64::from(*byte);
238            hash = hash.wrapping_mul(0x1000_0000_01b3);
239        }
240    };
241    mix(env!("CARGO_PKG_VERSION").as_bytes());
242    mix(&FORMAT_VERSION.to_le_bytes());
243    mix(&CLASSIFICATION_VERSION.to_le_bytes());
244    mix(&VALIDITY_VERSION.to_le_bytes());
245    hash
246}
247
248/// Write `index` to `path`, replacing any existing snapshot atomically.
249pub fn save(index: &Index, path: &Path) -> Result<()> {
250    if !crate::stored_state::entries_writable(index) {
251        return Err(Error::Snapshot(
252            "refusing to persist an index that is stale, reconciling, or incomplete".into(),
253        ));
254    }
255    // The loader refuses a table whose limits its scope does not claim, so writing one
256    // would publish a snapshot that every later open discards.
257    index.require_control_limits_in_scope(index.control_table().limits())?;
258    let mut buf: Vec<u8> = Vec::new();
259    buf.extend_from_slice(MAGIC);
260    buf.extend_from_slice(&FORMAT_VERSION.to_le_bytes());
261    buf.extend_from_slice(&engine_fingerprint().to_le_bytes());
262    buf.push(path_encoding());
263    debug_assert_eq!(buf.len(), WRITING_PASS_STARTED_AT_OFFSET);
264    buf.extend_from_slice(&index.writing_pass_started_at_ns().to_le_bytes());
265    // Every tier identity shares the prologue's engine fingerprint, which is this build's.
266    buf.extend_from_slice(&index.snapshot_identity().encode());
267
268    put_os_str(&mut buf, index.root_path().as_os_str())?;
269
270    // Pre-order, so a parent's record always precedes its children's and the loader can
271    // rebuild the tree in one forward pass with no fixups.
272    let mut records: Vec<(u32, EntryId)> = Vec::new();
273    let mut stack: Vec<(u32, EntryId)> = vec![(NO_PARENT, EntryId::ROOT)];
274    while let Some((parent_slot, id)) = stack.pop() {
275        let slot = u32::try_from(records.len())
276            .map_err(|_| Error::Snapshot("snapshot exceeds u32 entry capacity".into()))?;
277        records.push((parent_slot, id));
278        let children = index
279            .children_of(id)
280            .ok_or_else(|| Error::Snapshot("stale entry handle while saving".into()))?;
281        for (_, child) in children.rev() {
282            stack.push((slot, child));
283        }
284    }
285
286    let count = u64::try_from(records.len())
287        .map_err(|_| Error::Snapshot("snapshot entry count overflow".into()))?;
288    buf.extend_from_slice(&count.to_le_bytes());
289
290    for (parent_slot, id) in records {
291        buf.extend_from_slice(&parent_slot.to_le_bytes());
292        let kind = index
293            .kind_of(id)
294            .ok_or_else(|| Error::Snapshot("stale entry handle while saving".into()))?;
295        buf.push(kind as u8);
296        let name = index
297            .name_of(id)
298            .ok_or_else(|| Error::Snapshot("stale entry handle while saving".into()))?;
299        put_os_str(&mut buf, name)?;
300        let attrs = index
301            .attrs_of(id)
302            .ok_or_else(|| Error::Snapshot("stale entry handle while saving".into()))?;
303        buf.extend_from_slice(&attrs.size.to_le_bytes());
304        buf.extend_from_slice(&attrs.allocated.to_le_bytes());
305        buf.extend_from_slice(&attrs.mtime_ns.to_le_bytes());
306        buf.extend_from_slice(&attrs.ctime_ns.to_le_bytes());
307        buf.extend_from_slice(&attrs.inode.to_le_bytes());
308        buf.extend_from_slice(&attrs.dev.to_le_bytes());
309    }
310    put_controls(&mut buf, index.control_table())?;
311    publish(path, buf)
312}
313
314/// Seal a snapshot payload and write it to `path`, keeping an image already there that
315/// encodes the same facts.
316///
317/// Two passes over an unchanged tree encode identical images except for each pass's
318/// start, and rewriting for the stamp alone would give back the skip [`write_atomically`]
319/// exists for. A default run never reads its snapshot for a metadata query, so on that path
320/// the rewrite was the whole cost of having a cache: about 70 ms of a 375 ms run over 175k
321/// entries (exp-066), and a 14 MB write and `F_FULLFSYNC` on every run (exp-067). So an image at `path` that differs only in its stamp is kept, stamp
322/// included, and only its mtime moves. The kept stamp is the start of an earlier pass that
323/// wrote exactly these facts, which can understate how recently they were verified and
324/// never overstates it.
325fn publish(path: &Path, mut payload: Vec<u8>) -> Result<()> {
326    if keep_equivalent_image(path, &payload) {
327        return Ok(());
328    }
329    seal(&mut payload);
330    // Nothing at `path` is this image under any stamp, this pass's included, so comparing
331    // again before replacing it could only repeat the answer.
332    replace_atomically(path, &payload)
333}
334
335/// Keep the file at `path` when it is `payload` sealed under any pass's stamp, moving only
336/// its mtime, to the stamp in `payload`.
337///
338/// The mtime is cache-maintenance metadata, not filesystem-observation provenance: a copy
339/// or a touch must not change when the tree was observed. Moving it to the start of the pass
340/// that would have stamped the image lets the equivalent-image fast path retain its useful
341/// monotonic hint without changing the persisted observation stamp. It never moves backwards:
342/// a save of an index loaded under an older stamp leaves a later mtime in place.
343///
344/// One read through one handle, comparing every payload byte before computing the
345/// checksum, so a changed tree stops at its first difference and costs no checksum here,
346/// and an unchanged one costs the one checksum sealing it would have. The checksum is
347/// still verified before the file is kept: a file whose own checksum is wrong reads as
348/// absent, and keeping it would keep every later run cold.
349fn keep_equivalent_image(path: &Path, payload: &[u8]) -> bool {
350    const FOOTER_BYTES: usize = CHECKSUM_BYTES + TRAILER.len();
351    let Ok(mut file) = fs::File::open(path) else { return false };
352    let Ok(metadata) = file.metadata() else { return false };
353    let image_len = payload.len().checked_add(FOOTER_BYTES).and_then(|len| u64::try_from(len).ok());
354    if image_len != Some(metadata.len()) {
355        return false;
356    }
357    let (before_stamp, after_stamp) =
358        (&payload[..WRITING_PASS_STARTED_AT_OFFSET], &payload[IDENTITY_OFFSET..]);
359    let mut stamp = [0u8; 8];
360    if !reads_as(&mut file, before_stamp)
361        || file.read_exact(&mut stamp).is_err()
362        || !reads_as(&mut file, after_stamp)
363    {
364        return false;
365    }
366    let checksum =
367        ![before_stamp, &stamp[..], after_stamp].into_iter().fold(u32::MAX, crc32c_update);
368    let mut footer = [0u8; FOOTER_BYTES];
369    if file.read_exact(&mut footer).is_err()
370        || footer[..CHECKSUM_BYTES] != checksum.to_le_bytes()
371        || footer[CHECKSUM_BYTES..] != *TRAILER
372        // A file that grew under the comparison is not this image.
373        || !matches!(file.read(&mut [0u8; 1]), Ok(0))
374    {
375        return false;
376    }
377    let pass_started = payload
378        .get(WRITING_PASS_STARTED_AT_OFFSET..IDENTITY_OFFSET)
379        .and_then(|stamp| <[u8; 8]>::try_from(stamp).ok())
380        .map(i64::from_le_bytes)
381        .and_then(|nanos| u64::try_from(nanos).ok())
382        .and_then(|nanos| {
383            std::time::UNIX_EPOCH.checked_add(std::time::Duration::from_nanos(nanos))
384        });
385    if let Some(pass_started) = pass_started {
386        if !matches!(metadata.modified(), Ok(modified) if modified >= pass_started) {
387            // Best effort, as for any kept image: a stale mtime understates freshness.
388            let _ = touch(path, pass_started);
389        }
390    }
391    true
392}
393
394/// Append the checksum of every byte so far, then the trailer.
395fn seal(payload: &mut Vec<u8>) {
396    let checksum = crc32c(payload);
397    payload.extend_from_slice(&checksum.to_le_bytes());
398    payload.extend_from_slice(TRAILER);
399}
400
401/// Capture and persist a coherent shared-index image without holding its lock during
402/// serialization or filesystem I/O.
403pub fn save_handle(index: &IndexHandle, path: &Path) -> Result<()> {
404    let snapshot = index.snapshot()?;
405    save(&snapshot, path)
406}
407
408/// Load a snapshot, or return `None` when there is nothing usable at `path`.
409///
410/// Returns `None` — not an error — for a missing file, a foreign file, a version or
411/// engine-fingerprint mismatch, and a truncated or corrupt file. Every one of those
412/// means the same thing to a caller: there is no warm cache, so scan.
413pub fn load(path: &Path) -> Result<Option<Index>> {
414    load_with_size_limit(path, MAX_SNAPSHOT_BYTES)
415}
416
417/// Load a snapshot under the registry that will classify and analyze its entries.
418///
419/// A snapshot made under different rules is a clean cache miss. The snapshot carries
420/// only their derived identity, so the caller must supply the matching registry rather
421/// than allowing the loader to attach unrelated default rules.
422pub fn load_with_types(
423    path: &Path,
424    types: std::sync::Arc<crate::classify::TypeRegistry>,
425) -> Result<Option<Index>> {
426    load_with_types_and_size_limit(path, MAX_SNAPSHOT_BYTES, types, None).map(|outcome| {
427        match outcome {
428            LoadOutcome::Served { index, .. } => Some(index),
429            LoadOutcome::Refused { .. } | LoadOutcome::Absent => None,
430        }
431    })
432}
433
434/// Load a snapshot that can serve `wanted`, projecting observed control state away when
435/// the entry tier is equal and the request does not observe controls.
436pub fn load_serving(
437    path: &Path,
438    types: std::sync::Arc<crate::classify::TypeRegistry>,
439    wanted: SnapshotIdentity,
440) -> Result<LoadOutcome> {
441    load_with_types_and_size_limit(path, MAX_SNAPSHOT_BYTES, types, Some(wanted))
442}
443
444fn load_with_size_limit(path: &Path, max_snapshot_bytes: u64) -> Result<Option<Index>> {
445    load_with_types_and_size_limit(
446        path,
447        max_snapshot_bytes,
448        crate::classify::TypeRegistry::compiled_shared(),
449        None,
450    )
451    .map(|outcome| match outcome {
452        LoadOutcome::Served { index, .. } => Some(index),
453        LoadOutcome::Refused { .. } | LoadOutcome::Absent => None,
454    })
455}
456
457fn load_with_types_and_size_limit(
458    path: &Path,
459    max_snapshot_bytes: u64,
460    types: std::sync::Arc<crate::classify::TypeRegistry>,
461    wanted: Option<SnapshotIdentity>,
462) -> Result<LoadOutcome> {
463    let mut file = match fs::File::open(path) {
464        Ok(file) => file,
465        Err(e) if e.kind() == std::io::ErrorKind::NotFound => return Ok(LoadOutcome::Absent),
466        Err(e) => return Err(Error::io(path, e)),
467    };
468    let file_len = file.metadata().map_err(|e| Error::io(path, e))?.len();
469    let footer_bytes = CHECKSUM_BYTES
470        .checked_add(TRAILER.len())
471        .and_then(|bytes| u64::try_from(bytes).ok())
472        .ok_or_else(|| Error::Snapshot("snapshot footer size overflow".into()))?;
473    if file_len > max_snapshot_bytes || file_len < footer_bytes {
474        return Ok(LoadOutcome::Absent);
475    }
476
477    let footer_offset = i64::try_from(footer_bytes)
478        .map_err(|_| Error::Snapshot("snapshot footer size overflow".into()))?;
479    file.seek(SeekFrom::End(-footer_offset)).map_err(|e| Error::io(path, e))?;
480    let expected_checksum = match read_footer_checksum(&mut file) {
481        Ok(checksum) => checksum,
482        Err(error) if error.kind() == std::io::ErrorKind::UnexpectedEof => {
483            return Ok(LoadOutcome::Absent);
484        }
485        Err(error) => return Err(Error::io(path, error)),
486    };
487    let mut trailer = [0u8; TRAILER.len()];
488    match file.read_exact(&mut trailer) {
489        Ok(()) if &trailer == TRAILER => {}
490        Ok(()) => return Ok(LoadOutcome::Absent),
491        Err(e) if e.kind() == std::io::ErrorKind::UnexpectedEof => return Ok(LoadOutcome::Absent),
492        Err(e) => return Err(Error::io(path, e)),
493    }
494    let payload_len = file_len
495        .checked_sub(footer_bytes)
496        .ok_or_else(|| Error::Snapshot("snapshot length underflow".into()))?;
497    file.seek(SeekFrom::Start(0)).map_err(|e| Error::io(path, e))?;
498
499    // One pass, not two: the checksum accumulates as the parser consumes, instead of
500    // a separate full read of the image before parsing begins. The verdict still
501    // gates the data — the parsed index is returned only after the digest over the
502    // complete payload matches, so nothing is ever served from bytes that failed
503    // their checksum. What changes is the failure mode for structurally-valid
504    // corruption: the parser may do work before the mismatch is known, and the
505    // result is then discarded. Structural corruption is caught by the parser's own
506    // bounds and consistency checks exactly as before, fail-closed either way.
507    let mut reader = Crc32cReader::new(BufReader::new(file.take(payload_len)));
508    let outcome = parse_stream(&mut reader, payload_len, types, wanted);
509    match outcome {
510        Ok(outcome) => {
511            // A successful parse consumed every payload byte (the trailing-byte check
512            // proves it), so the running digest covers the whole image.
513            if reader.finish() == expected_checksum { Ok(outcome) } else { Ok(LoadOutcome::Absent) }
514        }
515        Err(ParseError::Invalid) => Ok(LoadOutcome::Absent),
516        Err(ParseError::Io(source)) => Err(Error::io(path, source)),
517    }
518}
519
520/// A reader that folds CRC-32C over every byte the caller consumes.
521struct Crc32cReader<R> {
522    inner: R,
523    state: u32,
524}
525
526impl<R: Read> Crc32cReader<R> {
527    fn new(inner: R) -> Self {
528        Self { inner, state: u32::MAX }
529    }
530
531    fn finish(&self) -> u32 {
532        !self.state
533    }
534}
535
536impl<R: Read> Read for Crc32cReader<R> {
537    fn read(&mut self, buf: &mut [u8]) -> std::io::Result<usize> {
538        let read = self.inner.read(buf)?;
539        self.state = crc32c_update(self.state, &buf[..read]);
540        Ok(read)
541    }
542}
543
544fn read_footer_checksum(reader: &mut impl Read) -> std::io::Result<u32> {
545    let mut bytes = [0u8; CHECKSUM_BYTES];
546    reader.read_exact(&mut bytes)?;
547    Ok(u32::from_le_bytes(bytes))
548}
549
550pub(crate) fn crc32c(bytes: &[u8]) -> u32 {
551    !crc32c_update(u32::MAX, bytes)
552}
553
554/// Slicing-by-8: fold eight input bytes per step through eight derived tables
555/// instead of one byte through one. Table 0 is the classic byte table, so the
556/// remainder loop and the 8-byte path share one source of truth; the digest is
557/// bit-identical to the byte-at-a-time form, which a test asserts over uneven
558/// lengths alongside the standard check value.
559fn crc32c_update(mut state: u32, bytes: &[u8]) -> u32 {
560    let mut chunks = bytes.chunks_exact(8);
561    for chunk in &mut chunks {
562        let low = (state ^ u32::from_le_bytes(chunk[..4].try_into().expect("chunk holds 8 bytes")))
563            .to_le_bytes();
564        let high =
565            u32::from_le_bytes(chunk[4..].try_into().expect("chunk holds 8 bytes")).to_le_bytes();
566        state = CRC32C_TABLES[7][usize::from(low[0])]
567            ^ CRC32C_TABLES[6][usize::from(low[1])]
568            ^ CRC32C_TABLES[5][usize::from(low[2])]
569            ^ CRC32C_TABLES[4][usize::from(low[3])]
570            ^ CRC32C_TABLES[3][usize::from(high[0])]
571            ^ CRC32C_TABLES[2][usize::from(high[1])]
572            ^ CRC32C_TABLES[1][usize::from(high[2])]
573            ^ CRC32C_TABLES[0][usize::from(high[3])];
574    }
575    for byte in chunks.remainder() {
576        let index = usize::from(state.to_le_bytes()[0] ^ *byte);
577        state = CRC32C_TABLES[0][index] ^ (state >> 8);
578    }
579    state
580}
581
582const fn make_crc32c_tables() -> [[u32; 256]; 8] {
583    let mut tables = [[0u32; 256]; 8];
584    let mut index = 0usize;
585    let mut value = 0u32;
586    while index < 256 {
587        let mut crc = value;
588        let mut bit = 0;
589        while bit < u8::BITS {
590            crc = (crc >> 1) ^ (CRC32C_POLYNOMIAL & 0u32.wrapping_sub(crc & 1));
591            bit += 1;
592        }
593        tables[0][index] = crc;
594        index += 1;
595        value += 1;
596    }
597    // tables[k] advances a byte's contribution k further positions: one more
598    // byte-table step applied to the previous table's value.
599    let mut table = 1usize;
600    while table < tables.len() {
601        let mut index = 0usize;
602        while index < 256 {
603            let previous = tables[table - 1][index];
604            tables[table][index] = tables[0][(previous & 0xFF) as usize] ^ (previous >> 8);
605            index += 1;
606        }
607        table += 1;
608    }
609    tables
610}
611
612/// Read only a snapshot's header, without materializing its index.
613///
614/// Returns `None` for anything this build cannot serve — absent, truncated, foreign,
615/// a different format version, or a mismatched engine fingerprint. Corrupt equals
616/// absent here exactly as it does on the load path: a caller asking what is in the cache
617/// must not be stopped by one unreadable file.
618pub fn read_header(path: &Path) -> Result<Option<crate::cache::SnapshotInfo>> {
619    Ok(match identify(path)? {
620        Some(Identity::Current(info)) => Some(info),
621        Some(Identity::Stale(_) | Identity::Foreign) | None => None,
622    })
623}
624
625/// What a file's leading bytes and trailer say it is.
626#[derive(Debug)]
627pub(crate) enum Identity {
628    /// A snapshot this build reads.
629    Current(crate::cache::SnapshotInfo),
630    /// It begins with the snapshot magic, so fdu wrote it, but this build cannot serve it.
631    Stale(crate::cache::StaleReason),
632    /// It does not begin with the snapshot magic.
633    Foreign,
634}
635
636/// Identify the file at `path` by its contents, or return `None` when nothing is there.
637///
638/// The magic alone decides whether a file is fdu's; the rest of the prologue decides
639/// whether this build can serve it. Every format fdu has written puts the version and the
640/// engine fingerprint at the same offsets after the magic, so a snapshot from another
641/// release is recognised as stale rather than mistaken for a foreign file. That is what
642/// lets the cache lifecycle reclaim the snapshots an upgrade leaves behind.
643///
644/// Opening follows a symbolic link at `path`, so a caller that must not follow one checks
645/// the path's own metadata first.
646pub(crate) fn identify(path: &Path) -> Result<Option<Identity>> {
647    let file = match fs::File::open(path) {
648        Ok(file) => file,
649        Err(error) if error.kind() == std::io::ErrorKind::NotFound => return Ok(None),
650        Err(error) => return Err(Error::io(path, error)),
651    };
652    let trailer_intact = has_intact_trailer(&file).map_err(|error| Error::io(path, error))?;
653    // The trailer check left the cursor at the end; the header lives at the start.
654    let mut file = file;
655    file.seek(SeekFrom::Start(0)).map_err(|error| Error::io(path, error))?;
656    identify_prologue(&mut BufReader::new(file), trailer_intact)
657        .map(Some)
658        .map_err(|error| Error::io(path, error))
659}
660
661/// Classify a file from its prologue. Only an I/O failure is an error; every malformed
662/// byte is an answer.
663fn identify_prologue(reader: &mut impl Read, trailer_intact: bool) -> io::Result<Identity> {
664    use crate::cache::StaleReason;
665
666    match read_array::<_, 8>(reader) {
667        Ok(magic) if magic == *MAGIC => {}
668        Ok(_) | Err(ParseError::Invalid) => return Ok(Identity::Foreign),
669        Err(ParseError::Io(error)) => return Err(error),
670    }
671    let Some(version) = invalid_as_none(read_u32(reader))? else {
672        return Ok(Identity::Stale(StaleReason::Unreadable));
673    };
674    match version.cmp(&FORMAT_VERSION) {
675        std::cmp::Ordering::Less => {
676            return Ok(Identity::Stale(StaleReason::OlderFormat { version }));
677        }
678        std::cmp::Ordering::Greater => {
679            return Ok(Identity::Stale(StaleReason::NewerFormat { version }));
680        }
681        std::cmp::Ordering::Equal => {}
682    }
683    let engine = match invalid_as_none(read_u64(reader))? {
684        Some(fingerprint) if fingerprint == engine_fingerprint() => fingerprint,
685        Some(_) => return Ok(Identity::Stale(StaleReason::OtherEngine)),
686        None => return Ok(Identity::Stale(StaleReason::Unreadable)),
687    };
688    if !trailer_intact {
689        // Truncation removes the tail and leaves the prologue readable, so a header-only
690        // check would call a half-written file current.
691        return Ok(Identity::Stale(StaleReason::Unreadable));
692    }
693    Ok(match invalid_as_none(parse_header_fields(reader, engine))? {
694        Some(header) => Identity::Current(crate::cache::SnapshotInfo {
695            root: header.root,
696            identity: header.identity,
697            entries: header.entries,
698        }),
699        None => Identity::Stale(StaleReason::Unreadable),
700    })
701}
702
703/// Separate a malformed value, which is an answer, from an I/O failure, which is not.
704fn invalid_as_none<T>(result: ParseResult<T>) -> io::Result<Option<T>> {
705    match result {
706        Ok(value) => Ok(Some(value)),
707        Err(ParseError::Invalid) => Ok(None),
708        Err(ParseError::Io(error)) => Err(error),
709    }
710}
711
712/// Whether a file ends with the snapshot trailer.
713///
714/// Cheap enough for a status listing — one seek and eight bytes — and it is exactly what
715/// a truncated write destroys.
716fn has_intact_trailer(file: &fs::File) -> io::Result<bool> {
717    let footer_bytes = CHECKSUM_BYTES + TRAILER.len();
718    let file_len = file.metadata()?.len();
719    if file_len < u64::try_from(footer_bytes).unwrap_or(u64::MAX) {
720        return Ok(false);
721    }
722
723    let mut handle = file;
724    handle.seek(SeekFrom::End(-(i64::try_from(TRAILER.len()).unwrap_or(0))))?;
725    let mut trailer = [0u8; TRAILER.len()];
726    match handle.read_exact(&mut trailer) {
727        Ok(()) => Ok(&trailer == TRAILER),
728        Err(error) if error.kind() == io::ErrorKind::UnexpectedEof => Ok(false),
729        Err(error) => Err(error),
730    }
731}
732
733/// The header fields after a snapshot's prologue.
734struct Header {
735    /// The start of the pass that last wrote the image.
736    ///
737    /// A later pass that verifies the same facts keeps the image and this stamp rather
738    /// than rewriting for the stamp alone, so the value is a lower bound on when the facts
739    /// were last verified: it can predate many verifying passes and never postdates the one
740    /// that wrote them. Until P1.4.4 stamps completed passes, reconciliation does not
741    /// advance it either.
742    writing_pass_started_at_ns: i64,
743    /// The identity of every tier the snapshot holds.
744    identity: SnapshotIdentity,
745    /// The absolute root the snapshot describes.
746    root: PathBuf,
747    /// How many entry records follow.
748    entries: u64,
749}
750
751/// Parse the header fields after the magic, format version, and `engine` fingerprint.
752fn parse_header_fields(reader: &mut impl Read, engine: u64) -> ParseResult<Header> {
753    if read_u8(reader)? != path_encoding() {
754        return Err(ParseError::Invalid);
755    }
756    let writing_pass_started_at_ns = read_i64(reader)?;
757    let identity =
758        SnapshotIdentity::decode(engine, &read_array::<_, SNAPSHOT_IDENTITY_BYTES>(reader)?)
759            .ok_or(ParseError::Invalid)?;
760    let root = PathBuf::from(read_os_string(reader)?);
761    let entries = read_u64(reader)?;
762    if entries == 0 || entries > MAX_SNAPSHOT_ENTRIES {
763        return Err(ParseError::Invalid);
764    }
765    Ok(Header { writing_pass_started_at_ns, identity, root, entries })
766}
767
768/// Parse a bounded payload. Records are applied one at a time so bootstrap paths are not
769/// retained in a second full-tree allocation.
770fn parse_stream(
771    reader: &mut impl Read,
772    payload_len: u64,
773    types: std::sync::Arc<crate::classify::TypeRegistry>,
774    wanted: Option<SnapshotIdentity>,
775) -> ParseResult<LoadOutcome> {
776    if read_array::<_, 8>(reader)? != *MAGIC {
777        return Err(ParseError::Invalid);
778    }
779    let engine = engine_fingerprint();
780    if read_u32(reader)? != FORMAT_VERSION || read_u64(reader)? != engine {
781        return Err(ParseError::Invalid);
782    }
783    let Header { writing_pass_started_at_ns, identity, root: root_path, entries: count } =
784        parse_header_fields(reader, engine)?;
785    let serves = wanted.map_or(Serves::Exact, |wanted| serves_snapshot(identity, wanted));
786    let mut scope = match (serves, wanted) {
787        (Serves::ProjectControlsOff, Some(wanted)) => wanted.scan_scope(),
788        _ => identity.scan_scope(),
789    };
790    if scope.type_rules_fingerprint != types.fingerprint() {
791        if serves != Serves::Refuse {
792            return Err(ParseError::Invalid);
793        }
794        // A foreign registry cannot be attached to a served index. For a refusal,
795        // reconstruct only to validate every record and the checksum, then discard it.
796        // The stored identity below remains unchanged for the caller's diagnostic.
797        //
798        // That costs a full index build for a diagnostic: every record is decoded and
799        // inserted so the checksum can be verified over the bytes it covers, and the
800        // result is dropped. It is paid only under a type-rules mismatch, which a
801        // cache-only request reports and every other policy answers by scanning, so it
802        // buys a truthful refusal (valid but foreign, rather than absent) at the price of
803        // one load on a path that has no answer anyway.
804        scope.type_rules_fingerprint = types.fingerprint();
805    }
806    let minimum_body = count
807        .checked_mul(u64::try_from(MIN_RECORD_BYTES).map_err(|_| ParseError::Invalid)?)
808        .ok_or(ParseError::Invalid)?;
809    if minimum_body > payload_len {
810        return Err(ParseError::Invalid);
811    }
812
813    let mut index = Index::new_with_scope_and_types(&root_path, scope, types);
814    // The facts below were verified by the pass that wrote them, not by this load.
815    index.set_writing_pass_started_at_ns(writing_pass_started_at_ns);
816    // Everything this loader inserts describes the tree as the snapshot found it, not
817    // as this process has seen it. Stamping the entries `Cached` is what lets a
818    // consumer paint them immediately and label them honestly; without it a loaded
819    // index claims to be fresh when nothing has been checked since the file was read.
820    index.set_applying_source(Source::Cached, writing_pass_started_at_ns);
821    // The record count is validated against the bytes actually present above, so it is
822    // safe to size from: reserving here removes the geometric regrowth of a 450k-element
823    // vector without letting a corrupt count drive the allocation.
824    let mut ids: Vec<EntryId> = Vec::with_capacity(usize::try_from(count).unwrap_or(0));
825    for slot in 0..count {
826        let parent_slot = read_u32(reader)?;
827        let kind = EntryKind::from_u8(read_u8(reader)?).ok_or(ParseError::Invalid)?;
828        let name = read_os_string(reader)?;
829        let attrs = Attrs {
830            size: read_u64(reader)?,
831            allocated: read_u64(reader)?,
832            mtime_ns: read_i64(reader)?,
833            ctime_ns: read_i64(reader)?,
834            inode: read_u64(reader)?,
835            dev: read_u64(reader)?,
836        };
837
838        if parent_slot == NO_PARENT {
839            if slot != 0 || kind != EntryKind::Dir || !name.is_empty() {
840                return Err(ParseError::Invalid);
841            }
842            index
843                .apply_baseline(&Observation::new(vec![Op::Upsert {
844                    path: PathBuf::new(),
845                    kind,
846                    attrs,
847                }]))
848                .map_err(|_| ParseError::Invalid)?;
849            ids.push(EntryId::ROOT);
850            continue;
851        }
852
853        let parent = *ids
854            .get(usize::try_from(parent_slot).map_err(|_| ParseError::Invalid)?)
855            .ok_or(ParseError::Invalid)?;
856        if !is_snapshot_name(&name) {
857            return Err(ParseError::Invalid);
858        }
859        // The parent's id is already in hand, so the record is inserted straight beneath
860        // it. Resolving a path to rediscover that parent, and then searching the parent's
861        // children to rediscover the id just created, were both work the format had
862        // already answered. A snapshot naming the same path twice, or parenting an entry
863        // to a non-directory, is corrupt, and `insert_loaded_child` fails closed on both.
864        let id = index.insert_loaded_child(parent, name, kind, attrs).ok_or(ParseError::Invalid)?;
865        ids.push(id);
866    }
867
868    let controls = read_controls(reader, identity.controls)?;
869    if serves == Serves::Exact {
870        index.install_controls(controls).map_err(|_| ParseError::Invalid)?;
871    }
872
873    let mut extra = [0u8; 1];
874    if reader.read(&mut extra).map_err(ParseError::Io)? != 0 {
875        return Err(ParseError::Invalid);
876    }
877    // Once, rather than once per record: the loaded tree is the process baseline, and
878    // nothing it inserted was journalled.
879    index.establish_baseline();
880    // Anything applied after the load is this process checking what the snapshot
881    // claimed, which is a revalidation rather than a first sighting.
882    index.set_applying_source(Source::Revalidated, 0);
883    // The image on disk is this index, so nothing is owed until a pass mutates it.
884    index.set_persistence_owed(false);
885    Ok(if serves == Serves::Refuse {
886        LoadOutcome::Refused { identity, root: root_path }
887    } else {
888        LoadOutcome::Served { index, stored: identity }
889    })
890}
891
892#[derive(Debug)]
893enum ParseError {
894    Invalid,
895    Io(std::io::Error),
896}
897
898type ParseResult<T> = std::result::Result<T, ParseError>;
899
900fn read_array<R: Read, const N: usize>(reader: &mut R) -> ParseResult<[u8; N]> {
901    let mut bytes = [0u8; N];
902    match reader.read_exact(&mut bytes) {
903        Ok(()) => Ok(bytes),
904        Err(error) if error.kind() == std::io::ErrorKind::UnexpectedEof => Err(ParseError::Invalid),
905        Err(error) => Err(ParseError::Io(error)),
906    }
907}
908
909fn read_u8(reader: &mut impl Read) -> ParseResult<u8> {
910    Ok(read_array::<_, 1>(reader)?[0])
911}
912
913fn read_u32(reader: &mut impl Read) -> ParseResult<u32> {
914    Ok(u32::from_le_bytes(read_array(reader)?))
915}
916
917fn read_u64(reader: &mut impl Read) -> ParseResult<u64> {
918    Ok(u64::from_le_bytes(read_array(reader)?))
919}
920
921fn read_i64(reader: &mut impl Read) -> ParseResult<i64> {
922    Ok(i64::from_le_bytes(read_array(reader)?))
923}
924
925fn read_bytes(reader: &mut impl Read) -> ParseResult<Vec<u8>> {
926    let len = read_u32(reader)?;
927    if len > MAX_PATH_BYTES {
928        return Err(ParseError::Invalid);
929    }
930    let mut bytes = vec![0u8; usize::try_from(len).map_err(|_| ParseError::Invalid)?];
931    match reader.read_exact(&mut bytes) {
932        Ok(()) => Ok(bytes),
933        Err(error) if error.kind() == std::io::ErrorKind::UnexpectedEof => Err(ParseError::Invalid),
934        Err(error) => Err(ParseError::Io(error)),
935    }
936}
937
938/// Read the control section, retained sources then refusals, for the control tier the
939/// header records.
940///
941/// Every source is admitted again under the header's limits. The charge does not depend on
942/// admission order, so a table a scan retained always fits; one that does not, a repeated
943/// path, a refusal naming a retained or repeated path, a refusal by a limit the header
944/// records as unbounded, or any source or refusal under a tier that observed nothing, was
945/// not written by [`save`].
946fn read_controls(
947    reader: &mut impl Read,
948    tier: ControlTierIdentity,
949) -> ParseResult<crate::control::ControlTable> {
950    let limits = match tier {
951        ControlTierIdentity::Observed { limits } => limits,
952        // Limits decide nothing when nothing was observed, and the section must be empty.
953        // So a loaded blind table holds the default limits where the scanning index kept
954        // the request's; no reader consults a blind table's limits, and a cache status or
955        // projection that reports them must not show that as a difference.
956        ControlTierIdentity::NotObserved => crate::control::ControlLimits::default(),
957    };
958    let control_count = read_u32(reader)?;
959    if control_count != 0 && !tier.is_observed() {
960        return Err(ParseError::Invalid);
961    }
962    if usize::try_from(control_count)
963        .map_err(|_| ParseError::Invalid)?
964        .saturating_mul(crate::control::CONTROL_SOURCE_OVERHEAD)
965        > SNAPSHOT_CONTROL_TABLE_CEILING
966    {
967        return Err(ParseError::Invalid);
968    }
969    let mut sources = Vec::new();
970    let mut source_bytes = 0_usize;
971    for _ in 0..control_count {
972        let path = PathBuf::from(read_os_string(reader)?);
973        let source = read_control_bytes(reader)?;
974        source_bytes = source_bytes.saturating_add(source.len());
975        if source_bytes > SNAPSHOT_CONTROL_TABLE_CEILING {
976            return Err(ParseError::Invalid);
977        }
978        sources.push((path, source));
979    }
980    let mut controls = crate::control::ControlTable::with_limits(limits);
981    for (path, source) in sources {
982        let admission = controls.upsert(&path, source).map_err(|_| ParseError::Invalid)?;
983        if admission != (crate::control::ControlAdmission::Retained { changed: true }) {
984            return Err(ParseError::Invalid);
985        }
986    }
987    if controls.retained_cost() > SNAPSHOT_CONTROL_TABLE_CEILING {
988        return Err(ParseError::Invalid);
989    }
990    let refused = read_u32(reader)?;
991    if refused != 0 && !tier.is_observed() {
992        return Err(ParseError::Invalid);
993    }
994    for _ in 0..refused {
995        let path = PathBuf::from(read_os_string(reader)?);
996        let reason = match read_u8(reader)? {
997            REFUSED_FOR_BUDGET => crate::control::ControlRefusalReason::Budget,
998            REFUSED_FOR_LINE_LIMIT => crate::control::ControlRefusalReason::LineLimit,
999            _ => return Err(ParseError::Invalid),
1000        };
1001        // An unbounded limit refuses nothing, so no save records a refusal by one.
1002        if !crate::control::is_control_file(&path)
1003            || controls.contains(&path)
1004            || limits.limit_for(reason).is_none()
1005        {
1006            return Err(ParseError::Invalid);
1007        }
1008        controls.record_refusal(&path, reason).map_err(|_| ParseError::Invalid)?;
1009    }
1010    Ok(controls)
1011}
1012
1013fn read_control_bytes(reader: &mut impl Read) -> ParseResult<Vec<u8>> {
1014    let len = read_u32(reader)?;
1015    if usize::try_from(len).map_err(|_| ParseError::Invalid)? > SNAPSHOT_CONTROL_TABLE_CEILING {
1016        return Err(ParseError::Invalid);
1017    }
1018    let mut bytes = vec![0u8; usize::try_from(len).map_err(|_| ParseError::Invalid)?];
1019    match reader.read_exact(&mut bytes) {
1020        Ok(()) => Ok(bytes),
1021        Err(error) if error.kind() == std::io::ErrorKind::UnexpectedEof => Err(ParseError::Invalid),
1022        Err(error) => Err(ParseError::Io(error)),
1023    }
1024}
1025
1026#[cfg(unix)]
1027fn read_os_string(reader: &mut impl Read) -> ParseResult<OsString> {
1028    Ok(os_string_from_bytes(&read_bytes(reader)?))
1029}
1030
1031#[cfg(not(unix))]
1032fn read_os_string(reader: &mut impl Read) -> ParseResult<OsString> {
1033    os_string_from_bytes(&read_bytes(reader)?).ok_or(ParseError::Invalid)
1034}
1035
1036fn put_bytes(buf: &mut Vec<u8>, bytes: &[u8]) -> Result<()> {
1037    let len = u32::try_from(bytes.len())
1038        .map_err(|_| Error::Snapshot("string too long for snapshot".into()))?;
1039    if len > MAX_PATH_BYTES {
1040        return Err(Error::Snapshot("path exceeds snapshot limit".into()));
1041    }
1042    buf.extend_from_slice(&len.to_le_bytes());
1043    buf.extend_from_slice(bytes);
1044    Ok(())
1045}
1046
1047/// Write the control section [`read_controls`] reads: sources, then refusals. The limits
1048/// they were admitted under are the header's control tier identity.
1049fn put_controls(buf: &mut Vec<u8>, controls: &crate::control::ControlTable) -> Result<()> {
1050    if controls.retained_cost() > SNAPSHOT_CONTROL_TABLE_CEILING
1051        || controls.source_bytes() > SNAPSHOT_CONTROL_TABLE_CEILING
1052    {
1053        return Err(Error::Snapshot(format!(
1054            "the control table retains {} bytes of charge, above the {} bytes a snapshot can \
1055             carry; set a control budget below it, or open without a cache",
1056            controls.retained_cost(),
1057            SNAPSHOT_CONTROL_TABLE_CEILING
1058        )));
1059    }
1060    let sources: Vec<_> = controls.sources().collect();
1061    let control_count = u32::try_from(sources.len())
1062        .map_err(|_| Error::Snapshot("control table exceeds u32 capacity".into()))?;
1063    buf.extend_from_slice(&control_count.to_le_bytes());
1064    for (path, source) in sources {
1065        put_os_str(buf, path.as_os_str())?;
1066        let len = u32::try_from(source.len())
1067            .map_err(|_| Error::Snapshot("control source exceeds u32 capacity".into()))?;
1068        buf.extend_from_slice(&len.to_le_bytes());
1069        buf.extend_from_slice(source);
1070    }
1071    let refused = u32::try_from(controls.refused_len())
1072        .map_err(|_| Error::Snapshot("refused controls exceed u32 capacity".into()))?;
1073    buf.extend_from_slice(&refused.to_le_bytes());
1074    for refusal in controls.refusals() {
1075        put_os_str(buf, refusal.path.as_os_str())?;
1076        buf.push(match refusal.reason {
1077            crate::control::ControlRefusalReason::Budget => REFUSED_FOR_BUDGET,
1078            crate::control::ControlRefusalReason::LineLimit => REFUSED_FOR_LINE_LIMIT,
1079        });
1080    }
1081    Ok(())
1082}
1083
1084/// A snapshot entry name is one filesystem component, spelled exactly as stored.
1085///
1086/// `Path::components` treats `a` and `a/` as the same `Normal("a")`. The child map
1087/// distinguishes the raw names, but a `PathBuf`-keyed restore map merges them and
1088/// shrinks the cache-only completeness denominator. The raw name must equal that
1089/// single normal component so a checksummed alias cannot load.
1090fn is_snapshot_name(name: &OsStr) -> bool {
1091    let mut components = Path::new(name).components();
1092    let Some(Component::Normal(component)) = components.next() else { return false };
1093    components.next().is_none() && component == name
1094}
1095
1096#[cfg(unix)]
1097pub(crate) fn path_encoding() -> u8 {
1098    PATH_ENCODING_UNIX_BYTES
1099}
1100
1101#[cfg(windows)]
1102pub(crate) fn path_encoding() -> u8 {
1103    PATH_ENCODING_WINDOWS_WIDE
1104}
1105
1106#[cfg(not(any(unix, windows)))]
1107pub(crate) fn path_encoding() -> u8 {
1108    PATH_ENCODING_UTF8
1109}
1110
1111#[cfg(unix)]
1112pub(crate) fn put_os_str(buf: &mut Vec<u8>, value: &OsStr) -> Result<()> {
1113    use std::os::unix::ffi::OsStrExt;
1114    put_bytes(buf, value.as_bytes())
1115}
1116
1117#[cfg(windows)]
1118pub(crate) fn put_os_str(buf: &mut Vec<u8>, value: &OsStr) -> Result<()> {
1119    use std::os::windows::ffi::OsStrExt;
1120    let mut bytes = Vec::new();
1121    for unit in value.encode_wide() {
1122        bytes.extend_from_slice(&unit.to_le_bytes());
1123    }
1124    put_bytes(buf, &bytes)
1125}
1126
1127#[cfg(not(any(unix, windows)))]
1128pub(crate) fn put_os_str(buf: &mut Vec<u8>, value: &OsStr) -> Result<()> {
1129    let text = value
1130        .to_str()
1131        .ok_or_else(|| Error::Snapshot("path is not valid UTF-8 on this platform".into()))?;
1132    put_bytes(buf, text.as_bytes())
1133}
1134
1135#[cfg(unix)]
1136fn os_string_from_bytes(bytes: &[u8]) -> OsString {
1137    use std::os::unix::ffi::OsStringExt;
1138    OsString::from_vec(bytes.to_vec())
1139}
1140
1141#[cfg(windows)]
1142fn os_string_from_bytes(bytes: &[u8]) -> Option<OsString> {
1143    use std::os::windows::ffi::OsStringExt;
1144    // Stable since 1.87, against an MSRV of 1.85 — see the note in content_cache.rs.
1145    if bytes.len() % std::mem::size_of::<u16>() != 0 {
1146        return None;
1147    }
1148    let units: Vec<u16> = bytes
1149        .chunks_exact(std::mem::size_of::<u16>())
1150        .map(|chunk| u16::from_le_bytes([chunk[0], chunk[1]]))
1151        .collect();
1152    Some(OsString::from_wide(&units))
1153}
1154
1155#[cfg(not(any(unix, windows)))]
1156fn os_string_from_bytes(bytes: &[u8]) -> Option<OsString> {
1157    Some(OsString::from(String::from_utf8(bytes.to_vec()).ok()?))
1158}
1159
1160/// Write to a sibling temporary file, then rename over the target.
1161///
1162/// Rename is atomic within a filesystem, so a reader either sees the whole old snapshot
1163/// or the whole new one. The temporary must be a sibling for that to hold — a rename
1164/// across filesystems is a copy, and copies are not atomic.
1165pub(crate) fn write_atomically(path: &Path, bytes: &[u8]) -> Result<()> {
1166    // Serialization is deterministic, so unchanged input encodes to the bytes already on
1167    // disk, and replacing them is a full write, an `F_FULLFSYNC`, and a rename that leave
1168    // the file exactly as it was. Comparing against the page-cached file costs a few
1169    // milliseconds and cannot go stale, because the bytes are the same bytes. The metadata
1170    // snapshot, whose images also differ in their pass-start stamp, compares in `publish`
1171    // instead, which records what the skip is worth (exp-066, exp-067).
1172    if keep_identical(path, bytes) {
1173        return Ok(());
1174    }
1175    replace_atomically(path, bytes)
1176}
1177
1178/// [`write_atomically`] without first comparing against the file at `path`, for a caller
1179/// that has already compared.
1180fn replace_atomically(path: &Path, bytes: &[u8]) -> Result<()> {
1181    let parent = parent_dir(path);
1182    fs::create_dir_all(parent).map_err(|e| Error::io(parent, e))?;
1183    let (tmp, mut file) = create_temp_file(path, parent)?;
1184    let write_then_sync = file.write_all(bytes).and_then(|()| file.sync_all());
1185    if let Err(e) = write_then_sync {
1186        let _ = fs::remove_file(&tmp);
1187        return Err(Error::io(&tmp, e));
1188    }
1189    drop(file);
1190
1191    if let Err(e) = fs::rename(&tmp, path) {
1192        let _ = fs::remove_file(&tmp);
1193        return Err(Error::io(path, e));
1194    }
1195    reap_stale_temporaries(parent, path, STALE_TEMP_AGE);
1196    Ok(())
1197}
1198
1199/// Leave `path` in place when it already holds exactly `bytes`, moving only its mtime to
1200/// now.
1201///
1202/// The bytes were just produced again, so now is when they were last known to be current,
1203/// and moving the mtime says so without writing the payload. Best effort: a stale mtime
1204/// understates that, which fails safe.
1205fn keep_identical(path: &Path, bytes: &[u8]) -> bool {
1206    if !same_bytes_on_disk(path, bytes) {
1207        return false;
1208    }
1209    let _ = touch(path, std::time::SystemTime::now());
1210    true
1211}
1212
1213/// Whether `path` already holds exactly `bytes`.
1214///
1215/// Streamed in 1 MiB pieces rather than read whole, so deciding not to write a 14 MB
1216/// snapshot does not cost a second 14 MB allocation beside the one being compared. Any
1217/// error -- missing file, short read, a file that grew under the comparison -- reads as
1218/// "different", and the caller writes; the comparison can only ever save work.
1219fn same_bytes_on_disk(path: &Path, bytes: &[u8]) -> bool {
1220    let Ok(mut file) = fs::File::open(path) else { return false };
1221    let Ok(metadata) = file.metadata() else { return false };
1222    if metadata.len() != u64::try_from(bytes.len()).unwrap_or(u64::MAX) {
1223        return false;
1224    }
1225    // A file that holds every byte and then more is not the same file.
1226    reads_as(&mut file, bytes) && matches!(file.read(&mut [0u8; 1]), Ok(0))
1227}
1228
1229/// Whether the next bytes `file` yields are exactly `bytes`.
1230///
1231/// Streamed in 1 MiB pieces, and stopping at the first piece that differs; any read error
1232/// or early end reads as "different".
1233fn reads_as(file: &mut fs::File, bytes: &[u8]) -> bool {
1234    let mut buffer = vec![0u8; bytes.len().min(1 << 20)];
1235    let mut offset = 0usize;
1236    while offset < bytes.len() {
1237        let want = buffer.len().min(bytes.len() - offset);
1238        let read = match file.read(&mut buffer[..want]) {
1239            Ok(0) | Err(_) => return false,
1240            Ok(read) => read,
1241        };
1242        let end = offset + read;
1243        if buffer[..read] != bytes[offset..end] {
1244            return false;
1245        }
1246        offset = end;
1247    }
1248    true
1249}
1250
1251/// Move `path`'s modification time to `when` without touching its contents.
1252fn touch(path: &Path, when: std::time::SystemTime) -> io::Result<()> {
1253    let file = OpenOptions::new().write(true).open(path)?;
1254    file.set_modified(when)
1255}
1256
1257/// Remove long-abandoned temporaries beside `path`.
1258///
1259/// Best effort and deliberately silent: this is housekeeping, and a snapshot that was
1260/// written correctly must not fail because a directory could not be tidied. Anything
1261/// that cannot be read or removed is left for the next writer.
1262fn reap_stale_temporaries(parent: &Path, path: &Path, older_than: std::time::Duration) {
1263    let Some(prefix) = temp_prefix(path) else { return };
1264    let Ok(entries) = fs::read_dir(parent) else { return };
1265    let now = std::time::SystemTime::now();
1266    for entry in entries.flatten() {
1267        let name = entry.file_name();
1268        if !name.as_encoded_bytes().starts_with(prefix.as_encoded_bytes()) {
1269            continue;
1270        }
1271        let stale = entry
1272            .metadata()
1273            .and_then(|meta| meta.modified())
1274            .ok()
1275            .and_then(|modified| now.duration_since(modified).ok())
1276            .is_some_and(|age| age >= older_than);
1277        if stale {
1278            let _ = fs::remove_file(entry.path());
1279        }
1280    }
1281}
1282
1283/// The shared leading portion of every temporary written for `path`.
1284///
1285/// Matching on this rather than on a full parse keeps the reaper from depending on the
1286/// pid, entropy, and sequence encoding: a temporary written by an older build with a
1287/// different suffix layout is still recognised and still collected.
1288fn temp_prefix(path: &Path) -> Option<OsString> {
1289    let mut prefix = OsString::from(".");
1290    prefix.push(path.file_name()?);
1291    prefix.push(".tmp.");
1292    Some(prefix)
1293}
1294
1295/// The directory a target lives in, as a path that can actually be opened.
1296///
1297/// A bare relative target such as `snap.fdu` has an *empty* parent rather than none, and
1298/// the empty path is not the current directory: `create_dir_all` and `rename` tolerate
1299/// it, but `read_dir` does not — which silently disabled the reaper for exactly those
1300/// targets while the write itself kept working.
1301fn parent_dir(path: &Path) -> &Path {
1302    match path.parent() {
1303        Some(parent) if !parent.as_os_str().is_empty() => parent,
1304        _ => Path::new("."),
1305    }
1306}
1307
1308/// The temporary name this process uses for `path` at `sequence`.
1309///
1310/// Split out so a test can plant a collision at a name the writer will actually try.
1311/// Guessing the shape does not work — the entropy is per process — and getting that
1312/// wrong is how the first version of the stale-temporary test came to exercise nothing.
1313fn temp_name(path: &Path, sequence: u64) -> OsString {
1314    let mut name = OsString::from(".");
1315    name.push(path.file_name().unwrap_or_else(|| OsStr::new("snapshot")));
1316    name.push(format!(".tmp.{}.{:016x}.{}", std::process::id(), *TEMP_FILE_ENTROPY, sequence));
1317    name
1318}
1319
1320fn create_temp_file(path: &Path, parent: &Path) -> Result<(PathBuf, fs::File)> {
1321    for _ in 0..MAX_TEMP_CREATE_ATTEMPTS {
1322        let sequence = NEXT_TEMP_FILE.fetch_add(1, Ordering::Relaxed);
1323        let tmp = parent.join(temp_name(path, sequence));
1324        let mut options = OpenOptions::new();
1325        options.write(true).create_new(true);
1326        #[cfg(unix)]
1327        {
1328            use std::os::unix::fs::OpenOptionsExt;
1329            options.mode(0o600);
1330        }
1331        match options.open(&tmp) {
1332            Ok(file) => return Ok((tmp, file)),
1333            Err(error) if error.kind() == std::io::ErrorKind::AlreadyExists => {}
1334            Err(error) => return Err(Error::io(&tmp, error)),
1335        }
1336    }
1337    Err(Error::io(
1338        parent,
1339        std::io::Error::new(
1340            std::io::ErrorKind::AlreadyExists,
1341            "could not reserve a unique snapshot temporary file",
1342        ),
1343    ))
1344}
1345
1346#[cfg(test)]
1347mod tests {
1348    use super::*;
1349    use crate::engine_contract::Observation;
1350    use crate::index::ExtTally;
1351
1352    fn attrs(size: u64, mtime_ns: i64) -> Attrs {
1353        Attrs {
1354            size,
1355            allocated: size.div_ceil(512) * 512,
1356            mtime_ns,
1357            ctime_ns: mtime_ns,
1358            inode: size.wrapping_mul(7).wrapping_add(1),
1359            dev: 3,
1360        }
1361    }
1362
1363    fn sample_index() -> Index {
1364        let mut index = Index::new("/some/root");
1365        index.apply_ok(&Observation::new(vec![
1366            Op::Upsert { path: PathBuf::from("src"), kind: EntryKind::Dir, attrs: attrs(0, 1) },
1367            Op::Upsert {
1368                path: PathBuf::from("src/main.rs"),
1369                kind: EntryKind::File,
1370                attrs: attrs(100, 10),
1371            },
1372            Op::Upsert {
1373                path: PathBuf::from("src/deep"),
1374                kind: EntryKind::Dir,
1375                attrs: attrs(0, 2),
1376            },
1377            Op::Upsert {
1378                path: PathBuf::from("src/deep/nested.rs"),
1379                kind: EntryKind::File,
1380                attrs: attrs(50, 20),
1381            },
1382            Op::Upsert {
1383                path: PathBuf::from("notes.md"),
1384                kind: EntryKind::File,
1385                attrs: attrs(7, 30),
1386            },
1387        ]));
1388        index
1389    }
1390
1391    fn entry_count_offset(bytes: &[u8]) -> usize {
1392        let root_len_at = ROOT_OFFSET;
1393        let root_len = u32::from_le_bytes(
1394            bytes[root_len_at..root_len_at + 4]
1395                .try_into()
1396                .expect("saved snapshot has a root length"),
1397        );
1398        root_len_at + 4 + usize::try_from(root_len).expect("root length fits usize")
1399    }
1400
1401    fn rewrite_checksum(bytes: &mut [u8]) {
1402        let payload_len = bytes.len() - CHECKSUM_BYTES - TRAILER.len();
1403        let checksum = crc32c(&bytes[..payload_len]);
1404        bytes[payload_len..payload_len + CHECKSUM_BYTES].copy_from_slice(&checksum.to_le_bytes());
1405    }
1406
1407    /// The encoded name field and following attributes for every entry record.
1408    fn entry_record_fields(bytes: &[u8]) -> Vec<(std::ops::Range<usize>, std::ops::Range<usize>)> {
1409        let count_at = entry_count_offset(bytes);
1410        let count = u64::from_le_bytes(
1411            bytes[count_at..count_at + 8].try_into().expect("saved snapshot has an entry count"),
1412        );
1413        let mut at = count_at + 8;
1414        let mut fields = Vec::new();
1415        for _ in 0..count {
1416            let name_len_at = at + 5;
1417            let name_len = usize::try_from(u32::from_le_bytes(
1418                bytes[name_len_at..name_len_at + 4]
1419                    .try_into()
1420                    .expect("saved entry has a name length"),
1421            ))
1422            .expect("name length fits usize");
1423            let name_end = name_len_at + 4 + name_len;
1424            fields.push((name_len_at..name_end, name_end..name_end + 6 * 8));
1425            at = name_end + 6 * 8;
1426        }
1427        fields
1428    }
1429
1430    fn replace_entry_name(image: &[u8], record: usize, name: &OsStr) -> Vec<u8> {
1431        let mut encoded = Vec::new();
1432        put_os_str(&mut encoded, name).expect("encode replacement name");
1433        let field = entry_record_fields(image)[record].0.clone();
1434        let mut rewritten = image.to_vec();
1435        rewritten.splice(field, encoded);
1436        rewrite_checksum(&mut rewritten);
1437        rewritten
1438    }
1439
1440    #[test]
1441    fn crc32c_matches_the_standard_check_value() {
1442        assert_eq!(crc32c(b"123456789"), 0xe306_9283);
1443    }
1444
1445    #[test]
1446    fn crc32c_slicing_matches_the_byte_reference_on_uneven_lengths() {
1447        // The 8-byte path and the remainder loop must agree with the classic
1448        // byte-at-a-time recurrence at every alignment, or an old snapshot's digest
1449        // stops verifying. The reference below IS that recurrence, against table 0.
1450        fn reference(bytes: &[u8]) -> u32 {
1451            let mut state = u32::MAX;
1452            for byte in bytes {
1453                let index = usize::from(state.to_le_bytes()[0] ^ *byte);
1454                state = CRC32C_TABLES[0][index] ^ (state >> 8);
1455            }
1456            !state
1457        }
1458        let mut data = Vec::new();
1459        let mut seed = 0x9e37_79b9u32;
1460        for length in [0usize, 1, 7, 8, 9, 15, 16, 63, 64, 65, 1000] {
1461            data.clear();
1462            for _ in 0..length {
1463                seed = seed.wrapping_mul(1_664_525).wrapping_add(1_013_904_223);
1464                data.push(seed.to_le_bytes()[0]);
1465            }
1466            assert_eq!(crc32c(&data), reference(&data), "length {length}");
1467        }
1468    }
1469
1470    #[test]
1471    fn snapshot_names_must_equal_their_single_normal_component() {
1472        assert!(is_snapshot_name(OsStr::new("notes.md")));
1473        assert!(!is_snapshot_name(OsStr::new("notes.md/")));
1474        assert!(!is_snapshot_name(OsStr::new("a/b")));
1475        assert!(!is_snapshot_name(OsStr::new("../bad")));
1476        assert!(!is_snapshot_name(OsStr::new("")));
1477        assert!(!is_snapshot_name(OsStr::new(".")));
1478        assert!(!is_snapshot_name(OsStr::new("..")));
1479        #[cfg(unix)]
1480        {
1481            use std::os::unix::ffi::OsStringExt;
1482            let native = OsString::from_vec(vec![b'n', 0x80]);
1483            assert!(is_snapshot_name(&native), "canonical validation is OsStr identity");
1484            let aliased = OsString::from_vec(vec![b'n', 0x80, b'/']);
1485            assert!(!is_snapshot_name(&aliased));
1486        }
1487    }
1488
1489    #[test]
1490    fn a_valid_forged_rename_loads_and_is_found_by_lookup() {
1491        let dir = tempfile::tempdir().expect("tempdir");
1492        let path = dir.path().join("snapshot.fdu");
1493        let mut index = Index::new("/some/root");
1494        index.apply_ok(&Observation::new(vec![Op::Upsert {
1495            path: PathBuf::from("valid"),
1496            kind: EntryKind::File,
1497            attrs: attrs(1, 1),
1498        }]));
1499        save(&index, &path).expect("save");
1500        let saved = fs::read(&path).expect("read");
1501        let forged = replace_entry_name(&saved, 1, OsStr::new("renamed"));
1502        fs::write(&path, forged).expect("write valid rename");
1503        let restored = load(&path).expect("load").expect("a valid rename is a snapshot");
1504        assert!(restored.lookup(Path::new("renamed")).is_some());
1505        assert!(restored.lookup(Path::new("valid")).is_none());
1506    }
1507
1508    #[test]
1509    fn noncanonical_entry_names_are_rejected_after_integrity_checks() {
1510        let dir = tempfile::tempdir().expect("tempdir");
1511        let path = dir.path().join("snapshot.fdu");
1512        let mut index = Index::new("/some/root");
1513        index.apply_ok(&Observation::new(vec![Op::Upsert {
1514            path: PathBuf::from("valid"),
1515            kind: EntryKind::File,
1516            attrs: attrs(1, 1),
1517        }]));
1518        save(&index, &path).expect("save");
1519        let saved = fs::read(&path).expect("read");
1520
1521        #[cfg(windows)]
1522        let names = ["a/", "a//", "a/.", "./a", r"a\", r"a\\", r"a\."];
1523        #[cfg(not(windows))]
1524        let names = ["a/", "a//", "a/.", "./a"];
1525        for name in names {
1526            let forged = replace_entry_name(&saved, 1, OsStr::new(name));
1527            fs::write(&path, forged).expect("write forged snapshot");
1528            assert!(load(&path).expect("malformed snapshot is a miss").is_none(), "{name:?}");
1529        }
1530    }
1531
1532    #[test]
1533    fn a_checksummed_name_alias_cannot_serve_cache_only_content() {
1534        let dir = tempfile::tempdir().expect("tempdir");
1535        let root = dir.path().join("root");
1536        fs::create_dir(&root).expect("create root");
1537        fs::write(root.join("a"), b"one two\n").expect("write first");
1538        fs::write(root.join("bb"), b"one two\n").expect("write second");
1539        let snapshot_path = dir.path().join("snapshot.fdu");
1540        let config = crate::OpenFixture {
1541            cache_path: Some(snapshot_path.clone()),
1542            policy: crate::CachePolicy::Auto,
1543            analysis: crate::content::AnalysisRequest {
1544                profile: crate::content::AnalysisSet::NONE.with_lines(),
1545                ..crate::content::AnalysisRequest::default()
1546            },
1547            ..crate::OpenFixture::default()
1548        };
1549        crate::open_fixture(&root, &config).expect("seed snapshot and sidecar");
1550
1551        let mut image = fs::read(&snapshot_path).expect("read snapshot");
1552        let fields = entry_record_fields(&image);
1553        assert_eq!(fields.len(), 3, "root and two files");
1554        let first_attrs = image[fields[1].1.clone()].to_vec();
1555        image[fields[2].1.clone()].copy_from_slice(&first_attrs);
1556        let forged = replace_entry_name(&image, 2, OsStr::new("a/"));
1557        fs::write(&snapshot_path, forged).expect("write checksummed alias");
1558
1559        let only = crate::OpenFixture { policy: crate::CachePolicy::Only, ..config };
1560        assert!(
1561            matches!(crate::open_fixture(&root, &only), Err(Error::Snapshot(_))),
1562            "a malformed snapshot cannot shrink the cache-only completeness denominator"
1563        );
1564    }
1565
1566    #[test]
1567    fn a_loaded_index_reports_cached_provenance_not_fresh() {
1568        // The gap that motivated the provenance model. A snapshot is complete when it
1569        // is written, so a loaded index used to claim `Fresh` — true of when the file
1570        // was made, and exactly backwards for a consumer painting on load, which needs
1571        // to know nothing has been checked since.
1572        let dir = tempfile::tempdir().expect("tempdir");
1573        let path = dir.path().join("snapshot.fdu");
1574        let original = sample_index();
1575        save(&original, &path).expect("save");
1576
1577        let restored = load(&path).expect("load").expect("snapshot present");
1578        let provenance =
1579            restored.provenance(Path::new("src/main.rs")).expect("the loaded entry is present");
1580        assert_eq!(provenance.source, crate::Source::Cached);
1581        assert!(!provenance.is_verified(), "nothing has been stat'd since the load");
1582        assert!(
1583            provenance.observed_at_ns > 0,
1584            "a cached value must say as of when, or a UI cannot label it"
1585        );
1586
1587        // A freshly scanned index is the contrasting case.
1588        assert_eq!(
1589            original.provenance(Path::new("src/main.rs")).expect("present").source,
1590            crate::Source::Scanned
1591        );
1592    }
1593
1594    #[test]
1595    fn a_loaded_root_reports_cached_provenance_not_fresh() {
1596        // The root is the one entry the child-only test above cannot cover, and it is
1597        // the one that matters most: whole-tree totals hang off it, so `fdu ~` reads
1598        // the root's provenance to label its headline number. `apply_upsert` handles
1599        // the root in a separate branch, and that branch used to skip the source stamp
1600        // entirely — leaving a snapshot-loaded root claiming `Scanned` and
1601        // `is_verified()`, the precise silent lie the model exists to prevent.
1602        let dir = tempfile::tempdir().expect("tempdir");
1603        let path = dir.path().join("snapshot.fdu");
1604        save(&sample_index(), &path).expect("save");
1605
1606        let restored = load(&path).expect("load").expect("snapshot present");
1607        let provenance = restored.provenance(Path::new("")).expect("the root is always present");
1608        assert_eq!(provenance.source, crate::Source::Cached, "the root came off disk too");
1609        assert!(!provenance.is_verified(), "nothing has been stat'd since the load");
1610        assert!(
1611            provenance.observed_at_ns > 0,
1612            "a cached total must say as of when, or a UI cannot label it"
1613        );
1614    }
1615
1616    #[test]
1617    fn cached_observation_time_comes_from_the_header_after_touch_or_copy() {
1618        use std::time::{Duration, UNIX_EPOCH};
1619
1620        let dir = tempfile::tempdir().expect("tempdir");
1621        let path = dir.path().join("snapshot.fdu");
1622        let copied = dir.path().join("copied.fdu");
1623        let mut original = sample_index();
1624        original.set_writing_pass_started_at_ns(1_000);
1625        save(&original, &path).expect("save");
1626
1627        // Cache files are routinely kept in place or copied between cache locations. Neither
1628        // operation observes the tree, so neither file mtime may become value provenance.
1629        touch(&path, UNIX_EPOCH + Duration::from_secs(10)).expect("touch cache image");
1630        fs::copy(&path, &copied).expect("copy cache image");
1631        touch(&copied, UNIX_EPOCH + Duration::from_secs(20)).expect("touch copied image");
1632
1633        for cache in [&path, &copied] {
1634            let restored = load(cache).expect("load").expect("present");
1635            for entry in [Path::new(""), Path::new("src/main.rs")] {
1636                assert_eq!(
1637                    restored.provenance(entry).expect("present").observed_at_ns,
1638                    1_000,
1639                    "{} uses its persisted pass start, not its cache-file mtime",
1640                    cache.display()
1641                );
1642            }
1643        }
1644    }
1645
1646    #[test]
1647    fn revalidating_a_loaded_index_promotes_entries_out_of_cached() {
1648        // The failure a reviewer caught on PR #6: entries were stamped only when
1649        // allocated, so a warm sweep could stat every entry and leave them all
1650        // reporting `Cached`. A consumer could then never clear a stale-value
1651        // indicator no matter how much verification ran, which defeats the point.
1652        let dir = tempfile::tempdir().expect("tempdir");
1653        let tree = dir.path().join("tree");
1654        std::fs::create_dir_all(tree.join("sub")).expect("create dirs");
1655        std::fs::write(tree.join("sub/file.txt"), b"contents").expect("write");
1656        let snapshot_path = dir.path().join("snapshot.fdu");
1657
1658        let config = crate::ScanConfig::default();
1659        let (original, report) = crate::scan::scan_into_index(&tree, &config).expect("scan");
1660        assert!(report.is_complete());
1661        save(&original, &snapshot_path).expect("save");
1662
1663        let mut restored = load(&snapshot_path).expect("load").expect("present");
1664        let target = Path::new("sub/file.txt");
1665        assert_eq!(
1666            restored.provenance(target).expect("present").source,
1667            crate::Source::Cached,
1668            "straight off disk, nothing has been checked"
1669        );
1670
1671        // A sweep that finds nothing changed still verified every entry it stat'd.
1672        let reconciled =
1673            crate::scan::reconcile(&mut restored, &config, &mut |_| {}).expect("reconcile");
1674        assert!(reconciled.is_complete());
1675        assert_eq!(reconciled.apply.updated, 0, "the tree did not change");
1676
1677        let provenance = restored.provenance(target).expect("present");
1678        assert_eq!(
1679            provenance.source,
1680            crate::Source::Revalidated,
1681            "an unchanged entry that was freshly stat'd has still been verified"
1682        );
1683        assert!(provenance.is_verified());
1684    }
1685
1686    /// Every entry, not just the root.
1687    ///
1688    /// The loader inserts beneath a known parent instead of replaying an observation, so
1689    /// it no longer shares a code path with the producer that built the original. Root
1690    /// totals cannot catch a roll-up that is wrong halfway down — the errors would have
1691    /// to cancel at the root to hide, but a single misplaced subtree does not touch the
1692    /// root at all. This walks both trees and compares each directory's own roll-up,
1693    /// each entry's kind, attributes and extension tallies, and the shape of the tree.
1694    #[test]
1695    fn round_trip_preserves_every_entry_not_just_the_root() {
1696        let dir = tempfile::tempdir().expect("tempdir");
1697        let tree = dir.path().join("tree");
1698        let snapshot_path = dir.path().join("cache").join("snap.fdu");
1699        // Depth and fan-out both matter: a parent-relative insert that mis-parents an
1700        // entry shows up as a wrong roll-up on an interior directory.
1701        for (relative, contents) in [
1702            ("a.rs", &b"fn main() {}"[..]),
1703            ("deep/one/two/three/leaf.txt", b"leaf"),
1704            ("deep/one/two/sibling.rs", b"sibling"),
1705            ("deep/one/other.md", b"# other"),
1706            ("wide/w1.txt", b"1"),
1707            ("wide/w2.txt", b"22"),
1708            ("wide/w3.rs", b"333"),
1709            ("empty/.keep", b""),
1710        ] {
1711            let path = tree.join(relative);
1712            fs::create_dir_all(path.parent().expect("parent")).expect("create dirs");
1713            fs::write(&path, contents).expect("write");
1714        }
1715
1716        let config = crate::ScanConfig::default();
1717        let (original, report) = crate::scan::scan_into_index(&tree, &config).expect("scan");
1718        assert!(report.is_complete());
1719        save(&original, &snapshot_path).expect("save");
1720        let restored = load(&snapshot_path).expect("load").expect("present");
1721
1722        assert_eq!(restored.len(), original.len(), "entry count");
1723
1724        // Walk the original and demand the same entry, in the same place, with the same
1725        // numbers, in the restored index.
1726        let mut stack = vec![crate::index::EntryId::ROOT];
1727        let mut compared = 0_u64;
1728        while let Some(id) = stack.pop() {
1729            let path = original.path_of(id).expect("original path");
1730            let mirrored = restored.lookup(&path).expect("restored entry at the same path");
1731            assert_eq!(
1732                original.kind_of(id),
1733                restored.kind_of(mirrored),
1734                "kind at {}",
1735                path.display()
1736            );
1737            assert_eq!(
1738                original.attrs_of(id),
1739                restored.attrs_of(mirrored),
1740                "attrs at {}",
1741                path.display()
1742            );
1743            // Roll-ups and children exist only on directories; a file legitimately has
1744            // neither, and demanding them would fail on the tree rather than the loader.
1745            if original.kind_of(id) == Some(crate::EntryKind::Dir) {
1746                let (before, after) = (
1747                    original.rollup_of(id).expect("original rollup"),
1748                    restored.rollup_of(mirrored).expect("restored rollup"),
1749                );
1750                assert_eq!(
1751                    (
1752                        before.files,
1753                        before.dirs,
1754                        before.bytes,
1755                        before.allocated,
1756                        before.newest_mtime_ns
1757                    ),
1758                    (after.files, after.dirs, after.bytes, after.allocated, after.newest_mtime_ns),
1759                    "rollup at {}",
1760                    path.display()
1761                );
1762                assert_eq!(before.by_ext, after.by_ext, "extension tallies at {}", path.display());
1763                let names: Vec<_> = original
1764                    .children_of(id)
1765                    .expect("original children")
1766                    .map(|(name, _)| name.to_os_string())
1767                    .collect();
1768                let mirrored_names: Vec<_> = restored
1769                    .children_of(mirrored)
1770                    .expect("restored children")
1771                    .map(|(name, _)| name.to_os_string())
1772                    .collect();
1773                assert_eq!(names, mirrored_names, "children of {}", path.display());
1774                stack.extend(original.children_of(id).expect("children").map(|(_, child)| child));
1775            }
1776            compared += 1;
1777        }
1778        assert_eq!(compared, original.len(), "every entry was compared");
1779
1780        // The loader must not leave the index looking like it has pending history.
1781        assert_eq!(restored.clock(), crate::Clock::ZERO);
1782        assert!(restored.since(crate::Clock::ZERO).commits.is_empty());
1783    }
1784
1785    #[test]
1786    fn round_trip_preserves_tree_and_rollups() {
1787        let dir = tempfile::tempdir().expect("tempdir");
1788        let path = dir.path().join("cache").join("snap.fdu");
1789        let original = sample_index();
1790
1791        save(&original, &path).expect("save");
1792        let restored = load(&path).expect("load").expect("snapshot present");
1793
1794        assert_eq!(restored.root_path(), Path::new("/some/root"));
1795        assert_eq!(restored.clock(), crate::Clock::ZERO);
1796        assert!(restored.since(crate::Clock::ZERO).commits.is_empty());
1797        assert_eq!(restored.len(), original.len());
1798        // Public roll-ups resolve assignment-ordered extension ids to stable names.
1799        let (restored_total, original_total) = (restored.total(), original.total());
1800        assert_eq!(
1801            (restored_total.files, restored_total.dirs, restored_total.bytes),
1802            (original_total.files, original_total.dirs, original_total.bytes)
1803        );
1804        assert_eq!(restored_total.allocated, original_total.allocated);
1805        assert_eq!(restored_total.newest_mtime_ns, original_total.newest_mtime_ns);
1806        assert_eq!(restored_total.by_ext, original_total.by_ext);
1807        assert!(!original.serving_indexes_enabled());
1808        assert!(!restored.serving_indexes_enabled());
1809        assert_eq!(restored.total().files, 3);
1810        assert_eq!(restored.total().dirs, 2);
1811        assert_eq!(restored.total().bytes, 157);
1812        assert_eq!(
1813            restored.total().by_ext[".rs"],
1814            ExtTally { files: 2, bytes: 150, allocated: 1024 }
1815        );
1816        assert_eq!(
1817            restored.attrs(Path::new("src/deep/nested.rs")),
1818            original.attrs(Path::new("src/deep/nested.rs"))
1819        );
1820    }
1821
1822    #[test]
1823    fn round_trip_preserves_exact_controls_and_fixed_partitions() {
1824        let dir = tempfile::tempdir().expect("tempdir");
1825        let path = dir.path().join("controls.fdu");
1826        let mut original =
1827            Index::new_with_scope("/some/root", crate::test_support::observing_controls());
1828        original.apply_ok(&Observation::new(vec![
1829            Op::Upsert {
1830                path: PathBuf::from(".gitignore"),
1831                kind: EntryKind::File,
1832                attrs: attrs(6, 1),
1833            },
1834            Op::Upsert {
1835                path: PathBuf::from("debug.log"),
1836                kind: EntryKind::File,
1837                attrs: attrs(10, 2),
1838            },
1839            Op::Upsert {
1840                path: PathBuf::from("keep.rs"),
1841                kind: EntryKind::File,
1842                attrs: attrs(20, 3),
1843            },
1844            Op::ControlUpsert { path: PathBuf::from(".gitignore"), source: b"*.log\n".to_vec() },
1845        ]));
1846
1847        save(&original, &path).expect("save");
1848        let restored = load(&path).expect("load").expect("snapshot present");
1849
1850        assert!(
1851            restored
1852                .controls()
1853                .expect("control state observed")
1854                .source_is(Path::new(".gitignore"), b"*.log\n")
1855        );
1856        assert_eq!(restored.controls().expect("control state observed").source_bytes(), 6);
1857        assert_eq!(
1858            restored.is_ignored(Path::new("debug.log")).expect("control state observed"),
1859            Some(true)
1860        );
1861        assert_eq!(
1862            restored.is_ignored(Path::new("keep.rs")).expect("control state observed"),
1863            Some(false)
1864        );
1865        let partitions = restored.partition_total().expect("control state observed");
1866        assert_eq!(partitions, original.partition_total().expect("control state observed"));
1867        assert_eq!(partitions.all.files, 3);
1868        assert_eq!(partitions.unignored.files, 2);
1869    }
1870
1871    #[test]
1872    fn removing_the_last_control_before_save_round_trips_an_empty_table() {
1873        let dir = tempfile::tempdir().expect("tempdir");
1874        let path = dir.path().join("no-controls.fdu");
1875        let mut original =
1876            Index::new_with_scope("/some/root", crate::test_support::observing_controls());
1877        original.apply_ok(&Observation::new(vec![
1878            Op::Upsert {
1879                path: PathBuf::from(".gitignore"),
1880                kind: EntryKind::File,
1881                attrs: attrs(6, 1),
1882            },
1883            Op::Upsert {
1884                path: PathBuf::from("debug.log"),
1885                kind: EntryKind::File,
1886                attrs: attrs(10, 2),
1887            },
1888            Op::ControlUpsert { path: PathBuf::from(".gitignore"), source: b"*.log\n".to_vec() },
1889        ]));
1890        original
1891            .apply_ok(&Observation::new(vec![Op::Remove { path: PathBuf::from(".gitignore") }]));
1892
1893        save(&original, &path).expect("save");
1894        let restored = load(&path).expect("load").expect("snapshot present");
1895
1896        assert!(restored.controls().expect("control state observed").is_empty());
1897        assert_eq!(
1898            restored.is_ignored(Path::new("debug.log")).expect("control state observed"),
1899            Some(false)
1900        );
1901        let partitions = restored.partition_total().expect("control state observed");
1902        assert_eq!(partitions.all, partitions.unignored);
1903    }
1904
1905    #[test]
1906    fn a_control_table_at_its_shared_bound_round_trips() {
1907        let dir = tempfile::tempdir().expect("tempdir");
1908        let path = dir.path().join("bounded-controls.fdu");
1909        let mut original =
1910            Index::new_with_scope("/some/root", crate::test_support::observing_controls());
1911        let source = crate::control::source_at_test_limit();
1912        original.apply_ok(&Observation::new(vec![Op::ControlUpsert {
1913            path: PathBuf::from(".gitignore"),
1914            source: source.clone(),
1915        }]));
1916        assert_eq!(
1917            original.controls().expect("control state observed").retained_cost(),
1918            crate::control::DEFAULT_CONTROL_BUDGET
1919        );
1920
1921        save(&original, &path).expect("save at bound");
1922        let restored = load(&path).expect("load").expect("snapshot present");
1923
1924        assert_eq!(
1925            restored.controls().expect("control state observed").retained_cost(),
1926            original.controls().expect("control state observed").retained_cost()
1927        );
1928        assert!(
1929            restored
1930                .controls()
1931                .expect("control state observed")
1932                .source_is(Path::new(".gitignore"), &source)
1933        );
1934    }
1935
1936    /// A control table must enforce the limits its scope was taken under. A hand-built index
1937    /// whose scope claims other limits is refused at save with a typed error, a snapshot
1938    /// whose control section its header's control tier would not admit fails closed at
1939    /// load, and `Index::new_with_config` builds an index whose table and scope agree.
1940    #[test]
1941    fn control_limits_that_disagree_with_the_scope_are_refused_at_save_and_load() {
1942        let dir = tempfile::tempdir().expect("tempdir");
1943        let path = dir.path().join("limits.fdu");
1944        let lifted = crate::ScanConfig {
1945            control_limits: crate::control::ControlLimits {
1946                budget: None,
1947                ..crate::control::ControlLimits::default()
1948            },
1949            ..crate::ScanConfig::default()
1950        };
1951
1952        let mismatched = Index::new_with_scope("/some/root", lifted.scope());
1953        let error = save(&mismatched, &path).expect_err("the scope claims no budget");
1954        assert!(
1955            matches!(
1956                error,
1957                Error::ControlLimitsOutsideScope { limits }
1958                    if limits == crate::control::ControlLimits::default()
1959            ),
1960            "{error}"
1961        );
1962        assert!(!path.exists(), "nothing is written");
1963
1964        let mut agreeing = Index::new_with_config("/some/root", &lifted);
1965        agreeing.apply_ok(&Observation::new(vec![Op::ControlUpsert {
1966            path: PathBuf::from(".gitignore"),
1967            source: b"*.log\n".to_vec(),
1968        }]));
1969        assert_eq!(agreeing.scope(), lifted.scope());
1970        save(&agreeing, &path).expect("save an index whose table enforces its scope's limits");
1971        let restored = load(&path).expect("load").expect("snapshot present");
1972        assert_eq!(restored.control_coverage(), agreeing.control_coverage());
1973        assert_eq!(restored.snapshot_identity(), lifted.snapshot_identity());
1974
1975        // The limits live only in the header's control tier. A CRC-valid snapshot whose
1976        // header claims a line limit the retained source exceeds, or claims no control
1977        // state beside a retained source, was written by no save, so it fails closed like
1978        // any other corruption.
1979        let saved = fs::read(&path).expect("read snapshot");
1980        let controls_at = IDENTITY_OFFSET + crate::stored_state::ENTRY_TIER_BYTES;
1981        let controls = controls_at..controls_at + crate::stored_state::CONTROL_TIER_BYTES;
1982        let blind = crate::ScanConfig { read_controls: false, ..lifted.clone() };
1983        let forge = |tier: ControlTierIdentity| {
1984            let mut forged = saved.clone();
1985            forged[controls.clone()].copy_from_slice(&tier.encode());
1986            rewrite_checksum(&mut forged);
1987            fs::write(&path, &forged).expect("write forged limits");
1988            (
1989                load(&path).expect("forged equals absent"),
1990                load_serving(&path, blind.types_shared(), blind.snapshot_identity())
1991                    .expect("projected forged equals absent"),
1992            )
1993        };
1994        let tight = crate::control::ControlLimits { line_limit: Some(1), ..lifted.control_limits };
1995        let (exact, projected) = forge(ControlTierIdentity::Observed { limits: tight });
1996        assert!(exact.is_none());
1997        assert!(matches!(projected, LoadOutcome::Absent));
1998        let (exact, projected) = forge(ControlTierIdentity::NotObserved);
1999        assert!(exact.is_none());
2000        assert!(matches!(projected, LoadOutcome::Absent));
2001        // The same splice with the saved tier restores, so the splice is not what failed.
2002        let (exact, projected) = forge(lifted.control_identity());
2003        assert!(exact.is_some());
2004        let LoadOutcome::Served { index: projected, stored } = projected else {
2005            panic!("the valid control payload projects");
2006        };
2007        assert_eq!(serves_snapshot(stored, blind.snapshot_identity()), Serves::ProjectControlsOff);
2008        assert_eq!(projected.snapshot_identity(), blind.snapshot_identity());
2009        assert_eq!(projected.scope(), blind.scope());
2010        assert!(matches!(projected.controls(), Err(Error::ControlStateNotObserved)));
2011    }
2012
2013    /// An index whose control table refused some sources, and its path-ordered tail.
2014    fn index_with_refused_controls() -> Index {
2015        let mut index =
2016            Index::new_with_scope("/some/root", crate::test_support::observing_controls());
2017        let mut line = vec![b'x'; crate::control::DEFAULT_CONTROL_LINE_LIMIT + 1];
2018        line.push(b'\n');
2019        index.apply_ok(&Observation::new(vec![
2020            Op::Upsert {
2021                path: PathBuf::from(".gitignore"),
2022                kind: EntryKind::File,
2023                attrs: attrs(6, 1),
2024            },
2025            Op::Upsert { path: PathBuf::from("a"), kind: EntryKind::Dir, attrs: attrs(0, 1) },
2026            Op::Upsert {
2027                path: PathBuf::from("a/.gitignore"),
2028                kind: EntryKind::File,
2029                attrs: attrs(1, 1),
2030            },
2031            Op::Upsert { path: PathBuf::from("b"), kind: EntryKind::Dir, attrs: attrs(0, 1) },
2032            Op::Upsert {
2033                path: PathBuf::from("b/.gitignore"),
2034                kind: EntryKind::File,
2035                attrs: attrs(1, 1),
2036            },
2037            Op::Upsert {
2038                path: PathBuf::from("b/debug.log"),
2039                kind: EntryKind::File,
2040                attrs: attrs(9, 1),
2041            },
2042            Op::ControlUpsert { path: PathBuf::from(".gitignore"), source: b"*.log\n".to_vec() },
2043            Op::ControlUpsert { path: PathBuf::from("a/.gitignore"), source: line },
2044            Op::ControlUpsert {
2045                path: PathBuf::from("b/.gitignore"),
2046                source: b"y\n".repeat(crate::control::DEFAULT_CONTROL_BUDGET / 2),
2047            },
2048        ]));
2049        index
2050    }
2051
2052    /// A snapshot of an index that refused control files reloads with the same coverage,
2053    /// rather than as an index whose every rule applied.
2054    #[test]
2055    fn a_snapshot_with_refused_controls_reloads_its_coverage_exactly() {
2056        let dir = tempfile::tempdir().expect("tempdir");
2057        let path = dir.path().join("refused.fdu");
2058        let original = index_with_refused_controls();
2059        let crate::control::ControlCoverage::Observed(coverage) = original.control_coverage()
2060        else {
2061            panic!("observed");
2062        };
2063        assert_eq!((coverage.applied, coverage.refused), (1, 2));
2064
2065        save(&original, &path).expect("save a partially covered index");
2066        let restored = load(&path).expect("load").expect("snapshot present");
2067
2068        assert_eq!(restored.control_coverage(), original.control_coverage());
2069        assert_eq!(
2070            restored.is_ignored(Path::new("b/debug.log")).expect("observed"),
2071            original.is_ignored(Path::new("b/debug.log")).expect("observed")
2072        );
2073        assert_eq!(
2074            restored.partition_total().expect("observed"),
2075            original.partition_total().expect("observed")
2076        );
2077    }
2078
2079    /// The refusal records are parsed as strictly as the rest: an unknown reason, and a
2080    /// refusal naming a retained source, are corrupt.
2081    #[test]
2082    fn corrupt_refusal_records_fail_closed() {
2083        let dir = tempfile::tempdir().expect("tempdir");
2084        let path = dir.path().join("refused.fdu");
2085        save(&index_with_refused_controls(), &path).expect("save");
2086        let saved = fs::read(&path).expect("read snapshot");
2087        let footer = CHECKSUM_BYTES + TRAILER.len();
2088        // The last refusal is `b/.gitignore` and its one-byte reason ends the payload.
2089        let reason_at = saved.len() - footer - 1;
2090        assert_eq!(saved[reason_at], REFUSED_FOR_BUDGET);
2091
2092        let mut unknown_reason = saved;
2093        unknown_reason[reason_at] = 9;
2094        rewrite_checksum(&mut unknown_reason);
2095        fs::write(&path, &unknown_reason).expect("write corrupt reason");
2096        assert!(load(&path).expect("corrupt equals absent").is_none());
2097
2098        // A refusal naming a retained source, or one already refused, is corrupt, and so is a
2099        // refusal by a limit the section records as unbounded, which refuses nothing. Built
2100        // with the platform's own path encoding, so the same bytes mean the same on every
2101        // target.
2102        let defaults = crate::control::ControlLimits::default();
2103        let section = |refusals: &[(&str, u8)]| {
2104            let mut section = Vec::new();
2105            section.extend_from_slice(&1_u32.to_le_bytes());
2106            put_os_str(&mut section, OsStr::new(".gitignore")).expect("retained path");
2107            section.extend_from_slice(&6_u32.to_le_bytes());
2108            section.extend_from_slice(b"*.log\n");
2109            let count = u32::try_from(refusals.len()).expect("few refusals");
2110            section.extend_from_slice(&count.to_le_bytes());
2111            for (refusal, reason) in refusals {
2112                put_os_str(&mut section, OsStr::new(refusal)).expect("refused path");
2113                section.push(*reason);
2114            }
2115            section
2116        };
2117        let no_budget = crate::control::ControlLimits { budget: None, ..defaults };
2118        let no_line_limit = crate::control::ControlLimits { line_limit: None, ..defaults };
2119        for (limits, refusals) in [
2120            (defaults, &[("a/.gitignore", REFUSED_FOR_BUDGET)][..]),
2121            (no_budget, &[("a/.gitignore", REFUSED_FOR_LINE_LIMIT)][..]),
2122            (no_line_limit, &[("a/.gitignore", REFUSED_FOR_BUDGET)][..]),
2123        ] {
2124            let tier = ControlTierIdentity::Observed { limits };
2125            let valid = read_controls(&mut section(refusals).as_slice(), tier)
2126                .expect("a well-formed control section");
2127            assert_eq!((valid.len(), valid.refused_len()), (1, 1));
2128        }
2129        for (limits, refusals) in [
2130            (defaults, &[(".gitignore", REFUSED_FOR_BUDGET)][..]),
2131            (
2132                defaults,
2133                &[("a/.gitignore", REFUSED_FOR_BUDGET), ("a/.gitignore", REFUSED_FOR_BUDGET)][..],
2134            ),
2135            (no_budget, &[("a/.gitignore", REFUSED_FOR_BUDGET)][..]),
2136            (no_line_limit, &[("a/.gitignore", REFUSED_FOR_LINE_LIMIT)][..]),
2137        ] {
2138            let tier = ControlTierIdentity::Observed { limits };
2139            assert!(
2140                matches!(
2141                    read_controls(&mut section(refusals).as_slice(), tier),
2142                    Err(ParseError::Invalid)
2143                ),
2144                "{limits:?} {refusals:?}"
2145            );
2146        }
2147        // A tier that observed nothing holds no source and no refusal.
2148        for refusals in [&[][..], &[("a/.gitignore", REFUSED_FOR_BUDGET)][..]] {
2149            assert!(matches!(
2150                read_controls(&mut section(refusals).as_slice(), ControlTierIdentity::NotObserved),
2151                Err(ParseError::Invalid)
2152            ));
2153        }
2154    }
2155
2156    /// A snapshot with no control state loads without walking the tree to reclassify it.
2157    ///
2158    /// Installing the loaded control table re-evaluated every entry's ignored bit, allocating
2159    /// a path per entry, even when the table was empty and no bit could move. That is every
2160    /// warm open of a tree with no `.gitignore`, and it scaled with the tree.
2161    #[test]
2162    fn loading_a_snapshot_without_controls_skips_the_reclassification_walk() {
2163        fn visits_while_loading(path: &Path) -> (u64, Index) {
2164            crate::index::RECLASSIFY_VISITS.with(|visits| visits.set(0));
2165            let restored = load(path).expect("load").expect("snapshot present");
2166            (crate::index::RECLASSIFY_VISITS.with(std::cell::Cell::get), restored)
2167        }
2168
2169        let dir = tempfile::tempdir().expect("tempdir");
2170        let mut original =
2171            Index::new_with_scope("/some/root", crate::test_support::observing_controls());
2172        let mut ops = vec![Op::Upsert {
2173            path: PathBuf::from("src"),
2174            kind: EntryKind::Dir,
2175            attrs: attrs(0, 1),
2176        }];
2177        ops.extend((0..64).map(|sequence| Op::Upsert {
2178            path: PathBuf::from(format!("src/file-{sequence:02}.rs")),
2179            kind: EntryKind::File,
2180            attrs: attrs(sequence + 1, 2),
2181        }));
2182        original.apply_baseline_ok(&Observation::new(ops));
2183        let plain = dir.path().join("plain.fdu");
2184        save(&original, &plain).expect("save");
2185
2186        let (visits, restored) = visits_while_loading(&plain);
2187
2188        assert_eq!(visits, 0, "an empty control table has nothing to reclassify");
2189        assert!(restored.control_table().is_empty());
2190        // The load still answers as the saved index did.
2191        assert_eq!(
2192            restored.is_ignored(Path::new("src/file-00.rs")).expect("control state observed"),
2193            Some(false)
2194        );
2195        assert_eq!(
2196            restored.partition_total().expect("control state observed"),
2197            original.partition_total().expect("control state observed")
2198        );
2199
2200        // The probe sees the walk when there is something to walk for.
2201        original.apply_ok(&Observation::new(vec![Op::ControlUpsert {
2202            path: PathBuf::from("src/.gitignore"),
2203            source: b"file-0*.rs\n".to_vec(),
2204        }]));
2205        let controlled = dir.path().join("controlled.fdu");
2206        save(&original, &controlled).expect("save");
2207
2208        let (visits, restored) = visits_while_loading(&controlled);
2209
2210        assert!(visits > 64, "{visits}");
2211        assert_eq!(
2212            restored.is_ignored(Path::new("src/file-00.rs")).expect("control state observed"),
2213            Some(true)
2214        );
2215        assert_eq!(
2216            restored.is_ignored(Path::new("src/file-10.rs")).expect("control state observed"),
2217            Some(false)
2218        );
2219    }
2220
2221    #[test]
2222    fn round_trip_handles_wide_directory_fanout() {
2223        // Load resolves each record's id from its parent. Doing that by scanning the
2224        // parent's children costs O(width^2) per directory, which is invisible on the
2225        // handful of siblings every other test uses and dominates a real one — a
2226        // node_modules or a Maildir is thousands wide.
2227        const CHILDREN: u64 = 4_096;
2228        let dir = tempfile::tempdir().expect("tempdir");
2229        let path = dir.path().join("wide.fdu");
2230        let mut original = Index::new("/some/root");
2231        let ops = (0..CHILDREN)
2232            .map(|sequence| Op::Upsert {
2233                path: PathBuf::from(format!("child-{sequence:04}.dat")),
2234                kind: EntryKind::File,
2235                attrs: attrs(sequence + 1, i64::try_from(sequence).expect("fanout fits i64")),
2236            })
2237            .collect();
2238        original.apply_baseline_ok(&Observation::new(ops));
2239
2240        save(&original, &path).expect("save wide snapshot");
2241        let restored = load(&path).expect("load wide snapshot").expect("snapshot present");
2242
2243        assert_eq!(restored.total().files, CHILDREN);
2244        assert_eq!(restored.len(), original.len());
2245        // The last child is the one a linear scan reaches last, so it is the one that
2246        // proves the resolution used the parent's map rather than its sibling order.
2247        assert_eq!(
2248            restored.attrs(Path::new("child-4095.dat")),
2249            original.attrs(Path::new("child-4095.dat"))
2250        );
2251    }
2252
2253    #[test]
2254    fn shared_save_captures_before_filesystem_io() {
2255        let dir = tempfile::tempdir().expect("tempdir");
2256        let path = dir.path().join("shared.fdu");
2257        let handle = IndexHandle::new(sample_index());
2258
2259        save_handle(&handle, &path).expect("save shared snapshot");
2260        handle
2261            .apply(&Observation::new(vec![Op::Upsert {
2262                path: PathBuf::from("after.txt"),
2263                kind: EntryKind::File,
2264                attrs: attrs(9, 40),
2265            }]))
2266            .expect("mutate after capture");
2267
2268        let restored = load(&path).expect("load").expect("snapshot present");
2269        assert!(restored.lookup(Path::new("after.txt")).is_none());
2270        assert!(handle.kind(Path::new("after.txt")).expect("query").is_some());
2271    }
2272
2273    #[test]
2274    fn missing_snapshot_is_absent_not_an_error() {
2275        let dir = tempfile::tempdir().expect("tempdir");
2276        let loaded = load(&dir.path().join("nope.fdu")).expect("load must not error");
2277        assert!(loaded.is_none());
2278    }
2279
2280    #[test]
2281    fn configured_size_limit_rejects_before_body_allocation() {
2282        let dir = tempfile::tempdir().expect("tempdir");
2283        let path = dir.path().join("snap.fdu");
2284        save(&sample_index(), &path).expect("save");
2285        let file_len = fs::metadata(&path).expect("metadata").len();
2286
2287        assert!(load_with_size_limit(&path, file_len - 1).expect("load must not error").is_none());
2288    }
2289
2290    #[test]
2291    fn foreign_file_is_treated_as_absent() {
2292        let dir = tempfile::tempdir().expect("tempdir");
2293        let path = dir.path().join("snap.fdu");
2294        fs::write(&path, b"this is not a snapshot at all").expect("write");
2295        assert!(load(&path).expect("load must not error").is_none());
2296    }
2297
2298    #[test]
2299    fn truncated_snapshot_is_treated_as_absent() {
2300        let dir = tempfile::tempdir().expect("tempdir");
2301        let path = dir.path().join("snap.fdu");
2302        save(&sample_index(), &path).expect("save");
2303
2304        let full = fs::read(&path).expect("read");
2305        for cut in [full.len() - 1, full.len() / 2, MAGIC.len() + 2] {
2306            fs::write(&path, &full[..cut]).expect("truncate");
2307            assert!(
2308                load(&path).expect("load must not error").is_none(),
2309                "a snapshot truncated to {cut} bytes must not parse"
2310            );
2311        }
2312    }
2313
2314    #[test]
2315    fn corrupt_body_with_intact_magic_and_trailer_is_rejected() {
2316        let dir = tempfile::tempdir().expect("tempdir");
2317        let path = dir.path().join("snap.fdu");
2318        save(&sample_index(), &path).expect("save");
2319
2320        let mut bytes = fs::read(&path).expect("read");
2321        // Claim far more entries than the body holds.
2322        let count_at = entry_count_offset(&bytes);
2323        bytes[count_at..count_at + 8].copy_from_slice(&u64::MAX.to_le_bytes());
2324        rewrite_checksum(&mut bytes);
2325        fs::write(&path, &bytes).expect("write");
2326
2327        assert!(load(&path).expect("load must not error").is_none());
2328    }
2329
2330    #[test]
2331    fn plausible_attribute_corruption_is_rejected() {
2332        let dir = tempfile::tempdir().expect("tempdir");
2333        let path = dir.path().join("snap.fdu");
2334        save(&sample_index(), &path).expect("save");
2335
2336        let mut bytes = fs::read(&path).expect("read");
2337        let records_at = entry_count_offset(&bytes) + 8;
2338        // The root record has an empty name, so its first attribute starts immediately
2339        // after parent, kind, and encoded-name length. Changing a low size byte keeps the
2340        // image structurally valid and would silently alter the restored state without an
2341        // integrity check.
2342        let root_size_at = records_at + 4 + 1 + 4;
2343        bytes[root_size_at] ^= 1;
2344        fs::write(&path, &bytes).expect("write");
2345
2346        assert!(load(&path).expect("load must not error").is_none());
2347    }
2348
2349    #[test]
2350    fn entry_names_with_path_components_are_rejected() {
2351        let dir = tempfile::tempdir().expect("tempdir");
2352        let path = dir.path().join("snap.fdu");
2353        save(&sample_index(), &path).expect("save");
2354
2355        let mut bytes = fs::read(&path).expect("read");
2356        let count_at = entry_count_offset(&bytes);
2357        let records_at = count_at + 8;
2358        let first_child_name_len_at = records_at + MIN_RECORD_BYTES + 4 + 1;
2359        let old_name_len = u32::from_le_bytes(
2360            bytes[first_child_name_len_at..first_child_name_len_at + 4]
2361                .try_into()
2362                .expect("saved snapshot has a child name length"),
2363        );
2364        let old_name_end = first_child_name_len_at
2365            + 4
2366            + usize::try_from(old_name_len).expect("name length fits usize");
2367        let mut invalid_name = Vec::new();
2368        put_os_str(&mut invalid_name, OsStr::new("../bad")).expect("encode invalid name");
2369        bytes.splice(first_child_name_len_at..old_name_end, invalid_name);
2370        rewrite_checksum(&mut bytes);
2371        fs::write(&path, &bytes).expect("write");
2372
2373        assert!(load(&path).expect("load must not error").is_none());
2374    }
2375
2376    #[test]
2377    fn oversized_declared_path_is_rejected_before_allocation() {
2378        let dir = tempfile::tempdir().expect("tempdir");
2379        let path = dir.path().join("snap.fdu");
2380        save(&sample_index(), &path).expect("save");
2381
2382        let mut bytes = fs::read(&path).expect("read");
2383        let root_len_at = ROOT_OFFSET;
2384        bytes[root_len_at..root_len_at + 4].copy_from_slice(&(MAX_PATH_BYTES + 1).to_le_bytes());
2385        rewrite_checksum(&mut bytes);
2386        fs::write(&path, &bytes).expect("write");
2387
2388        assert!(load(&path).expect("load must not error").is_none());
2389    }
2390
2391    #[test]
2392    fn engine_fingerprint_mismatch_discards_the_snapshot() {
2393        let dir = tempfile::tempdir().expect("tempdir");
2394        let path = dir.path().join("snap.fdu");
2395        save(&sample_index(), &path).expect("save");
2396
2397        let mut bytes = fs::read(&path).expect("read");
2398        let fp_at = MAGIC.len() + 4;
2399        bytes[fp_at] ^= 0xff;
2400        rewrite_checksum(&mut bytes);
2401        fs::write(&path, &bytes).expect("write");
2402
2403        assert!(load(&path).expect("load must not error").is_none());
2404    }
2405
2406    #[test]
2407    fn save_replaces_an_existing_snapshot_and_leaves_no_temp_files() {
2408        let dir = tempfile::tempdir().expect("tempdir");
2409        let path = dir.path().join("snap.fdu");
2410
2411        save(&sample_index(), &path).expect("first save");
2412        let mut smaller = Index::new("/some/root");
2413        smaller.apply_ok(&Observation::new(vec![Op::Upsert {
2414            path: PathBuf::from("only.txt"),
2415            kind: EntryKind::File,
2416            attrs: attrs(1, 1),
2417        }]));
2418        save(&smaller, &path).expect("second save");
2419
2420        let restored = load(&path).expect("load").expect("present");
2421        assert_eq!(restored.total().files, 1);
2422
2423        let leftovers: Vec<_> = fs::read_dir(dir.path())
2424            .expect("read_dir")
2425            .filter_map(std::result::Result::ok)
2426            .map(|e| e.file_name().to_string_lossy().into_owned())
2427            .filter(|name| name != "snap.fdu")
2428            .collect();
2429        assert!(leftovers.is_empty(), "temp files left behind: {leftovers:?}");
2430    }
2431
2432    #[test]
2433    fn an_abandoned_temporary_is_collected_once_it_is_old_enough() {
2434        // Per-process entropy means no future writer will ever generate a corpse's
2435        // name again, so nothing would otherwise reclaim it: unique names turn an
2436        // occasional collision into permanent litter, one whole snapshot image at a
2437        // time. The reaper closes that, and the age threshold is what makes it safe
2438        // without a liveness check — pid-based liveness is unportable and wrong under
2439        // pid reuse.
2440        //
2441        // The threshold is a parameter so both sides are testable without setting
2442        // mtimes, which the standard library cannot do and which is not worth a
2443        // dependency.
2444        let dir = tempfile::tempdir().expect("tempdir");
2445        let path = dir.path().join("snap.fdu");
2446        let prefix = temp_prefix(&path).expect("a named target has a prefix");
2447
2448        let mut corpse = prefix.clone();
2449        corpse.push("111.0123456789abcdef.7");
2450        let unrelated = OsString::from("notes.txt");
2451        fs::write(dir.path().join(&corpse), b"a writer killed long ago").expect("plant");
2452        fs::write(dir.path().join(&unrelated), b"not ours").expect("plant");
2453
2454        // The shipped threshold spares anything that could still be in flight.
2455        reap_stale_temporaries(dir.path(), &path, STALE_TEMP_AGE);
2456        assert!(dir.path().join(&corpse).exists(), "a fresh corpse must be left alone");
2457
2458        // Past the threshold it is collected, and only it.
2459        reap_stale_temporaries(dir.path(), &path, std::time::Duration::ZERO);
2460        assert!(!dir.path().join(&corpse).exists(), "an old corpse must be collected");
2461        assert!(dir.path().join(&unrelated).exists(), "unrelated files are never touched");
2462    }
2463
2464    #[test]
2465    fn the_reaper_only_matches_its_own_targets_temporaries() {
2466        // The prefix carries the target's file name, so two snapshots sharing a
2467        // directory cannot collect each other's work in progress.
2468        let mine = temp_prefix(Path::new("/cache/snap.fdu")).expect("prefix");
2469        let theirs = temp_prefix(Path::new("/cache/other.fdu")).expect("prefix");
2470        assert_eq!(mine, OsString::from(".snap.fdu.tmp."));
2471        assert_ne!(mine, theirs);
2472        assert!(temp_prefix(Path::new("/")).is_none(), "a rootless path has no name");
2473    }
2474
2475    #[test]
2476    fn a_stale_temporary_does_not_block_a_later_write() {
2477        // A killed writer leaves its temporary behind and nothing reaps it until it is
2478        // a day old, so a later write can find a corpse sitting exactly where it wants
2479        // to go. `O_EXCL` plus the retry loop is what makes that safe: the collision is
2480        // detected and stepped over, never shared.
2481        //
2482        // The corpse is planted at names this process will really try, via `temp_name`.
2483        // An earlier version of this test guessed the shape and planted
2484        // `.snap.fdu.tmp.{pid}.0`, which the entropy in the name means the writer can
2485        // never generate — so it collided with nothing and proved nothing, while
2486        // reading as though it had.
2487        let dir = tempfile::tempdir().expect("tempdir");
2488        let path = dir.path().join("snap.fdu");
2489
2490        // Block a window of upcoming sequence values. A window rather than one name
2491        // because the counter is process-global and other tests draw from it too;
2492        // whichever value this write lands on, it starts inside the blocked range.
2493        let first = NEXT_TEMP_FILE.load(Ordering::Relaxed);
2494        let planted: Vec<OsString> =
2495            (first..first + 32).map(|sequence| temp_name(&path, sequence)).collect();
2496        for name in &planted {
2497            fs::write(dir.path().join(name), b"corpse").expect("plant a stale temporary");
2498        }
2499
2500        write_atomically(&path, b"payload").expect("write past the stale temporaries");
2501        assert_eq!(fs::read(&path).expect("read back"), b"payload");
2502
2503        // Every corpse survives: the writer stepped over them rather than reusing or
2504        // removing one, and they are far too young for the reaper.
2505        let mut survivors: Vec<_> = fs::read_dir(dir.path())
2506            .expect("read_dir")
2507            .filter_map(std::result::Result::ok)
2508            .map(|entry| entry.file_name())
2509            .filter(|name| name != "snap.fdu")
2510            .collect();
2511        survivors.sort();
2512        let mut expected = planted;
2513        expected.sort();
2514        assert_eq!(survivors, expected, "the write must step over corpses, not consume them");
2515    }
2516
2517    #[test]
2518    fn an_identical_payload_leaves_the_file_in_place() {
2519        // The default command encodes an unchanged tree to the bytes already on disk.
2520        // Replacing them was a full write, an fsync and a rename for nothing; now the
2521        // file is left alone, which on a filesystem is observable as the same inode.
2522        let dir = tempfile::tempdir().expect("tempdir");
2523        let path = dir.path().join("snap.fdu");
2524        let payload: Vec<u8> =
2525            (0u32..(3 << 20)).map(|i| u8::try_from(i % 251).unwrap_or(0)).collect();
2526
2527        write_atomically(&path, &payload).expect("first write");
2528        let first = fs::metadata(&path).expect("metadata");
2529        let before = std::time::SystemTime::now();
2530        write_atomically(&path, &payload).expect("identical write");
2531        let second = fs::metadata(&path).expect("metadata");
2532
2533        assert_eq!(fs::read(&path).expect("read back"), payload);
2534        assert_same_file(&first, &second);
2535        // The "as of" moved even though the bytes did not: the loader reads the mtime
2536        // as every cached entry's observation time, and the tree was just verified.
2537        assert!(second.modified().expect("mtime") >= before);
2538        // No temporary was created and abandoned on the way to deciding not to write.
2539        let extras = fs::read_dir(dir.path())
2540            .expect("read_dir")
2541            .filter_map(std::result::Result::ok)
2542            .filter(|entry| entry.file_name() != "snap.fdu")
2543            .count();
2544        assert_eq!(extras, 0);
2545    }
2546
2547    #[test]
2548    fn a_different_payload_of_the_same_length_is_written() {
2549        // Length is the cheap pre-check, not the decision: two snapshots of the same
2550        // tree with one mtime changed are the same length and must not be confused.
2551        let dir = tempfile::tempdir().expect("tempdir");
2552        let path = dir.path().join("snap.fdu");
2553        write_atomically(&path, b"payload-a").expect("first write");
2554        write_atomically(&path, b"payload-b").expect("second write");
2555        assert_eq!(fs::read(&path).expect("read back"), b"payload-b");
2556
2557        // And a file that holds the payload as a prefix is not the payload.
2558        fs::write(&path, b"payload-b-and-more").expect("lengthen");
2559        assert!(!same_bytes_on_disk(&path, b"payload-b"));
2560        assert!(!same_bytes_on_disk(&path, b"payload-b-and-mor"));
2561        assert!(same_bytes_on_disk(&path, b"payload-b-and-more"));
2562        assert!(!same_bytes_on_disk(&dir.path().join("absent"), b""));
2563    }
2564
2565    #[cfg(unix)]
2566    fn assert_same_file(first: &fs::Metadata, second: &fs::Metadata) {
2567        use std::os::unix::fs::MetadataExt;
2568        assert_eq!(first.ino(), second.ino(), "the snapshot was replaced, not left in place");
2569    }
2570
2571    #[cfg(not(unix))]
2572    fn assert_same_file(first: &fs::Metadata, second: &fs::Metadata) {
2573        // Creation time survives a skipped write and changes across a rename of a fresh
2574        // temporary, on the platforms that report it.
2575        if let (Ok(a), Ok(b)) = (first.created(), second.created()) {
2576            assert_eq!(a, b, "the snapshot was replaced, not left in place");
2577        }
2578    }
2579
2580    #[test]
2581    fn a_bare_filename_target_resolves_to_an_openable_directory() {
2582        // `Path::parent` returns `Some("")` for a bare name, not `None`, so an
2583        // `unwrap_or(".")` fallback never fires. The write still worked — `rename`
2584        // accepts the empty path — but `read_dir("")` fails, so the reaper returned
2585        // immediately and collected nothing for those targets.
2586        assert_eq!(parent_dir(Path::new("snap.fdu")), Path::new("."));
2587        assert!(fs::read_dir(parent_dir(Path::new("snap.fdu"))).is_ok(), "must be openable");
2588        assert_eq!(parent_dir(Path::new("/cache/snap.fdu")), Path::new("/cache"));
2589        assert_eq!(parent_dir(Path::new("cache/snap.fdu")), Path::new("cache"));
2590    }
2591
2592    #[test]
2593    fn a_temporary_name_is_unique_per_sequence_and_carries_process_entropy() {
2594        // The two components that make a corpse unreachable by a future process, and a
2595        // writer unable to collide with itself.
2596        let path = Path::new("/cache/snap.fdu");
2597        assert_ne!(temp_name(path, 0), temp_name(path, 1), "the counter separates files");
2598        let name = temp_name(path, 0).to_string_lossy().into_owned();
2599        assert!(name.starts_with(".snap.fdu.tmp."), "reaper prefix must match: {name}");
2600        assert!(
2601            name.contains(&format!("{:016x}", *TEMP_FILE_ENTROPY)),
2602            "entropy must be in the name, or a recycled pid regenerates a corpse: {name}"
2603        );
2604    }
2605
2606    #[test]
2607    fn concurrent_atomic_writes_do_not_share_a_temporary_file() {
2608        use std::sync::{Arc, Barrier};
2609
2610        const WRITERS: usize = 8;
2611        const PAYLOAD_BYTES: usize = 1024 * 1024;
2612
2613        let dir = tempfile::tempdir().expect("tempdir");
2614        let path = dir.path().join("snap.fdu");
2615        let barrier = Arc::new(Barrier::new(WRITERS));
2616        let results = std::thread::scope(|scope| {
2617            let handles: Vec<_> = (0..WRITERS)
2618                .map(|writer| {
2619                    let barrier = Arc::clone(&barrier);
2620                    let path = path.clone();
2621                    scope.spawn(move || {
2622                        let byte = u8::try_from(writer + 1).expect("writer id fits");
2623                        let bytes = vec![byte; PAYLOAD_BYTES];
2624                        barrier.wait();
2625                        write_atomically(&path, &bytes)
2626                    })
2627                })
2628                .collect();
2629            handles.into_iter().map(std::thread::ScopedJoinHandle::join).collect::<Vec<_>>()
2630        });
2631
2632        for result in results {
2633            result.expect("writer thread did not panic").expect("concurrent atomic write");
2634        }
2635        let final_bytes = fs::read(&path).expect("read final image");
2636        assert_eq!(final_bytes.len(), PAYLOAD_BYTES);
2637        assert!(final_bytes.iter().all(|byte| *byte == final_bytes[0]));
2638
2639        let leftovers: Vec<_> = fs::read_dir(dir.path())
2640            .expect("read_dir")
2641            .filter_map(std::result::Result::ok)
2642            .map(|entry| entry.file_name())
2643            .filter(|name| name != "snap.fdu")
2644            .collect();
2645        assert!(leftovers.is_empty(), "temp files left behind: {leftovers:?}");
2646    }
2647
2648    #[test]
2649    fn concurrent_snapshot_reader_sees_only_a_complete_old_or_new_image() {
2650        use std::sync::{Arc, Barrier};
2651
2652        let dir = tempfile::tempdir().expect("tempdir");
2653        let path = dir.path().join("snap.fdu");
2654        let mut old_index = Index::new("/some/root");
2655        old_index.apply_ok(&Observation::new(vec![Op::Upsert {
2656            path: PathBuf::from("old.txt"),
2657            kind: EntryKind::File,
2658            attrs: attrs(11, 1),
2659        }]));
2660        let mut new_index = Index::new("/some/root");
2661        new_index.apply_ok(&Observation::new(vec![
2662            Op::Upsert {
2663                path: PathBuf::from("new-a.txt"),
2664                kind: EntryKind::File,
2665                attrs: attrs(20, 2),
2666            },
2667            Op::Upsert {
2668                path: PathBuf::from("new-b.txt"),
2669                kind: EntryKind::File,
2670                attrs: attrs(30, 3),
2671            },
2672        ]));
2673        save(&old_index, &path).expect("save old image");
2674
2675        let start: Arc<Barrier> = Arc::new(Barrier::new(2));
2676        let (done_tx, done_rx) = std::sync::mpsc::sync_channel(1);
2677        std::thread::scope(|scope| {
2678            let writer_start: Arc<Barrier> = Arc::clone(&start);
2679            let writer_path: PathBuf = path.clone();
2680            scope.spawn(move || {
2681                writer_start.wait();
2682                done_tx.send(save(&new_index, &writer_path)).expect("report snapshot write");
2683            });
2684
2685            let reader_start: Arc<Barrier> = Arc::clone(&start);
2686            let reader_path: PathBuf = path.clone();
2687            scope.spawn(move || {
2688                reader_start.wait();
2689                let deadline: std::time::Instant =
2690                    std::time::Instant::now() + std::time::Duration::from_secs(10);
2691                loop {
2692                    assert!(
2693                        std::time::Instant::now() < deadline,
2694                        "snapshot replacement did not finish before the deadline"
2695                    );
2696                    let image: Index =
2697                        load(&reader_path).expect("load during replacement").expect("image");
2698                    match image.total().files {
2699                        1 => {
2700                            assert_eq!(image.total().bytes, 11);
2701                            assert!(image.lookup(Path::new("old.txt")).is_some());
2702                            assert!(image.lookup(Path::new("new-a.txt")).is_none());
2703                        }
2704                        2 => {
2705                            assert_eq!(image.total().bytes, 50);
2706                            assert!(image.lookup(Path::new("old.txt")).is_none());
2707                            assert!(image.lookup(Path::new("new-a.txt")).is_some());
2708                            assert!(image.lookup(Path::new("new-b.txt")).is_some());
2709                        }
2710                        partial => panic!("reader observed partial image with {partial} files"),
2711                    }
2712
2713                    match done_rx.try_recv() {
2714                        Ok(result) => {
2715                            result.expect("replace snapshot");
2716                            break;
2717                        }
2718                        Err(std::sync::mpsc::TryRecvError::Empty) => {}
2719                        Err(std::sync::mpsc::TryRecvError::Disconnected) => {
2720                            panic!("snapshot writer disconnected")
2721                        }
2722                    }
2723                }
2724            });
2725        });
2726
2727        let final_image: Index = load(&path).expect("load final").expect("final image");
2728        assert_eq!(final_image.total().files, 2);
2729        assert_eq!(final_image.total().bytes, 50);
2730    }
2731
2732    #[cfg(unix)]
2733    #[test]
2734    fn saved_snapshot_is_owner_readable_only() {
2735        use std::os::unix::fs::PermissionsExt;
2736
2737        let dir = tempfile::tempdir().expect("tempdir");
2738        let path = dir.path().join("snap.fdu");
2739        save(&sample_index(), &path).expect("save");
2740
2741        let mode = fs::metadata(&path).expect("metadata").permissions().mode() & 0o777;
2742        assert_eq!(mode, 0o600);
2743    }
2744
2745    #[test]
2746    fn empty_index_round_trips() {
2747        let dir = tempfile::tempdir().expect("tempdir");
2748        let path = dir.path().join("snap.fdu");
2749        let empty = Index::new("/some/root");
2750
2751        save(&empty, &path).expect("save");
2752        let restored = load(&path).expect("load").expect("present");
2753        assert!(restored.is_empty());
2754        assert_eq!(restored.total(), empty.total());
2755    }
2756
2757    #[test]
2758    fn partial_index_is_never_persisted() {
2759        let dir = tempfile::tempdir().expect("tempdir");
2760        let path = dir.path().join("partial.fdu");
2761        save(&Index::new("/root"), &path).expect("complete baseline");
2762        let complete_bytes = fs::read(&path).expect("read complete snapshot");
2763
2764        let mut index = Index::new("/root");
2765        index.set_initial_freshness(false);
2766
2767        assert!(matches!(save(&index, &path), Err(Error::Snapshot(_))));
2768        assert_eq!(fs::read(&path).expect("old snapshot remains"), complete_bytes);
2769    }
2770
2771    #[test]
2772    fn semantic_scan_scope_round_trips() {
2773        let dir = tempfile::tempdir().expect("tempdir");
2774        let path = dir.path().join("snap.fdu");
2775        let entries = crate::EntryTierIdentity {
2776            engine: engine_fingerprint(),
2777            scope: crate::EntryScope {
2778                max_depth: Some(7),
2779                follow_symlinks: false,
2780                one_filesystem: true,
2781                hidden_fingerprint: 5,
2782                exclude_special: true,
2783            },
2784            type_rules_fingerprint: crate::classify::type_rule_fingerprint(),
2785            reducers_fingerprint: 33,
2786        };
2787        // The default control limits, which the table of an index built with
2788        // `new_with_scope` enforces; any other limits are refused at save.
2789        for controls in [
2790            ControlTierIdentity::Observed { limits: crate::control::ControlLimits::default() },
2791            ControlTierIdentity::NotObserved,
2792        ] {
2793            let identity = SnapshotIdentity { entries, controls };
2794            let index = Index::new_with_scope("/some/root", identity.scan_scope());
2795
2796            save(&index, &path).expect("save");
2797            let restored = load(&path).expect("load").expect("present");
2798            assert_eq!(restored.snapshot_identity(), identity);
2799            assert_eq!(restored.scope(), identity.scan_scope());
2800            let info = read_header(&path).expect("read header").expect("current");
2801            assert_eq!(info.identity, identity);
2802            assert_eq!(info.scope(), identity.scan_scope());
2803        }
2804    }
2805
2806    /// Reading control files decides which entries are ignored, never which entries exist,
2807    /// so the snapshots of one tree with observation on and off record one entry tier, byte
2808    /// for byte, and differ only in the control tier.
2809    #[test]
2810    fn controls_on_and_off_snapshots_share_the_entry_identity() {
2811        let tree = tempfile::tempdir().expect("tree");
2812        fs::write(tree.path().join(".gitignore"), b"*.log\n").expect("write");
2813        fs::write(tree.path().join("debug.log"), b"log").expect("write");
2814        fs::create_dir(tree.path().join("src")).expect("mkdir");
2815        fs::write(tree.path().join("src/main.rs"), b"fn main() {}\n").expect("write");
2816        let dir = tempfile::tempdir().expect("tempdir");
2817        let observed = crate::ScanConfig::default();
2818        let blind = crate::ScanConfig { read_controls: false, ..observed.clone() };
2819
2820        let saved = [(&observed, "on.fdu"), (&blind, "off.fdu")].map(|(config, name)| {
2821            let (index, _) = crate::scan::scan_into_index(tree.path(), config).expect("scan");
2822            let path = dir.path().join(name);
2823            save(&index, &path).expect("save");
2824            let restored = load(&path).expect("load").expect("present");
2825            assert_eq!(restored.snapshot_identity(), config.snapshot_identity());
2826            (restored.snapshot_identity(), fs::read(&path).expect("read"))
2827        });
2828        let [(on, on_bytes), (off, off_bytes)] = saved;
2829
2830        assert_eq!(on.entries, off.entries);
2831        assert_ne!(on.controls, off.controls);
2832        let entry_tier = IDENTITY_OFFSET..IDENTITY_OFFSET + crate::stored_state::ENTRY_TIER_BYTES;
2833        let control_tier = entry_tier.end..ROOT_OFFSET;
2834        assert_eq!(on_bytes[entry_tier.clone()], off_bytes[entry_tier]);
2835        assert_ne!(on_bytes[control_tier.clone()], off_bytes[control_tier]);
2836
2837        let projected = load_serving(
2838            &dir.path().join("on.fdu"),
2839            blind.types_shared(),
2840            blind.snapshot_identity(),
2841        )
2842        .expect("load projection");
2843        let LoadOutcome::Served { index: projected, stored } = projected else {
2844            panic!("an observed snapshot should project to the blind request");
2845        };
2846        assert_eq!(serves_snapshot(stored, blind.snapshot_identity()), Serves::ProjectControlsOff);
2847        let (cold, _) = crate::scan::scan_into_index(tree.path(), &blind).expect("blind scan");
2848        assert_eq!(projected.snapshot_identity(), blind.snapshot_identity());
2849        assert_eq!(projected.scope(), blind.scope());
2850        assert_eq!(projected.len(), cold.len());
2851        assert_eq!(projected.total(), cold.total());
2852        assert!(matches!(projected.controls(), Err(Error::ControlStateNotObserved)));
2853        assert_eq!(
2854            projected.path_state(Path::new("debug.log")),
2855            cold.path_state(Path::new("debug.log"))
2856        );
2857
2858        let mut corrupt = on_bytes;
2859        corrupt[WRITING_PASS_STARTED_AT_OFFSET] ^= 1;
2860        fs::write(dir.path().join("on.fdu"), corrupt).expect("corrupt checksum");
2861        assert!(matches!(
2862            load_serving(
2863                &dir.path().join("on.fdu"),
2864                blind.types_shared(),
2865                blind.snapshot_identity(),
2866            )
2867            .expect("corrupt projection is absent"),
2868            LoadOutcome::Absent
2869        ));
2870    }
2871
2872    /// A format-4 snapshot, laid out as that format wrote it, is fdu's and stale: the
2873    /// prologue kept its offsets, so the cache lifecycle names its version rather than
2874    /// calling it foreign, and loading it is a miss.
2875    #[test]
2876    fn a_v4_snapshot_is_older_format() {
2877        const V4: u32 = 4;
2878        let dir = tempfile::tempdir().expect("tempdir");
2879        let path = dir.path().join("v4.fdu");
2880
2881        let mut image = MAGIC.to_vec();
2882        image.extend_from_slice(&V4.to_le_bytes());
2883        // The v4 engine fingerprint. Recognition reads the version before the engine, so a
2884        // format-4 image is older whatever these eight bytes hold.
2885        image.extend_from_slice(&0x4444_4444_4444_4444_u64.to_le_bytes());
2886        image.push(path_encoding());
2887        // The v4 scope: depth, flags, and the hidden, ignore-rules, type-rules, and reducer
2888        // fingerprints.
2889        image.extend_from_slice(&u64::MAX.to_le_bytes());
2890        image.push(0);
2891        let scope = crate::ScanConfig::default().scope();
2892        for fingerprint in [
2893            scope.hidden_fingerprint,
2894            scope.ignore_rules_fingerprint,
2895            scope.type_rules_fingerprint,
2896            scope.reducers_fingerprint,
2897        ] {
2898            image.extend_from_slice(&fingerprint.to_le_bytes());
2899        }
2900        put_os_str(&mut image, OsStr::new("/some/root")).expect("root");
2901        image.extend_from_slice(&1_u64.to_le_bytes());
2902        image.extend_from_slice(&NO_PARENT.to_le_bytes());
2903        image.push(EntryKind::Dir as u8);
2904        put_os_str(&mut image, OsStr::new("")).expect("root name");
2905        image.extend_from_slice(&[0; 6 * 8]);
2906        // The v4 control section: no sources, both default limits, no refusals.
2907        image.extend_from_slice(&0_u32.to_le_bytes());
2908        for limit in [crate::DEFAULT_CONTROL_BUDGET, crate::DEFAULT_CONTROL_LINE_LIMIT] {
2909            image.push(1);
2910            image.extend_from_slice(&u64::try_from(limit).expect("limit").to_le_bytes());
2911        }
2912        image.extend_from_slice(&0_u32.to_le_bytes());
2913        let checksum = crc32c(&image);
2914        image.extend_from_slice(&checksum.to_le_bytes());
2915        image.extend_from_slice(TRAILER);
2916        fs::write(&path, &image).expect("write v4 image");
2917
2918        assert!(matches!(
2919            identify(&path).expect("identify"),
2920            Some(Identity::Stale(crate::cache::StaleReason::OlderFormat { version: V4 }))
2921        ));
2922        assert!(read_header(&path).expect("read header").is_none());
2923        assert!(load(&path).expect("load").is_none());
2924        assert_eq!(
2925            crate::cache::cache_status(&path).expect("status").state,
2926            crate::cache::CacheState::Stale(crate::cache::StaleReason::OlderFormat { version: V4 })
2927        );
2928    }
2929
2930    /// A later pass over an unchanged tree encodes the same facts under a newer pass start.
2931    /// The image already on disk is kept, its stamp included, rather than rewritten for the
2932    /// stamp alone, and its mtime moves to the later pass's start, never backwards; a
2933    /// changed tree is written with the new pass's stamp.
2934    #[test]
2935    fn a_later_pass_over_the_same_facts_keeps_the_snapshot_in_place() {
2936        use std::time::{Duration, UNIX_EPOCH};
2937
2938        let dir = tempfile::tempdir().expect("tempdir");
2939        let path = dir.path().join("snap.fdu");
2940        let stamp_at = |bytes: &[u8]| {
2941            i64::from_le_bytes(
2942                bytes[WRITING_PASS_STARTED_AT_OFFSET..IDENTITY_OFFSET].try_into().expect("stamp"),
2943            )
2944        };
2945        let mtime = || fs::metadata(&path).expect("metadata").modified().expect("mtime");
2946        let nanos = |time: std::time::SystemTime| {
2947            i64::try_from(time.duration_since(UNIX_EPOCH).expect("after the epoch").as_nanos())
2948                .expect("nanoseconds")
2949        };
2950
2951        let mut first = sample_index();
2952        first.set_writing_pass_started_at_ns(1_000);
2953        save(&first, &path).expect("first save");
2954        let written = fs::metadata(&path).expect("metadata");
2955        assert_eq!(stamp_at(&fs::read(&path).expect("read")), 1_000);
2956        assert_eq!(
2957            load(&path).expect("load").expect("present").writing_pass_started_at_ns(),
2958            1_000
2959        );
2960
2961        // A pass that started after the image was written, on a whole second so every
2962        // filesystem's mtime holds the instant exactly.
2963        let later_started = UNIX_EPOCH
2964            + Duration::from_secs(
2965                mtime().duration_since(UNIX_EPOCH).expect("after the epoch").as_secs() + 1,
2966            );
2967        let mut later = sample_index();
2968        later.set_writing_pass_started_at_ns(nanos(later_started));
2969        save(&later, &path).expect("unchanged save");
2970        let kept = fs::metadata(&path).expect("metadata");
2971        assert_same_file(&written, &kept);
2972        assert_eq!(kept.modified().expect("mtime"), later_started, "the kept image's as-of");
2973        assert_eq!(stamp_at(&fs::read(&path).expect("read")), 1_000);
2974        assert!(load(&path).expect("load").is_some(), "the kept image is still valid");
2975
2976        // An equivalent save under an older stamp, as of an index loaded from an earlier
2977        // image, keeps the later mtime, which was already true of these facts.
2978        save(&first, &path).expect("older unchanged save");
2979        assert_same_file(&written, &fs::metadata(&path).expect("metadata"));
2980        assert_eq!(mtime(), later_started);
2981
2982        // A kept image must still verify: one whose own checksum is wrong reads as absent,
2983        // so it is replaced rather than kept under an older stamp forever.
2984        let mut corrupt = fs::read(&path).expect("read");
2985        let checksum_at = corrupt.len() - TRAILER.len() - CHECKSUM_BYTES;
2986        corrupt[checksum_at] ^= 0xff;
2987        fs::write(&path, &corrupt).expect("corrupt the checksum");
2988        assert!(load(&path).expect("load").is_none());
2989        save(&later, &path).expect("save over a corrupt image");
2990        assert_eq!(stamp_at(&fs::read(&path).expect("read")), nanos(later_started));
2991        assert!(load(&path).expect("load").is_some());
2992
2993        let mut changed = sample_index();
2994        changed.set_writing_pass_started_at_ns(3_000);
2995        changed.apply_ok(&Observation::new(vec![Op::Upsert {
2996            path: PathBuf::from("notes.md"),
2997            kind: EntryKind::File,
2998            attrs: attrs(8, 30),
2999        }]));
3000        save(&changed, &path).expect("changed save");
3001        let rewritten = fs::read(&path).expect("read");
3002        assert_eq!(stamp_at(&rewritten), 3_000);
3003        let restored = load(&path).expect("load").expect("present");
3004        assert_eq!(restored.writing_pass_started_at_ns(), 3_000);
3005        assert_eq!(restored.total(), changed.total());
3006    }
3007
3008    #[cfg(unix)]
3009    #[test]
3010    fn non_utf8_names_round_trip_without_aliasing() {
3011        use std::os::unix::ffi::OsStringExt;
3012
3013        let first = PathBuf::from(OsString::from_vec(vec![b'n', 0x80]));
3014        let second = PathBuf::from(OsString::from_vec(vec![b'n', 0x81]));
3015        let mut root = PathBuf::from("/some");
3016        root.push(OsString::from_vec(vec![b'r', 0x82]));
3017        let mut index = Index::new(&root);
3018        index.apply_baseline_ok(&Observation::new(vec![
3019            Op::Upsert { path: first.clone(), kind: EntryKind::File, attrs: attrs(10, 1) },
3020            Op::Upsert { path: second.clone(), kind: EntryKind::File, attrs: attrs(20, 2) },
3021        ]));
3022
3023        let dir = tempfile::tempdir().expect("tempdir");
3024        let path = dir.path().join("snap.fdu");
3025        save(&index, &path).expect("save");
3026        let restored = load(&path).expect("load").expect("present");
3027
3028        assert_eq!(restored.root_path(), root);
3029        assert_eq!(restored.total().files, 2);
3030        assert_eq!(restored.total().bytes, 30);
3031        assert!(restored.lookup(&first).is_some());
3032        assert!(restored.lookup(&second).is_some());
3033        assert!(!restored.serving_indexes_enabled());
3034    }
3035}