Skip to main content

mkit_core/
worktree.rs

1//! Worktree → tree-object builder.
2//!
3//! Walks a directory, applies `.mkitignore`, hashes each file as a
4//! [`Blob`](crate::object::Blob), recurses on subdirectories, validates
5//! symlink targets against path-traversal, and writes a single root
6//! [`Tree`] into the supplied [`ObjectStore`].
7//!
8//! Notes:
9//!
10//! - Files at or below [`CHUNK_THRESHOLD`] are stored as a single
11//!   [`Blob`](crate::object::Blob). Files above the threshold are
12//!   chunked with [`crate::chunker::FastCdc::v1`]; each chunk is
13//!   stored as a `Blob` and the file is represented by a
14//!   [`ChunkedBlob`] manifest whose hash
15//!   is what lands in the parent tree.
16//! - We never follow symlinks while walking. Linux/macOS `read_link`
17//!   reports the target verbatim and we hash it as a blob.
18
19use std::fs;
20use std::io::{self, Read};
21use std::path::{Path, PathBuf};
22
23use crate::chunker::{ChunkIterator, ChunkReader, FastCdc};
24use crate::hash::Hash;
25use crate::ignore::{self, IgnoreList};
26use crate::index::{self, Index};
27use crate::object::{ChunkedBlob, EntryMode, Object, Tree, TreeEntry};
28use crate::serialize;
29use crate::store::{ObjectSink, ObjectStore};
30
31/// Files larger than this go through the chunker (1 MiB).
32pub const CHUNK_THRESHOLD: u64 = 1024 * 1024;
33
34/// Hard cap on a single file (1 GiB).
35pub const MAX_FILE_BYTES: u64 = 1024 * 1024 * 1024;
36
37/// Errors returned by this module.
38#[derive(Debug, thiserror::Error)]
39pub enum WorktreeError {
40    /// `read_link` returned a target that fails [`validate_symlink_target`].
41    #[error("symlink target '{0}' is invalid (absolute or contains '..')")]
42    InvalidSymlinkTarget(String),
43    /// File exceeded [`MAX_FILE_BYTES`].
44    #[error("file '{0}' exceeds the {MAX_FILE_BYTES} byte limit")]
45    FileTooLarge(PathBuf),
46    /// Path component had non-UTF-8 bytes; tree entry names must be UTF-8.
47    #[error("path component is not valid UTF-8")]
48    InvalidUtf8,
49    /// Underlying I/O failure.
50    #[error(transparent)]
51    Io(#[from] io::Error),
52    /// Error encoding/serialising an object on its way into the store.
53    #[error(transparent)]
54    Object(#[from] crate::object::MkitError),
55    /// Error returned by the object store.
56    #[error(transparent)]
57    Store(#[from] crate::store::StoreError),
58    /// A `hash_chunks` callback passed to
59    /// [`store_large_file_streaming_with`]/[`hash_file_with_metadata_with`]
60    /// returned a different number of hashes than the batch it was given —
61    /// a contract violation by the caller (see those functions' docs),
62    /// never a normal runtime condition. Reported as a typed error rather
63    /// than a panic since the caller here is `mkit-cli`, a
64    /// long(er)-running process where a hard panic is a worse failure mode
65    /// than a clean, exit-coded error.
66    #[error("hash_chunks callback returned {actual} hashes for a {expected}-chunk batch")]
67    ChunkBatchLengthMismatch { expected: usize, actual: usize },
68}
69
70/// Result alias used throughout this module.
71pub type WorktreeResult<T> = Result<T, WorktreeError>;
72
73mod blob;
74pub use blob::{LoadedBlob, content_eq, content_eq_bytes, content_fingerprint, read_blob};
75
76/// Validate a symlink target: must be relative and contain no `..`
77/// segments.
78#[must_use]
79pub fn validate_symlink_target(target: &str) -> bool {
80    if target.is_empty() {
81        return false;
82    }
83    if target.starts_with('/') {
84        return false;
85    }
86    for part in target.split('/') {
87        if part == ".." {
88            return false;
89        }
90    }
91    true
92}
93
94/// A hash-time stat observation: while building a tree we re-hashed
95/// `path` (its cache was absent or racy-smudged) and the result equals
96/// the staging index's hash — so the stat captured from the OPENED file
97/// descriptor *before* its content was read proves the entry clean.
98/// `status` consumes these to heal the stat cache without ever pairing
99/// a post-verification stat with a pre-verification hash (the unsound
100/// verify-then-stat order).
101#[derive(Debug, Clone, PartialEq, Eq)]
102pub struct StatObservation {
103    /// Repo-relative path, `/`-separated (index path form).
104    pub path: String,
105    /// The content hash the re-hash produced (== the index entry's).
106    pub object_hash: Hash,
107    /// Stat fields captured from the opened fd before the read, in
108    /// [`stat_cache_fields`] order.
109    pub mtime_ns: u64,
110    pub size: u64,
111    pub ino: u64,
112    pub ctime_ns: u64,
113}
114
115/// Build a tree object for `dir` and its subdirectories. Honours the
116/// `.gitignore` + `.mkitignore` ignore files loaded from `dir`.
117///
118/// Ignore rules only exclude **untracked** content: a path that is tracked
119/// (or whose subtree holds tracked content) is always included even if it
120/// matches an ignore rule, so a tracked file matching `.gitignore` is never
121/// dropped from the worktree snapshot (which would misreport it as a deletion
122/// in status/diff). The staging index at `<dir>/.mkit/index` provides the
123/// tracked set; an absent index means nothing is tracked.
124///
125/// # Errors
126/// See [`WorktreeError`].
127pub fn build_tree<S: ObjectSink + Sync + ?Sized>(sink: &S, dir: &Path) -> WorktreeResult<Hash> {
128    build_tree_filtered(sink, dir, None)
129}
130
131/// Like [`build_tree`], but the caller supplies the authoritative tracked
132/// set (`index`). Callers that seed their index from `HEAD` when no index
133/// file exists yet (status, restore safety) MUST pass it here so a tracked
134/// file that matches an ignore rule is not dropped right after a checkout.
135/// `None` falls back to the on-disk `<dir>/.mkit/index` (empty if absent).
136///
137/// # Errors
138/// See [`WorktreeError`].
139pub fn build_tree_filtered<S: ObjectSink + Sync + ?Sized>(
140    sink: &S,
141    dir: &Path,
142    index: Option<&Index>,
143) -> WorktreeResult<Hash> {
144    build_tree_filtered_observed(sink, dir, index, &mut Vec::new())
145}
146
147/// [`build_tree_filtered`] that additionally reports every
148/// [`StatObservation`] (file re-hashed to a hash matching its index
149/// entry) into `observations`, so callers can heal the stat cache from
150/// hash-time stats.
151///
152/// # Errors
153/// See [`WorktreeError`].
154pub fn build_tree_filtered_observed<S: ObjectSink + Sync + ?Sized>(
155    sink: &S,
156    dir: &Path,
157    index: Option<&Index>,
158    observations: &mut Vec<StatObservation>,
159) -> WorktreeResult<Hash> {
160    build_tree_observed_impl(sink, None, dir, index, observations)
161}
162
163/// Build a snapshot with read access to newly written and indexed objects.
164/// On a cache miss, equal bytes preserve the staged object identity even when
165/// the current chunker would choose a different representation.
166///
167/// # Errors
168/// See [`WorktreeError`]; unreadable staged content is an error.
169pub fn build_tree_filtered_observed_with_source<S: ObjectSink + Sync + ?Sized>(
170    sink: &S,
171    source: &(dyn crate::store::ObjectSource + Sync),
172    dir: &Path,
173    index: Option<&Index>,
174    observations: &mut Vec<StatObservation>,
175) -> WorktreeResult<Hash> {
176    build_tree_observed_impl(sink, Some(source), dir, index, observations)
177}
178
179fn build_tree_observed_impl<S: ObjectSink + Sync + ?Sized>(
180    sink: &S,
181    source: Option<&(dyn crate::store::ObjectSource + Sync)>,
182    dir: &Path,
183    index: Option<&Index>,
184    observations: &mut Vec<StatObservation>,
185) -> WorktreeResult<Hash> {
186    let ignores = ignore::load(dir).map_err(|e| match e {
187        crate::ignore::IgnoreError::Io(io) => WorktreeError::Io(io),
188        crate::ignore::IgnoreError::FileTooLarge => {
189            WorktreeError::Io(io::Error::other("ignore file exceeds 1 MiB"))
190        }
191    })?;
192    // Tracked set for ignore exemption: the caller's index if given, else the
193    // on-disk index (missing/unreadable = empty = nothing tracked).
194    let loaded;
195    let index = if let Some(i) = index {
196        i
197    } else {
198        // Single-layout assumption: `dir` is treated as a classic
199        // single-worktree root. Linked-worktree callers (#493 Phase 1+)
200        // must pass `Some(index)` or this fallback reads the wrong
201        // index; the CLI always passes the discovered layout's index.
202        loaded = index::read_index(&crate::layout::RepoLayout::single(dir))
203            .map_err(|e| WorktreeError::Io(io::Error::other(e)))?;
204        &loaded
205    };
206    // O(1) per-file entry lookups; `Index::find_entry` is a linear scan
207    // and the walk consults it once per regular file.
208    let by_path: std::collections::HashMap<&str, &crate::index::IndexEntry> =
209        index.entries.iter().map(|e| (e.path.as_str(), e)).collect();
210    build_tree_inner(
211        sink,
212        source,
213        dir,
214        "",
215        &ignores,
216        index,
217        &by_path,
218        false,
219        observations,
220    )
221}
222
223fn retained_content_hash(
224    source: Option<&(dyn crate::store::ObjectSource + Sync)>,
225    indexed: Option<&crate::index::IndexEntry>,
226    fresh: Hash,
227) -> WorktreeResult<Hash> {
228    if let (Some(entry), Some(source)) = (indexed, source)
229        && entry.status != crate::index::EntryStatus::Removed
230        && content_eq(source, &entry.object_hash, &fresh)?
231    {
232        return Ok(entry.object_hash);
233    }
234    Ok(fresh)
235}
236
237/// `rel_dir` is the path of `dir` relative to the repo root (empty at the
238/// root), so ignore patterns can be matched against full repo-relative paths
239/// rather than bare basenames. `parent_ignored` carries down whether an
240/// ancestor directory is ignored (git "everything under an excluded dir is
241/// excluded"); `index` is the tracked set used to exempt tracked content.
242#[allow(clippy::too_many_arguments)]
243fn build_tree_inner<S: ObjectSink + Sync + ?Sized>(
244    sink: &S,
245    source: Option<&(dyn crate::store::ObjectSource + Sync)>,
246    dir: &Path,
247    rel_dir: &str,
248    ignores: &IgnoreList,
249    index: &Index,
250    by_path: &std::collections::HashMap<&str, &crate::index::IndexEntry>,
251    parent_ignored: bool,
252    observations: &mut Vec<StatObservation>,
253) -> WorktreeResult<Hash> {
254    let mut entries: Vec<TreeEntry> = Vec::new();
255    // Cache-miss regular files collected during the walk below, hashed
256    // together (sequentially or fanned out) once the walk finishes —
257    // see `hash_pending_files`.
258    let mut misses: Vec<PendingFileHash<'_>> = Vec::new();
259
260    for entry in fs::read_dir(dir)? {
261        let entry = entry?;
262        let file_name = entry.file_name();
263        let name_str = file_name
264            .to_str()
265            .ok_or(WorktreeError::InvalidUtf8)?
266            .to_string();
267        // `symlink_metadata` does not follow symlinks.
268        let meta = entry.path().symlink_metadata()?;
269        let is_dir = meta.is_dir();
270        let rel_path = if rel_dir.is_empty() {
271            name_str.clone()
272        } else {
273            format!("{rel_dir}/{name_str}")
274        };
275        // Exclude ignored content, but only when it is UNTRACKED — a tracked
276        // path (or a dir holding tracked content) is always kept so status/
277        // diff see it. An ignored dir with tracked content is descended into
278        // (carrying the ignored bit) so its untracked children stay excluded.
279        let entry_ignored = parent_ignored || ignores.is_ignored(&rel_path, is_dir);
280        if entry_ignored && !index.tracks_path_or_descendant(&rel_path) {
281            continue;
282        }
283
284        let name_bytes = name_str.as_bytes();
285        if !TreeEntry::validate_name(name_bytes) {
286            return Err(WorktreeError::Io(io::Error::new(
287                io::ErrorKind::InvalidInput,
288                format!("invalid tree entry name: {name_str:?}"),
289            )));
290        }
291
292        if meta.file_type().is_file() {
293            // Stat cache: when the staging index proves this file's
294            // content via mtime+size+ino+ctime (+exec class), reuse the
295            // staged hash without opening the file — O(stat) instead of
296            // O(content) for unchanged files. The object was stored at
297            // `add` time, so the tree reference stays resolvable.
298            let indexed = by_path.get(rel_path.as_str()).copied();
299            let cached = indexed.filter(|e| stat_matches(e, &meta));
300            if let Some(e) = cached {
301                entries.push(TreeEntry {
302                    name: name_str.into_bytes(),
303                    mode: entry_mode_from_file_metadata(&meta),
304                    object_hash: e.object_hash,
305                });
306            } else {
307                // Placeholder, overwritten by `hash_pending_files` below
308                // once every miss in this directory has been hashed —
309                // its slot is this entry's fixed position in `entries`.
310                let slot = entries.len();
311                entries.push(TreeEntry {
312                    name: name_str.into_bytes(),
313                    mode: EntryMode::Blob,
314                    object_hash: crate::hash::ZERO,
315                });
316                misses.push(PendingFileHash {
317                    slot,
318                    abs_path: entry.path(),
319                    rel_path,
320                    indexed,
321                });
322            }
323        } else if meta.file_type().is_dir() {
324            // A directory on disk at a path tracked as a *file* shadows that
325            // tracked entry. git reports only the tracked-side deletion and
326            // suppresses the directory's contents as untracked (#288); mirror
327            // that by leaving the whole subtree out of the snapshot, so the
328            // tracked file reads as deleted and nothing inside surfaces.
329            if index.has_tracked_file_at(&rel_path) {
330                continue;
331            }
332            let h = build_tree_inner(
333                sink,
334                source,
335                &entry.path(),
336                &rel_path,
337                ignores,
338                index,
339                by_path,
340                entry_ignored,
341                observations,
342            )?;
343            entries.push(TreeEntry {
344                name: name_str.into_bytes(),
345                mode: EntryMode::Tree,
346                object_hash: h,
347            });
348        } else if meta.file_type().is_symlink() {
349            let target = fs::read_link(entry.path())?;
350            let target_str = target
351                .to_str()
352                .ok_or(WorktreeError::InvalidUtf8)?
353                .to_string();
354            if !validate_symlink_target(&target_str) {
355                return Err(WorktreeError::InvalidSymlinkTarget(target_str));
356            }
357            let target_bytes = target_str.as_bytes();
358            let prologue = serialize::blob_prologue(target_bytes.len())?;
359            let h = sink.put_parts(&[&prologue, target_bytes])?;
360            entries.push(TreeEntry {
361                name: name_str.into_bytes(),
362                mode: EntryMode::Symlink,
363                object_hash: h,
364            });
365        } else {
366            // Block / char / fifo / socket — silently skip.
367        }
368    }
369
370    if !misses.is_empty() {
371        hash_pending_files(sink, source, &misses, &mut entries, observations)?;
372    }
373
374    entries.sort_by(|a, b| a.name.cmp(&b.name));
375    let tree = Object::Tree(Tree { entries });
376    let bytes = serialize::serialize(&tree)?;
377    Ok(sink.put(&bytes)?)
378}
379
380/// One cache-miss regular file discovered by [`build_tree_inner`]'s walk
381/// (its stat cache was absent or racy-smudged): still needs
382/// [`hash_file_with_metadata`], off the main walk thread once there's
383/// enough of them in one directory to amortize the dispatch. `'e`
384/// borrows from the same [`Index`] `by_path` does.
385struct PendingFileHash<'e> {
386    /// This miss's fixed position in `build_tree_inner`'s `entries` —
387    /// stable regardless of how the batch below is chunked or ordered,
388    /// so results always land back on the right entry.
389    slot: usize,
390    abs_path: PathBuf,
391    rel_path: String,
392    indexed: Option<&'e crate::index::IndexEntry>,
393}
394
395/// Below this many cache-miss files per available thread, hashing them
396/// in [`build_tree_inner`]'s own loop wins over thread-spawn overhead —
397/// same crossover shape as [`probe_staged_objects`]. The value borrows
398/// `mkit-cli`'s `commands::add::HASH_FANOUT_FILES_PER_THREAD`
399/// (measured on PR #951's Slack thread): both fan out the exact same
400/// per-item cost, `hash_file_with_metadata`'s open+read+BLAKE3, so the
401/// same crossover point applies here. `worktree_walk_fanout` is this
402/// path's own regression guard, not a from-scratch tuning.
403#[cfg(not(target_arch = "wasm32"))]
404const FILE_HASH_MISSES_PER_THREAD: usize = 8;
405
406/// Hash every cache-miss file in `misses` (see [`PendingFileHash`]),
407/// filling in `entries[slot]` and pushing a [`StatObservation`] for each
408/// one whose fresh hash proves the stat cache healable. Sequential
409/// below a thread-count-scaled threshold, fanned out across scoped
410/// worker threads at or above it — [`probe_staged_objects_parallel`]'s
411/// shape (`mkit-core` stays dependency-neutral and wasm-clean, so this
412/// uses `std::thread::scope` directly rather than rayon). Every file's
413/// hash result is written back to `entries`/`observations` in `misses`'
414/// original directory-walk order regardless of which thread computed
415/// it, so the output is byte-identical to the old fully-sequential
416/// loop; only which of several simultaneous I/O errors is reported can
417/// vary between runs — the same accepted nondeterminism
418/// `probe_staged_objects_parallel` already has.
419fn hash_pending_files<S: ObjectSink + Sync + ?Sized>(
420    sink: &S,
421    source: Option<&(dyn crate::store::ObjectSource + Sync)>,
422    misses: &[PendingFileHash<'_>],
423    entries: &mut [TreeEntry],
424    observations: &mut Vec<StatObservation>,
425) -> WorktreeResult<()> {
426    #[cfg(not(target_arch = "wasm32"))]
427    {
428        let threads = available_threads();
429        if threads > 1 && misses.len() >= FILE_HASH_MISSES_PER_THREAD.saturating_mul(threads) {
430            let results = hash_pending_files_parallel(sink, source, misses, threads)?;
431            apply_hash_results(misses, results, entries, observations);
432            return Ok(());
433        }
434    }
435    let results = misses
436        .iter()
437        .map(|m| hash_one_pending_file(sink, source, m))
438        .collect::<WorktreeResult<Vec<_>>>()?;
439    apply_hash_results(misses, results, entries, observations);
440    Ok(())
441}
442
443/// Hash one [`PendingFileHash`]: pure function of `sink`/`source` and
444/// the miss itself, safe to call concurrently across a batch — shared
445/// by [`hash_pending_files`]'s sequential loop and
446/// [`hash_pending_files_parallel`]'s worker threads.
447fn hash_one_pending_file<S: ObjectSink + ?Sized>(
448    sink: &S,
449    source: Option<&(dyn crate::store::ObjectSource + Sync)>,
450    m: &PendingFileHash<'_>,
451) -> WorktreeResult<(Hash, EntryMode, Option<StatObservation>)> {
452    let (mut h, opened_meta) = hash_file_with_metadata(sink, &m.abs_path)?;
453    h = retained_content_hash(source, m.indexed, h)?;
454    // Cache miss that re-hashed back to the staged hash: report the
455    // observation (stat captured from the opened fd BEFORE the content
456    // read) so callers can heal the racy-smudged cache soundly.
457    let observation = if let Some(e) = m.indexed
458        && e.object_hash == h
459    {
460        let (mtime_ns, size, ino, ctime_ns) = stat_cache_fields(&opened_meta);
461        Some(StatObservation {
462            path: m.rel_path.clone(),
463            object_hash: h,
464            mtime_ns,
465            size,
466            ino,
467            ctime_ns,
468        })
469    } else {
470        None
471    };
472    Ok((h, entry_mode_from_file_metadata(&opened_meta), observation))
473}
474
475/// Parallel branch of [`hash_pending_files`]: [`chunked_scoped_map`]
476/// over `misses`, hashing each one via [`hash_one_pending_file`].
477#[cfg(not(target_arch = "wasm32"))]
478fn hash_pending_files_parallel<S: ObjectSink + Sync + ?Sized>(
479    sink: &S,
480    source: Option<&(dyn crate::store::ObjectSource + Sync)>,
481    misses: &[PendingFileHash<'_>],
482    threads: usize,
483) -> WorktreeResult<Vec<(Hash, EntryMode, Option<StatObservation>)>> {
484    chunked_scoped_map(misses, threads, |m| hash_one_pending_file(sink, source, m))
485}
486
487/// Write [`hash_pending_files`]'s per-miss results back into `entries`
488/// and `observations`, in `misses`' original directory-walk order.
489fn apply_hash_results(
490    misses: &[PendingFileHash<'_>],
491    results: Vec<(Hash, EntryMode, Option<StatObservation>)>,
492    entries: &mut [TreeEntry],
493    observations: &mut Vec<StatObservation>,
494) {
495    for (m, (hash, mode, observation)) in misses.iter().zip(results) {
496        entries[m.slot].object_hash = hash;
497        entries[m.slot].mode = mode;
498        if let Some(obs) = observation {
499            observations.push(obs);
500        }
501    }
502}
503
504/// Build a tree object from an [`Index`] (the staging area).
505///
506/// Walks the flat list of entries, groups them by directory, and
507/// recursively materialises sub-tree objects so the on-disk shape
508/// matches what [`build_tree`] would produce for the same set of
509/// paths. Entries with [`crate::index::EntryStatus::Removed`] are
510/// excluded; everything else maps to an [`EntryMode`] one-to-one.
511///
512/// A file entry (Blob/Executable) may address either a single
513/// [`Blob`](crate::object::Blob) or, for content above
514/// [`CHUNK_THRESHOLD`], a [`ChunkedBlob`]
515/// manifest — exactly the two shapes `store_file_object` (and hence
516/// `add`/`hash_file`/`build_tree`) can produce. Symlink entries must be
517/// a single `Blob`. Any other object kind under a file entry is rejected.
518///
519/// # Errors
520/// - [`WorktreeError::Io`] on a [`crate::object::TreeEntry::validate_name`]
521///   failure (the path's leaf segment is reserved or alias-prone), or
522///   when a file entry points at a non-blob/non-chunked-blob object.
523/// - Wraps [`crate::MkitError`] surfaced by `serialize` / `store.write`.
524pub fn build_tree_from_index(
525    store: &ObjectStore,
526    index: &crate::index::Index,
527) -> WorktreeResult<Hash> {
528    // The convenience wrapper publishes a durable tree (commit/merge/
529    // rebase/…), so it integrity-verifies staged objects by default.
530    build_tree_from_index_with(store, store, index, true)
531}
532
533/// [`build_tree_from_index`] writing tree objects through `sink` —
534/// pass a [`WriteBatch`](crate::batch::WriteBatch) to amortise the
535/// flush cost of all materialised trees into the batch's single commit.
536/// `store` is still needed read-only to validate that staged hashes
537/// point at blob-shaped objects (a sink cannot read).
538///
539/// `verify` selects the staged-object integrity check. With `verify =
540/// true` (every path that publishes a durable tree) each referenced
541/// object is read and re-hashed before the tree is materialised, so a
542/// corrupt staged object can never be published. With `verify = false`
543/// (ephemeral status/diff snapshots that publish nothing durable) only
544/// the 6-byte prologue is read for the blob-shape check — the read path
545/// still integrity-verifies the object whenever it is actually used.
546///
547/// # Errors
548/// See [`build_tree_from_index`].
549#[allow(clippy::items_after_statements, clippy::too_many_lines)]
550pub fn build_tree_from_index_with<S: ObjectSink + ?Sized>(
551    store: &ObjectStore,
552    sink: &S,
553    index: &crate::index::Index,
554    verify: bool,
555) -> WorktreeResult<Hash> {
556    use crate::index::EntryStatus;
557
558    // Build an in-memory directory tree. Each node is either a leaf
559    // (one staged blob/symlink) or a directory containing children.
560    #[derive(Default)]
561    struct Node {
562        // Subdirectory name → child node.
563        children: std::collections::BTreeMap<String, Node>,
564        // Leaf entries directly under this dir: name → (mode, hash).
565        leaves: std::collections::BTreeMap<String, (EntryMode, Hash)>,
566    }
567
568    let mut root = Node::default();
569    let mut seen_paths = std::collections::HashSet::with_capacity(index.entries.len());
570
571    // Pass 1: validate paths/status and collect the surviving entries'
572    // (path, mode, staged hash) — cheap, pure-CPU work, no store access.
573    // The original single-pass loop bailed on the *first* entry (in
574    // index order) with any problem — duplicate path, reserved status,
575    // or bad staged object — never even looking at later entries. A
576    // path/status problem here defers its error via `deferred_status_err`
577    // and stops collecting instead of returning immediately, so pass 2
578    // below still gets a chance to surface an *earlier* staged-object
579    // problem among the entries collected before the stopping point —
580    // preserving that same index-order precedence across both error
581    // classes. Entries at or after the stopping point are unreachable
582    // either way (the original loop never got there), so excluding them
583    // from `kept` changes nothing observable.
584    let mut kept: Vec<(&str, EntryMode, Hash)> = Vec::with_capacity(index.entries.len());
585    let mut deferred_status_err: Option<WorktreeError> = None;
586    for entry in &index.entries {
587        if !seen_paths.insert(entry.path.as_str()) {
588            deferred_status_err = Some(WorktreeError::Io(io::Error::other(format!(
589                "duplicate index path: '{}'",
590                entry.path
591            ))));
592            break;
593        }
594        if entry.status == EntryStatus::Removed {
595            continue;
596        }
597        let mode = match entry.status {
598            EntryStatus::Blob => EntryMode::Blob,
599            EntryStatus::Executable => EntryMode::Executable,
600            EntryStatus::Symlink => EntryMode::Symlink,
601            EntryStatus::Tree => {
602                // Reserved-but-unused per SPEC-INDEX §3. Reject for
603                // now; if a subtree-staging design lands later it
604                // can populate this branch.
605                deferred_status_err = Some(WorktreeError::Io(io::Error::other(
606                    "index entry uses reserved Tree status (subtree staging not implemented)",
607                )));
608                break;
609            }
610            EntryStatus::Removed => unreachable!("filtered above"),
611        };
612        kept.push((entry.path.as_str(), mode, entry.object_hash));
613    }
614
615    // Pass 2: one batched staged-object check (fetch the type, then the
616    // same blob-shape check the original loop ran inline right after
617    // its own per-entry fetch) for every entry pass 1 kept, instead of
618    // interleaving it one-at-a-time with the pass-3 tree walk below.
619    // `probe_staged_objects` fans this out across threads once there's
620    // enough work to amortize it (native builds only) — a `status`/
621    // `diff` snapshot over a many-file repo was paying one serialized
622    // `open`+`read`(+`close`) per tracked file here, dominating wall
623    // time well before hashing or tree-materialization cost did. Its
624    // sequential (default) path still short-circuits on the first
625    // failing entry in `kept`'s (index) order, so — combined with pass
626    // 1 stopping at the same point the original loop would have —
627    // whichever of the two errors is earlier in index order is the one
628    // that surfaces, exactly matching the original interleaved loop.
629    probe_staged_objects(store, &kept, verify)?;
630    if let Some(err) = deferred_status_err {
631        return Err(err);
632    }
633
634    // Pass 3: walk each surviving, already-validated entry into the
635    // in-memory node tree.
636    for (path, mode, object_hash) in kept {
637        // Split "a/b/c.txt" into ["a", "b"] + "c.txt".
638        let segments: Vec<&str> = path.split('/').collect();
639        let Some((leaf, dirs)) = segments.split_last() else {
640            return Err(WorktreeError::Io(io::Error::other("empty index path")));
641        };
642        if leaf.is_empty() {
643            return Err(WorktreeError::Io(io::Error::other(
644                "trailing slash in index path",
645            )));
646        }
647
648        let mut node = &mut root;
649        let mut walked = String::new();
650        for seg in dirs {
651            if seg.is_empty() {
652                return Err(WorktreeError::Io(io::Error::other(
653                    "empty path segment in index",
654                )));
655            }
656            // Collision: this segment was previously staged as a blob
657            // (e.g. earlier index entry was `a` as a file, this one
658            // is `a/b`). Tree object format requires unique entry
659            // names per directory; emitting both would produce an
660            // invalid tree the deserializer rejects under its strict
661            // ascending-name rule.
662            if node.leaves.contains_key(*seg) {
663                let conflicting = if walked.is_empty() {
664                    (*seg).to_string()
665                } else {
666                    format!("{walked}/{seg}")
667                };
668                return Err(WorktreeError::Io(io::Error::other(format!(
669                    "index path conflict: '{conflicting}' is staged as both a file and a directory"
670                ))));
671            }
672            walked = if walked.is_empty() {
673                (*seg).to_string()
674            } else {
675                format!("{walked}/{seg}")
676            };
677            node = node.children.entry((*seg).to_string()).or_default();
678        }
679        // The reverse collision: this entry's leaf name already exists
680        // as a child directory under the same parent (an earlier
681        // entry staged `a/b` and now this one stages `a` as a file).
682        if node.children.contains_key(*leaf) {
683            let conflicting = if walked.is_empty() {
684                (*leaf).to_string()
685            } else {
686                format!("{walked}/{leaf}")
687            };
688            return Err(WorktreeError::Io(io::Error::other(format!(
689                "index path conflict: '{conflicting}' is staged as both a file and a directory"
690            ))));
691        }
692        if node
693            .leaves
694            .insert((*leaf).to_string(), (mode, object_hash))
695            .is_some()
696        {
697            let duplicate = if walked.is_empty() {
698                (*leaf).to_string()
699            } else {
700                format!("{walked}/{leaf}")
701            };
702            return Err(WorktreeError::Io(io::Error::other(format!(
703                "duplicate index path: '{duplicate}'"
704            ))));
705        }
706    }
707
708    fn write_node<S: ObjectSink + ?Sized>(sink: &S, node: &Node) -> WorktreeResult<Hash> {
709        let mut entries: Vec<TreeEntry> = Vec::new();
710
711        // Subdirectories first (alphabetical via BTreeMap).
712        for (name, child) in &node.children {
713            let h = write_node(sink, child)?;
714            let bytes = name.as_bytes().to_vec();
715            if !crate::object::TreeEntry::validate_name(&bytes) {
716                return Err(WorktreeError::Io(io::Error::other(format!(
717                    "invalid tree entry name: {name:?}"
718                ))));
719            }
720            entries.push(TreeEntry {
721                name: bytes,
722                mode: EntryMode::Tree,
723                object_hash: h,
724            });
725        }
726
727        // Then leaves.
728        for (name, (mode, hash)) in &node.leaves {
729            let bytes = name.as_bytes().to_vec();
730            if !crate::object::TreeEntry::validate_name(&bytes) {
731                return Err(WorktreeError::Io(io::Error::other(format!(
732                    "invalid tree entry name: {name:?}"
733                ))));
734            }
735            entries.push(TreeEntry {
736                name: bytes,
737                mode: *mode,
738                object_hash: *hash,
739            });
740        }
741
742        // Tree-entry order is name-ascending per SPEC-OBJECTS §4.
743        entries.sort_by(|a, b| a.name.cmp(&b.name));
744        let tree = Object::Tree(Tree { entries });
745        let bytes = serialize::serialize(&tree)?;
746        Ok(sink.put(&bytes)?)
747    }
748
749    write_node(sink, &root)
750}
751
752/// `std::thread::available_parallelism()`, cached process-wide — the
753/// core count never changes during a process's lifetime, so every
754/// native fan-out crossover check in this module (there are several,
755/// and [`build_tree_inner`]'s recurses once per directory) shares one
756/// syscall instead of paying it again at every call site/recursion.
757#[cfg(not(target_arch = "wasm32"))]
758fn available_threads() -> usize {
759    static THREADS: std::sync::OnceLock<usize> = std::sync::OnceLock::new();
760    *THREADS
761        .get_or_init(|| std::thread::available_parallelism().map_or(1, std::num::NonZeroUsize::get))
762}
763
764/// Shared chunk+scoped-thread+join scaffolding behind this module's two
765/// order-preserving, short-circuit-on-first-error native fan-outs
766/// ([`probe_staged_objects_parallel`], [`hash_pending_files_parallel`]):
767/// split `items` into `threads` contiguous chunks, run each chunk's
768/// items through `f` sequentially on its own scoped thread, and join.
769/// `std::thread::scope` (not a persistent pool) keeps `mkit-core`
770/// dependency-neutral and wasm-clean.
771///
772/// (`pack::stage_raw_entries_parallel` looks similar but isn't folded
773/// in here: it *defers* rather than short-circuits on an item error, via
774/// a `first_failure` atomic shared across every chunk so a later pack
775/// position can still be skipped once an earlier one fails — genuinely
776/// different result semantics from this helper's per-chunk
777/// short-circuit, not just a different call shape.)
778///
779/// Every item's `f` output lands in the returned `Vec` at that item's
780/// original index in `items`, regardless of which thread computed it.
781/// A chunk stops calling `f` on its own remaining items as soon as one
782/// of them errors, but a concurrently-running chunk is not signalled to
783/// stop early; on any error, which chunk's is the one returned is
784/// unspecified — the same accepted nondeterminism
785/// `probe_staged_objects_parallel` had before this helper existed.
786///
787/// # Panics
788/// Propagates a worker thread's panic once `std::thread::scope` joins
789/// it, rather than swallowing it.
790#[cfg(not(target_arch = "wasm32"))]
791fn chunked_scoped_map<T, U, E>(
792    items: &[T],
793    threads: usize,
794    f: impl Fn(&T) -> Result<U, E> + Sync,
795) -> Result<Vec<U>, E>
796where
797    T: Sync,
798    U: Send,
799    E: Send,
800{
801    let chunk_size = items.len().div_ceil(threads).max(1);
802    let mut out: Vec<Option<U>> = (0..items.len()).map(|_| None).collect();
803    let mut first_err: Option<E> = None;
804
805    std::thread::scope(|scope| {
806        let f = &f;
807        let handles: Vec<_> = items
808            .chunks(chunk_size)
809            .enumerate()
810            .map(|(chunk_idx, chunk)| {
811                let base = chunk_idx * chunk_size;
812                scope.spawn(move || {
813                    let mut results = Vec::with_capacity(chunk.len());
814                    for it in chunk {
815                        match f(it) {
816                            Ok(u) => results.push(u),
817                            Err(e) => return (base, results, Some(e)),
818                        }
819                    }
820                    (base, results, None)
821                })
822            })
823            .collect();
824        for handle in handles {
825            let (base, results, err) = handle
826                .join()
827                .expect("chunked fan-out worker thread panicked");
828            for (i, u) in results.into_iter().enumerate() {
829                out[base + i] = Some(u);
830            }
831            if let Some(e) = err
832                && first_err.is_none()
833            {
834                first_err = Some(e);
835            }
836        }
837    });
838
839    match first_err {
840        Some(e) => Err(e),
841        None => Ok(out
842            .into_iter()
843            .map(|o| o.expect("every slot filled when no error was reported"))
844            .collect()),
845    }
846}
847
848/// Staged-object check for every `(path, mode, hash)` triple in
849/// `entries`, in order — [`build_tree_from_index_with`]'s batched pass-2
850/// step. Below a small-batch threshold this runs a plain sequential
851/// loop (short-circuiting on `entries`' first failure, same as a plain
852/// `for` loop would); at or above it (native builds only — wasm32 has
853/// no threads) fans out across a scoped thread pool sized to the
854/// machine, same shape as [`crate::pack`]'s
855/// `stage_raw_entries`/`stage_raw_entries_parallel`: each check is one
856/// `open`+`read`(+`close`) syscall triplet on an independent loose
857/// object file, so the calls parallelize with none of the ordering or
858/// shared-mutable-state concerns a write path has to account for.
859fn probe_staged_objects(
860    store: &ObjectStore,
861    entries: &[(&str, EntryMode, Hash)],
862    verify: bool,
863) -> WorktreeResult<()> {
864    #[cfg(not(target_arch = "wasm32"))]
865    {
866        // Below this many entries per available thread, thread-spawn
867        // overhead isn't worth it — same crossover shape as
868        // `pack::stage_raw_entries`, tuned here by the
869        // `status_snapshot` bench.
870        const ENTRIES_PER_THREAD: usize = 32;
871        let threads = available_threads();
872        if threads > 1 && entries.len() >= ENTRIES_PER_THREAD.saturating_mul(threads) {
873            return probe_staged_objects_parallel(store, entries, verify, threads);
874        }
875    }
876    entries.iter().try_for_each(|&(path, mode, hash)| {
877        check_one_staged_object(store, path, mode, hash, verify)
878    })
879}
880
881/// Parallel branch of [`probe_staged_objects`]: [`chunked_scoped_map`]
882/// over `entries`, checking each one via [`check_one_staged_object`] and
883/// discarding its (unit) results — only whether any entry errored
884/// matters here.
885///
886/// On an index with more than one entry pointing at a missing,
887/// malformed, or wrong-shape object, the specific error surfaced here
888/// depends on chunk/thread scheduling, not index position —
889/// `chunked_scoped_map`'s accepted nondeterminism. The tree build is
890/// rejected either way; only which error is reported can vary between
891/// runs.
892#[cfg(not(target_arch = "wasm32"))]
893fn probe_staged_objects_parallel(
894    store: &ObjectStore,
895    entries: &[(&str, EntryMode, Hash)],
896    verify: bool,
897    threads: usize,
898) -> WorktreeResult<()> {
899    chunked_scoped_map(entries, threads, |&(path, mode, hash)| {
900        check_one_staged_object(store, path, mode, hash, verify)
901    })
902    .map(|_: Vec<()>| ())
903}
904
905// Pure per-entry step shared by both branches of
906// `probe_staged_objects`: fetch `hash`'s object-type tag —
907// `verify_object_type` (full read + rehash) when `verify`, else the
908// cheap prologue-only `object_type` — then run the same blob-shape
909// check the original single-pass loop ran inline right after its own
910// per-entry fetch.
911//
912// A regular file (Blob/Executable) may be stored as a single Blob or,
913// for content above CHUNK_THRESHOLD, a ChunkedBlob manifest —
914// `add`/`hash_file`/`build_tree` all route through
915// `store_file_object`. A Symlink is always a single Blob (its target
916// path). Accept both blob shapes for file entries so the commit/index
917// path agrees with the worktree-hashing path; a tree/commit/etc.
918// under a file entry is still rejected. Publishing paths (`verify`)
919// read + re-hash the staged object so a tree never references a
920// corrupt blob; the read path's hash check is the same one `add`
921// passed, so this only catches post-`add` corruption. Ephemeral
922// status/diff snapshots skip it — re-reading every staged blob on
923// every status dominates large repos of small files, and they
924// publish nothing durable.
925fn check_one_staged_object(
926    store: &ObjectStore,
927    path: &str,
928    mode: EntryMode,
929    hash: Hash,
930    verify: bool,
931) -> WorktreeResult<()> {
932    let object_type = if verify {
933        store.verify_object_type(&hash)?
934    } else {
935        store.object_type(&hash)?
936    };
937    match object_type {
938        crate::object::ObjectType::Blob => Ok(()),
939        crate::object::ObjectType::ChunkedBlob if mode != EntryMode::Symlink => Ok(()),
940        other => Err(WorktreeError::Io(io::Error::other(format!(
941            "index entry '{path}' points to a non-blob object (got {})",
942            other.name()
943        )))),
944    }
945}
946
947/// Read a file from disk, hash it, store it, and return the
948/// content-address of the resulting object.
949///
950/// Files at or below [`CHUNK_THRESHOLD`] become a single
951/// [`Blob`](crate::object::Blob). Files above the threshold are split
952/// with [`FastCdc::v1`]; each chunk is stored as a `Blob`, and the
953/// file is represented by a [`ChunkedBlob`]
954/// manifest whose hash is returned and lands in the parent tree. See
955/// `SPEC-FASTCDC.md` and `SPEC-OBJECTS.md` §7.
956///
957/// # Errors
958/// See [`WorktreeError`].
959pub fn hash_file<S: ObjectSink + ?Sized>(sink: &S, path: &Path) -> WorktreeResult<Hash> {
960    hash_file_with_metadata(sink, path).map(|(hash, _)| hash)
961}
962
963/// Read a regular file without following the final path component on
964/// Unix, enforcing [`MAX_FILE_BYTES`] against both the opened handle's
965/// metadata and the actual bytes read.
966pub fn read_regular_file_bounded(path: &Path) -> WorktreeResult<(fs::Metadata, Vec<u8>)> {
967    let mut file = open_regular_file(path)?;
968    let meta = file.metadata()?;
969    if !meta.file_type().is_file() {
970        return Err(WorktreeError::Io(io::Error::new(
971            io::ErrorKind::InvalidInput,
972            "path is not a regular file",
973        )));
974    }
975    if meta.len() > MAX_FILE_BYTES {
976        return Err(WorktreeError::FileTooLarge(path.to_path_buf()));
977    }
978    let initial_capacity = usize::try_from(meta.len().min(CHUNK_THRESHOLD))
979        .map_err(|_| WorktreeError::FileTooLarge(path.to_path_buf()))?;
980    let mut data = Vec::with_capacity(initial_capacity);
981    file.by_ref()
982        .take(MAX_FILE_BYTES + 1)
983        .read_to_end(&mut data)?;
984    if u64::try_from(data.len()).unwrap_or(u64::MAX) > MAX_FILE_BYTES {
985        return Err(WorktreeError::FileTooLarge(path.to_path_buf()));
986    }
987    Ok((meta, data))
988}
989
990/// Hash and store a regular file, returning its content-address and the
991/// [`fs::Metadata`] observed when it was opened (for stat-cache use by
992/// callers like `mkit add`).
993///
994/// Files at or below [`CHUNK_THRESHOLD`] are read fully into memory — a
995/// single [`Blob`](crate::object::Blob) needs its bytes contiguous
996/// anyway, and the threshold keeps this small (1 MiB). Files above the
997/// threshold are streamed chunk-by-chunk directly from the open file
998/// handle instead of first buffering the whole file (issue #828):
999/// ingest memory is bounded by one `FastCdc` window regardless of the
1000/// file's total size.
1001///
1002/// # Errors
1003/// See [`WorktreeError`].
1004pub fn hash_file_with_metadata<S: ObjectSink + ?Sized>(
1005    sink: &S,
1006    path: &Path,
1007) -> WorktreeResult<(Hash, fs::Metadata)> {
1008    hash_file_with_metadata_using(sink, path, store_large_file_streaming)
1009}
1010
1011/// [`hash_file_with_metadata`], but with an explicit `hash_chunks` batch
1012/// callback in place of the built-in sequential loop over
1013/// [`store_chunk_blob`] — see [`store_large_file_streaming_with`], which
1014/// this delegates to for files above [`CHUNK_THRESHOLD`].
1015///
1016/// # Errors
1017/// See [`WorktreeError`].
1018pub fn hash_file_with_metadata_with<S: ObjectSink + ?Sized>(
1019    sink: &S,
1020    path: &Path,
1021    hash_chunks: impl FnMut(&S, &[Vec<u8>]) -> WorktreeResult<Vec<Hash>>,
1022) -> WorktreeResult<(Hash, fs::Metadata)> {
1023    hash_file_with_metadata_using(sink, path, |sink, reader, path| {
1024        store_large_file_streaming_with(sink, reader, path, hash_chunks)
1025    })
1026}
1027
1028/// Share file opening, size checks, and metadata capture between sequential
1029/// and batched ingest. Only the large-file streaming strategy differs.
1030fn hash_file_with_metadata_using<S: ObjectSink + ?Sized>(
1031    sink: &S,
1032    path: &Path,
1033    stream: impl FnOnce(&S, io::Take<fs::File>, &Path) -> WorktreeResult<Hash>,
1034) -> WorktreeResult<(Hash, fs::Metadata)> {
1035    let mut file = open_regular_file(path)?;
1036    let meta = file.metadata()?;
1037    if !meta.file_type().is_file() {
1038        return Err(WorktreeError::Io(io::Error::new(
1039            io::ErrorKind::InvalidInput,
1040            "path is not a regular file",
1041        )));
1042    }
1043    if meta.len() > MAX_FILE_BYTES {
1044        return Err(WorktreeError::FileTooLarge(path.to_path_buf()));
1045    }
1046
1047    if meta.len() <= CHUNK_THRESHOLD {
1048        let initial_capacity = usize::try_from(meta.len())
1049            .map_err(|_| WorktreeError::FileTooLarge(path.to_path_buf()))?;
1050        let mut data = Vec::with_capacity(initial_capacity);
1051        file.by_ref()
1052            .take(MAX_FILE_BYTES + 1)
1053            .read_to_end(&mut data)?;
1054        if u64::try_from(data.len()).unwrap_or(u64::MAX) > MAX_FILE_BYTES {
1055            return Err(WorktreeError::FileTooLarge(path.to_path_buf()));
1056        }
1057        let hash = store_file_object(sink, &data)?;
1058        return Ok((hash, meta));
1059    }
1060
1061    let hash = stream(sink, file.take(MAX_FILE_BYTES + 1), path)?;
1062    Ok((hash, meta))
1063}
1064
1065/// Number of chunks [`store_large_file_streaming_with`] buffers before
1066/// handing a batch to its `hash_chunks` callback. Bounds the extra
1067/// memory a fan-out `hash_chunks` needs to hold resident to at most
1068/// `STREAM_HASH_BATCH * chunker::MAX_SIZE` (16 MiB at the current 256
1069/// KiB `MAX_SIZE`) regardless of the file's total size — the same
1070/// "independent of file size" bound issue #828 gave the streaming
1071/// reader itself, just sized for a batch instead of one chunk.
1072const STREAM_HASH_BATCH: usize = 64;
1073
1074/// Store one already-cut chunk as a canonical Blob object, returning its
1075/// content address. The per-chunk unit [`store_large_file_streaming_with`]
1076/// fans out over — hashing (BLAKE3, via `put_parts`) and staging a temp
1077/// file are both independent of every other chunk once boundaries are
1078/// known.
1079///
1080/// # Errors
1081/// See [`WorktreeError`].
1082pub fn store_chunk_blob<S: ObjectSink + ?Sized>(sink: &S, chunk: &[u8]) -> WorktreeResult<Hash> {
1083    let prologue = serialize::blob_prologue(chunk.len())?;
1084    Ok(sink.put_parts(&[&prologue, chunk])?)
1085}
1086
1087/// Sequential ingest borrows each chunk directly from the reader window.
1088/// It does not allocate owned chunks or retain a batch between writes.
1089fn store_large_file_streaming<S: ObjectSink + ?Sized, R: Read>(
1090    sink: &S,
1091    reader: R,
1092    path: &Path,
1093) -> WorktreeResult<Hash> {
1094    let mut chunker = ChunkReader::new(FastCdc::v1(), reader);
1095    let mut chunks = Vec::new();
1096    let mut total_size: u64 = 0;
1097    while let Some(chunk) = chunker.next_chunk_ref()? {
1098        total_size = total_size
1099            .checked_add(chunk.len() as u64)
1100            .filter(|&t| t <= MAX_FILE_BYTES)
1101            .ok_or_else(|| WorktreeError::FileTooLarge(path.to_path_buf()))?;
1102        chunks.push(store_chunk_blob(sink, chunk)?);
1103    }
1104    store_chunk_manifest(sink, total_size, chunks)
1105}
1106
1107/// Both streaming strategies use the same canonical manifest encoding.
1108fn store_chunk_manifest<S: ObjectSink + ?Sized>(
1109    sink: &S,
1110    total_size: u64,
1111    chunks: Vec<Hash>,
1112) -> WorktreeResult<Hash> {
1113    let manifest = Object::ChunkedBlob(ChunkedBlob {
1114        total_size,
1115        chunk_size: 0, // 0 = content-defined (FastCDC) per SPEC-OBJECTS §7
1116        chunks,
1117    });
1118    let manifest_bytes = serialize::serialize(&manifest)?;
1119    Ok(sink.put(&manifest_bytes)?)
1120}
1121
1122/// Call `hash_chunks(sink, batch)` and enforce its documented contract —
1123/// exactly one [`Hash`](tyalias@Hash) per input chunk — before the caller ever sees the
1124/// result, returning [`WorktreeError::ChunkBatchLengthMismatch`] instead of
1125/// silently letting a short/long result desync the manifest's `chunks`
1126/// list from the file's actual chunk sequence. [`store_large_file_streaming_with`]'s
1127/// only caller of `hash_chunks`; kept as a thin wrapper so the check lives
1128/// in exactly one place rather than being repeated at both call sites.
1129fn checked_hash_chunks<S: ObjectSink + ?Sized>(
1130    hash_chunks: &mut impl FnMut(&S, &[Vec<u8>]) -> WorktreeResult<Vec<Hash>>,
1131    sink: &S,
1132    batch: &[Vec<u8>],
1133) -> WorktreeResult<Vec<Hash>> {
1134    let hashes = hash_chunks(sink, batch)?;
1135    if hashes.len() == batch.len() {
1136        Ok(hashes)
1137    } else {
1138        Err(WorktreeError::ChunkBatchLengthMismatch {
1139            expected: batch.len(),
1140            actual: hashes.len(),
1141        })
1142    }
1143}
1144
1145/// Store a large (> [`CHUNK_THRESHOLD`]) file's content as a
1146/// [`ChunkedBlob`] manifest, streaming chunks directly from `reader`
1147/// instead of requiring the whole file resident in memory first (issue
1148/// #828). Bounds ingest memory to one fixed `ChunkReader` window
1149/// (1 MiB), an owned batch of at most 64 chunks (16 MiB), and the
1150/// growing chunk-hash list (32 bytes/chunk). The reader and batch stay
1151/// bounded independently of total file size.
1152///
1153/// `path` is used only to name the file in a [`WorktreeError::FileTooLarge`]
1154/// error if `reader` yields more than [`MAX_FILE_BYTES`].
1155///
1156/// Takes an explicit `hash_chunks` batch callback in place of a built-in
1157/// sequential loop over [`store_chunk_blob`] — `mkit-core` has no
1158/// thread-pool dependency of its own — it stays usable from wasm
1159/// targets, which have no OS threads — so the fan-out decision lives
1160/// with the caller instead of living in this function, the same shape
1161/// as [`crate::transfer::plan_pack_with`]'s `encode_deltas` callback.
1162/// Cutting chunk boundaries from `reader` is inherently sequential (each
1163/// cut's start is the previous cut's end), but once a batch of up to
1164/// `STREAM_HASH_BATCH` chunks is in hand, hashing and storing each one
1165/// is independent of every other chunk in the batch — `mkit-cli`'s
1166/// native, rayon-backed `add` path passes a parallel `hash_chunks`
1167/// (mirroring the pack-compression/signature-verification/delta-encoding
1168/// fan-outs already in `mkit-cli`). [`hash_file_with_metadata`] uses a
1169/// separate sequential path that borrows chunks without batching.
1170///
1171/// `hash_chunks` receives each batch in file order and MUST return
1172/// exactly one [`Hash`](tyalias@Hash) per input chunk, in the same order — the
1173/// manifest's chunk list, and therefore the file's content address,
1174/// depends on that order.
1175///
1176/// Cutting chunk boundaries is otherwise strictly sequential (each cut's
1177/// start is the previous cut's end), but cutting *batch N+1* has no data
1178/// dependency on *hashing batch N* — only on the reader position, which
1179/// the cutter alone advances. Native builds (see
1180/// `store_large_file_streaming_pipelined`) overlap the two phases
1181/// across a scoped worker thread instead of running them strictly one
1182/// after the other; wasm32 (no threads) falls back to
1183/// `store_large_file_streaming_batched`, the original phase-serial
1184/// loop. (Both are private, cfg-gated to exactly one of the two
1185/// targets, so neither is a valid intra-doc link here — a native `cargo
1186/// doc` build never sees `store_large_file_streaming_batched` at all,
1187/// and vice versa on wasm32.)
1188///
1189/// # Errors
1190/// See [`WorktreeError`].
1191pub fn store_large_file_streaming_with<S: ObjectSink + ?Sized, R: Read + Send>(
1192    sink: &S,
1193    reader: R,
1194    path: &Path,
1195    mut hash_chunks: impl FnMut(&S, &[Vec<u8>]) -> WorktreeResult<Vec<Hash>>,
1196) -> WorktreeResult<Hash> {
1197    #[cfg(not(target_arch = "wasm32"))]
1198    {
1199        store_large_file_streaming_pipelined(sink, reader, path, &mut hash_chunks)
1200    }
1201    #[cfg(target_arch = "wasm32")]
1202    {
1203        store_large_file_streaming_batched(sink, reader, path, &mut hash_chunks)
1204    }
1205}
1206
1207/// Original phase-serial implementation of
1208/// [`store_large_file_streaming_with`]: cut one batch, then hash it via
1209/// `hash_chunks`, then cut the next. wasm32 (no threads to pipeline
1210/// across) uses this directly; native builds use
1211/// `store_large_file_streaming_pipelined` instead (native-only; not a
1212/// valid intra-doc link from a wasm32 doc build).
1213#[cfg(target_arch = "wasm32")]
1214fn store_large_file_streaming_batched<S: ObjectSink + ?Sized, R: Read>(
1215    sink: &S,
1216    reader: R,
1217    path: &Path,
1218    hash_chunks: &mut impl FnMut(&S, &[Vec<u8>]) -> WorktreeResult<Vec<Hash>>,
1219) -> WorktreeResult<Hash> {
1220    let mut chunker = ChunkReader::new(FastCdc::v1(), reader);
1221    let mut chunks = Vec::new();
1222    let mut total_size: u64 = 0;
1223    let mut batch: Vec<Vec<u8>> = Vec::with_capacity(STREAM_HASH_BATCH);
1224    while let Some(chunk) = chunker.next_chunk()? {
1225        total_size = total_size
1226            .checked_add(chunk.len() as u64)
1227            .filter(|&t| t <= MAX_FILE_BYTES)
1228            .ok_or_else(|| WorktreeError::FileTooLarge(path.to_path_buf()))?;
1229        batch.push(chunk);
1230        if batch.len() == STREAM_HASH_BATCH {
1231            chunks.extend(checked_hash_chunks(hash_chunks, sink, &batch)?);
1232            batch.clear();
1233        }
1234    }
1235    if !batch.is_empty() {
1236        chunks.extend(checked_hash_chunks(hash_chunks, sink, &batch)?);
1237    }
1238
1239    store_chunk_manifest(sink, total_size, chunks)
1240}
1241
1242/// Pipelined implementation of [`store_large_file_streaming_with`]: a
1243/// scoped worker thread runs `reader` through the `FastCdc` cutter and
1244/// hands off each full [`STREAM_HASH_BATCH`]-sized batch (plus the final
1245/// partial one) over a rendezvous (zero-capacity) channel, while this
1246/// (the calling) thread receives batches and runs `hash_chunks` on them
1247/// — so cutting batch N+1 overlaps hashing batch N instead of waiting
1248/// for it. A zero-capacity channel's `send` blocks until the matching
1249/// `recv`, so the cutter can have at most one batch *fully built* ahead
1250/// of the one currently being hashed (held on its own stack, blocked on
1251/// send) — never two, the way a buffered channel would allow by letting
1252/// it dequeue immediately and start a third batch. That caps the extra
1253/// memory this adds at one more `STREAM_HASH_BATCH` (matching this
1254/// module's existing "independent of file size" bound) while still
1255/// overlapping the two phases fully: the blocked send/recv handshake
1256/// itself is a single unblocked pair of syscalls, not a wait for work.
1257///
1258/// The cutter thread performs the exact same running-total overflow
1259/// check `store_large_file_streaming_batched` (wasm32-only; not a valid
1260/// intra-doc link here) does, so an oversized
1261/// file is still rejected as soon as the cutter itself detects it
1262/// (before hashing catches up) rather than only once every chunk has
1263/// been received; the receiving thread separately sums each received
1264/// chunk's length into `total_size` for the final manifest, which is
1265/// always consistent with the cutter's own total since it receives
1266/// every chunk the cutter decided to keep.
1267///
1268/// If `hash_chunks` (or the overflow check) errors, this thread returns
1269/// early and drops its end of the channel; the cutter's next blocked
1270/// [`std::sync::mpsc::SyncSender::send`] then fails and it exits — no
1271/// deadlock, no unjoined thread (`std::thread::scope` joins it before
1272/// returning). If the cutter itself panics, `std::thread::scope`
1273/// resumes that panic here once joined, the same behavior
1274/// [`crate::pack::PackReader::read`]'s own scoped fan-out relies on.
1275#[cfg(not(target_arch = "wasm32"))]
1276fn store_large_file_streaming_pipelined<S: ObjectSink + ?Sized, R: Read + Send>(
1277    sink: &S,
1278    reader: R,
1279    path: &Path,
1280    hash_chunks: &mut impl FnMut(&S, &[Vec<u8>]) -> WorktreeResult<Vec<Hash>>,
1281) -> WorktreeResult<Hash> {
1282    use std::sync::mpsc;
1283
1284    let (tx, rx) = mpsc::sync_channel::<WorktreeResult<Vec<Vec<u8>>>>(0);
1285
1286    let (chunks, total_size) = std::thread::scope(|scope| -> WorktreeResult<(Vec<Hash>, u64)> {
1287        scope.spawn(move || {
1288            let mut chunker = ChunkReader::new(FastCdc::v1(), reader);
1289            let mut batch: Vec<Vec<u8>> = Vec::with_capacity(STREAM_HASH_BATCH);
1290            let mut running_total: u64 = 0;
1291            loop {
1292                let chunk = match chunker.next_chunk() {
1293                    Ok(Some(chunk)) => chunk,
1294                    Ok(None) => {
1295                        if !batch.is_empty() {
1296                            let _ = tx.send(Ok(batch));
1297                        }
1298                        return;
1299                    }
1300                    Err(e) => {
1301                        let _ = tx.send(Err(WorktreeError::from(e)));
1302                        return;
1303                    }
1304                };
1305                running_total = if let Some(t) = running_total
1306                    .checked_add(chunk.len() as u64)
1307                    .filter(|&t| t <= MAX_FILE_BYTES)
1308                {
1309                    t
1310                } else {
1311                    let _ = tx.send(Err(WorktreeError::FileTooLarge(path.to_path_buf())));
1312                    return;
1313                };
1314                batch.push(chunk);
1315                if batch.len() == STREAM_HASH_BATCH {
1316                    let full = std::mem::replace(&mut batch, Vec::with_capacity(STREAM_HASH_BATCH));
1317                    if tx.send(Ok(full)).is_err() {
1318                        return;
1319                    }
1320                }
1321            }
1322        });
1323
1324        // On any error, keep draining (discarding) the channel instead of
1325        // returning immediately: with a rendezvous channel the cutter's
1326        // *next* `send` — for a batch already cut, or one it hasn't
1327        // finished cutting yet — has no buffer to land in and blocks
1328        // until a matching `recv`, so an early return here that stops
1329        // calling `recv` would deadlock `thread::scope`'s join waiting
1330        // for a cutter that is itself waiting for a `recv` that will
1331        // never come. Draining lets every future send succeed (or the
1332        // cutter hits its own error/EOF and exits on its own), so the
1333        // channel always disconnects and this loop always terminates.
1334        let mut chunks = Vec::new();
1335        let mut total_size: u64 = 0;
1336        let mut first_err: Option<WorktreeError> = None;
1337        while let Ok(received) = rx.recv() {
1338            if first_err.is_some() {
1339                continue;
1340            }
1341            match received.and_then(|batch| {
1342                for chunk in &batch {
1343                    total_size += chunk.len() as u64;
1344                }
1345                checked_hash_chunks(hash_chunks, sink, &batch)
1346            }) {
1347                Ok(hashes) => chunks.extend(hashes),
1348                Err(e) => first_err = Some(e),
1349            }
1350        }
1351        match first_err {
1352            Some(e) => Err(e),
1353            None => Ok((chunks, total_size)),
1354        }
1355    })?;
1356
1357    store_chunk_manifest(sink, total_size, chunks)
1358}
1359
1360/// Store a regular file's bytes as the canonical object and return its
1361/// content-address.
1362///
1363/// This is the single source of truth for how file content maps to an
1364/// object hash, shared by [`hash_file`], [`build_tree`], and `mkit add`
1365/// so all three agree on the representation:
1366///
1367/// - At or below [`CHUNK_THRESHOLD`]: a single
1368///   [`Blob`](crate::object::Blob).
1369/// - Above the threshold: `FastCdc::v1` chunks, each stored as a `Blob`,
1370///   addressed by a [`ChunkedBlob`] manifest.
1371///
1372/// # Errors
1373/// See [`WorktreeError`].
1374pub fn store_file_object<S: ObjectSink + ?Sized>(sink: &S, data: &[u8]) -> WorktreeResult<Hash> {
1375    if u64::try_from(data.len()).unwrap_or(u64::MAX) <= CHUNK_THRESHOLD {
1376        // Zero-copy: the canonical Blob bytes are `prologue ‖ data`
1377        // (pinned to serialize() by proptest), so the sink can hash and
1378        // write straight from the source buffer.
1379        let prologue = serialize::blob_prologue(data.len())?;
1380        return Ok(sink.put_parts(&[&prologue, data])?);
1381    }
1382
1383    // Large file: split with FastCDC v1 via the public ChunkIterator,
1384    // store each chunk as a Blob, and assemble a ChunkedBlob manifest.
1385    // Per-manifest chunk count is bounded by serialize::MAX_CHUNKS
1386    // (1_000_000); MAX_FILE_BYTES (1 GiB) ÷ FastCDC MIN_SIZE (16 KiB)
1387    // = ~65k, well under the cap.
1388    let total_size = data.len() as u64;
1389    let chunks: Vec<Hash> = ChunkIterator::new(FastCdc::v1(), data)
1390        .map(|b| {
1391            let chunk = &data[b.offset..b.offset + b.length];
1392            let prologue = serialize::blob_prologue(chunk.len())?;
1393            Ok::<_, WorktreeError>(sink.put_parts(&[&prologue, chunk])?)
1394        })
1395        .collect::<Result<_, _>>()?;
1396
1397    let manifest = Object::ChunkedBlob(ChunkedBlob {
1398        total_size,
1399        chunk_size: 0, // 0 = content-defined (FastCDC) per SPEC-OBJECTS §7
1400        chunks,
1401    });
1402    let manifest_bytes = serialize::serialize(&manifest)?;
1403    Ok(sink.put(&manifest_bytes)?)
1404}
1405
1406/// Build the canonical [`ChunkedBlob`] manifest for `data`: split with
1407/// [`FastCdc::v1`], address each chunk by its **Blob object id** — the hash of
1408/// the chunk's canonical Blob bytes (`blob_prologue ‖ chunk`), i.e. the id the
1409/// chunk Blob is stored under — and set `chunk_size = 0` (content-defined).
1410///
1411/// This is the single source of the manifest recipe: [`hash_file_object`]
1412/// (read-only change detection) and the `mkit-wasm` encoder both build through
1413/// it, and [`store_file_object`] writes the chunk Blobs under these same ids,
1414/// so every path agrees on the manifest — and thus on its merkle root.
1415///
1416/// # Errors
1417/// [`WorktreeError::Object`] if a chunk length exceeds the wire-format cap.
1418pub fn chunked_blob_from_bytes(data: &[u8]) -> WorktreeResult<ChunkedBlob> {
1419    let chunks: Vec<Hash> = ChunkIterator::new(FastCdc::v1(), data)
1420        .map(|b| {
1421            let chunk = &data[b.offset..b.offset + b.length];
1422            // Blob object id = BLAKE3(blob_prologue ‖ chunk) — the id
1423            // store_file_object writes this chunk Blob under (the prologue ‖
1424            // payload split is pinned to serialize(Blob) by proptest).
1425            let prologue = serialize::blob_prologue(chunk.len())?;
1426            let mut hasher = crate::hash::Hasher::new();
1427            hasher.update(&prologue);
1428            hasher.update(chunk);
1429            Ok::<_, WorktreeError>(hasher.finalize())
1430        })
1431        .collect::<Result<_, _>>()?;
1432    Ok(ChunkedBlob {
1433        total_size: data.len() as u64,
1434        chunk_size: 0,
1435        chunks,
1436    })
1437}
1438
1439/// Content-address `data` exactly as [`store_file_object`] would,
1440/// **without storing anything**. Backs change detection (`status`, `rm`,
1441/// restore safety checks) where only the answer "would this file hash to X?"
1442/// is needed — writing objects there would turn a read-only query into store
1443/// mutation. Equivalence with `store_file_object` is pinned by test.
1444///
1445/// # Errors
1446/// [`WorktreeError::Object`] if a length exceeds the wire-format cap.
1447pub fn hash_file_object(data: &[u8]) -> WorktreeResult<Hash> {
1448    if u64::try_from(data.len()).unwrap_or(u64::MAX) <= CHUNK_THRESHOLD {
1449        let prologue = serialize::blob_prologue(data.len())?;
1450        let mut hasher = crate::hash::Hasher::new();
1451        hasher.update(&prologue);
1452        hasher.update(data);
1453        return Ok(hasher.finalize());
1454    }
1455    // A ChunkedBlob is addressed by its merkle BMT root, matching what the
1456    // sink stores it under. This read-only mirror must agree with the write
1457    // path or change detection breaks.
1458    Ok(crate::merkle::compute_chunked_id(&chunked_blob_from_bytes(
1459        data,
1460    )?))
1461}
1462
1463/// A file's mtime as nanoseconds since the Unix epoch, saturating; `0`
1464/// (the "no cache" sentinel) when the mtime is unavailable or predates
1465/// the epoch.
1466#[must_use]
1467pub fn mtime_nanos(meta: &fs::Metadata) -> u64 {
1468    meta.modified()
1469        .ok()
1470        .and_then(|t| t.duration_since(std::time::UNIX_EPOCH).ok())
1471        .map_or(0, |d| u64::try_from(d.as_nanos()).unwrap_or(u64::MAX))
1472}
1473
1474/// The full stat-cache observation for `meta`, in index-entry field
1475/// order: `(mtime_ns, size, ino, ctime_ns)`. The single producer-side
1476/// dual of [`stat_matches`] — every site that records the cache uses
1477/// this so the recorded and compared field sets can never drift.
1478/// `ino`/`ctime_ns` are 0 (= don't check) on platforms without them.
1479#[must_use]
1480pub fn stat_cache_fields(meta: &fs::Metadata) -> (u64, u64, u64, u64) {
1481    #[cfg(unix)]
1482    let (ino, ctime_ns) = {
1483        use std::os::unix::fs::MetadataExt;
1484        let ctime_ns = u64::try_from(meta.ctime())
1485            .ok()
1486            .and_then(|s| s.checked_mul(1_000_000_000))
1487            .and_then(|ns| ns.checked_add(u64::try_from(meta.ctime_nsec()).unwrap_or(0)))
1488            .unwrap_or(0);
1489        (meta.ino(), ctime_ns)
1490    };
1491    #[cfg(not(unix))]
1492    let (ino, ctime_ns) = (0u64, 0u64);
1493    (mtime_nanos(meta), meta.len(), ino, ctime_ns)
1494}
1495
1496/// True iff `meta` proves the worktree file behind `entry` is
1497/// byte-identical to `entry.object_hash` without reading it: the cached
1498/// mtime is nonzero (cache present, not racy-smudged) and equal, the
1499/// size is equal, the inode and ctime match when recorded (catching
1500/// replace-by-rename and `touch -r`-style timestamp restoration —
1501/// ctime cannot be set from userspace), and the live mode's exec class
1502/// matches the staged status. Symlink entries never stat-match — the
1503/// target re-read is cheap and `meta` semantics differ.
1504#[must_use]
1505pub fn stat_matches(entry: &crate::index::IndexEntry, meta: &fs::Metadata) -> bool {
1506    use crate::index::EntryStatus;
1507    if entry.mtime_ns == 0 || !meta.is_file() {
1508        return false;
1509    }
1510    let (mtime_ns, size, ino, ctime_ns) = stat_cache_fields(meta);
1511    if size != entry.size || mtime_ns != entry.mtime_ns {
1512        return false;
1513    }
1514    // ino/ctime: compare only when both sides have a value — a v2 entry
1515    // recorded on a platform without them (0) stays usable elsewhere.
1516    if entry.ino != 0 && ino != 0 && ino != entry.ino {
1517        return false;
1518    }
1519    if entry.ctime_ns != 0 && ctime_ns != 0 && ctime_ns != entry.ctime_ns {
1520        return false;
1521    }
1522    match entry.status {
1523        // On non-unix the exec bit is not observable in the filesystem
1524        // mode, so the recorded status is the source of truth — both
1525        // classes stat-match (previously Executable entries could never
1526        // match there, silently defeating the cache).
1527        #[cfg(not(unix))]
1528        EntryStatus::Blob | EntryStatus::Executable => true,
1529        #[cfg(unix)]
1530        EntryStatus::Blob => entry_mode_from_file_metadata(meta) == EntryMode::Blob,
1531        #[cfg(unix)]
1532        EntryStatus::Executable => entry_mode_from_file_metadata(meta) == EntryMode::Executable,
1533        EntryStatus::Symlink | EntryStatus::Removed | EntryStatus::Tree => false,
1534    }
1535}
1536
1537#[cfg(unix)]
1538fn open_regular_file(path: &Path) -> io::Result<fs::File> {
1539    use std::os::unix::fs::OpenOptionsExt;
1540
1541    fs::OpenOptions::new()
1542        .read(true)
1543        .custom_flags(libc::O_NOFOLLOW)
1544        .open(path)
1545}
1546
1547#[cfg(not(unix))]
1548fn open_regular_file(path: &Path) -> io::Result<fs::File> {
1549    // Best-effort direct-symlink rejection on platforms without the
1550    // Unix O_NOFOLLOW path. This does not close the swap race, but it
1551    // keeps normal symlinks from being treated as regular files.
1552    let meta = path.symlink_metadata()?;
1553    if !meta.file_type().is_file() {
1554        return Err(io::Error::new(
1555            io::ErrorKind::InvalidInput,
1556            "path is not a regular file",
1557        ));
1558    }
1559    fs::File::open(path)
1560}
1561
1562#[cfg(unix)]
1563fn entry_mode_from_file_metadata(meta: &fs::Metadata) -> EntryMode {
1564    use std::os::unix::fs::PermissionsExt;
1565
1566    if meta.permissions().mode() & 0o111 != 0 {
1567        EntryMode::Executable
1568    } else {
1569        EntryMode::Blob
1570    }
1571}
1572
1573#[cfg(not(unix))]
1574fn entry_mode_from_file_metadata(_meta: &fs::Metadata) -> EntryMode {
1575    EntryMode::Blob
1576}
1577
1578#[cfg(test)]
1579mod tests {
1580    use super::*;
1581    use crate::object::ObjectType;
1582    use tempfile::TempDir;
1583
1584    fn fresh_store() -> (TempDir, ObjectStore) {
1585        let dir = TempDir::new().unwrap();
1586        let store = ObjectStore::init(&crate::layout::RepoLayout::single(dir.path())).unwrap();
1587        (dir, store)
1588    }
1589
1590    #[test]
1591    fn validate_symlink_targets() {
1592        assert!(validate_symlink_target("hello"));
1593        assert!(validate_symlink_target("sub/dir/file"));
1594        assert!(!validate_symlink_target(""));
1595        assert!(!validate_symlink_target("/etc/passwd"));
1596        assert!(!validate_symlink_target("../escape"));
1597        assert!(!validate_symlink_target("a/../b"));
1598    }
1599
1600    #[test]
1601    fn build_tree_from_empty_dir() {
1602        let (_sd, store) = fresh_store();
1603        let work = TempDir::new().unwrap();
1604        let h = build_tree(&store, work.path()).unwrap();
1605        let obj = store.read_object(&h).unwrap();
1606        match obj {
1607            Object::Tree(t) => assert_eq!(t.entries.len(), 0),
1608            other => panic!("expected tree, got {other:?}"),
1609        }
1610    }
1611
1612    #[test]
1613    fn build_tree_with_single_file() {
1614        let (_sd, store) = fresh_store();
1615        let work = TempDir::new().unwrap();
1616        fs::write(work.path().join("hello.txt"), b"hello world").unwrap();
1617        let h = build_tree(&store, work.path()).unwrap();
1618        let obj = store.read_object(&h).unwrap();
1619        let Object::Tree(t) = obj else {
1620            panic!("expected tree");
1621        };
1622        assert_eq!(t.entries.len(), 1);
1623        assert_eq!(t.entries[0].name.as_slice(), b"hello.txt");
1624        assert_eq!(t.entries[0].mode, EntryMode::Blob);
1625        let blob_obj = store.read_object(&t.entries[0].object_hash).unwrap();
1626        let Object::Blob(b) = blob_obj else {
1627            panic!("expected blob");
1628        };
1629        assert_eq!(b.data, b"hello world");
1630    }
1631
1632    #[cfg(unix)]
1633    #[test]
1634    fn build_tree_marks_executable_regular_files() {
1635        use std::os::unix::fs::PermissionsExt;
1636
1637        let (_sd, store) = fresh_store();
1638        let work = TempDir::new().unwrap();
1639        let script = work.path().join("run.sh");
1640        fs::write(&script, b"#!/bin/sh\n").unwrap();
1641        let mut perms = fs::metadata(&script).unwrap().permissions();
1642        perms.set_mode(perms.mode() | 0o111);
1643        fs::set_permissions(&script, perms).unwrap();
1644
1645        let h = build_tree(&store, work.path()).unwrap();
1646        let Object::Tree(t) = store.read_object(&h).unwrap() else {
1647            panic!("expected tree");
1648        };
1649        assert_eq!(t.entries[0].name.as_slice(), b"run.sh");
1650        assert_eq!(t.entries[0].mode, EntryMode::Executable);
1651    }
1652
1653    #[cfg(unix)]
1654    #[test]
1655    fn build_tree_rejects_invalid_entry_name_before_writing_tree() {
1656        let (_sd, store) = fresh_store();
1657        let work = TempDir::new().unwrap();
1658        fs::write(work.path().join("bad."), b"bad name").unwrap();
1659
1660        let err = build_tree(&store, work.path()).unwrap_err();
1661        assert!(matches!(err, WorktreeError::Io(_)));
1662    }
1663
1664    #[cfg(unix)]
1665    #[test]
1666    fn hash_file_rejects_final_component_symlink() {
1667        use std::os::unix::fs::symlink;
1668
1669        let (_sd, store) = fresh_store();
1670        let work = TempDir::new().unwrap();
1671        fs::write(work.path().join("target.txt"), b"target").unwrap();
1672        symlink("target.txt", work.path().join("link.txt")).unwrap();
1673
1674        let err = hash_file(&store, &work.path().join("link.txt")).unwrap_err();
1675        assert!(matches!(err, WorktreeError::Io(_)));
1676    }
1677
1678    #[test]
1679    fn build_tree_with_nested_directories() {
1680        let (_sd, store) = fresh_store();
1681        let work = TempDir::new().unwrap();
1682        fs::write(work.path().join("a.txt"), b"file a").unwrap();
1683        fs::create_dir(work.path().join("subdir")).unwrap();
1684        fs::write(work.path().join("subdir/b.txt"), b"file b").unwrap();
1685        let h = build_tree(&store, work.path()).unwrap();
1686        let obj = store.read_object(&h).unwrap();
1687        let Object::Tree(t) = obj else {
1688            panic!("expected tree");
1689        };
1690        assert_eq!(t.entries.len(), 2);
1691        // Sorted lex: a.txt first, subdir second.
1692        assert_eq!(t.entries[0].name.as_slice(), b"a.txt");
1693        assert_eq!(t.entries[1].name.as_slice(), b"subdir");
1694        assert_eq!(t.entries[1].mode, EntryMode::Tree);
1695        let sub = store.read_object(&t.entries[1].object_hash).unwrap();
1696        let Object::Tree(st) = sub else {
1697            panic!("expected tree");
1698        };
1699        assert_eq!(st.entries.len(), 1);
1700        assert_eq!(st.entries[0].name.as_slice(), b"b.txt");
1701    }
1702
1703    #[test]
1704    fn build_tree_skips_mkit_directory() {
1705        let (_sd, store) = fresh_store();
1706        let work = TempDir::new().unwrap();
1707        fs::create_dir(work.path().join(".mkit")).unwrap();
1708        fs::write(work.path().join(".mkit/should_skip"), b"").unwrap();
1709        fs::write(work.path().join("keep.txt"), b"kept").unwrap();
1710        let h = build_tree(&store, work.path()).unwrap();
1711        let obj = store.read_object(&h).unwrap();
1712        let Object::Tree(t) = obj else {
1713            panic!("expected tree");
1714        };
1715        assert_eq!(t.entries.len(), 1);
1716        assert_eq!(t.entries[0].name.as_slice(), b"keep.txt");
1717    }
1718
1719    #[test]
1720    fn build_tree_is_deterministic() {
1721        let (_sd, store) = fresh_store();
1722        let work = TempDir::new().unwrap();
1723        fs::write(work.path().join("z.txt"), b"z").unwrap();
1724        fs::write(work.path().join("a.txt"), b"a").unwrap();
1725        let h1 = build_tree(&store, work.path()).unwrap();
1726        let h2 = build_tree(&store, work.path()).unwrap();
1727        assert_eq!(h1, h2);
1728    }
1729
1730    #[test]
1731    fn build_tree_respects_mkitignore() {
1732        let (_sd, store) = fresh_store();
1733        let work = TempDir::new().unwrap();
1734        fs::write(work.path().join(".mkitignore"), b"*.log\n").unwrap();
1735        fs::write(work.path().join("keep.txt"), b"kept").unwrap();
1736        fs::write(work.path().join("debug.log"), b"ignored").unwrap();
1737        let h = build_tree(&store, work.path()).unwrap();
1738        let obj = store.read_object(&h).unwrap();
1739        let Object::Tree(t) = obj else {
1740            panic!("expected tree");
1741        };
1742        // .mkitignore + keep.txt, but not debug.log.
1743        assert_eq!(t.entries.len(), 2);
1744        assert_eq!(t.entries[0].name.as_slice(), b".mkitignore");
1745        assert_eq!(t.entries[1].name.as_slice(), b"keep.txt");
1746    }
1747
1748    #[cfg(unix)]
1749    #[test]
1750    fn rejects_invalid_symlink_targets() {
1751        use std::os::unix::fs::symlink;
1752        let (_sd, store) = fresh_store();
1753        let work = TempDir::new().unwrap();
1754        symlink("/etc/passwd", work.path().join("bad-link")).unwrap();
1755        let err = build_tree(&store, work.path()).unwrap_err();
1756        assert!(matches!(err, WorktreeError::InvalidSymlinkTarget(_)));
1757    }
1758
1759    #[cfg(unix)]
1760    #[test]
1761    fn rejects_dotdot_symlink_targets() {
1762        use std::os::unix::fs::symlink;
1763        let (_sd, store) = fresh_store();
1764        let work = TempDir::new().unwrap();
1765        symlink("../../etc/passwd", work.path().join("bad-link")).unwrap();
1766        let err = build_tree(&store, work.path()).unwrap_err();
1767        assert!(matches!(err, WorktreeError::InvalidSymlinkTarget(_)));
1768    }
1769
1770    #[test]
1771    fn small_file_stays_as_regular_blob() {
1772        let (_sd, store) = fresh_store();
1773        let work = TempDir::new().unwrap();
1774        fs::write(work.path().join("small.txt"), b"hello world").unwrap();
1775        let h = build_tree(&store, work.path()).unwrap();
1776        let obj = store.read_object(&h).unwrap();
1777        let Object::Tree(t) = obj else {
1778            panic!("expected tree");
1779        };
1780        let entry = store.read_object(&t.entries[0].object_hash).unwrap();
1781        assert_eq!(entry.object_type(), ObjectType::Blob);
1782    }
1783
1784    #[test]
1785    fn large_file_becomes_chunked_blob() {
1786        // File > CHUNK_THRESHOLD should land as a ChunkedBlob manifest
1787        // pointing at one Blob per FastCDC chunk. We pseudo-randomize
1788        // the buffer so FastCDC sees real boundary candidates instead
1789        // of running the entire file as one max-sized chunk.
1790        let (_sd, store) = fresh_store();
1791        let work = TempDir::new().unwrap();
1792        let n = usize::try_from(CHUNK_THRESHOLD).unwrap() + 256 * 1024;
1793        let mut big = Vec::with_capacity(n);
1794        let mut state: u64 = 0x00C0_FFEE;
1795        for _ in 0..n {
1796            // splitmix64-ish; same construction as the gear table seed.
1797            state = state.wrapping_add(0x9E37_79B9_7F4A_7C15);
1798            let mut z = state;
1799            z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
1800            z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
1801            z ^= z >> 31;
1802            big.push((z & 0xFF) as u8);
1803        }
1804        fs::write(work.path().join("big.bin"), &big).unwrap();
1805
1806        let tree_hash = build_tree(&store, work.path()).unwrap();
1807        let Object::Tree(t) = store.read_object(&tree_hash).unwrap() else {
1808            panic!("expected tree");
1809        };
1810        assert_eq!(t.entries.len(), 1);
1811
1812        let entry_hash = t.entries[0].object_hash;
1813        let entry = store.read_object(&entry_hash).unwrap();
1814        let Object::ChunkedBlob(manifest) = entry else {
1815            panic!("expected chunked_blob, got {entry:?}");
1816        };
1817
1818        assert_eq!(manifest.total_size, n as u64);
1819        assert_eq!(manifest.chunk_size, 0, "0 = content-defined (FastCDC)");
1820        assert!(!manifest.chunks.is_empty());
1821        // Every chunk hash must resolve to a Blob in the store, and
1822        // the concatenation must reproduce the original file bytes.
1823        let mut reassembled: Vec<u8> = Vec::with_capacity(n);
1824        for h in &manifest.chunks {
1825            let Object::Blob(b) = store.read_object(h).unwrap() else {
1826                panic!("chunk did not resolve to a Blob");
1827            };
1828            reassembled.extend_from_slice(&b.data);
1829        }
1830        assert_eq!(reassembled, big, "chunks must round-trip the source");
1831    }
1832
1833    #[test]
1834    fn hash_file_with_metadata_streaming_matches_store_file_object_in_memory() {
1835        // The streaming disk path (issue #828) must be content-address
1836        // equivalent to the original whole-buffer path for the same
1837        // bytes — same test data/PRNG shape as `large_file_becomes_chunked_blob`.
1838        let work = TempDir::new().unwrap();
1839        let n = usize::try_from(CHUNK_THRESHOLD).unwrap() + 700 * 1024;
1840        let mut big = Vec::with_capacity(n);
1841        let mut state: u64 = 0xFACE_FEED;
1842        for _ in 0..n {
1843            state = state.wrapping_add(0x9E37_79B9_7F4A_7C15);
1844            let mut z = state;
1845            z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
1846            z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
1847            z ^= z >> 31;
1848            big.push((z & 0xFF) as u8);
1849        }
1850        let path = work.path().join("big.bin");
1851        fs::write(&path, &big).unwrap();
1852
1853        let (_sd1, store1) = fresh_store();
1854        let (streamed_hash, meta) = hash_file_with_metadata(&store1, &path).unwrap();
1855        assert_eq!(meta.len(), n as u64);
1856
1857        let (_sd2, store2) = fresh_store();
1858        let in_memory_hash = store_file_object(&store2, &big).unwrap();
1859
1860        assert_eq!(
1861            streamed_hash, in_memory_hash,
1862            "streaming a file from disk must produce the same content-address \
1863             as chunking the fully-buffered bytes"
1864        );
1865
1866        // And the objects it actually wrote must round-trip.
1867        let Object::ChunkedBlob(manifest) = store1.read_object(&streamed_hash).unwrap() else {
1868            panic!("expected chunked_blob");
1869        };
1870        let mut reassembled = Vec::with_capacity(n);
1871        for h in &manifest.chunks {
1872            let Object::Blob(b) = store1.read_object(h).unwrap() else {
1873                panic!("chunk did not resolve to a Blob");
1874            };
1875            reassembled.extend_from_slice(&b.data);
1876        }
1877        assert_eq!(reassembled, big);
1878    }
1879
1880    #[test]
1881    fn streaming_rejects_miscounted_hashes_in_full_and_partial_batches() {
1882        for size in [
1883            crate::chunker::MAX_SIZE,
1884            STREAM_HASH_BATCH * crate::chunker::MAX_SIZE + 1,
1885        ] {
1886            let (_dir, sink) = fresh_store();
1887            let data = vec![0u8; size];
1888            let mut expected = 0;
1889            let err = store_large_file_streaming_with(
1890                &sink,
1891                io::Cursor::new(data),
1892                Path::new("batch.bin"),
1893                |_sink, batch| {
1894                    expected = batch.len();
1895                    Ok(Vec::new())
1896                },
1897            )
1898            .unwrap_err();
1899            assert!(expected > 0);
1900            assert!(matches!(err, WorktreeError::ChunkBatchLengthMismatch {
1901                expected: n, actual: 0
1902            } if n == expected));
1903        }
1904    }
1905
1906    #[test]
1907    fn streaming_pipeline_errors_promptly_when_hash_chunks_fails_on_an_early_batch() {
1908        // Regression test for a real deadlock in the native pipelined
1909        // path (`store_large_file_streaming_pipelined`): on a rendezvous
1910        // channel, if the receiving side stops calling `recv` as soon as
1911        // it sees an error, the cutter thread's *next* `send` — for a
1912        // batch it already cut, or is about to — has no buffer to land
1913        // in and blocks forever waiting for a `recv` that will never
1914        // come, hanging `thread::scope`'s join on that thread. Three
1915        // full batches (well beyond `STREAM_HASH_BATCH`) ensures the
1916        // cutter still has more batches to send after the one that
1917        // triggers the very first error. Runs the call on its own
1918        // thread with a bounded `recv_timeout` so a regression fails
1919        // this test instead of hanging the whole suite.
1920        let size = 3 * STREAM_HASH_BATCH * crate::chunker::MAX_SIZE;
1921        let (_dir, sink) = fresh_store();
1922        let data = vec![0u8; size];
1923
1924        let (done_tx, done_rx) = std::sync::mpsc::channel();
1925        std::thread::spawn(move || {
1926            let result = store_large_file_streaming_with(
1927                &sink,
1928                io::Cursor::new(data),
1929                Path::new("batch.bin"),
1930                |_sink, _batch| Err(WorktreeError::InvalidUtf8),
1931            );
1932            let _ = done_tx.send(matches!(result, Err(WorktreeError::InvalidUtf8)));
1933        });
1934
1935        let errored_correctly = done_rx
1936            .recv_timeout(std::time::Duration::from_secs(20))
1937            .expect(
1938                "store_large_file_streaming_with deadlocked instead of \
1939                 returning promptly after an early hash_chunks error",
1940            );
1941        assert!(errored_correctly);
1942    }
1943
1944    #[test]
1945    fn hash_file_with_metadata_with_batch_fanout_matches_sequential() {
1946        // A `hash_chunks` callback that reorders its own work internally
1947        // (simulating a rayon fan-out that hashes a batch out of order)
1948        // must still produce the same content-address as the sequential
1949        // default, as long as it returns results in input order — this
1950        // pins the `_with` contract `hash_pending` (mkit-cli) relies on.
1951        // Large enough for several `STREAM_HASH_BATCH`-sized (64-chunk)
1952        // batches at the ~64 KiB average chunk size.
1953        let n = usize::try_from(CHUNK_THRESHOLD).unwrap() + 6 * 1024 * 1024;
1954        let mut big = Vec::with_capacity(n);
1955        let mut state: u64 = 0xABCD_EF01;
1956        for _ in 0..n {
1957            state = state.wrapping_add(0x9E37_79B9_7F4A_7C15);
1958            let mut z = state;
1959            z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
1960            z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
1961            z ^= z >> 31;
1962            big.push((z & 0xFF) as u8);
1963        }
1964        let work = TempDir::new().unwrap();
1965        let path = work.path().join("big.bin");
1966        fs::write(&path, &big).unwrap();
1967
1968        let (_sd1, sequential_store) = fresh_store();
1969        let (sequential_hash, _) = hash_file_with_metadata(&sequential_store, &path).unwrap();
1970
1971        let (_sd2, fanout_store) = fresh_store();
1972        let (fanout_hash, _) = hash_file_with_metadata_with(&fanout_store, &path, |sink, batch| {
1973            // Hash in reverse, then un-reverse before returning — proves
1974            // the contract cares about the *returned* order, not the
1975            // order chunks are actually processed in.
1976            let mut out: Vec<Hash> = batch
1977                .iter()
1978                .rev()
1979                .map(|chunk| store_chunk_blob(sink, chunk))
1980                .collect::<WorktreeResult<_>>()?;
1981            out.reverse();
1982            Ok(out)
1983        })
1984        .unwrap();
1985
1986        assert_eq!(
1987            sequential_hash, fanout_hash,
1988            "an out-of-order-processing hash_chunks callback must still match \
1989             the sequential default when it returns results in input order"
1990        );
1991    }
1992
1993    // ---- build_tree_from_index — the staging-area path -------------
1994
1995    use crate::index::{EntryStatus, Index, IndexEntry};
1996
1997    fn write_blob(store: &ObjectStore, bytes: &[u8]) -> Hash {
1998        let blob = Object::Blob(crate::object::Blob {
1999            data: bytes.to_vec(),
2000        });
2001        let body = serialize::serialize(&blob).unwrap();
2002        store.write(&body).unwrap()
2003    }
2004
2005    #[test]
2006    fn from_index_empty_returns_empty_tree() {
2007        let (_sd, store) = fresh_store();
2008        let idx = Index::new();
2009        let h = build_tree_from_index(&store, &idx).unwrap();
2010        let Object::Tree(t) = store.read_object(&h).unwrap() else {
2011            panic!("expected tree");
2012        };
2013        assert!(t.entries.is_empty());
2014    }
2015
2016    #[test]
2017    fn from_index_single_file_at_root() {
2018        let (_sd, store) = fresh_store();
2019        let blob_hash = write_blob(&store, b"hello world");
2020        let mut idx = Index::new();
2021        idx.entries.push(IndexEntry {
2022            path: "hello.txt".into(),
2023            status: EntryStatus::Blob,
2024            object_hash: blob_hash,
2025            mtime_ns: 0,
2026            size: 0,
2027            ino: 0,
2028            ctime_ns: 0,
2029        });
2030        let h = build_tree_from_index(&store, &idx).unwrap();
2031        let Object::Tree(t) = store.read_object(&h).unwrap() else {
2032            panic!();
2033        };
2034        assert_eq!(t.entries.len(), 1);
2035        assert_eq!(t.entries[0].name, b"hello.txt");
2036        assert_eq!(t.entries[0].mode, EntryMode::Blob);
2037        assert_eq!(t.entries[0].object_hash, blob_hash);
2038    }
2039
2040    #[test]
2041    fn from_index_nested_paths_build_subtrees() {
2042        let (_sd, store) = fresh_store();
2043        let a = write_blob(&store, b"file a");
2044        let b = write_blob(&store, b"file b");
2045        let mut idx = Index::new();
2046        idx.entries.push(IndexEntry {
2047            path: "a.txt".into(),
2048            status: EntryStatus::Blob,
2049            object_hash: a,
2050            mtime_ns: 0,
2051            size: 0,
2052            ino: 0,
2053            ctime_ns: 0,
2054        });
2055        idx.entries.push(IndexEntry {
2056            path: "subdir/b.txt".into(),
2057            status: EntryStatus::Blob,
2058            object_hash: b,
2059            mtime_ns: 0,
2060            size: 0,
2061            ino: 0,
2062            ctime_ns: 0,
2063        });
2064        let root_hash = build_tree_from_index(&store, &idx).unwrap();
2065        let Object::Tree(root) = store.read_object(&root_hash).unwrap() else {
2066            panic!();
2067        };
2068        assert_eq!(root.entries.len(), 2);
2069        assert_eq!(root.entries[0].name, b"a.txt");
2070        assert_eq!(root.entries[0].mode, EntryMode::Blob);
2071        assert_eq!(root.entries[1].name, b"subdir");
2072        assert_eq!(root.entries[1].mode, EntryMode::Tree);
2073
2074        let Object::Tree(sub) = store.read_object(&root.entries[1].object_hash).unwrap() else {
2075            panic!();
2076        };
2077        assert_eq!(sub.entries.len(), 1);
2078        assert_eq!(sub.entries[0].name, b"b.txt");
2079        assert_eq!(sub.entries[0].object_hash, b);
2080    }
2081
2082    #[test]
2083    fn from_index_removed_entries_are_skipped() {
2084        let (_sd, store) = fresh_store();
2085        let a = write_blob(&store, b"keep me");
2086        let mut idx = Index::new();
2087        idx.entries.push(IndexEntry {
2088            path: "keep.txt".into(),
2089            status: EntryStatus::Blob,
2090            object_hash: a,
2091            mtime_ns: 0,
2092            size: 0,
2093            ino: 0,
2094            ctime_ns: 0,
2095        });
2096        idx.entries.push(IndexEntry {
2097            path: "drop.txt".into(),
2098            status: EntryStatus::Removed,
2099            object_hash: [0; 32],
2100            mtime_ns: 0,
2101            size: 0,
2102            ino: 0,
2103            ctime_ns: 0,
2104        });
2105        let h = build_tree_from_index(&store, &idx).unwrap();
2106        let Object::Tree(t) = store.read_object(&h).unwrap() else {
2107            panic!();
2108        };
2109        assert_eq!(t.entries.len(), 1);
2110        assert_eq!(t.entries[0].name, b"keep.txt");
2111    }
2112
2113    #[test]
2114    fn from_index_executable_and_symlink_modes_pass_through() {
2115        let (_sd, store) = fresh_store();
2116        let exec = write_blob(&store, b"#!/bin/sh");
2117        let link = write_blob(&store, b"target.txt");
2118        let mut idx = Index::new();
2119        idx.entries.push(IndexEntry {
2120            path: "run.sh".into(),
2121            status: EntryStatus::Executable,
2122            object_hash: exec,
2123            mtime_ns: 0,
2124            size: 0,
2125            ino: 0,
2126            ctime_ns: 0,
2127        });
2128        idx.entries.push(IndexEntry {
2129            path: "link".into(),
2130            status: EntryStatus::Symlink,
2131            object_hash: link,
2132            mtime_ns: 0,
2133            size: 0,
2134            ino: 0,
2135            ctime_ns: 0,
2136        });
2137        let h = build_tree_from_index(&store, &idx).unwrap();
2138        let Object::Tree(t) = store.read_object(&h).unwrap() else {
2139            panic!();
2140        };
2141        let by_name: std::collections::HashMap<&[u8], &TreeEntry> =
2142            t.entries.iter().map(|e| (e.name.as_slice(), e)).collect();
2143        assert_eq!(by_name[&b"run.sh"[..]].mode, EntryMode::Executable);
2144        assert_eq!(by_name[&b"link"[..]].mode, EntryMode::Symlink);
2145    }
2146
2147    #[test]
2148    fn from_index_entries_are_sorted_by_name() {
2149        let (_sd, store) = fresh_store();
2150        let a = write_blob(&store, b"x");
2151        let mut idx = Index::new();
2152        // Insert out-of-order; the on-disk Tree must still be sorted
2153        // (SPEC-OBJECTS §4 normative).
2154        idx.entries.push(IndexEntry {
2155            path: "z.txt".into(),
2156            status: EntryStatus::Blob,
2157            object_hash: a,
2158            mtime_ns: 0,
2159            size: 0,
2160            ino: 0,
2161            ctime_ns: 0,
2162        });
2163        idx.entries.push(IndexEntry {
2164            path: "a.txt".into(),
2165            status: EntryStatus::Blob,
2166            object_hash: a,
2167            mtime_ns: 0,
2168            size: 0,
2169            ino: 0,
2170            ctime_ns: 0,
2171        });
2172        idx.entries.push(IndexEntry {
2173            path: "m.txt".into(),
2174            status: EntryStatus::Blob,
2175            object_hash: a,
2176            mtime_ns: 0,
2177            size: 0,
2178            ino: 0,
2179            ctime_ns: 0,
2180        });
2181        let h = build_tree_from_index(&store, &idx).unwrap();
2182        let Object::Tree(t) = store.read_object(&h).unwrap() else {
2183            panic!();
2184        };
2185        let names: Vec<&[u8]> = t.entries.iter().map(|e| e.name.as_slice()).collect();
2186        assert_eq!(names, vec![&b"a.txt"[..], b"m.txt", b"z.txt"]);
2187    }
2188
2189    #[test]
2190    fn from_index_rejects_trailing_slash() {
2191        let (_sd, store) = fresh_store();
2192        let h = write_blob(&store, b"x");
2193        let mut idx = Index::new();
2194        idx.entries.push(IndexEntry {
2195            path: "dir/".into(),
2196            status: EntryStatus::Blob,
2197            object_hash: h,
2198            mtime_ns: 0,
2199            size: 0,
2200            ino: 0,
2201            ctime_ns: 0,
2202        });
2203        let err = build_tree_from_index(&store, &idx).unwrap_err();
2204        assert!(matches!(err, WorktreeError::Io(_)));
2205    }
2206
2207    #[test]
2208    fn from_index_rejects_empty_segment() {
2209        let (_sd, store) = fresh_store();
2210        let h = write_blob(&store, b"x");
2211        let mut idx = Index::new();
2212        idx.entries.push(IndexEntry {
2213            path: "a//b.txt".into(),
2214            status: EntryStatus::Blob,
2215            object_hash: h,
2216            mtime_ns: 0,
2217            size: 0,
2218            ino: 0,
2219            ctime_ns: 0,
2220        });
2221        let err = build_tree_from_index(&store, &idx).unwrap_err();
2222        assert!(matches!(err, WorktreeError::Io(_)));
2223    }
2224
2225    #[test]
2226    fn from_index_rejects_reserved_name() {
2227        let (_sd, store) = fresh_store();
2228        let h = write_blob(&store, b"x");
2229        let mut idx = Index::new();
2230        // ".mkit" is rejected by TreeEntry::validate_name as repo
2231        // metadata aliasing.
2232        idx.entries.push(IndexEntry {
2233            path: ".mkit".into(),
2234            status: EntryStatus::Blob,
2235            object_hash: h,
2236            mtime_ns: 0,
2237            size: 0,
2238            ino: 0,
2239            ctime_ns: 0,
2240        });
2241        let err = build_tree_from_index(&store, &idx).unwrap_err();
2242        assert!(matches!(err, WorktreeError::Io(_)));
2243    }
2244
2245    /// The most important invariant: for a worktree whose contents
2246    /// match the index entry-for-entry, `build_tree` and
2247    /// `build_tree_from_index` MUST produce the identical root hash.
2248    /// If this drifts, attestations signed under one path won't
2249    /// verify against trees built under the other.
2250    #[test]
2251    fn from_index_matches_build_tree_for_equivalent_worktree() {
2252        let (_sd, store) = fresh_store();
2253
2254        // Build the same content two ways:
2255        //   1. drop files on disk, call build_tree.
2256        //   2. write blobs to the store directly, populate an index,
2257        //      call build_tree_from_index.
2258        let work = TempDir::new().unwrap();
2259        fs::write(work.path().join("a.txt"), b"alpha").unwrap();
2260        fs::create_dir(work.path().join("dir")).unwrap();
2261        fs::write(work.path().join("dir/b.txt"), b"beta").unwrap();
2262        fs::write(work.path().join("dir/c.txt"), b"gamma").unwrap();
2263        let worktree_root = build_tree(&store, work.path()).unwrap();
2264
2265        let a = write_blob(&store, b"alpha");
2266        let b = write_blob(&store, b"beta");
2267        let c = write_blob(&store, b"gamma");
2268        let mut idx = Index::new();
2269        idx.entries.push(IndexEntry {
2270            path: "a.txt".into(),
2271            status: EntryStatus::Blob,
2272            object_hash: a,
2273            mtime_ns: 0,
2274            size: 0,
2275            ino: 0,
2276            ctime_ns: 0,
2277        });
2278        idx.entries.push(IndexEntry {
2279            path: "dir/b.txt".into(),
2280            status: EntryStatus::Blob,
2281            object_hash: b,
2282            mtime_ns: 0,
2283            size: 0,
2284            ino: 0,
2285            ctime_ns: 0,
2286        });
2287        idx.entries.push(IndexEntry {
2288            path: "dir/c.txt".into(),
2289            status: EntryStatus::Blob,
2290            object_hash: c,
2291            mtime_ns: 0,
2292            size: 0,
2293            ino: 0,
2294            ctime_ns: 0,
2295        });
2296        let index_root = build_tree_from_index(&store, &idx).unwrap();
2297
2298        assert_eq!(
2299            worktree_root, index_root,
2300            "build_tree_from_index must produce the same root hash as build_tree for equivalent contents"
2301        );
2302    }
2303
2304    #[test]
2305    fn from_index_deeply_nested_paths_build_chain_of_subtrees() {
2306        let (_sd, store) = fresh_store();
2307        let h = write_blob(&store, b"deep");
2308        let mut idx = Index::new();
2309        idx.entries.push(IndexEntry {
2310            path: "a/b/c/d/e.txt".into(),
2311            status: EntryStatus::Blob,
2312            object_hash: h,
2313            mtime_ns: 0,
2314            size: 0,
2315            ino: 0,
2316            ctime_ns: 0,
2317        });
2318        let root = build_tree_from_index(&store, &idx).unwrap();
2319        let Object::Tree(t) = store.read_object(&root).unwrap() else {
2320            panic!();
2321        };
2322        assert_eq!(t.entries.len(), 1);
2323        assert_eq!(t.entries[0].name, b"a");
2324        assert_eq!(t.entries[0].mode, EntryMode::Tree);
2325        // Walk down to the leaf.
2326        let mut cursor = t.entries[0].object_hash;
2327        for seg in [b"b" as &[u8], b"c", b"d"] {
2328            let Object::Tree(t) = store.read_object(&cursor).unwrap() else {
2329                panic!();
2330            };
2331            assert_eq!(t.entries.len(), 1);
2332            assert_eq!(t.entries[0].name, seg);
2333            cursor = t.entries[0].object_hash;
2334        }
2335        let Object::Tree(t) = store.read_object(&cursor).unwrap() else {
2336            panic!();
2337        };
2338        assert_eq!(t.entries[0].name, b"e.txt");
2339        assert_eq!(t.entries[0].object_hash, h);
2340    }
2341
2342    /// Path-collision: an index that stakes the same name as both a
2343    /// blob and a directory MUST be rejected. Without the check the
2344    /// builder would happily emit two `TreeEntries` with name `a`
2345    /// (one Blob, one Tree), which the deserializer rejects under
2346    /// its strict ascending-name rule. We catch it earlier with a
2347    /// clearer error so the user knows which path needs unstaging.
2348    /// (Reviewer finding 2 on PR #103.)
2349    #[test]
2350    fn from_index_rejects_blob_then_subdir_collision() {
2351        let (_sd, store) = fresh_store();
2352        let h = write_blob(&store, b"x");
2353        let mut idx = Index::new();
2354        idx.entries.push(IndexEntry {
2355            path: "a".into(),
2356            status: EntryStatus::Blob,
2357            object_hash: h,
2358            mtime_ns: 0,
2359            size: 0,
2360            ino: 0,
2361            ctime_ns: 0,
2362        });
2363        idx.entries.push(IndexEntry {
2364            path: "a/b".into(),
2365            status: EntryStatus::Blob,
2366            object_hash: h,
2367            mtime_ns: 0,
2368            size: 0,
2369            ino: 0,
2370            ctime_ns: 0,
2371        });
2372        let err = build_tree_from_index(&store, &idx).unwrap_err();
2373        let msg = format!("{err}");
2374        assert!(
2375            msg.contains("conflict") || msg.contains("collision") || msg.contains("'a'"),
2376            "expected collision error mentioning the path, got: {msg}"
2377        );
2378    }
2379
2380    /// Same collision in the opposite stage order: subdir entry
2381    /// staged first, then a blob at the parent.
2382    #[test]
2383    fn from_index_rejects_subdir_then_blob_collision() {
2384        let (_sd, store) = fresh_store();
2385        let h = write_blob(&store, b"x");
2386        let mut idx = Index::new();
2387        idx.entries.push(IndexEntry {
2388            path: "a/b".into(),
2389            status: EntryStatus::Blob,
2390            object_hash: h,
2391            mtime_ns: 0,
2392            size: 0,
2393            ino: 0,
2394            ctime_ns: 0,
2395        });
2396        idx.entries.push(IndexEntry {
2397            path: "a".into(),
2398            status: EntryStatus::Blob,
2399            object_hash: h,
2400            mtime_ns: 0,
2401            size: 0,
2402            ino: 0,
2403            ctime_ns: 0,
2404        });
2405        assert!(build_tree_from_index(&store, &idx).is_err());
2406    }
2407
2408    #[test]
2409    fn from_index_rejects_duplicate_exact_path() {
2410        let (_sd, store) = fresh_store();
2411        let a = write_blob(&store, b"a");
2412        let b = write_blob(&store, b"b");
2413        let mut idx = Index::new();
2414        idx.entries.push(IndexEntry {
2415            path: "same.txt".into(),
2416            status: EntryStatus::Blob,
2417            object_hash: a,
2418            mtime_ns: 0,
2419            size: 0,
2420            ino: 0,
2421            ctime_ns: 0,
2422        });
2423        idx.entries.push(IndexEntry {
2424            path: "same.txt".into(),
2425            status: EntryStatus::Blob,
2426            object_hash: b,
2427            mtime_ns: 0,
2428            size: 0,
2429            ino: 0,
2430            ctime_ns: 0,
2431        });
2432
2433        let err = build_tree_from_index(&store, &idx).unwrap_err();
2434        let msg = format!("{err}");
2435        assert!(msg.contains("duplicate index path"), "got: {msg}");
2436    }
2437
2438    #[test]
2439    fn from_index_rejects_duplicate_removed_and_live_path() {
2440        let (_sd, store) = fresh_store();
2441        let h = write_blob(&store, b"live");
2442        let mut idx = Index::new();
2443        idx.entries.push(IndexEntry {
2444            path: "same.txt".into(),
2445            status: EntryStatus::Removed,
2446            object_hash: [0; 32],
2447            mtime_ns: 0,
2448            size: 0,
2449            ino: 0,
2450            ctime_ns: 0,
2451        });
2452        idx.entries.push(IndexEntry {
2453            path: "same.txt".into(),
2454            status: EntryStatus::Blob,
2455            object_hash: h,
2456            mtime_ns: 0,
2457            size: 0,
2458            ino: 0,
2459            ctime_ns: 0,
2460        });
2461
2462        let err = build_tree_from_index(&store, &idx).unwrap_err();
2463        let msg = format!("{err}");
2464        assert!(msg.contains("duplicate index path"), "got: {msg}");
2465    }
2466
2467    /// All-Removed index → empty root tree, NOT an error.
2468    /// (Reviewer finding 1 on PR #103.) `staged_count()` excludes
2469    /// Removed entries by design; the tree builder does too. The
2470    /// resulting empty tree is a valid commit target — applying a
2471    /// removals-only changeset to a tree that previously contained
2472    /// those paths produces an empty root.
2473    #[test]
2474    fn from_index_all_removed_produces_empty_tree() {
2475        let (_sd, store) = fresh_store();
2476        let mut idx = Index::new();
2477        idx.entries.push(IndexEntry {
2478            path: "gone.txt".into(),
2479            status: EntryStatus::Removed,
2480            object_hash: [0; 32],
2481            mtime_ns: 0,
2482            size: 0,
2483            ino: 0,
2484            ctime_ns: 0,
2485        });
2486        let h = build_tree_from_index(&store, &idx).unwrap();
2487        let Object::Tree(t) = store.read_object(&h).unwrap() else {
2488            panic!();
2489        };
2490        assert!(t.entries.is_empty());
2491    }
2492
2493    /// Sanity: `ObjectType::Tree` is what we materialise. Pin so a
2494    /// future enum reshuffle catches us.
2495    #[test]
2496    fn from_index_root_is_a_tree_object() {
2497        let (_sd, store) = fresh_store();
2498        let idx = Index::new();
2499        let h = build_tree_from_index(&store, &idx).unwrap();
2500        let obj = store.read_object(&h).unwrap();
2501        assert_eq!(obj.object_type(), ObjectType::Tree);
2502    }
2503
2504    #[test]
2505    fn from_index_rejects_missing_blob_object() {
2506        let (_sd, store) = fresh_store();
2507        let mut idx = Index::new();
2508        idx.entries.push(IndexEntry {
2509            path: "missing.txt".into(),
2510            status: EntryStatus::Blob,
2511            object_hash: [42; 32],
2512            mtime_ns: 0,
2513            size: 0,
2514            ino: 0,
2515            ctime_ns: 0,
2516        });
2517
2518        let err = build_tree_from_index(&store, &idx).unwrap_err();
2519        assert!(matches!(err, WorktreeError::Store(_)));
2520    }
2521
2522    #[test]
2523    fn from_index_rejects_non_blob_object_for_blob_status() {
2524        let (_sd, store) = fresh_store();
2525        let tree = Object::Tree(Tree { entries: vec![] });
2526        let body = serialize::serialize(&tree).unwrap();
2527        let tree_hash = store.write(&body).unwrap();
2528        let mut idx = Index::new();
2529        idx.entries.push(IndexEntry {
2530            path: "not-a-blob.txt".into(),
2531            status: EntryStatus::Blob,
2532            object_hash: tree_hash,
2533            mtime_ns: 0,
2534            size: 0,
2535            ino: 0,
2536            ctime_ns: 0,
2537        });
2538
2539        let err = build_tree_from_index(&store, &idx).unwrap_err();
2540        let msg = format!("{err}");
2541        assert!(
2542            msg.contains("non-blob"),
2543            "expected non-blob index object error, got: {msg}"
2544        );
2545    }
2546
2547    /// A file entry whose object is a `ChunkedBlob` (the canonical
2548    /// representation for > `CHUNK_THRESHOLD` content) is accepted by the
2549    /// commit/index tree builder, NOT rejected as "non-blob" (#203). The
2550    /// resulting tree carries an `EntryMode::Blob` pointing at the
2551    /// manifest, exactly as `build_tree` produces for a large worktree
2552    /// file.
2553    #[test]
2554    fn from_index_accepts_chunked_blob_for_file_entry() {
2555        let (_sd, store) = fresh_store();
2556        // Build a > CHUNK_THRESHOLD file's content and store it via the
2557        // shared object path (lands as a ChunkedBlob).
2558        let n = usize::try_from(CHUNK_THRESHOLD).unwrap() + 256 * 1024;
2559        let mut big = Vec::with_capacity(n);
2560        let mut state: u64 = 0x00C0_FFEE;
2561        for _ in 0..n {
2562            state = state.wrapping_add(0x9E37_79B9_7F4A_7C15);
2563            let mut z = state;
2564            z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
2565            z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
2566            z ^= z >> 31;
2567            big.push((z & 0xFF) as u8);
2568        }
2569        let chunked_hash = store_file_object(&store, &big).unwrap();
2570        assert!(
2571            matches!(
2572                store.read_object(&chunked_hash).unwrap(),
2573                Object::ChunkedBlob(_)
2574            ),
2575            "fixture must be a ChunkedBlob"
2576        );
2577
2578        let mut idx = Index::new();
2579        idx.entries.push(IndexEntry {
2580            path: "big.bin".into(),
2581            status: EntryStatus::Blob,
2582            object_hash: chunked_hash,
2583            mtime_ns: 0,
2584            size: 0,
2585            ino: 0,
2586            ctime_ns: 0,
2587        });
2588        let root = build_tree_from_index(&store, &idx).unwrap();
2589        let Object::Tree(t) = store.read_object(&root).unwrap() else {
2590            panic!("expected tree");
2591        };
2592        assert_eq!(t.entries.len(), 1);
2593        assert_eq!(t.entries[0].name, b"big.bin");
2594        assert_eq!(t.entries[0].mode, EntryMode::Blob);
2595        assert_eq!(t.entries[0].object_hash, chunked_hash);
2596        // Reassembly via the shared helper round-trips the source bytes.
2597        assert_eq!(read_blob(&store, &chunked_hash).unwrap(), big);
2598    }
2599
2600    /// SPEC-OBJECTS §7: "The concatenated length MUST equal `total_size`."
2601    /// A manifest with valid chunks but a forged `total_size` (stored under
2602    /// its real merkle id — the forgery changes the id, not its validity as
2603    /// a store entry) must fail reassembly, not silently return
2604    /// wrong-length content.
2605    #[test]
2606    fn read_blob_rejects_chunked_total_size_mismatch() {
2607        let (_sd, store) = fresh_store();
2608        let chunk = serialize::serialize(&Object::Blob(crate::object::Blob {
2609            data: b"twelve bytes".to_vec(),
2610        }))
2611        .unwrap();
2612        let chunk_hash = store.write(&chunk).unwrap();
2613        let manifest = Object::ChunkedBlob(ChunkedBlob {
2614            total_size: 999,
2615            chunk_size: 0,
2616            chunks: vec![chunk_hash],
2617        });
2618        let h = store
2619            .write(&serialize::serialize(&manifest).unwrap())
2620            .unwrap();
2621        let err = read_blob(&store, &h).unwrap_err();
2622        assert!(
2623            matches!(
2624                err,
2625                WorktreeError::Object(crate::object::MkitError::ChunkedBlobSizeMismatch {
2626                    expected: 999,
2627                    actual: 12,
2628                })
2629            ),
2630            "expected ChunkedBlobSizeMismatch, got {err:?}"
2631        );
2632    }
2633
2634    /// A symlink entry MUST still address a single `Blob` (its target
2635    /// path); a `ChunkedBlob` under a symlink entry is rejected.
2636    #[test]
2637    fn from_index_rejects_chunked_blob_for_symlink_entry() {
2638        let (_sd, store) = fresh_store();
2639        let n = usize::try_from(CHUNK_THRESHOLD).unwrap() + 256 * 1024;
2640        let big = vec![0xABu8; n];
2641        let chunked_hash = store_file_object(&store, &big).unwrap();
2642        let mut idx = Index::new();
2643        idx.entries.push(IndexEntry {
2644            path: "link".into(),
2645            status: EntryStatus::Symlink,
2646            object_hash: chunked_hash,
2647            mtime_ns: 0,
2648            size: 0,
2649            ino: 0,
2650            ctime_ns: 0,
2651        });
2652        let err = build_tree_from_index(&store, &idx).unwrap_err();
2653        assert!(format!("{err}").contains("non-blob"));
2654    }
2655
2656    // ---- batched / zero-copy ingest ----------------------------------
2657
2658    /// Same chunked input through a batch and through the plain store
2659    /// must produce the identical manifest hash and identical readable
2660    /// bytes — this transitively pins the zero-copy `put_parts` chunk
2661    /// path to the golden `ChunkedBlob` vectors.
2662    #[test]
2663    fn store_file_object_via_batch_equals_via_store() {
2664        // 3 MiB of varied bytes: above CHUNK_THRESHOLD, multiple chunks.
2665        let data: Vec<u8> = (0..3 * 1024 * 1024u32)
2666            .map(|i| u8::try_from((i.wrapping_mul(2_654_435_761)) % 251).unwrap())
2667            .collect();
2668
2669        let (_d1, store1) = fresh_store();
2670        let h_store = store_file_object(&store1, &data).unwrap();
2671
2672        let (_d2, store2) = fresh_store();
2673        let batch = store2.batch();
2674        let h_batch = store_file_object(&batch, &data).unwrap();
2675        batch.commit().unwrap();
2676
2677        assert_eq!(h_store, h_batch, "sink choice must not change hashes");
2678        assert_eq!(
2679            read_blob(&store1, &h_store).unwrap(),
2680            read_blob(&store2, &h_batch).unwrap(),
2681        );
2682
2683        // Small (single-blob) shape too.
2684        let small = b"under the chunk threshold";
2685        let h1 = store_file_object(&store1, small).unwrap();
2686        let batch2 = store2.batch();
2687        let h2 = store_file_object(&batch2, small).unwrap();
2688        batch2.commit().unwrap();
2689        assert_eq!(h1, h2);
2690    }
2691
2692    /// Committing a staged index must cost exactly one full flush no
2693    /// matter how many tree objects it materialises.
2694    #[test]
2695    fn build_tree_from_index_with_batch_single_flush() {
2696        use crate::batch::testing::{Ev, RecordingSyncer};
2697        use crate::index::{EntryStatus, Index, IndexEntry};
2698        use std::sync::Arc;
2699
2700        let (_sd, mut store) = fresh_store();
2701        // Stage 20 files across nested dirs (many tree objects).
2702        let mut idx = Index::default();
2703        for i in 0..20 {
2704            let blob = Object::Blob(crate::object::Blob {
2705                data: format!("file {i}").into_bytes(),
2706            });
2707            let bytes = serialize::serialize(&blob).unwrap();
2708            let h = store.write(&bytes).unwrap();
2709            idx.entries.push(IndexEntry {
2710                status: EntryStatus::Blob,
2711                object_hash: h,
2712                path: format!("d{}/sub/f{i}.txt", i % 5),
2713                mtime_ns: 0,
2714                size: 0,
2715                ino: 0,
2716                ctime_ns: 0,
2717            });
2718        }
2719
2720        let rec = Arc::new(RecordingSyncer::default());
2721        store.set_syncer(rec.clone());
2722
2723        let batch = store.batch();
2724        let tree_h = build_tree_from_index_with(&store, &batch, &idx, true).unwrap();
2725        batch.commit().unwrap();
2726
2727        let fulls = rec
2728            .events()
2729            .iter()
2730            .filter(|e| matches!(e, Ev::Full(_)))
2731            .count();
2732        assert_eq!(fulls, 2, "tree materialisation flush cost must be constant");
2733        assert!(store.read_object(&tree_h).is_ok());
2734
2735        // Equivalence: the per-object path yields the same root hash.
2736        let (_sd2, store2) = fresh_store();
2737        for i in 0..20 {
2738            let blob = Object::Blob(crate::object::Blob {
2739                data: format!("file {i}").into_bytes(),
2740            });
2741            store2.write(&serialize::serialize(&blob).unwrap()).unwrap();
2742        }
2743        assert_eq!(tree_h, build_tree_from_index(&store2, &idx).unwrap());
2744    }
2745
2746    /// A staged object corrupted after `add` must NOT be publishable: the
2747    /// verifying path (commit and friends) rejects it, while the cheap
2748    /// non-verifying path (status/diff snapshots) still accepts the shape.
2749    #[test]
2750    fn build_tree_from_index_verify_rejects_corrupt_staged_object() {
2751        use crate::index::{EntryStatus, Index, IndexEntry};
2752
2753        let (_sd, store) = fresh_store();
2754        let blob = Object::Blob(crate::object::Blob {
2755            data: b"hello".to_vec(),
2756        });
2757        let h = store.write(&serialize::serialize(&blob).unwrap()).unwrap();
2758        let mut idx = Index::default();
2759        idx.entries.push(IndexEntry {
2760            status: EntryStatus::Blob,
2761            object_hash: h,
2762            path: "a.txt".to_string(),
2763            mtime_ns: 0,
2764            size: 0,
2765            ino: 0,
2766            ctime_ns: 0,
2767        });
2768
2769        // Clean object: both paths succeed and agree.
2770        assert!(build_tree_from_index_with(&store, &store, &idx, true).is_ok());
2771        assert!(build_tree_from_index_with(&store, &store, &idx, false).is_ok());
2772
2773        // Corrupt a payload byte past the 6-byte prologue (the prologue
2774        // shape stays valid, so only a re-hash can catch it).
2775        let path = store.path_for(&h);
2776        let mut bytes = std::fs::read(&path).unwrap();
2777        let i = bytes.len() - 1;
2778        bytes[i] ^= 0xFF;
2779        std::fs::write(&path, &bytes).unwrap();
2780
2781        // Verifying path refuses to publish the corrupt object…
2782        assert!(
2783            build_tree_from_index_with(&store, &store, &idx, true).is_err(),
2784            "commit-path tree build must reject a corrupt staged object"
2785        );
2786        // …but the cheap snapshot path still passes the prologue shape check.
2787        assert!(
2788            build_tree_from_index_with(&store, &store, &idx, false).is_ok(),
2789            "status/diff snapshot path keeps the cheap prologue-only check"
2790        );
2791    }
2792
2793    /// The batched pass-2 staged-object check (`probe_staged_objects`)
2794    /// must not change *which* error an index with more than one kind
2795    /// of problem reports — the sequential (default, non-parallel)
2796    /// path has to surface the same error the original single
2797    /// interleaved loop would have: whichever entry comes first in
2798    /// index order, checked dup-path, then status, then staged-object
2799    /// shape, in that sub-order. Here entry 0's staged hash was never
2800    /// written (fails the staged-object check) while entry 1 uses the
2801    /// reserved `Tree` status (fails the pass-1 status check) — entry
2802    /// 0 comes first, so its error must win even though pass 1 alone
2803    /// would reach entry 1's problem without ever touching the store.
2804    #[test]
2805    fn build_tree_from_index_earlier_object_error_wins_over_later_status_error() {
2806        use crate::index::{EntryStatus, Index, IndexEntry};
2807
2808        let (_sd, store) = fresh_store();
2809        let mut idx = Index::default();
2810        idx.entries.push(IndexEntry {
2811            status: EntryStatus::Blob,
2812            object_hash: [0xAB; 32],
2813            path: "a.txt".to_string(),
2814            mtime_ns: 0,
2815            size: 0,
2816            ino: 0,
2817            ctime_ns: 0,
2818        });
2819        idx.entries.push(IndexEntry {
2820            status: EntryStatus::Tree,
2821            object_hash: [0; 32],
2822            path: "b".to_string(),
2823            mtime_ns: 0,
2824            size: 0,
2825            ino: 0,
2826            ctime_ns: 0,
2827        });
2828
2829        let err = build_tree_from_index_with(&store, &store, &idx, false)
2830            .expect_err("index has no valid entries");
2831        let msg = err.to_string();
2832        assert!(
2833            msg.contains("ab") || msg.to_lowercase().contains("not found"),
2834            "expected entry 0's missing-object error, got: {msg}"
2835        );
2836        assert!(
2837            !msg.contains("reserved Tree status"),
2838            "entry 1's status error must not preempt entry 0's earlier object error, got: {msg}"
2839        );
2840    }
2841
2842    // ---- pure hashing + stat cache ------------------------------------
2843
2844    /// `hash_file_object` must agree with `store_file_object` on every
2845    /// input shape (single blob, chunked) without touching any store.
2846    #[test]
2847    fn hash_file_object_equals_store_file_object() {
2848        let threshold = usize::try_from(CHUNK_THRESHOLD).unwrap();
2849        for len in [0usize, 1, 1024, threshold, 3 * 1024 * 1024] {
2850            let data: Vec<u8> = (0..len)
2851                .map(|i| u8::try_from((i * 31 + 7) % 251).unwrap())
2852                .collect();
2853            let (_sd, store) = fresh_store();
2854            let stored = store_file_object(&store, &data).unwrap();
2855            let pure = hash_file_object(&data).unwrap();
2856            assert_eq!(stored, pure, "len {len}: pure hash must match stored hash");
2857        }
2858    }
2859
2860    #[test]
2861    fn hash_file_object_writes_nothing() {
2862        let (_sd, store) = fresh_store();
2863        let data = vec![0xAB; 2 * 1024 * 1024]; // chunked shape
2864        let _ = hash_file_object(&data).unwrap();
2865        assert!(
2866            store.iter_object_hashes().unwrap().is_empty(),
2867            "pure hashing must not create objects"
2868        );
2869    }
2870
2871    fn meta_of(p: &Path) -> fs::Metadata {
2872        p.symlink_metadata().unwrap()
2873    }
2874
2875    #[test]
2876    fn stat_matches_requires_nonzero_mtime_and_equal_fields() {
2877        let work = TempDir::new().unwrap();
2878        let f = work.path().join("a.txt");
2879        fs::write(&f, b"hello").unwrap();
2880        let meta = meta_of(&f);
2881        let entry = crate::index::IndexEntry {
2882            path: "a.txt".into(),
2883            status: crate::index::EntryStatus::Blob,
2884            object_hash: crate::hash::hash(b"irrelevant"),
2885            mtime_ns: mtime_nanos(&meta),
2886            size: meta.len(),
2887            ino: 0,
2888            ctime_ns: 0,
2889        };
2890        assert!(stat_matches(&entry, &meta));
2891
2892        // Zero mtime sentinel: never matches.
2893        let mut zeroed = entry.clone();
2894        zeroed.mtime_ns = 0;
2895        assert!(!stat_matches(&zeroed, &meta), "zero sentinel must re-hash");
2896
2897        // Size mismatch.
2898        let mut wrong_size = entry.clone();
2899        wrong_size.size += 1;
2900        assert!(!stat_matches(&wrong_size, &meta));
2901
2902        // Mtime mismatch.
2903        let mut wrong_time = entry.clone();
2904        wrong_time.mtime_ns ^= 1;
2905        assert!(!stat_matches(&wrong_time, &meta));
2906    }
2907
2908    #[cfg(unix)]
2909    #[test]
2910    fn stat_matches_detects_exec_bit_flip() {
2911        use std::os::unix::fs::PermissionsExt;
2912        let work = TempDir::new().unwrap();
2913        let f = work.path().join("run.sh");
2914        fs::write(&f, b"#!/bin/sh\n").unwrap();
2915        let meta = meta_of(&f);
2916        let entry = crate::index::IndexEntry {
2917            path: "run.sh".into(),
2918            status: crate::index::EntryStatus::Blob,
2919            object_hash: crate::hash::hash(b"x"),
2920            mtime_ns: mtime_nanos(&meta),
2921            size: meta.len(),
2922            ino: 0,
2923            ctime_ns: 0,
2924        };
2925        assert!(stat_matches(&entry, &meta));
2926        // chmod +x without touching content, then restore the mtime so
2927        // ONLY the mode differs — the exec-class check must still fire.
2928        let mtime = meta.modified().unwrap();
2929        fs::set_permissions(&f, fs::Permissions::from_mode(0o755)).unwrap();
2930        let f_handle = fs::File::options().write(true).open(&f).unwrap();
2931        f_handle
2932            .set_times(fs::FileTimes::new().set_modified(mtime))
2933            .unwrap();
2934        drop(f_handle);
2935        let meta2 = meta_of(&f);
2936        assert_eq!(mtime_nanos(&meta2), entry.mtime_ns, "mtime restored");
2937        assert!(
2938            !stat_matches(&entry, &meta2),
2939            "exec-bit flip must invalidate a Blob-status cache hit"
2940        );
2941    }
2942
2943    /// The killer observable: a stat-matched file is NEVER opened. With
2944    /// the file made unreadable (chmod 000), the tree build still
2945    /// succeeds and reuses the staged hash.
2946    #[cfg(unix)]
2947    #[test]
2948    fn build_tree_reuses_hash_on_stat_match_without_reading_file() {
2949        use std::os::unix::fs::PermissionsExt;
2950        let (_sd, store) = fresh_store();
2951        let work = TempDir::new().unwrap();
2952        let f = work.path().join("locked.txt");
2953        fs::write(&f, b"cached content").unwrap();
2954
2955        let staged_hash = store_file_object(&store, b"cached content").unwrap();
2956        let meta = meta_of(&f);
2957        let idx = crate::index::Index::from_entries(vec![crate::index::IndexEntry {
2958            path: "locked.txt".into(),
2959            status: crate::index::EntryStatus::Blob,
2960            object_hash: staged_hash,
2961            mtime_ns: mtime_nanos(&meta),
2962            size: meta.len(),
2963            ino: 0,
2964            ctime_ns: 0,
2965        }]);
2966
2967        // Make any read attempt error out.
2968        fs::set_permissions(&f, fs::Permissions::from_mode(0o000)).unwrap();
2969        let result = build_tree_filtered(&store, work.path(), Some(&idx));
2970        fs::set_permissions(&f, fs::Permissions::from_mode(0o644)).unwrap();
2971        let tree_h = result.expect("stat match must skip the file read");
2972
2973        let Object::Tree(t) = store.read_object(&tree_h).unwrap() else {
2974            panic!("expected tree");
2975        };
2976        assert_eq!(t.entries.len(), 1);
2977        assert_eq!(t.entries[0].object_hash, staged_hash);
2978
2979        // Same content hashed normally yields the identical tree.
2980        let (_sd2, store2) = fresh_store();
2981        let f2_dir = TempDir::new().unwrap();
2982        fs::write(f2_dir.path().join("locked.txt"), b"cached content").unwrap();
2983        let plain = build_tree(&store2, f2_dir.path()).unwrap();
2984        assert_eq!(plain, tree_h, "cache hit must not change tree hashes");
2985    }
2986
2987    /// `hash_pending_files_parallel`'s fan-out branch only fires once a
2988    /// single directory has at least `FILE_HASH_MISSES_PER_THREAD *`
2989    /// the machine's thread count worth of cache-miss files — every
2990    /// other `build_tree` test in this module stays well under that, so
2991    /// none of them ever exercise it. 4096 distinct-content files is
2992    /// comfortably above that threshold on any real machine (including
2993    /// CI runners with far more than 4 cores), forcing the parallel
2994    /// branch to run rather than just its sequential fallback.
2995    ///
2996    /// Every file's expected content is unique and includes its index,
2997    /// so a slot/result misalignment in the parallel path (a chunk's
2998    /// results landing at the wrong base offset, or `entries`/`misses`
2999    /// getting out of step) would surface here as an entry whose hash
3000    /// belongs to a different file than its name says.
3001    #[test]
3002    fn build_tree_parallel_fanout_assigns_each_hash_to_the_right_file() {
3003        const N: usize = 4096;
3004
3005        let (_sd, store) = fresh_store();
3006        let work = TempDir::new().unwrap();
3007        for i in 0..N {
3008            fs::write(
3009                work.path().join(format!("f{i:04}.txt")),
3010                format!("content {i}"),
3011            )
3012            .unwrap();
3013        }
3014
3015        // No index passed: every file is untracked, so every one is a
3016        // cache miss (never a stat-match short-circuit) regardless of
3017        // stat-cache behavior.
3018        let h = build_tree(&store, work.path()).unwrap();
3019        let Object::Tree(t) = store.read_object(&h).unwrap() else {
3020            panic!("expected tree");
3021        };
3022        assert_eq!(t.entries.len(), N);
3023
3024        let mut seen_hashes = std::collections::HashSet::with_capacity(N);
3025        for (i, entry) in t.entries.iter().enumerate() {
3026            // Tree entries are sorted by name, and zero-padded decimal
3027            // names sort in the same order as their numeric index.
3028            let expected_name = format!("f{i:04}.txt");
3029            assert_eq!(
3030                entry.name,
3031                expected_name.as_bytes(),
3032                "entry {i} out of order"
3033            );
3034
3035            let expected_content = format!("content {i}");
3036            let expected_hash = hash_file_object(expected_content.as_bytes()).unwrap();
3037            assert_eq!(
3038                entry.object_hash, expected_hash,
3039                "entry {i} ({expected_name}) has the wrong hash — parallel slot misalignment?"
3040            );
3041            assert!(
3042                seen_hashes.insert(entry.object_hash),
3043                "entry {i} repeats another entry's hash (or the ZERO placeholder)"
3044            );
3045        }
3046    }
3047
3048    /// Replace-by-rename with preserved mtime+size must be caught by
3049    /// the inode check: the replacement file has a different ino.
3050    #[cfg(unix)]
3051    #[test]
3052    fn stat_mismatch_on_inode_rehashes() {
3053        let work = TempDir::new().unwrap();
3054        let f = work.path().join("swap.txt");
3055        fs::write(&f, b"original").unwrap();
3056        let meta = meta_of(&f);
3057        let (mtime_ns, size, ino, ctime_ns) = stat_cache_fields(&meta);
3058        let entry = crate::index::IndexEntry {
3059            path: "swap.txt".into(),
3060            status: crate::index::EntryStatus::Blob,
3061            object_hash: crate::hash::hash(b"original"),
3062            mtime_ns,
3063            size,
3064            ino,
3065            ctime_ns,
3066        };
3067        assert!(stat_matches(&entry, &meta));
3068
3069        // Same-size replacement via rename with timestamps restored —
3070        // the tar -x / rsync -t / mv-of-prepared-file shape.
3071        let staging = work.path().join(".swap.new");
3072        fs::write(&staging, b"REPLACED").unwrap(); // same 8-byte size
3073        let fh = fs::File::options().write(true).open(&staging).unwrap();
3074        fh.set_times(fs::FileTimes::new().set_modified(meta.modified().unwrap()))
3075            .unwrap();
3076        drop(fh);
3077        fs::rename(&staging, &f).unwrap();
3078        let meta2 = meta_of(&f);
3079        assert_eq!(meta2.len(), entry.size, "size preserved by the swap");
3080        assert!(
3081            !stat_matches(&entry, &meta2),
3082            "a renamed-in replacement must not stat-match (ino differs)"
3083        );
3084    }
3085
3086    /// A recorded ctime that disagrees with the live one must miss —
3087    /// ctime cannot be restored from userspace, so `touch -r` after an
3088    /// in-place edit is caught even when mtime+size+ino all match.
3089    #[test]
3090    fn stat_mismatch_on_ctime_rehashes() {
3091        let work = TempDir::new().unwrap();
3092        let f = work.path().join("touched.txt");
3093        fs::write(&f, b"content").unwrap();
3094        let meta = meta_of(&f);
3095        let (mtime_ns, size, ino, ctime_ns) = stat_cache_fields(&meta);
3096        if ctime_ns == 0 {
3097            return; // platform without ctime — check not applicable
3098        }
3099        let entry = crate::index::IndexEntry {
3100            path: "touched.txt".into(),
3101            status: crate::index::EntryStatus::Blob,
3102            object_hash: crate::hash::hash(b"content"),
3103            mtime_ns,
3104            size,
3105            ino,
3106            ctime_ns: ctime_ns ^ 1,
3107        };
3108        assert!(
3109            !stat_matches(&entry, &meta),
3110            "ctime disagreement must invalidate the cache"
3111        );
3112    }
3113
3114    /// The worktree walk must report hash-time observations for entries
3115    /// whose cache was absent but whose content re-hashed to the staged
3116    /// hash — and the observation must carry the fd-stat, enabling the
3117    /// status command to heal the cache soundly.
3118    #[test]
3119    fn build_tree_observed_reports_clean_rehashes() {
3120        let (_sd, store) = fresh_store();
3121        let work = TempDir::new().unwrap();
3122        fs::write(work.path().join("clean.txt"), b"clean bytes").unwrap();
3123        fs::write(work.path().join("dirty.txt"), b"new content").unwrap();
3124
3125        let clean_hash = store_file_object(&store, b"clean bytes").unwrap();
3126        let stale_hash = crate::hash::hash(b"old content");
3127        let idx = crate::index::Index::from_entries(vec![
3128            crate::index::IndexEntry {
3129                path: "clean.txt".into(),
3130                status: crate::index::EntryStatus::Blob,
3131                object_hash: clean_hash,
3132                mtime_ns: 0, // racy-smudged: forces a re-hash
3133                size: 0,
3134                ino: 0,
3135                ctime_ns: 0,
3136            },
3137            crate::index::IndexEntry {
3138                path: "dirty.txt".into(),
3139                status: crate::index::EntryStatus::Blob,
3140                object_hash: stale_hash,
3141                mtime_ns: 0,
3142                size: 0,
3143                ino: 0,
3144                ctime_ns: 0,
3145            },
3146        ]);
3147        let mut obs = Vec::new();
3148        build_tree_filtered_observed(&store, work.path(), Some(&idx), &mut obs).unwrap();
3149
3150        assert_eq!(obs.len(), 1, "only the verified-clean entry is observed");
3151        let o = &obs[0];
3152        assert_eq!(o.path, "clean.txt");
3153        assert_eq!(o.object_hash, clean_hash);
3154        let meta = meta_of(&work.path().join("clean.txt"));
3155        let (mtime_ns, size, _ino, _ctime) = stat_cache_fields(&meta);
3156        assert_eq!(o.mtime_ns, mtime_ns, "observation carries the fd stat");
3157        assert_eq!(o.size, size);
3158    }
3159
3160    /// A stat MISMATCH must fall back to re-hashing the live content.
3161    #[test]
3162    fn build_tree_rehashes_on_stat_mismatch() {
3163        let (_sd, store) = fresh_store();
3164        let work = TempDir::new().unwrap();
3165        let f = work.path().join("changed.txt");
3166        fs::write(&f, b"new content").unwrap();
3167        let stale_hash = crate::hash::hash(b"not the real object");
3168        let meta = meta_of(&f);
3169        let idx = crate::index::Index::from_entries(vec![crate::index::IndexEntry {
3170            path: "changed.txt".into(),
3171            status: crate::index::EntryStatus::Blob,
3172            object_hash: stale_hash,
3173            // size deliberately wrong → mismatch → re-hash.
3174            mtime_ns: mtime_nanos(&meta),
3175            size: meta.len() + 1,
3176            ino: 0,
3177            ctime_ns: 0,
3178        }]);
3179        let tree_h = build_tree_filtered(&store, work.path(), Some(&idx)).unwrap();
3180        let Object::Tree(t) = store.read_object(&tree_h).unwrap() else {
3181            panic!("expected tree");
3182        };
3183        assert_ne!(
3184            t.entries[0].object_hash, stale_hash,
3185            "mismatched stat must not reuse the stale hash"
3186        );
3187        assert_eq!(
3188            t.entries[0].object_hash,
3189            store_file_object(&store, b"new content").unwrap()
3190        );
3191    }
3192}