use std::fs;
use std::io::{self, Read};
use std::path::{Path, PathBuf};
use crate::chunker::{ChunkIterator, ChunkReader, FastCdc};
use crate::hash::Hash;
use crate::ignore::{self, IgnoreList};
use crate::index::{self, Index};
use crate::object::{ChunkedBlob, EntryMode, Object, Tree, TreeEntry};
use crate::serialize;
use crate::store::{ObjectSink, ObjectStore};
pub const CHUNK_THRESHOLD: u64 = 1024 * 1024;
pub const MAX_FILE_BYTES: u64 = 1024 * 1024 * 1024;
#[derive(Debug, thiserror::Error)]
pub enum WorktreeError {
#[error("symlink target '{0}' is invalid (absolute or contains '..')")]
InvalidSymlinkTarget(String),
#[error("file '{0}' exceeds the {MAX_FILE_BYTES} byte limit")]
FileTooLarge(PathBuf),
#[error("path component is not valid UTF-8")]
InvalidUtf8,
#[error(transparent)]
Io(#[from] io::Error),
#[error(transparent)]
Object(#[from] crate::object::MkitError),
#[error(transparent)]
Store(#[from] crate::store::StoreError),
#[error("hash_chunks callback returned {actual} hashes for a {expected}-chunk batch")]
ChunkBatchLengthMismatch { expected: usize, actual: usize },
}
pub type WorktreeResult<T> = Result<T, WorktreeError>;
mod blob;
pub use blob::{LoadedBlob, content_eq, content_eq_bytes, content_fingerprint, read_blob};
#[must_use]
pub fn validate_symlink_target(target: &str) -> bool {
if target.is_empty() {
return false;
}
if target.starts_with('/') {
return false;
}
for part in target.split('/') {
if part == ".." {
return false;
}
}
true
}
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct StatObservation {
pub path: String,
pub object_hash: Hash,
pub mtime_ns: u64,
pub size: u64,
pub ino: u64,
pub ctime_ns: u64,
}
pub fn build_tree<S: ObjectSink + Sync + ?Sized>(sink: &S, dir: &Path) -> WorktreeResult<Hash> {
build_tree_filtered(sink, dir, None)
}
pub fn build_tree_filtered<S: ObjectSink + Sync + ?Sized>(
sink: &S,
dir: &Path,
index: Option<&Index>,
) -> WorktreeResult<Hash> {
build_tree_filtered_observed(sink, dir, index, &mut Vec::new())
}
pub fn build_tree_filtered_observed<S: ObjectSink + Sync + ?Sized>(
sink: &S,
dir: &Path,
index: Option<&Index>,
observations: &mut Vec<StatObservation>,
) -> WorktreeResult<Hash> {
build_tree_observed_impl(sink, None, dir, index, observations)
}
pub fn build_tree_filtered_observed_with_source<S: ObjectSink + Sync + ?Sized>(
sink: &S,
source: &(dyn crate::store::ObjectSource + Sync),
dir: &Path,
index: Option<&Index>,
observations: &mut Vec<StatObservation>,
) -> WorktreeResult<Hash> {
build_tree_observed_impl(sink, Some(source), dir, index, observations)
}
fn build_tree_observed_impl<S: ObjectSink + Sync + ?Sized>(
sink: &S,
source: Option<&(dyn crate::store::ObjectSource + Sync)>,
dir: &Path,
index: Option<&Index>,
observations: &mut Vec<StatObservation>,
) -> WorktreeResult<Hash> {
let ignores = ignore::load(dir).map_err(|e| match e {
crate::ignore::IgnoreError::Io(io) => WorktreeError::Io(io),
crate::ignore::IgnoreError::FileTooLarge => {
WorktreeError::Io(io::Error::other("ignore file exceeds 1 MiB"))
}
})?;
let loaded;
let index = if let Some(i) = index {
i
} else {
loaded = index::read_index(&crate::layout::RepoLayout::single(dir))
.map_err(|e| WorktreeError::Io(io::Error::other(e)))?;
&loaded
};
let by_path: std::collections::HashMap<&str, &crate::index::IndexEntry> =
index.entries.iter().map(|e| (e.path.as_str(), e)).collect();
build_tree_inner(
sink,
source,
dir,
"",
&ignores,
index,
&by_path,
false,
observations,
)
}
fn retained_content_hash(
source: Option<&(dyn crate::store::ObjectSource + Sync)>,
indexed: Option<&crate::index::IndexEntry>,
fresh: Hash,
) -> WorktreeResult<Hash> {
if let (Some(entry), Some(source)) = (indexed, source)
&& entry.status != crate::index::EntryStatus::Removed
&& content_eq(source, &entry.object_hash, &fresh)?
{
return Ok(entry.object_hash);
}
Ok(fresh)
}
#[allow(clippy::too_many_arguments)]
fn build_tree_inner<S: ObjectSink + Sync + ?Sized>(
sink: &S,
source: Option<&(dyn crate::store::ObjectSource + Sync)>,
dir: &Path,
rel_dir: &str,
ignores: &IgnoreList,
index: &Index,
by_path: &std::collections::HashMap<&str, &crate::index::IndexEntry>,
parent_ignored: bool,
observations: &mut Vec<StatObservation>,
) -> WorktreeResult<Hash> {
let mut entries: Vec<TreeEntry> = Vec::new();
let mut misses: Vec<PendingFileHash<'_>> = Vec::new();
for entry in fs::read_dir(dir)? {
let entry = entry?;
let file_name = entry.file_name();
let name_str = file_name
.to_str()
.ok_or(WorktreeError::InvalidUtf8)?
.to_string();
let meta = entry.path().symlink_metadata()?;
let is_dir = meta.is_dir();
let rel_path = if rel_dir.is_empty() {
name_str.clone()
} else {
format!("{rel_dir}/{name_str}")
};
let entry_ignored = parent_ignored || ignores.is_ignored(&rel_path, is_dir);
if entry_ignored && !index.tracks_path_or_descendant(&rel_path) {
continue;
}
let name_bytes = name_str.as_bytes();
if !TreeEntry::validate_name(name_bytes) {
return Err(WorktreeError::Io(io::Error::new(
io::ErrorKind::InvalidInput,
format!("invalid tree entry name: {name_str:?}"),
)));
}
if meta.file_type().is_file() {
let indexed = by_path.get(rel_path.as_str()).copied();
let cached = indexed.filter(|e| stat_matches(e, &meta));
if let Some(e) = cached {
entries.push(TreeEntry {
name: name_str.into_bytes(),
mode: entry_mode_from_file_metadata(&meta),
object_hash: e.object_hash,
});
} else {
let slot = entries.len();
entries.push(TreeEntry {
name: name_str.into_bytes(),
mode: EntryMode::Blob,
object_hash: crate::hash::ZERO,
});
misses.push(PendingFileHash {
slot,
abs_path: entry.path(),
rel_path,
indexed,
});
}
} else if meta.file_type().is_dir() {
if index.has_tracked_file_at(&rel_path) {
continue;
}
let h = build_tree_inner(
sink,
source,
&entry.path(),
&rel_path,
ignores,
index,
by_path,
entry_ignored,
observations,
)?;
entries.push(TreeEntry {
name: name_str.into_bytes(),
mode: EntryMode::Tree,
object_hash: h,
});
} else if meta.file_type().is_symlink() {
let target = fs::read_link(entry.path())?;
let target_str = target
.to_str()
.ok_or(WorktreeError::InvalidUtf8)?
.to_string();
if !validate_symlink_target(&target_str) {
return Err(WorktreeError::InvalidSymlinkTarget(target_str));
}
let target_bytes = target_str.as_bytes();
let prologue = serialize::blob_prologue(target_bytes.len())?;
let h = sink.put_parts(&[&prologue, target_bytes])?;
entries.push(TreeEntry {
name: name_str.into_bytes(),
mode: EntryMode::Symlink,
object_hash: h,
});
} else {
}
}
if !misses.is_empty() {
hash_pending_files(sink, source, &misses, &mut entries, observations)?;
}
entries.sort_by(|a, b| a.name.cmp(&b.name));
let tree = Object::Tree(Tree { entries });
let bytes = serialize::serialize(&tree)?;
Ok(sink.put(&bytes)?)
}
struct PendingFileHash<'e> {
slot: usize,
abs_path: PathBuf,
rel_path: String,
indexed: Option<&'e crate::index::IndexEntry>,
}
#[cfg(not(target_arch = "wasm32"))]
const FILE_HASH_MISSES_PER_THREAD: usize = 8;
fn hash_pending_files<S: ObjectSink + Sync + ?Sized>(
sink: &S,
source: Option<&(dyn crate::store::ObjectSource + Sync)>,
misses: &[PendingFileHash<'_>],
entries: &mut [TreeEntry],
observations: &mut Vec<StatObservation>,
) -> WorktreeResult<()> {
#[cfg(not(target_arch = "wasm32"))]
{
let threads = available_threads();
if threads > 1 && misses.len() >= FILE_HASH_MISSES_PER_THREAD.saturating_mul(threads) {
let results = hash_pending_files_parallel(sink, source, misses, threads)?;
apply_hash_results(misses, results, entries, observations);
return Ok(());
}
}
let results = misses
.iter()
.map(|m| hash_one_pending_file(sink, source, m))
.collect::<WorktreeResult<Vec<_>>>()?;
apply_hash_results(misses, results, entries, observations);
Ok(())
}
fn hash_one_pending_file<S: ObjectSink + ?Sized>(
sink: &S,
source: Option<&(dyn crate::store::ObjectSource + Sync)>,
m: &PendingFileHash<'_>,
) -> WorktreeResult<(Hash, EntryMode, Option<StatObservation>)> {
let (mut h, opened_meta) = hash_file_with_metadata(sink, &m.abs_path)?;
h = retained_content_hash(source, m.indexed, h)?;
let observation = if let Some(e) = m.indexed
&& e.object_hash == h
{
let (mtime_ns, size, ino, ctime_ns) = stat_cache_fields(&opened_meta);
Some(StatObservation {
path: m.rel_path.clone(),
object_hash: h,
mtime_ns,
size,
ino,
ctime_ns,
})
} else {
None
};
Ok((h, entry_mode_from_file_metadata(&opened_meta), observation))
}
#[cfg(not(target_arch = "wasm32"))]
fn hash_pending_files_parallel<S: ObjectSink + Sync + ?Sized>(
sink: &S,
source: Option<&(dyn crate::store::ObjectSource + Sync)>,
misses: &[PendingFileHash<'_>],
threads: usize,
) -> WorktreeResult<Vec<(Hash, EntryMode, Option<StatObservation>)>> {
chunked_scoped_map(misses, threads, |m| hash_one_pending_file(sink, source, m))
}
fn apply_hash_results(
misses: &[PendingFileHash<'_>],
results: Vec<(Hash, EntryMode, Option<StatObservation>)>,
entries: &mut [TreeEntry],
observations: &mut Vec<StatObservation>,
) {
for (m, (hash, mode, observation)) in misses.iter().zip(results) {
entries[m.slot].object_hash = hash;
entries[m.slot].mode = mode;
if let Some(obs) = observation {
observations.push(obs);
}
}
}
pub fn build_tree_from_index(
store: &ObjectStore,
index: &crate::index::Index,
) -> WorktreeResult<Hash> {
build_tree_from_index_with(store, store, index, true)
}
#[allow(clippy::items_after_statements, clippy::too_many_lines)]
pub fn build_tree_from_index_with<S: ObjectSink + ?Sized>(
store: &ObjectStore,
sink: &S,
index: &crate::index::Index,
verify: bool,
) -> WorktreeResult<Hash> {
use crate::index::EntryStatus;
#[derive(Default)]
struct Node {
children: std::collections::BTreeMap<String, Node>,
leaves: std::collections::BTreeMap<String, (EntryMode, Hash)>,
}
let mut root = Node::default();
let mut seen_paths = std::collections::HashSet::with_capacity(index.entries.len());
let mut kept: Vec<(&str, EntryMode, Hash)> = Vec::with_capacity(index.entries.len());
let mut deferred_status_err: Option<WorktreeError> = None;
for entry in &index.entries {
if !seen_paths.insert(entry.path.as_str()) {
deferred_status_err = Some(WorktreeError::Io(io::Error::other(format!(
"duplicate index path: '{}'",
entry.path
))));
break;
}
if entry.status == EntryStatus::Removed {
continue;
}
let mode = match entry.status {
EntryStatus::Blob => EntryMode::Blob,
EntryStatus::Executable => EntryMode::Executable,
EntryStatus::Symlink => EntryMode::Symlink,
EntryStatus::Tree => {
deferred_status_err = Some(WorktreeError::Io(io::Error::other(
"index entry uses reserved Tree status (subtree staging not implemented)",
)));
break;
}
EntryStatus::Removed => unreachable!("filtered above"),
};
kept.push((entry.path.as_str(), mode, entry.object_hash));
}
probe_staged_objects(store, &kept, verify)?;
if let Some(err) = deferred_status_err {
return Err(err);
}
for (path, mode, object_hash) in kept {
let segments: Vec<&str> = path.split('/').collect();
let Some((leaf, dirs)) = segments.split_last() else {
return Err(WorktreeError::Io(io::Error::other("empty index path")));
};
if leaf.is_empty() {
return Err(WorktreeError::Io(io::Error::other(
"trailing slash in index path",
)));
}
let mut node = &mut root;
let mut walked = String::new();
for seg in dirs {
if seg.is_empty() {
return Err(WorktreeError::Io(io::Error::other(
"empty path segment in index",
)));
}
if node.leaves.contains_key(*seg) {
let conflicting = if walked.is_empty() {
(*seg).to_string()
} else {
format!("{walked}/{seg}")
};
return Err(WorktreeError::Io(io::Error::other(format!(
"index path conflict: '{conflicting}' is staged as both a file and a directory"
))));
}
walked = if walked.is_empty() {
(*seg).to_string()
} else {
format!("{walked}/{seg}")
};
node = node.children.entry((*seg).to_string()).or_default();
}
if node.children.contains_key(*leaf) {
let conflicting = if walked.is_empty() {
(*leaf).to_string()
} else {
format!("{walked}/{leaf}")
};
return Err(WorktreeError::Io(io::Error::other(format!(
"index path conflict: '{conflicting}' is staged as both a file and a directory"
))));
}
if node
.leaves
.insert((*leaf).to_string(), (mode, object_hash))
.is_some()
{
let duplicate = if walked.is_empty() {
(*leaf).to_string()
} else {
format!("{walked}/{leaf}")
};
return Err(WorktreeError::Io(io::Error::other(format!(
"duplicate index path: '{duplicate}'"
))));
}
}
fn write_node<S: ObjectSink + ?Sized>(sink: &S, node: &Node) -> WorktreeResult<Hash> {
let mut entries: Vec<TreeEntry> = Vec::new();
for (name, child) in &node.children {
let h = write_node(sink, child)?;
let bytes = name.as_bytes().to_vec();
if !crate::object::TreeEntry::validate_name(&bytes) {
return Err(WorktreeError::Io(io::Error::other(format!(
"invalid tree entry name: {name:?}"
))));
}
entries.push(TreeEntry {
name: bytes,
mode: EntryMode::Tree,
object_hash: h,
});
}
for (name, (mode, hash)) in &node.leaves {
let bytes = name.as_bytes().to_vec();
if !crate::object::TreeEntry::validate_name(&bytes) {
return Err(WorktreeError::Io(io::Error::other(format!(
"invalid tree entry name: {name:?}"
))));
}
entries.push(TreeEntry {
name: bytes,
mode: *mode,
object_hash: *hash,
});
}
entries.sort_by(|a, b| a.name.cmp(&b.name));
let tree = Object::Tree(Tree { entries });
let bytes = serialize::serialize(&tree)?;
Ok(sink.put(&bytes)?)
}
write_node(sink, &root)
}
#[cfg(not(target_arch = "wasm32"))]
fn available_threads() -> usize {
static THREADS: std::sync::OnceLock<usize> = std::sync::OnceLock::new();
*THREADS
.get_or_init(|| std::thread::available_parallelism().map_or(1, std::num::NonZeroUsize::get))
}
#[cfg(not(target_arch = "wasm32"))]
fn chunked_scoped_map<T, U, E>(
items: &[T],
threads: usize,
f: impl Fn(&T) -> Result<U, E> + Sync,
) -> Result<Vec<U>, E>
where
T: Sync,
U: Send,
E: Send,
{
let chunk_size = items.len().div_ceil(threads).max(1);
let mut out: Vec<Option<U>> = (0..items.len()).map(|_| None).collect();
let mut first_err: Option<E> = None;
std::thread::scope(|scope| {
let f = &f;
let handles: Vec<_> = items
.chunks(chunk_size)
.enumerate()
.map(|(chunk_idx, chunk)| {
let base = chunk_idx * chunk_size;
scope.spawn(move || {
let mut results = Vec::with_capacity(chunk.len());
for it in chunk {
match f(it) {
Ok(u) => results.push(u),
Err(e) => return (base, results, Some(e)),
}
}
(base, results, None)
})
})
.collect();
for handle in handles {
let (base, results, err) = handle
.join()
.expect("chunked fan-out worker thread panicked");
for (i, u) in results.into_iter().enumerate() {
out[base + i] = Some(u);
}
if let Some(e) = err
&& first_err.is_none()
{
first_err = Some(e);
}
}
});
match first_err {
Some(e) => Err(e),
None => Ok(out
.into_iter()
.map(|o| o.expect("every slot filled when no error was reported"))
.collect()),
}
}
fn probe_staged_objects(
store: &ObjectStore,
entries: &[(&str, EntryMode, Hash)],
verify: bool,
) -> WorktreeResult<()> {
#[cfg(not(target_arch = "wasm32"))]
{
const ENTRIES_PER_THREAD: usize = 32;
let threads = available_threads();
if threads > 1 && entries.len() >= ENTRIES_PER_THREAD.saturating_mul(threads) {
return probe_staged_objects_parallel(store, entries, verify, threads);
}
}
entries.iter().try_for_each(|&(path, mode, hash)| {
check_one_staged_object(store, path, mode, hash, verify)
})
}
#[cfg(not(target_arch = "wasm32"))]
fn probe_staged_objects_parallel(
store: &ObjectStore,
entries: &[(&str, EntryMode, Hash)],
verify: bool,
threads: usize,
) -> WorktreeResult<()> {
chunked_scoped_map(entries, threads, |&(path, mode, hash)| {
check_one_staged_object(store, path, mode, hash, verify)
})
.map(|_: Vec<()>| ())
}
fn check_one_staged_object(
store: &ObjectStore,
path: &str,
mode: EntryMode,
hash: Hash,
verify: bool,
) -> WorktreeResult<()> {
let object_type = if verify {
store.verify_object_type(&hash)?
} else {
store.object_type(&hash)?
};
match object_type {
crate::object::ObjectType::Blob => Ok(()),
crate::object::ObjectType::ChunkedBlob if mode != EntryMode::Symlink => Ok(()),
other => Err(WorktreeError::Io(io::Error::other(format!(
"index entry '{path}' points to a non-blob object (got {})",
other.name()
)))),
}
}
pub fn hash_file<S: ObjectSink + ?Sized>(sink: &S, path: &Path) -> WorktreeResult<Hash> {
hash_file_with_metadata(sink, path).map(|(hash, _)| hash)
}
pub fn read_regular_file_bounded(path: &Path) -> WorktreeResult<(fs::Metadata, Vec<u8>)> {
let mut file = open_regular_file(path)?;
let meta = file.metadata()?;
if !meta.file_type().is_file() {
return Err(WorktreeError::Io(io::Error::new(
io::ErrorKind::InvalidInput,
"path is not a regular file",
)));
}
if meta.len() > MAX_FILE_BYTES {
return Err(WorktreeError::FileTooLarge(path.to_path_buf()));
}
let initial_capacity = usize::try_from(meta.len().min(CHUNK_THRESHOLD))
.map_err(|_| WorktreeError::FileTooLarge(path.to_path_buf()))?;
let mut data = Vec::with_capacity(initial_capacity);
file.by_ref()
.take(MAX_FILE_BYTES + 1)
.read_to_end(&mut data)?;
if u64::try_from(data.len()).unwrap_or(u64::MAX) > MAX_FILE_BYTES {
return Err(WorktreeError::FileTooLarge(path.to_path_buf()));
}
Ok((meta, data))
}
pub fn hash_file_with_metadata<S: ObjectSink + ?Sized>(
sink: &S,
path: &Path,
) -> WorktreeResult<(Hash, fs::Metadata)> {
hash_file_with_metadata_using(sink, path, store_large_file_streaming)
}
pub fn hash_file_with_metadata_with<S: ObjectSink + ?Sized>(
sink: &S,
path: &Path,
hash_chunks: impl FnMut(&S, &[Vec<u8>]) -> WorktreeResult<Vec<Hash>>,
) -> WorktreeResult<(Hash, fs::Metadata)> {
hash_file_with_metadata_using(sink, path, |sink, reader, path| {
store_large_file_streaming_with(sink, reader, path, hash_chunks)
})
}
fn hash_file_with_metadata_using<S: ObjectSink + ?Sized>(
sink: &S,
path: &Path,
stream: impl FnOnce(&S, io::Take<fs::File>, &Path) -> WorktreeResult<Hash>,
) -> WorktreeResult<(Hash, fs::Metadata)> {
let mut file = open_regular_file(path)?;
let meta = file.metadata()?;
if !meta.file_type().is_file() {
return Err(WorktreeError::Io(io::Error::new(
io::ErrorKind::InvalidInput,
"path is not a regular file",
)));
}
if meta.len() > MAX_FILE_BYTES {
return Err(WorktreeError::FileTooLarge(path.to_path_buf()));
}
if meta.len() <= CHUNK_THRESHOLD {
let initial_capacity = usize::try_from(meta.len())
.map_err(|_| WorktreeError::FileTooLarge(path.to_path_buf()))?;
let mut data = Vec::with_capacity(initial_capacity);
file.by_ref()
.take(MAX_FILE_BYTES + 1)
.read_to_end(&mut data)?;
if u64::try_from(data.len()).unwrap_or(u64::MAX) > MAX_FILE_BYTES {
return Err(WorktreeError::FileTooLarge(path.to_path_buf()));
}
let hash = store_file_object(sink, &data)?;
return Ok((hash, meta));
}
let hash = stream(sink, file.take(MAX_FILE_BYTES + 1), path)?;
Ok((hash, meta))
}
const STREAM_HASH_BATCH: usize = 64;
pub fn store_chunk_blob<S: ObjectSink + ?Sized>(sink: &S, chunk: &[u8]) -> WorktreeResult<Hash> {
let prologue = serialize::blob_prologue(chunk.len())?;
Ok(sink.put_parts(&[&prologue, chunk])?)
}
fn store_large_file_streaming<S: ObjectSink + ?Sized, R: Read>(
sink: &S,
reader: R,
path: &Path,
) -> WorktreeResult<Hash> {
let mut chunker = ChunkReader::new(FastCdc::v1(), reader);
let mut chunks = Vec::new();
let mut total_size: u64 = 0;
while let Some(chunk) = chunker.next_chunk_ref()? {
total_size = total_size
.checked_add(chunk.len() as u64)
.filter(|&t| t <= MAX_FILE_BYTES)
.ok_or_else(|| WorktreeError::FileTooLarge(path.to_path_buf()))?;
chunks.push(store_chunk_blob(sink, chunk)?);
}
store_chunk_manifest(sink, total_size, chunks)
}
fn store_chunk_manifest<S: ObjectSink + ?Sized>(
sink: &S,
total_size: u64,
chunks: Vec<Hash>,
) -> WorktreeResult<Hash> {
let manifest = Object::ChunkedBlob(ChunkedBlob {
total_size,
chunk_size: 0, chunks,
});
let manifest_bytes = serialize::serialize(&manifest)?;
Ok(sink.put(&manifest_bytes)?)
}
fn checked_hash_chunks<S: ObjectSink + ?Sized>(
hash_chunks: &mut impl FnMut(&S, &[Vec<u8>]) -> WorktreeResult<Vec<Hash>>,
sink: &S,
batch: &[Vec<u8>],
) -> WorktreeResult<Vec<Hash>> {
let hashes = hash_chunks(sink, batch)?;
if hashes.len() == batch.len() {
Ok(hashes)
} else {
Err(WorktreeError::ChunkBatchLengthMismatch {
expected: batch.len(),
actual: hashes.len(),
})
}
}
pub fn store_large_file_streaming_with<S: ObjectSink + ?Sized, R: Read + Send>(
sink: &S,
reader: R,
path: &Path,
mut hash_chunks: impl FnMut(&S, &[Vec<u8>]) -> WorktreeResult<Vec<Hash>>,
) -> WorktreeResult<Hash> {
#[cfg(not(target_arch = "wasm32"))]
{
store_large_file_streaming_pipelined(sink, reader, path, &mut hash_chunks)
}
#[cfg(target_arch = "wasm32")]
{
store_large_file_streaming_batched(sink, reader, path, &mut hash_chunks)
}
}
#[cfg(target_arch = "wasm32")]
fn store_large_file_streaming_batched<S: ObjectSink + ?Sized, R: Read>(
sink: &S,
reader: R,
path: &Path,
hash_chunks: &mut impl FnMut(&S, &[Vec<u8>]) -> WorktreeResult<Vec<Hash>>,
) -> WorktreeResult<Hash> {
let mut chunker = ChunkReader::new(FastCdc::v1(), reader);
let mut chunks = Vec::new();
let mut total_size: u64 = 0;
let mut batch: Vec<Vec<u8>> = Vec::with_capacity(STREAM_HASH_BATCH);
while let Some(chunk) = chunker.next_chunk()? {
total_size = total_size
.checked_add(chunk.len() as u64)
.filter(|&t| t <= MAX_FILE_BYTES)
.ok_or_else(|| WorktreeError::FileTooLarge(path.to_path_buf()))?;
batch.push(chunk);
if batch.len() == STREAM_HASH_BATCH {
chunks.extend(checked_hash_chunks(hash_chunks, sink, &batch)?);
batch.clear();
}
}
if !batch.is_empty() {
chunks.extend(checked_hash_chunks(hash_chunks, sink, &batch)?);
}
store_chunk_manifest(sink, total_size, chunks)
}
#[cfg(not(target_arch = "wasm32"))]
fn store_large_file_streaming_pipelined<S: ObjectSink + ?Sized, R: Read + Send>(
sink: &S,
reader: R,
path: &Path,
hash_chunks: &mut impl FnMut(&S, &[Vec<u8>]) -> WorktreeResult<Vec<Hash>>,
) -> WorktreeResult<Hash> {
use std::sync::mpsc;
let (tx, rx) = mpsc::sync_channel::<WorktreeResult<Vec<Vec<u8>>>>(0);
let (chunks, total_size) = std::thread::scope(|scope| -> WorktreeResult<(Vec<Hash>, u64)> {
scope.spawn(move || {
let mut chunker = ChunkReader::new(FastCdc::v1(), reader);
let mut batch: Vec<Vec<u8>> = Vec::with_capacity(STREAM_HASH_BATCH);
let mut running_total: u64 = 0;
loop {
let chunk = match chunker.next_chunk() {
Ok(Some(chunk)) => chunk,
Ok(None) => {
if !batch.is_empty() {
let _ = tx.send(Ok(batch));
}
return;
}
Err(e) => {
let _ = tx.send(Err(WorktreeError::from(e)));
return;
}
};
running_total = if let Some(t) = running_total
.checked_add(chunk.len() as u64)
.filter(|&t| t <= MAX_FILE_BYTES)
{
t
} else {
let _ = tx.send(Err(WorktreeError::FileTooLarge(path.to_path_buf())));
return;
};
batch.push(chunk);
if batch.len() == STREAM_HASH_BATCH {
let full = std::mem::replace(&mut batch, Vec::with_capacity(STREAM_HASH_BATCH));
if tx.send(Ok(full)).is_err() {
return;
}
}
}
});
let mut chunks = Vec::new();
let mut total_size: u64 = 0;
let mut first_err: Option<WorktreeError> = None;
while let Ok(received) = rx.recv() {
if first_err.is_some() {
continue;
}
match received.and_then(|batch| {
for chunk in &batch {
total_size += chunk.len() as u64;
}
checked_hash_chunks(hash_chunks, sink, &batch)
}) {
Ok(hashes) => chunks.extend(hashes),
Err(e) => first_err = Some(e),
}
}
match first_err {
Some(e) => Err(e),
None => Ok((chunks, total_size)),
}
})?;
store_chunk_manifest(sink, total_size, chunks)
}
pub fn store_file_object<S: ObjectSink + ?Sized>(sink: &S, data: &[u8]) -> WorktreeResult<Hash> {
if u64::try_from(data.len()).unwrap_or(u64::MAX) <= CHUNK_THRESHOLD {
let prologue = serialize::blob_prologue(data.len())?;
return Ok(sink.put_parts(&[&prologue, data])?);
}
let total_size = data.len() as u64;
let chunks: Vec<Hash> = ChunkIterator::new(FastCdc::v1(), data)
.map(|b| {
let chunk = &data[b.offset..b.offset + b.length];
let prologue = serialize::blob_prologue(chunk.len())?;
Ok::<_, WorktreeError>(sink.put_parts(&[&prologue, chunk])?)
})
.collect::<Result<_, _>>()?;
let manifest = Object::ChunkedBlob(ChunkedBlob {
total_size,
chunk_size: 0, chunks,
});
let manifest_bytes = serialize::serialize(&manifest)?;
Ok(sink.put(&manifest_bytes)?)
}
pub fn chunked_blob_from_bytes(data: &[u8]) -> WorktreeResult<ChunkedBlob> {
let chunks: Vec<Hash> = ChunkIterator::new(FastCdc::v1(), data)
.map(|b| {
let chunk = &data[b.offset..b.offset + b.length];
let prologue = serialize::blob_prologue(chunk.len())?;
let mut hasher = crate::hash::Hasher::new();
hasher.update(&prologue);
hasher.update(chunk);
Ok::<_, WorktreeError>(hasher.finalize())
})
.collect::<Result<_, _>>()?;
Ok(ChunkedBlob {
total_size: data.len() as u64,
chunk_size: 0,
chunks,
})
}
pub fn hash_file_object(data: &[u8]) -> WorktreeResult<Hash> {
if u64::try_from(data.len()).unwrap_or(u64::MAX) <= CHUNK_THRESHOLD {
let prologue = serialize::blob_prologue(data.len())?;
let mut hasher = crate::hash::Hasher::new();
hasher.update(&prologue);
hasher.update(data);
return Ok(hasher.finalize());
}
Ok(crate::merkle::compute_chunked_id(&chunked_blob_from_bytes(
data,
)?))
}
#[must_use]
pub fn mtime_nanos(meta: &fs::Metadata) -> u64 {
meta.modified()
.ok()
.and_then(|t| t.duration_since(std::time::UNIX_EPOCH).ok())
.map_or(0, |d| u64::try_from(d.as_nanos()).unwrap_or(u64::MAX))
}
#[must_use]
pub fn stat_cache_fields(meta: &fs::Metadata) -> (u64, u64, u64, u64) {
#[cfg(unix)]
let (ino, ctime_ns) = {
use std::os::unix::fs::MetadataExt;
let ctime_ns = u64::try_from(meta.ctime())
.ok()
.and_then(|s| s.checked_mul(1_000_000_000))
.and_then(|ns| ns.checked_add(u64::try_from(meta.ctime_nsec()).unwrap_or(0)))
.unwrap_or(0);
(meta.ino(), ctime_ns)
};
#[cfg(not(unix))]
let (ino, ctime_ns) = (0u64, 0u64);
(mtime_nanos(meta), meta.len(), ino, ctime_ns)
}
#[must_use]
pub fn stat_matches(entry: &crate::index::IndexEntry, meta: &fs::Metadata) -> bool {
use crate::index::EntryStatus;
if entry.mtime_ns == 0 || !meta.is_file() {
return false;
}
let (mtime_ns, size, ino, ctime_ns) = stat_cache_fields(meta);
if size != entry.size || mtime_ns != entry.mtime_ns {
return false;
}
if entry.ino != 0 && ino != 0 && ino != entry.ino {
return false;
}
if entry.ctime_ns != 0 && ctime_ns != 0 && ctime_ns != entry.ctime_ns {
return false;
}
match entry.status {
#[cfg(not(unix))]
EntryStatus::Blob | EntryStatus::Executable => true,
#[cfg(unix)]
EntryStatus::Blob => entry_mode_from_file_metadata(meta) == EntryMode::Blob,
#[cfg(unix)]
EntryStatus::Executable => entry_mode_from_file_metadata(meta) == EntryMode::Executable,
EntryStatus::Symlink | EntryStatus::Removed | EntryStatus::Tree => false,
}
}
#[cfg(unix)]
fn open_regular_file(path: &Path) -> io::Result<fs::File> {
use std::os::unix::fs::OpenOptionsExt;
fs::OpenOptions::new()
.read(true)
.custom_flags(libc::O_NOFOLLOW)
.open(path)
}
#[cfg(not(unix))]
fn open_regular_file(path: &Path) -> io::Result<fs::File> {
let meta = path.symlink_metadata()?;
if !meta.file_type().is_file() {
return Err(io::Error::new(
io::ErrorKind::InvalidInput,
"path is not a regular file",
));
}
fs::File::open(path)
}
#[cfg(unix)]
fn entry_mode_from_file_metadata(meta: &fs::Metadata) -> EntryMode {
use std::os::unix::fs::PermissionsExt;
if meta.permissions().mode() & 0o111 != 0 {
EntryMode::Executable
} else {
EntryMode::Blob
}
}
#[cfg(not(unix))]
fn entry_mode_from_file_metadata(_meta: &fs::Metadata) -> EntryMode {
EntryMode::Blob
}
#[cfg(test)]
mod tests {
use super::*;
use crate::object::ObjectType;
use tempfile::TempDir;
fn fresh_store() -> (TempDir, ObjectStore) {
let dir = TempDir::new().unwrap();
let store = ObjectStore::init(&crate::layout::RepoLayout::single(dir.path())).unwrap();
(dir, store)
}
#[test]
fn validate_symlink_targets() {
assert!(validate_symlink_target("hello"));
assert!(validate_symlink_target("sub/dir/file"));
assert!(!validate_symlink_target(""));
assert!(!validate_symlink_target("/etc/passwd"));
assert!(!validate_symlink_target("../escape"));
assert!(!validate_symlink_target("a/../b"));
}
#[test]
fn build_tree_from_empty_dir() {
let (_sd, store) = fresh_store();
let work = TempDir::new().unwrap();
let h = build_tree(&store, work.path()).unwrap();
let obj = store.read_object(&h).unwrap();
match obj {
Object::Tree(t) => assert_eq!(t.entries.len(), 0),
other => panic!("expected tree, got {other:?}"),
}
}
#[test]
fn build_tree_with_single_file() {
let (_sd, store) = fresh_store();
let work = TempDir::new().unwrap();
fs::write(work.path().join("hello.txt"), b"hello world").unwrap();
let h = build_tree(&store, work.path()).unwrap();
let obj = store.read_object(&h).unwrap();
let Object::Tree(t) = obj else {
panic!("expected tree");
};
assert_eq!(t.entries.len(), 1);
assert_eq!(t.entries[0].name.as_slice(), b"hello.txt");
assert_eq!(t.entries[0].mode, EntryMode::Blob);
let blob_obj = store.read_object(&t.entries[0].object_hash).unwrap();
let Object::Blob(b) = blob_obj else {
panic!("expected blob");
};
assert_eq!(b.data, b"hello world");
}
#[cfg(unix)]
#[test]
fn build_tree_marks_executable_regular_files() {
use std::os::unix::fs::PermissionsExt;
let (_sd, store) = fresh_store();
let work = TempDir::new().unwrap();
let script = work.path().join("run.sh");
fs::write(&script, b"#!/bin/sh\n").unwrap();
let mut perms = fs::metadata(&script).unwrap().permissions();
perms.set_mode(perms.mode() | 0o111);
fs::set_permissions(&script, perms).unwrap();
let h = build_tree(&store, work.path()).unwrap();
let Object::Tree(t) = store.read_object(&h).unwrap() else {
panic!("expected tree");
};
assert_eq!(t.entries[0].name.as_slice(), b"run.sh");
assert_eq!(t.entries[0].mode, EntryMode::Executable);
}
#[cfg(unix)]
#[test]
fn build_tree_rejects_invalid_entry_name_before_writing_tree() {
let (_sd, store) = fresh_store();
let work = TempDir::new().unwrap();
fs::write(work.path().join("bad."), b"bad name").unwrap();
let err = build_tree(&store, work.path()).unwrap_err();
assert!(matches!(err, WorktreeError::Io(_)));
}
#[cfg(unix)]
#[test]
fn hash_file_rejects_final_component_symlink() {
use std::os::unix::fs::symlink;
let (_sd, store) = fresh_store();
let work = TempDir::new().unwrap();
fs::write(work.path().join("target.txt"), b"target").unwrap();
symlink("target.txt", work.path().join("link.txt")).unwrap();
let err = hash_file(&store, &work.path().join("link.txt")).unwrap_err();
assert!(matches!(err, WorktreeError::Io(_)));
}
#[test]
fn build_tree_with_nested_directories() {
let (_sd, store) = fresh_store();
let work = TempDir::new().unwrap();
fs::write(work.path().join("a.txt"), b"file a").unwrap();
fs::create_dir(work.path().join("subdir")).unwrap();
fs::write(work.path().join("subdir/b.txt"), b"file b").unwrap();
let h = build_tree(&store, work.path()).unwrap();
let obj = store.read_object(&h).unwrap();
let Object::Tree(t) = obj else {
panic!("expected tree");
};
assert_eq!(t.entries.len(), 2);
assert_eq!(t.entries[0].name.as_slice(), b"a.txt");
assert_eq!(t.entries[1].name.as_slice(), b"subdir");
assert_eq!(t.entries[1].mode, EntryMode::Tree);
let sub = store.read_object(&t.entries[1].object_hash).unwrap();
let Object::Tree(st) = sub else {
panic!("expected tree");
};
assert_eq!(st.entries.len(), 1);
assert_eq!(st.entries[0].name.as_slice(), b"b.txt");
}
#[test]
fn build_tree_skips_mkit_directory() {
let (_sd, store) = fresh_store();
let work = TempDir::new().unwrap();
fs::create_dir(work.path().join(".mkit")).unwrap();
fs::write(work.path().join(".mkit/should_skip"), b"").unwrap();
fs::write(work.path().join("keep.txt"), b"kept").unwrap();
let h = build_tree(&store, work.path()).unwrap();
let obj = store.read_object(&h).unwrap();
let Object::Tree(t) = obj else {
panic!("expected tree");
};
assert_eq!(t.entries.len(), 1);
assert_eq!(t.entries[0].name.as_slice(), b"keep.txt");
}
#[test]
fn build_tree_is_deterministic() {
let (_sd, store) = fresh_store();
let work = TempDir::new().unwrap();
fs::write(work.path().join("z.txt"), b"z").unwrap();
fs::write(work.path().join("a.txt"), b"a").unwrap();
let h1 = build_tree(&store, work.path()).unwrap();
let h2 = build_tree(&store, work.path()).unwrap();
assert_eq!(h1, h2);
}
#[test]
fn build_tree_respects_mkitignore() {
let (_sd, store) = fresh_store();
let work = TempDir::new().unwrap();
fs::write(work.path().join(".mkitignore"), b"*.log\n").unwrap();
fs::write(work.path().join("keep.txt"), b"kept").unwrap();
fs::write(work.path().join("debug.log"), b"ignored").unwrap();
let h = build_tree(&store, work.path()).unwrap();
let obj = store.read_object(&h).unwrap();
let Object::Tree(t) = obj else {
panic!("expected tree");
};
assert_eq!(t.entries.len(), 2);
assert_eq!(t.entries[0].name.as_slice(), b".mkitignore");
assert_eq!(t.entries[1].name.as_slice(), b"keep.txt");
}
#[cfg(unix)]
#[test]
fn rejects_invalid_symlink_targets() {
use std::os::unix::fs::symlink;
let (_sd, store) = fresh_store();
let work = TempDir::new().unwrap();
symlink("/etc/passwd", work.path().join("bad-link")).unwrap();
let err = build_tree(&store, work.path()).unwrap_err();
assert!(matches!(err, WorktreeError::InvalidSymlinkTarget(_)));
}
#[cfg(unix)]
#[test]
fn rejects_dotdot_symlink_targets() {
use std::os::unix::fs::symlink;
let (_sd, store) = fresh_store();
let work = TempDir::new().unwrap();
symlink("../../etc/passwd", work.path().join("bad-link")).unwrap();
let err = build_tree(&store, work.path()).unwrap_err();
assert!(matches!(err, WorktreeError::InvalidSymlinkTarget(_)));
}
#[test]
fn small_file_stays_as_regular_blob() {
let (_sd, store) = fresh_store();
let work = TempDir::new().unwrap();
fs::write(work.path().join("small.txt"), b"hello world").unwrap();
let h = build_tree(&store, work.path()).unwrap();
let obj = store.read_object(&h).unwrap();
let Object::Tree(t) = obj else {
panic!("expected tree");
};
let entry = store.read_object(&t.entries[0].object_hash).unwrap();
assert_eq!(entry.object_type(), ObjectType::Blob);
}
#[test]
fn large_file_becomes_chunked_blob() {
let (_sd, store) = fresh_store();
let work = TempDir::new().unwrap();
let n = usize::try_from(CHUNK_THRESHOLD).unwrap() + 256 * 1024;
let mut big = Vec::with_capacity(n);
let mut state: u64 = 0x00C0_FFEE;
for _ in 0..n {
state = state.wrapping_add(0x9E37_79B9_7F4A_7C15);
let mut z = state;
z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
z ^= z >> 31;
big.push((z & 0xFF) as u8);
}
fs::write(work.path().join("big.bin"), &big).unwrap();
let tree_hash = build_tree(&store, work.path()).unwrap();
let Object::Tree(t) = store.read_object(&tree_hash).unwrap() else {
panic!("expected tree");
};
assert_eq!(t.entries.len(), 1);
let entry_hash = t.entries[0].object_hash;
let entry = store.read_object(&entry_hash).unwrap();
let Object::ChunkedBlob(manifest) = entry else {
panic!("expected chunked_blob, got {entry:?}");
};
assert_eq!(manifest.total_size, n as u64);
assert_eq!(manifest.chunk_size, 0, "0 = content-defined (FastCDC)");
assert!(!manifest.chunks.is_empty());
let mut reassembled: Vec<u8> = Vec::with_capacity(n);
for h in &manifest.chunks {
let Object::Blob(b) = store.read_object(h).unwrap() else {
panic!("chunk did not resolve to a Blob");
};
reassembled.extend_from_slice(&b.data);
}
assert_eq!(reassembled, big, "chunks must round-trip the source");
}
#[test]
fn hash_file_with_metadata_streaming_matches_store_file_object_in_memory() {
let work = TempDir::new().unwrap();
let n = usize::try_from(CHUNK_THRESHOLD).unwrap() + 700 * 1024;
let mut big = Vec::with_capacity(n);
let mut state: u64 = 0xFACE_FEED;
for _ in 0..n {
state = state.wrapping_add(0x9E37_79B9_7F4A_7C15);
let mut z = state;
z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
z ^= z >> 31;
big.push((z & 0xFF) as u8);
}
let path = work.path().join("big.bin");
fs::write(&path, &big).unwrap();
let (_sd1, store1) = fresh_store();
let (streamed_hash, meta) = hash_file_with_metadata(&store1, &path).unwrap();
assert_eq!(meta.len(), n as u64);
let (_sd2, store2) = fresh_store();
let in_memory_hash = store_file_object(&store2, &big).unwrap();
assert_eq!(
streamed_hash, in_memory_hash,
"streaming a file from disk must produce the same content-address \
as chunking the fully-buffered bytes"
);
let Object::ChunkedBlob(manifest) = store1.read_object(&streamed_hash).unwrap() else {
panic!("expected chunked_blob");
};
let mut reassembled = Vec::with_capacity(n);
for h in &manifest.chunks {
let Object::Blob(b) = store1.read_object(h).unwrap() else {
panic!("chunk did not resolve to a Blob");
};
reassembled.extend_from_slice(&b.data);
}
assert_eq!(reassembled, big);
}
#[test]
fn streaming_rejects_miscounted_hashes_in_full_and_partial_batches() {
for size in [
crate::chunker::MAX_SIZE,
STREAM_HASH_BATCH * crate::chunker::MAX_SIZE + 1,
] {
let (_dir, sink) = fresh_store();
let data = vec![0u8; size];
let mut expected = 0;
let err = store_large_file_streaming_with(
&sink,
io::Cursor::new(data),
Path::new("batch.bin"),
|_sink, batch| {
expected = batch.len();
Ok(Vec::new())
},
)
.unwrap_err();
assert!(expected > 0);
assert!(matches!(err, WorktreeError::ChunkBatchLengthMismatch {
expected: n, actual: 0
} if n == expected));
}
}
#[test]
fn streaming_pipeline_errors_promptly_when_hash_chunks_fails_on_an_early_batch() {
let size = 3 * STREAM_HASH_BATCH * crate::chunker::MAX_SIZE;
let (_dir, sink) = fresh_store();
let data = vec![0u8; size];
let (done_tx, done_rx) = std::sync::mpsc::channel();
std::thread::spawn(move || {
let result = store_large_file_streaming_with(
&sink,
io::Cursor::new(data),
Path::new("batch.bin"),
|_sink, _batch| Err(WorktreeError::InvalidUtf8),
);
let _ = done_tx.send(matches!(result, Err(WorktreeError::InvalidUtf8)));
});
let errored_correctly = done_rx
.recv_timeout(std::time::Duration::from_secs(20))
.expect(
"store_large_file_streaming_with deadlocked instead of \
returning promptly after an early hash_chunks error",
);
assert!(errored_correctly);
}
#[test]
fn hash_file_with_metadata_with_batch_fanout_matches_sequential() {
let n = usize::try_from(CHUNK_THRESHOLD).unwrap() + 6 * 1024 * 1024;
let mut big = Vec::with_capacity(n);
let mut state: u64 = 0xABCD_EF01;
for _ in 0..n {
state = state.wrapping_add(0x9E37_79B9_7F4A_7C15);
let mut z = state;
z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
z ^= z >> 31;
big.push((z & 0xFF) as u8);
}
let work = TempDir::new().unwrap();
let path = work.path().join("big.bin");
fs::write(&path, &big).unwrap();
let (_sd1, sequential_store) = fresh_store();
let (sequential_hash, _) = hash_file_with_metadata(&sequential_store, &path).unwrap();
let (_sd2, fanout_store) = fresh_store();
let (fanout_hash, _) = hash_file_with_metadata_with(&fanout_store, &path, |sink, batch| {
let mut out: Vec<Hash> = batch
.iter()
.rev()
.map(|chunk| store_chunk_blob(sink, chunk))
.collect::<WorktreeResult<_>>()?;
out.reverse();
Ok(out)
})
.unwrap();
assert_eq!(
sequential_hash, fanout_hash,
"an out-of-order-processing hash_chunks callback must still match \
the sequential default when it returns results in input order"
);
}
use crate::index::{EntryStatus, Index, IndexEntry};
fn write_blob(store: &ObjectStore, bytes: &[u8]) -> Hash {
let blob = Object::Blob(crate::object::Blob {
data: bytes.to_vec(),
});
let body = serialize::serialize(&blob).unwrap();
store.write(&body).unwrap()
}
#[test]
fn from_index_empty_returns_empty_tree() {
let (_sd, store) = fresh_store();
let idx = Index::new();
let h = build_tree_from_index(&store, &idx).unwrap();
let Object::Tree(t) = store.read_object(&h).unwrap() else {
panic!("expected tree");
};
assert!(t.entries.is_empty());
}
#[test]
fn from_index_single_file_at_root() {
let (_sd, store) = fresh_store();
let blob_hash = write_blob(&store, b"hello world");
let mut idx = Index::new();
idx.entries.push(IndexEntry {
path: "hello.txt".into(),
status: EntryStatus::Blob,
object_hash: blob_hash,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
let h = build_tree_from_index(&store, &idx).unwrap();
let Object::Tree(t) = store.read_object(&h).unwrap() else {
panic!();
};
assert_eq!(t.entries.len(), 1);
assert_eq!(t.entries[0].name, b"hello.txt");
assert_eq!(t.entries[0].mode, EntryMode::Blob);
assert_eq!(t.entries[0].object_hash, blob_hash);
}
#[test]
fn from_index_nested_paths_build_subtrees() {
let (_sd, store) = fresh_store();
let a = write_blob(&store, b"file a");
let b = write_blob(&store, b"file b");
let mut idx = Index::new();
idx.entries.push(IndexEntry {
path: "a.txt".into(),
status: EntryStatus::Blob,
object_hash: a,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
idx.entries.push(IndexEntry {
path: "subdir/b.txt".into(),
status: EntryStatus::Blob,
object_hash: b,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
let root_hash = build_tree_from_index(&store, &idx).unwrap();
let Object::Tree(root) = store.read_object(&root_hash).unwrap() else {
panic!();
};
assert_eq!(root.entries.len(), 2);
assert_eq!(root.entries[0].name, b"a.txt");
assert_eq!(root.entries[0].mode, EntryMode::Blob);
assert_eq!(root.entries[1].name, b"subdir");
assert_eq!(root.entries[1].mode, EntryMode::Tree);
let Object::Tree(sub) = store.read_object(&root.entries[1].object_hash).unwrap() else {
panic!();
};
assert_eq!(sub.entries.len(), 1);
assert_eq!(sub.entries[0].name, b"b.txt");
assert_eq!(sub.entries[0].object_hash, b);
}
#[test]
fn from_index_removed_entries_are_skipped() {
let (_sd, store) = fresh_store();
let a = write_blob(&store, b"keep me");
let mut idx = Index::new();
idx.entries.push(IndexEntry {
path: "keep.txt".into(),
status: EntryStatus::Blob,
object_hash: a,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
idx.entries.push(IndexEntry {
path: "drop.txt".into(),
status: EntryStatus::Removed,
object_hash: [0; 32],
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
let h = build_tree_from_index(&store, &idx).unwrap();
let Object::Tree(t) = store.read_object(&h).unwrap() else {
panic!();
};
assert_eq!(t.entries.len(), 1);
assert_eq!(t.entries[0].name, b"keep.txt");
}
#[test]
fn from_index_executable_and_symlink_modes_pass_through() {
let (_sd, store) = fresh_store();
let exec = write_blob(&store, b"#!/bin/sh");
let link = write_blob(&store, b"target.txt");
let mut idx = Index::new();
idx.entries.push(IndexEntry {
path: "run.sh".into(),
status: EntryStatus::Executable,
object_hash: exec,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
idx.entries.push(IndexEntry {
path: "link".into(),
status: EntryStatus::Symlink,
object_hash: link,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
let h = build_tree_from_index(&store, &idx).unwrap();
let Object::Tree(t) = store.read_object(&h).unwrap() else {
panic!();
};
let by_name: std::collections::HashMap<&[u8], &TreeEntry> =
t.entries.iter().map(|e| (e.name.as_slice(), e)).collect();
assert_eq!(by_name[&b"run.sh"[..]].mode, EntryMode::Executable);
assert_eq!(by_name[&b"link"[..]].mode, EntryMode::Symlink);
}
#[test]
fn from_index_entries_are_sorted_by_name() {
let (_sd, store) = fresh_store();
let a = write_blob(&store, b"x");
let mut idx = Index::new();
idx.entries.push(IndexEntry {
path: "z.txt".into(),
status: EntryStatus::Blob,
object_hash: a,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
idx.entries.push(IndexEntry {
path: "a.txt".into(),
status: EntryStatus::Blob,
object_hash: a,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
idx.entries.push(IndexEntry {
path: "m.txt".into(),
status: EntryStatus::Blob,
object_hash: a,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
let h = build_tree_from_index(&store, &idx).unwrap();
let Object::Tree(t) = store.read_object(&h).unwrap() else {
panic!();
};
let names: Vec<&[u8]> = t.entries.iter().map(|e| e.name.as_slice()).collect();
assert_eq!(names, vec![&b"a.txt"[..], b"m.txt", b"z.txt"]);
}
#[test]
fn from_index_rejects_trailing_slash() {
let (_sd, store) = fresh_store();
let h = write_blob(&store, b"x");
let mut idx = Index::new();
idx.entries.push(IndexEntry {
path: "dir/".into(),
status: EntryStatus::Blob,
object_hash: h,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
let err = build_tree_from_index(&store, &idx).unwrap_err();
assert!(matches!(err, WorktreeError::Io(_)));
}
#[test]
fn from_index_rejects_empty_segment() {
let (_sd, store) = fresh_store();
let h = write_blob(&store, b"x");
let mut idx = Index::new();
idx.entries.push(IndexEntry {
path: "a//b.txt".into(),
status: EntryStatus::Blob,
object_hash: h,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
let err = build_tree_from_index(&store, &idx).unwrap_err();
assert!(matches!(err, WorktreeError::Io(_)));
}
#[test]
fn from_index_rejects_reserved_name() {
let (_sd, store) = fresh_store();
let h = write_blob(&store, b"x");
let mut idx = Index::new();
idx.entries.push(IndexEntry {
path: ".mkit".into(),
status: EntryStatus::Blob,
object_hash: h,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
let err = build_tree_from_index(&store, &idx).unwrap_err();
assert!(matches!(err, WorktreeError::Io(_)));
}
#[test]
fn from_index_matches_build_tree_for_equivalent_worktree() {
let (_sd, store) = fresh_store();
let work = TempDir::new().unwrap();
fs::write(work.path().join("a.txt"), b"alpha").unwrap();
fs::create_dir(work.path().join("dir")).unwrap();
fs::write(work.path().join("dir/b.txt"), b"beta").unwrap();
fs::write(work.path().join("dir/c.txt"), b"gamma").unwrap();
let worktree_root = build_tree(&store, work.path()).unwrap();
let a = write_blob(&store, b"alpha");
let b = write_blob(&store, b"beta");
let c = write_blob(&store, b"gamma");
let mut idx = Index::new();
idx.entries.push(IndexEntry {
path: "a.txt".into(),
status: EntryStatus::Blob,
object_hash: a,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
idx.entries.push(IndexEntry {
path: "dir/b.txt".into(),
status: EntryStatus::Blob,
object_hash: b,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
idx.entries.push(IndexEntry {
path: "dir/c.txt".into(),
status: EntryStatus::Blob,
object_hash: c,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
let index_root = build_tree_from_index(&store, &idx).unwrap();
assert_eq!(
worktree_root, index_root,
"build_tree_from_index must produce the same root hash as build_tree for equivalent contents"
);
}
#[test]
fn from_index_deeply_nested_paths_build_chain_of_subtrees() {
let (_sd, store) = fresh_store();
let h = write_blob(&store, b"deep");
let mut idx = Index::new();
idx.entries.push(IndexEntry {
path: "a/b/c/d/e.txt".into(),
status: EntryStatus::Blob,
object_hash: h,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
let root = build_tree_from_index(&store, &idx).unwrap();
let Object::Tree(t) = store.read_object(&root).unwrap() else {
panic!();
};
assert_eq!(t.entries.len(), 1);
assert_eq!(t.entries[0].name, b"a");
assert_eq!(t.entries[0].mode, EntryMode::Tree);
let mut cursor = t.entries[0].object_hash;
for seg in [b"b" as &[u8], b"c", b"d"] {
let Object::Tree(t) = store.read_object(&cursor).unwrap() else {
panic!();
};
assert_eq!(t.entries.len(), 1);
assert_eq!(t.entries[0].name, seg);
cursor = t.entries[0].object_hash;
}
let Object::Tree(t) = store.read_object(&cursor).unwrap() else {
panic!();
};
assert_eq!(t.entries[0].name, b"e.txt");
assert_eq!(t.entries[0].object_hash, h);
}
#[test]
fn from_index_rejects_blob_then_subdir_collision() {
let (_sd, store) = fresh_store();
let h = write_blob(&store, b"x");
let mut idx = Index::new();
idx.entries.push(IndexEntry {
path: "a".into(),
status: EntryStatus::Blob,
object_hash: h,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
idx.entries.push(IndexEntry {
path: "a/b".into(),
status: EntryStatus::Blob,
object_hash: h,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
let err = build_tree_from_index(&store, &idx).unwrap_err();
let msg = format!("{err}");
assert!(
msg.contains("conflict") || msg.contains("collision") || msg.contains("'a'"),
"expected collision error mentioning the path, got: {msg}"
);
}
#[test]
fn from_index_rejects_subdir_then_blob_collision() {
let (_sd, store) = fresh_store();
let h = write_blob(&store, b"x");
let mut idx = Index::new();
idx.entries.push(IndexEntry {
path: "a/b".into(),
status: EntryStatus::Blob,
object_hash: h,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
idx.entries.push(IndexEntry {
path: "a".into(),
status: EntryStatus::Blob,
object_hash: h,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
assert!(build_tree_from_index(&store, &idx).is_err());
}
#[test]
fn from_index_rejects_duplicate_exact_path() {
let (_sd, store) = fresh_store();
let a = write_blob(&store, b"a");
let b = write_blob(&store, b"b");
let mut idx = Index::new();
idx.entries.push(IndexEntry {
path: "same.txt".into(),
status: EntryStatus::Blob,
object_hash: a,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
idx.entries.push(IndexEntry {
path: "same.txt".into(),
status: EntryStatus::Blob,
object_hash: b,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
let err = build_tree_from_index(&store, &idx).unwrap_err();
let msg = format!("{err}");
assert!(msg.contains("duplicate index path"), "got: {msg}");
}
#[test]
fn from_index_rejects_duplicate_removed_and_live_path() {
let (_sd, store) = fresh_store();
let h = write_blob(&store, b"live");
let mut idx = Index::new();
idx.entries.push(IndexEntry {
path: "same.txt".into(),
status: EntryStatus::Removed,
object_hash: [0; 32],
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
idx.entries.push(IndexEntry {
path: "same.txt".into(),
status: EntryStatus::Blob,
object_hash: h,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
let err = build_tree_from_index(&store, &idx).unwrap_err();
let msg = format!("{err}");
assert!(msg.contains("duplicate index path"), "got: {msg}");
}
#[test]
fn from_index_all_removed_produces_empty_tree() {
let (_sd, store) = fresh_store();
let mut idx = Index::new();
idx.entries.push(IndexEntry {
path: "gone.txt".into(),
status: EntryStatus::Removed,
object_hash: [0; 32],
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
let h = build_tree_from_index(&store, &idx).unwrap();
let Object::Tree(t) = store.read_object(&h).unwrap() else {
panic!();
};
assert!(t.entries.is_empty());
}
#[test]
fn from_index_root_is_a_tree_object() {
let (_sd, store) = fresh_store();
let idx = Index::new();
let h = build_tree_from_index(&store, &idx).unwrap();
let obj = store.read_object(&h).unwrap();
assert_eq!(obj.object_type(), ObjectType::Tree);
}
#[test]
fn from_index_rejects_missing_blob_object() {
let (_sd, store) = fresh_store();
let mut idx = Index::new();
idx.entries.push(IndexEntry {
path: "missing.txt".into(),
status: EntryStatus::Blob,
object_hash: [42; 32],
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
let err = build_tree_from_index(&store, &idx).unwrap_err();
assert!(matches!(err, WorktreeError::Store(_)));
}
#[test]
fn from_index_rejects_non_blob_object_for_blob_status() {
let (_sd, store) = fresh_store();
let tree = Object::Tree(Tree { entries: vec![] });
let body = serialize::serialize(&tree).unwrap();
let tree_hash = store.write(&body).unwrap();
let mut idx = Index::new();
idx.entries.push(IndexEntry {
path: "not-a-blob.txt".into(),
status: EntryStatus::Blob,
object_hash: tree_hash,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
let err = build_tree_from_index(&store, &idx).unwrap_err();
let msg = format!("{err}");
assert!(
msg.contains("non-blob"),
"expected non-blob index object error, got: {msg}"
);
}
#[test]
fn from_index_accepts_chunked_blob_for_file_entry() {
let (_sd, store) = fresh_store();
let n = usize::try_from(CHUNK_THRESHOLD).unwrap() + 256 * 1024;
let mut big = Vec::with_capacity(n);
let mut state: u64 = 0x00C0_FFEE;
for _ in 0..n {
state = state.wrapping_add(0x9E37_79B9_7F4A_7C15);
let mut z = state;
z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
z ^= z >> 31;
big.push((z & 0xFF) as u8);
}
let chunked_hash = store_file_object(&store, &big).unwrap();
assert!(
matches!(
store.read_object(&chunked_hash).unwrap(),
Object::ChunkedBlob(_)
),
"fixture must be a ChunkedBlob"
);
let mut idx = Index::new();
idx.entries.push(IndexEntry {
path: "big.bin".into(),
status: EntryStatus::Blob,
object_hash: chunked_hash,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
let root = build_tree_from_index(&store, &idx).unwrap();
let Object::Tree(t) = store.read_object(&root).unwrap() else {
panic!("expected tree");
};
assert_eq!(t.entries.len(), 1);
assert_eq!(t.entries[0].name, b"big.bin");
assert_eq!(t.entries[0].mode, EntryMode::Blob);
assert_eq!(t.entries[0].object_hash, chunked_hash);
assert_eq!(read_blob(&store, &chunked_hash).unwrap(), big);
}
#[test]
fn read_blob_rejects_chunked_total_size_mismatch() {
let (_sd, store) = fresh_store();
let chunk = serialize::serialize(&Object::Blob(crate::object::Blob {
data: b"twelve bytes".to_vec(),
}))
.unwrap();
let chunk_hash = store.write(&chunk).unwrap();
let manifest = Object::ChunkedBlob(ChunkedBlob {
total_size: 999,
chunk_size: 0,
chunks: vec![chunk_hash],
});
let h = store
.write(&serialize::serialize(&manifest).unwrap())
.unwrap();
let err = read_blob(&store, &h).unwrap_err();
assert!(
matches!(
err,
WorktreeError::Object(crate::object::MkitError::ChunkedBlobSizeMismatch {
expected: 999,
actual: 12,
})
),
"expected ChunkedBlobSizeMismatch, got {err:?}"
);
}
#[test]
fn from_index_rejects_chunked_blob_for_symlink_entry() {
let (_sd, store) = fresh_store();
let n = usize::try_from(CHUNK_THRESHOLD).unwrap() + 256 * 1024;
let big = vec![0xABu8; n];
let chunked_hash = store_file_object(&store, &big).unwrap();
let mut idx = Index::new();
idx.entries.push(IndexEntry {
path: "link".into(),
status: EntryStatus::Symlink,
object_hash: chunked_hash,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
let err = build_tree_from_index(&store, &idx).unwrap_err();
assert!(format!("{err}").contains("non-blob"));
}
#[test]
fn store_file_object_via_batch_equals_via_store() {
let data: Vec<u8> = (0..3 * 1024 * 1024u32)
.map(|i| u8::try_from((i.wrapping_mul(2_654_435_761)) % 251).unwrap())
.collect();
let (_d1, store1) = fresh_store();
let h_store = store_file_object(&store1, &data).unwrap();
let (_d2, store2) = fresh_store();
let batch = store2.batch();
let h_batch = store_file_object(&batch, &data).unwrap();
batch.commit().unwrap();
assert_eq!(h_store, h_batch, "sink choice must not change hashes");
assert_eq!(
read_blob(&store1, &h_store).unwrap(),
read_blob(&store2, &h_batch).unwrap(),
);
let small = b"under the chunk threshold";
let h1 = store_file_object(&store1, small).unwrap();
let batch2 = store2.batch();
let h2 = store_file_object(&batch2, small).unwrap();
batch2.commit().unwrap();
assert_eq!(h1, h2);
}
#[test]
fn build_tree_from_index_with_batch_single_flush() {
use crate::batch::testing::{Ev, RecordingSyncer};
use crate::index::{EntryStatus, Index, IndexEntry};
use std::sync::Arc;
let (_sd, mut store) = fresh_store();
let mut idx = Index::default();
for i in 0..20 {
let blob = Object::Blob(crate::object::Blob {
data: format!("file {i}").into_bytes(),
});
let bytes = serialize::serialize(&blob).unwrap();
let h = store.write(&bytes).unwrap();
idx.entries.push(IndexEntry {
status: EntryStatus::Blob,
object_hash: h,
path: format!("d{}/sub/f{i}.txt", i % 5),
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
}
let rec = Arc::new(RecordingSyncer::default());
store.set_syncer(rec.clone());
let batch = store.batch();
let tree_h = build_tree_from_index_with(&store, &batch, &idx, true).unwrap();
batch.commit().unwrap();
let fulls = rec
.events()
.iter()
.filter(|e| matches!(e, Ev::Full(_)))
.count();
assert_eq!(fulls, 2, "tree materialisation flush cost must be constant");
assert!(store.read_object(&tree_h).is_ok());
let (_sd2, store2) = fresh_store();
for i in 0..20 {
let blob = Object::Blob(crate::object::Blob {
data: format!("file {i}").into_bytes(),
});
store2.write(&serialize::serialize(&blob).unwrap()).unwrap();
}
assert_eq!(tree_h, build_tree_from_index(&store2, &idx).unwrap());
}
#[test]
fn build_tree_from_index_verify_rejects_corrupt_staged_object() {
use crate::index::{EntryStatus, Index, IndexEntry};
let (_sd, store) = fresh_store();
let blob = Object::Blob(crate::object::Blob {
data: b"hello".to_vec(),
});
let h = store.write(&serialize::serialize(&blob).unwrap()).unwrap();
let mut idx = Index::default();
idx.entries.push(IndexEntry {
status: EntryStatus::Blob,
object_hash: h,
path: "a.txt".to_string(),
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
assert!(build_tree_from_index_with(&store, &store, &idx, true).is_ok());
assert!(build_tree_from_index_with(&store, &store, &idx, false).is_ok());
let path = store.path_for(&h);
let mut bytes = std::fs::read(&path).unwrap();
let i = bytes.len() - 1;
bytes[i] ^= 0xFF;
std::fs::write(&path, &bytes).unwrap();
assert!(
build_tree_from_index_with(&store, &store, &idx, true).is_err(),
"commit-path tree build must reject a corrupt staged object"
);
assert!(
build_tree_from_index_with(&store, &store, &idx, false).is_ok(),
"status/diff snapshot path keeps the cheap prologue-only check"
);
}
#[test]
fn build_tree_from_index_earlier_object_error_wins_over_later_status_error() {
use crate::index::{EntryStatus, Index, IndexEntry};
let (_sd, store) = fresh_store();
let mut idx = Index::default();
idx.entries.push(IndexEntry {
status: EntryStatus::Blob,
object_hash: [0xAB; 32],
path: "a.txt".to_string(),
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
idx.entries.push(IndexEntry {
status: EntryStatus::Tree,
object_hash: [0; 32],
path: "b".to_string(),
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
});
let err = build_tree_from_index_with(&store, &store, &idx, false)
.expect_err("index has no valid entries");
let msg = err.to_string();
assert!(
msg.contains("ab") || msg.to_lowercase().contains("not found"),
"expected entry 0's missing-object error, got: {msg}"
);
assert!(
!msg.contains("reserved Tree status"),
"entry 1's status error must not preempt entry 0's earlier object error, got: {msg}"
);
}
#[test]
fn hash_file_object_equals_store_file_object() {
let threshold = usize::try_from(CHUNK_THRESHOLD).unwrap();
for len in [0usize, 1, 1024, threshold, 3 * 1024 * 1024] {
let data: Vec<u8> = (0..len)
.map(|i| u8::try_from((i * 31 + 7) % 251).unwrap())
.collect();
let (_sd, store) = fresh_store();
let stored = store_file_object(&store, &data).unwrap();
let pure = hash_file_object(&data).unwrap();
assert_eq!(stored, pure, "len {len}: pure hash must match stored hash");
}
}
#[test]
fn hash_file_object_writes_nothing() {
let (_sd, store) = fresh_store();
let data = vec![0xAB; 2 * 1024 * 1024]; let _ = hash_file_object(&data).unwrap();
assert!(
store.iter_object_hashes().unwrap().is_empty(),
"pure hashing must not create objects"
);
}
fn meta_of(p: &Path) -> fs::Metadata {
p.symlink_metadata().unwrap()
}
#[test]
fn stat_matches_requires_nonzero_mtime_and_equal_fields() {
let work = TempDir::new().unwrap();
let f = work.path().join("a.txt");
fs::write(&f, b"hello").unwrap();
let meta = meta_of(&f);
let entry = crate::index::IndexEntry {
path: "a.txt".into(),
status: crate::index::EntryStatus::Blob,
object_hash: crate::hash::hash(b"irrelevant"),
mtime_ns: mtime_nanos(&meta),
size: meta.len(),
ino: 0,
ctime_ns: 0,
};
assert!(stat_matches(&entry, &meta));
let mut zeroed = entry.clone();
zeroed.mtime_ns = 0;
assert!(!stat_matches(&zeroed, &meta), "zero sentinel must re-hash");
let mut wrong_size = entry.clone();
wrong_size.size += 1;
assert!(!stat_matches(&wrong_size, &meta));
let mut wrong_time = entry.clone();
wrong_time.mtime_ns ^= 1;
assert!(!stat_matches(&wrong_time, &meta));
}
#[cfg(unix)]
#[test]
fn stat_matches_detects_exec_bit_flip() {
use std::os::unix::fs::PermissionsExt;
let work = TempDir::new().unwrap();
let f = work.path().join("run.sh");
fs::write(&f, b"#!/bin/sh\n").unwrap();
let meta = meta_of(&f);
let entry = crate::index::IndexEntry {
path: "run.sh".into(),
status: crate::index::EntryStatus::Blob,
object_hash: crate::hash::hash(b"x"),
mtime_ns: mtime_nanos(&meta),
size: meta.len(),
ino: 0,
ctime_ns: 0,
};
assert!(stat_matches(&entry, &meta));
let mtime = meta.modified().unwrap();
fs::set_permissions(&f, fs::Permissions::from_mode(0o755)).unwrap();
let f_handle = fs::File::options().write(true).open(&f).unwrap();
f_handle
.set_times(fs::FileTimes::new().set_modified(mtime))
.unwrap();
drop(f_handle);
let meta2 = meta_of(&f);
assert_eq!(mtime_nanos(&meta2), entry.mtime_ns, "mtime restored");
assert!(
!stat_matches(&entry, &meta2),
"exec-bit flip must invalidate a Blob-status cache hit"
);
}
#[cfg(unix)]
#[test]
fn build_tree_reuses_hash_on_stat_match_without_reading_file() {
use std::os::unix::fs::PermissionsExt;
let (_sd, store) = fresh_store();
let work = TempDir::new().unwrap();
let f = work.path().join("locked.txt");
fs::write(&f, b"cached content").unwrap();
let staged_hash = store_file_object(&store, b"cached content").unwrap();
let meta = meta_of(&f);
let idx = crate::index::Index::from_entries(vec![crate::index::IndexEntry {
path: "locked.txt".into(),
status: crate::index::EntryStatus::Blob,
object_hash: staged_hash,
mtime_ns: mtime_nanos(&meta),
size: meta.len(),
ino: 0,
ctime_ns: 0,
}]);
fs::set_permissions(&f, fs::Permissions::from_mode(0o000)).unwrap();
let result = build_tree_filtered(&store, work.path(), Some(&idx));
fs::set_permissions(&f, fs::Permissions::from_mode(0o644)).unwrap();
let tree_h = result.expect("stat match must skip the file read");
let Object::Tree(t) = store.read_object(&tree_h).unwrap() else {
panic!("expected tree");
};
assert_eq!(t.entries.len(), 1);
assert_eq!(t.entries[0].object_hash, staged_hash);
let (_sd2, store2) = fresh_store();
let f2_dir = TempDir::new().unwrap();
fs::write(f2_dir.path().join("locked.txt"), b"cached content").unwrap();
let plain = build_tree(&store2, f2_dir.path()).unwrap();
assert_eq!(plain, tree_h, "cache hit must not change tree hashes");
}
#[test]
fn build_tree_parallel_fanout_assigns_each_hash_to_the_right_file() {
const N: usize = 4096;
let (_sd, store) = fresh_store();
let work = TempDir::new().unwrap();
for i in 0..N {
fs::write(
work.path().join(format!("f{i:04}.txt")),
format!("content {i}"),
)
.unwrap();
}
let h = build_tree(&store, work.path()).unwrap();
let Object::Tree(t) = store.read_object(&h).unwrap() else {
panic!("expected tree");
};
assert_eq!(t.entries.len(), N);
let mut seen_hashes = std::collections::HashSet::with_capacity(N);
for (i, entry) in t.entries.iter().enumerate() {
let expected_name = format!("f{i:04}.txt");
assert_eq!(
entry.name,
expected_name.as_bytes(),
"entry {i} out of order"
);
let expected_content = format!("content {i}");
let expected_hash = hash_file_object(expected_content.as_bytes()).unwrap();
assert_eq!(
entry.object_hash, expected_hash,
"entry {i} ({expected_name}) has the wrong hash — parallel slot misalignment?"
);
assert!(
seen_hashes.insert(entry.object_hash),
"entry {i} repeats another entry's hash (or the ZERO placeholder)"
);
}
}
#[cfg(unix)]
#[test]
fn stat_mismatch_on_inode_rehashes() {
let work = TempDir::new().unwrap();
let f = work.path().join("swap.txt");
fs::write(&f, b"original").unwrap();
let meta = meta_of(&f);
let (mtime_ns, size, ino, ctime_ns) = stat_cache_fields(&meta);
let entry = crate::index::IndexEntry {
path: "swap.txt".into(),
status: crate::index::EntryStatus::Blob,
object_hash: crate::hash::hash(b"original"),
mtime_ns,
size,
ino,
ctime_ns,
};
assert!(stat_matches(&entry, &meta));
let staging = work.path().join(".swap.new");
fs::write(&staging, b"REPLACED").unwrap(); let fh = fs::File::options().write(true).open(&staging).unwrap();
fh.set_times(fs::FileTimes::new().set_modified(meta.modified().unwrap()))
.unwrap();
drop(fh);
fs::rename(&staging, &f).unwrap();
let meta2 = meta_of(&f);
assert_eq!(meta2.len(), entry.size, "size preserved by the swap");
assert!(
!stat_matches(&entry, &meta2),
"a renamed-in replacement must not stat-match (ino differs)"
);
}
#[test]
fn stat_mismatch_on_ctime_rehashes() {
let work = TempDir::new().unwrap();
let f = work.path().join("touched.txt");
fs::write(&f, b"content").unwrap();
let meta = meta_of(&f);
let (mtime_ns, size, ino, ctime_ns) = stat_cache_fields(&meta);
if ctime_ns == 0 {
return; }
let entry = crate::index::IndexEntry {
path: "touched.txt".into(),
status: crate::index::EntryStatus::Blob,
object_hash: crate::hash::hash(b"content"),
mtime_ns,
size,
ino,
ctime_ns: ctime_ns ^ 1,
};
assert!(
!stat_matches(&entry, &meta),
"ctime disagreement must invalidate the cache"
);
}
#[test]
fn build_tree_observed_reports_clean_rehashes() {
let (_sd, store) = fresh_store();
let work = TempDir::new().unwrap();
fs::write(work.path().join("clean.txt"), b"clean bytes").unwrap();
fs::write(work.path().join("dirty.txt"), b"new content").unwrap();
let clean_hash = store_file_object(&store, b"clean bytes").unwrap();
let stale_hash = crate::hash::hash(b"old content");
let idx = crate::index::Index::from_entries(vec![
crate::index::IndexEntry {
path: "clean.txt".into(),
status: crate::index::EntryStatus::Blob,
object_hash: clean_hash,
mtime_ns: 0, size: 0,
ino: 0,
ctime_ns: 0,
},
crate::index::IndexEntry {
path: "dirty.txt".into(),
status: crate::index::EntryStatus::Blob,
object_hash: stale_hash,
mtime_ns: 0,
size: 0,
ino: 0,
ctime_ns: 0,
},
]);
let mut obs = Vec::new();
build_tree_filtered_observed(&store, work.path(), Some(&idx), &mut obs).unwrap();
assert_eq!(obs.len(), 1, "only the verified-clean entry is observed");
let o = &obs[0];
assert_eq!(o.path, "clean.txt");
assert_eq!(o.object_hash, clean_hash);
let meta = meta_of(&work.path().join("clean.txt"));
let (mtime_ns, size, _ino, _ctime) = stat_cache_fields(&meta);
assert_eq!(o.mtime_ns, mtime_ns, "observation carries the fd stat");
assert_eq!(o.size, size);
}
#[test]
fn build_tree_rehashes_on_stat_mismatch() {
let (_sd, store) = fresh_store();
let work = TempDir::new().unwrap();
let f = work.path().join("changed.txt");
fs::write(&f, b"new content").unwrap();
let stale_hash = crate::hash::hash(b"not the real object");
let meta = meta_of(&f);
let idx = crate::index::Index::from_entries(vec![crate::index::IndexEntry {
path: "changed.txt".into(),
status: crate::index::EntryStatus::Blob,
object_hash: stale_hash,
mtime_ns: mtime_nanos(&meta),
size: meta.len() + 1,
ino: 0,
ctime_ns: 0,
}]);
let tree_h = build_tree_filtered(&store, work.path(), Some(&idx)).unwrap();
let Object::Tree(t) = store.read_object(&tree_h).unwrap() else {
panic!("expected tree");
};
assert_ne!(
t.entries[0].object_hash, stale_hash,
"mismatched stat must not reuse the stale hash"
);
assert_eq!(
t.entries[0].object_hash,
store_file_object(&store, b"new content").unwrap()
);
}
}