Skip to main content

mkit_core/worktree/
blob.rs

1//! Blob-content I/O: reading back what the worktree walker stored.
2//!
3//! A file's content lives either in a single inline [`Blob`](crate::object::Blob)
4//! or behind a [`ChunkedBlob`](crate::object::ChunkedBlob) manifest. [`LoadedBlob`]
5//! is the one-read-per-object view over both shapes; [`read_blob`] is the one-shot
6//! full-content convenience on top of it. Extracted from `worktree.rs` (#633) —
7//! this cluster reconstructs content and has nothing to do with walking a worktree.
8
9use std::borrow::Cow;
10use std::io;
11
12use crate::hash::Hash;
13use crate::object::{ChunkedBlob, Object};
14
15use super::{WorktreeError, WorktreeResult};
16
17/// Reassemble the full byte content of a `Blob` or `ChunkedBlob` object
18/// addressed by `hash`.
19///
20/// A plain [`Blob`](crate::object::Blob) returns its bytes directly. A
21/// [`ChunkedBlob`] manifest is reassembled
22/// by concatenating each referenced chunk (every chunk must itself be a
23/// `Blob`). This is the shared counterpart to [`store_file_object`](super::store_file_object) and
24/// backs `mkit cat`, `mkit diff`, conflict rendering, and blame so they
25/// all reconstruct large-file content the same way.
26///
27/// # Errors
28/// - [`WorktreeError::Store`] if `hash` or any chunk is missing.
29/// - [`WorktreeError::Io`] if `hash` (or a chunk) resolves to an object
30///   that is neither a `Blob` nor a `ChunkedBlob` of `Blob`s.
31/// - [`WorktreeError::Object`] if the concatenated chunks do not total
32///   the manifest's `total_size` (SPEC-OBJECTS §7).
33pub fn read_blob<S: crate::store::ObjectSource + ?Sized>(
34    store: &S,
35    hash: &Hash,
36) -> WorktreeResult<Vec<u8>> {
37    LoadedBlob::load(store, hash)?.into_content(store)
38}
39
40/// A blob's top-level object, read from the store exactly once and held so
41/// a caller needing several views of the same blob — byte length, bounded
42/// prefix, full content — pays a single top-level `read_object` instead of
43/// one per view. `diff --stat` is the motivating caller: its text-vs-binary
44/// sniff needs a prefix, then either the length (binary row) or the full
45/// content (text row), and taking each view through a separate store-level
46/// read re-read (and re-hash-verified) the same object two times per
47/// changed file, a per-entry cost that dominates a many-small-files
48/// diffstat (#624).
49///
50/// Chunk objects are still read on demand: [`Self::len`] reads none,
51/// [`Self::prefix`] reads only leading chunks, [`Self::into_content`]
52/// reads them all.
53///
54/// One read stays undeduped by design: [`Self::prefix`] on a chunked blob
55/// reads the leading chunk(s) to sniff, and [`Self::into_content`] reads
56/// every chunk (including that same leading one) to reassemble — a caller
57/// doing both, as `diff --stat` does, reads the first chunk twice. Caching
58/// it would need either interior mutability or a consuming-prefix API, to
59/// save an O(1) read on an O(n-chunks) path; not worth it (#624).
60#[derive(Debug)]
61pub enum LoadedBlob {
62    /// An inline [`Blob`](crate::object::Blob): its full content, already
63    /// in hand.
64    Inline(Vec<u8>),
65    /// A [`ChunkedBlob`] manifest: content lives in chunk objects, read
66    /// only when a view needs them.
67    Chunked(ChunkedBlob),
68}
69
70impl LoadedBlob {
71    /// Read the top-level object addressed by `hash` — one store read.
72    ///
73    /// # Errors
74    /// - [`WorktreeError::Store`] if `hash` is missing.
75    /// - [`WorktreeError::Io`] if `hash` resolves to an object that is
76    ///   neither a `Blob` nor a `ChunkedBlob`.
77    pub fn load<S: crate::store::ObjectSource + ?Sized>(
78        store: &S,
79        hash: &Hash,
80    ) -> WorktreeResult<Self> {
81        match store.read_object(hash)? {
82            Object::Blob(b) => Ok(Self::Inline(b.data)),
83            Object::ChunkedBlob(manifest) => Ok(Self::Chunked(manifest)),
84            other => Err(not_a_blob("object", hash, &other)),
85        }
86    }
87
88    /// Content byte length from what is already in hand — an inline blob's
89    /// data length, a chunked blob's manifest `total_size` — no chunk
90    /// reads. `total_size` is trustworthy without re-verifying against the
91    /// chunks: every reassembly path enforces it via
92    /// [`ChunkedBlob::check_reassembled_size`], so a manifest with a wrong
93    /// `total_size` cannot have been durably written (#550).
94    #[must_use]
95    pub fn len(&self) -> u64 {
96        match self {
97            Self::Inline(data) => data.len() as u64,
98            Self::Chunked(manifest) => manifest.total_size,
99        }
100    }
101
102    /// Whether the content is zero bytes long.
103    #[must_use]
104    pub fn is_empty(&self) -> bool {
105        self.len() == 0
106    }
107
108    /// Up to `max_len` leading content bytes. Inline data is borrowed (no
109    /// copy, no reads); a chunked blob reads only as many leading chunks
110    /// as it takes to cover `max_len`, then stops — one chunk in practice,
111    /// since every chunk but possibly the last is at least
112    /// [`crate::chunker::MIN_SIZE`] bytes, well above the sniff windows
113    /// callers use (e.g. [`crate::ops::diff::BINARY_SNIFF_LEN`]).
114    ///
115    /// # Errors
116    /// - [`WorktreeError::Store`] if a needed chunk is missing.
117    /// - [`WorktreeError::Io`] if a needed chunk is not a `Blob`.
118    pub fn prefix<S: crate::store::ObjectSource + ?Sized>(
119        &self,
120        store: &S,
121        max_len: usize,
122    ) -> WorktreeResult<Cow<'_, [u8]>> {
123        match self {
124            Self::Inline(data) => Ok(Cow::Borrowed(&data[..data.len().min(max_len)])),
125            Self::Chunked(manifest) => {
126                let cap = usize::try_from(manifest.total_size)
127                    .unwrap_or(max_len)
128                    .min(max_len);
129                let mut data = Vec::with_capacity(cap);
130                for chunk in &manifest.chunks {
131                    if data.len() >= max_len {
132                        break;
133                    }
134                    data.extend_from_slice(&read_chunk(store, chunk)?);
135                }
136                data.truncate(max_len);
137                Ok(Cow::Owned(data))
138            }
139        }
140    }
141
142    /// The full content: inline data as-is (no further reads), a chunked
143    /// blob reassembled by reading every chunk, with the result length
144    /// enforced against the manifest's `total_size` (SPEC-OBJECTS §7).
145    ///
146    /// # Errors
147    /// - [`WorktreeError::Store`] if a chunk is missing.
148    /// - [`WorktreeError::Io`] if a chunk is not a `Blob`.
149    /// - [`WorktreeError::Object`] if the concatenated chunks do not total
150    ///   the manifest's `total_size`.
151    pub fn into_content<S: crate::store::ObjectSource + ?Sized>(
152        self,
153        store: &S,
154    ) -> WorktreeResult<Vec<u8>> {
155        match self {
156            Self::Inline(data) => Ok(data),
157            Self::Chunked(manifest) => {
158                let mut data =
159                    Vec::with_capacity(usize::try_from(manifest.total_size).unwrap_or(0));
160                for chunk in &manifest.chunks {
161                    data.extend_from_slice(&read_chunk(store, chunk)?);
162                }
163                manifest.check_reassembled_size(data.len())?;
164                Ok(data)
165            }
166        }
167    }
168
169    /// The empty blob: what a diff side with no object (an add's old side,
170    /// a delete's new side) loads as. Zero length, empty prefix, empty
171    /// content — no store reads.
172    #[must_use]
173    pub fn empty() -> Self {
174        Self::Inline(Vec::new())
175    }
176}
177
178/// One chunk of a [`ChunkedBlob`]: must deserialize to an inline `Blob`.
179fn read_chunk<S: crate::store::ObjectSource + ?Sized>(
180    store: &S,
181    chunk: &Hash,
182) -> WorktreeResult<Vec<u8>> {
183    match store.read_object(chunk)? {
184        Object::Blob(b) => Ok(b.data),
185        other => Err(not_a_blob("chunk", chunk, &other)),
186    }
187}
188
189/// Read one manifest chunk's raw bytes, requiring a `Blob`, in
190/// [`crate::store::StoreError`]'s domain rather than [`WorktreeError`]'s
191/// (see [`read_chunk`] for the `WorktreeError` counterpart, kept separate
192/// since its error also carries the hash and the historical `"chunk ..."`
193/// wording). The single "read a manifest chunk" primitive shared by every
194/// `StoreError`-domain caller — [`content_fingerprint`],
195/// [`ContentCursor::remaining`], and [`chunked_content_eq`] — that
196/// previously each hand-rolled the same match.
197fn read_blob_chunk<S: crate::store::ObjectSource + ?Sized>(
198    store: &S,
199    chunk: &Hash,
200) -> Result<Vec<u8>, crate::store::StoreError> {
201    match store.read_object(chunk)? {
202        Object::Blob(b) => Ok(b.data),
203        _ => Err(crate::store::StoreError::Io(io::Error::other(
204            "manifest chunk is not a Blob",
205        ))),
206    }
207}
208
209/// The "expected a blob, found something else" error shared by every
210/// [`LoadedBlob`] read path; `what` is `"object"` for a top-level hash and
211/// `"chunk"` for a manifest chunk, preserving the historical wording of
212/// both messages.
213fn not_a_blob(what: &str, hash: &Hash, got: &Object) -> WorktreeError {
214    WorktreeError::Io(io::Error::other(format!(
215        "{what} {} is not a blob (got {})",
216        crate::hash::to_hex(hash),
217        got.object_type().name()
218    )))
219}
220
221/// Representation-independent content identity: byte length and raw BLAKE3.
222/// Reads each chunk once and verifies the complete manifest length. Memory is
223/// bounded by the manifest and one chunk, independent of reassembled size.
224///
225/// # Errors
226/// Missing, corrupt, wrong-type chunks and inconsistent lengths are errors.
227pub fn content_fingerprint<S: crate::store::ObjectSource + ?Sized>(
228    store: &S,
229    hash: &Hash,
230) -> Result<(u64, Hash), crate::store::StoreError> {
231    use crate::store::StoreError;
232    let mut hasher = crate::hash::Hasher::new();
233    match store.read_object(hash)? {
234        Object::Blob(b) => {
235            hasher.update(&b.data);
236            Ok((b.data.len() as u64, hasher.finalize()))
237        }
238        Object::ChunkedBlob(manifest) => {
239            let mut size = 0usize;
240            for chunk in &manifest.chunks {
241                let data = read_blob_chunk(store, chunk)?;
242                size = size
243                    .checked_add(data.len())
244                    .ok_or(StoreError::ObjectTooLarge)?;
245                hasher.update(&data);
246            }
247            manifest.check_reassembled_size(size)?;
248            Ok((size as u64, hasher.finalize()))
249        }
250        _ => Err(StoreError::Io(io::Error::other(
251            "content object is not a Blob or ChunkedBlob",
252        ))),
253    }
254}
255
256/// Compare file content independently of inline/chunked storage layout.
257/// Equal object IDs are a fast path; different IDs require verified content.
258///
259/// A `ChunkedBlob`-vs-`ChunkedBlob` pair (the large-file case) takes a
260/// further internal fast path (`chunked_content_eq`, private to this
261/// module), which skips reading any chunk both sides reference by the
262/// same hash instead of reassembling and byte-comparing every chunk of
263/// both blobs — see that function's doc for what it trusts and what it
264/// still fully verifies. Any other pairing
265/// (inline vs inline, or a mixed inline/chunked pair) keeps the exact
266/// byte-cursor walk this function has always used.
267///
268/// The chunked fast path's "same hash means same bytes, skip the read"
269/// trust holds only as far as `store: &S`'s reads are actually
270/// hash-verified — true of every `ObjectSource` in this crate today
271/// except [`crate::store::DisplaySource`] (used for read-only diff/show
272/// rendering, and not passed to `content_eq` by any current caller). A
273/// future caller doing so would trade that verification away for this
274/// fast path exactly as much as it already does for every other generic
275/// `ObjectSource` consumer.
276///
277/// # Errors
278/// Propagates errors from [`content_fingerprint`].
279pub fn content_eq<S: crate::store::ObjectSource + ?Sized>(
280    store: &S,
281    a: &Hash,
282    b: &Hash,
283) -> Result<bool, crate::store::StoreError> {
284    if a == b {
285        return Ok(true);
286    }
287    let obj_a = store.read_object(a)?;
288    let obj_b = store.read_object(b)?;
289    if let (Object::ChunkedBlob(ma), Object::ChunkedBlob(mb)) = (&obj_a, &obj_b) {
290        return chunked_content_eq(store, ma, mb);
291    }
292    let mut left = ContentCursor::from_object(obj_a)?;
293    let mut right = ContentCursor::from_object(obj_b)?;
294    let mut equal = true;
295    loop {
296        let a = left.remaining(store)?;
297        let b = right.remaining(store)?;
298        if a.is_empty() && b.is_empty() {
299            return Ok(equal);
300        }
301        if a.is_empty() || b.is_empty() {
302            equal = false;
303            let alen = a.len();
304            let blen = b.len();
305            left.offset += alen;
306            right.offset += blen;
307        } else {
308            let count = a.len().min(b.len());
309            equal &= a[..count] == b[..count];
310            left.offset += count;
311            right.offset += count;
312        }
313    }
314}
315
316/// Compare two [`ChunkedBlob`] manifests for content equality, skipping
317/// any chunk both sides reference by the same content-addressed hash:
318/// identical hash means identical bytes by construction — the same trust
319/// [`content_eq`]'s own `a == b` whole-object fast path already relies
320/// on, just applied per chunk instead of per object. Only chunks that
321/// diverge (a different hash at the same aligned position) are actually
322/// read and byte-compared, so a change confined to part of a large file
323/// costs only that part on both sides, not the whole file. An in-place
324/// edit of unchanged length walks straight past the shared, unaffected
325/// tail with no reads once the edited region resyncs by hash again.
326///
327/// `total_size` is trusted without re-summing actual chunk bytes only for
328/// a side whose chunks were *partly* skipped by id — the same trust
329/// [`LoadedBlob::len`] already documents: every reassembly path enforces
330/// it via [`ChunkedBlob::check_reassembled_size`], so a manifest with a
331/// wrong `total_size` cannot have been durably written through mkit's own
332/// writers (#550). A side every one of whose chunks gets actually read
333/// (no id ever matched a boundary on that side — the case an append or a
334/// wholly-rewritten file hits) has its real byte count checked against
335/// its declared `total_size` before returning, exactly like the old
336/// byte-cursor walk did, and errors the same way
337/// ([`crate::object::MkitError::ChunkedBlobSizeMismatch`]) if they
338/// disagree. Only a side that *did* skip at least one chunk by id skips
339/// this check for that side, since the skipped chunks' real lengths were
340/// never read to sum: two manifests whose declared `total_size` fields
341/// agree with each other but not with their own chunks, and whose chunk
342/// sequences also line up entirely (or partly, from the first divergence
343/// on) by hash, can still slip through unnoticed on that account. An
344/// append or truncation, meanwhile, needs zero chunk reads at all before
345/// even reaching this check: the two sides' declared sizes differ, so
346/// `content_eq` returns `Ok(false)` above before the merge walk starts.
347///
348/// # Errors
349/// [`crate::store::StoreError`] if a chunk that must be read (either
350/// side has no matching chunk to skip against) is missing, corrupt, or
351/// not a `Blob`; [`crate::object::MkitError::ChunkedBlobSizeMismatch`]
352/// if a side with no skipped chunks has a `total_size` that disagrees
353/// with its chunks' real byte sum.
354fn chunked_content_eq<S: crate::store::ObjectSource + ?Sized>(
355    store: &S,
356    ma: &ChunkedBlob,
357    mb: &ChunkedBlob,
358) -> Result<bool, crate::store::StoreError> {
359    if ma.total_size != mb.total_size {
360        return Ok(false);
361    }
362
363    let mut ia = 0usize;
364    let mut ib = 0usize;
365    let mut buf_a: Vec<u8> = Vec::new();
366    let mut buf_b: Vec<u8> = Vec::new();
367    let mut pos_a = 0usize;
368    let mut pos_b = 0usize;
369    let mut equal = true;
370    // Real bytes actually read (never chunks skipped by id) per side,
371    // and whether any chunk on that side WAS skipped by id — see the
372    // doc comment above for what these guard.
373    let mut read_a: u64 = 0;
374    let mut read_b: u64 = 0;
375    let mut skipped_a = false;
376    let mut skipped_b = false;
377
378    loop {
379        // Both cursors sit at a chunk boundary: skip a run of chunks
380        // that match by id, with no read on either side.
381        while pos_a == buf_a.len()
382            && pos_b == buf_b.len()
383            && ia < ma.chunks.len()
384            && ib < mb.chunks.len()
385            && ma.chunks[ia] == mb.chunks[ib]
386        {
387            ia += 1;
388            ib += 1;
389            skipped_a = true;
390            skipped_b = true;
391        }
392
393        if pos_a == buf_a.len() {
394            buf_a = match ma.chunks.get(ia) {
395                Some(h) => {
396                    ia += 1;
397                    let data = read_blob_chunk(store, h)?;
398                    read_a = read_a
399                        .checked_add(data.len() as u64)
400                        .ok_or(crate::store::StoreError::ObjectTooLarge)?;
401                    data
402                }
403                None => Vec::new(),
404            };
405            pos_a = 0;
406        }
407        if pos_b == buf_b.len() {
408            buf_b = match mb.chunks.get(ib) {
409                Some(h) => {
410                    ib += 1;
411                    let data = read_blob_chunk(store, h)?;
412                    read_b = read_b
413                        .checked_add(data.len() as u64)
414                        .ok_or(crate::store::StoreError::ObjectTooLarge)?;
415                    data
416                }
417                None => Vec::new(),
418            };
419            pos_b = 0;
420        }
421
422        let a_rest = &buf_a[pos_a..];
423        let b_rest = &buf_b[pos_b..];
424        if a_rest.is_empty() && b_rest.is_empty() {
425            if ia >= ma.chunks.len() && ib >= mb.chunks.len() {
426                if !skipped_a && read_a != ma.total_size {
427                    return Err(crate::object::MkitError::ChunkedBlobSizeMismatch {
428                        expected: ma.total_size,
429                        actual: read_a,
430                    }
431                    .into());
432                }
433                if !skipped_b && read_b != mb.total_size {
434                    return Err(crate::object::MkitError::ChunkedBlobSizeMismatch {
435                        expected: mb.total_size,
436                        actual: read_b,
437                    }
438                    .into());
439                }
440                return Ok(equal);
441            }
442            // A zero-length trailing chunk on one or both sides; loop
443            // again to pull the next one via the boundary checks above.
444            continue;
445        }
446
447        let count = a_rest.len().min(b_rest.len());
448        if count == 0 {
449            // One side has run out of chunks for good; drain the other
450            // (matching the byte-cursor path's behavior) so a read
451            // error later in its remaining chunks still surfaces.
452            equal = false;
453            pos_a = buf_a.len();
454            pos_b = buf_b.len();
455            continue;
456        }
457        if a_rest[..count] != b_rest[..count] {
458            equal = false;
459        }
460        pos_a += count;
461        pos_b += count;
462    }
463}
464
465/// Compare stored content to bytes without reassembling chunked storage.
466///
467/// # Errors
468/// Propagates errors from [`content_fingerprint`].
469pub fn content_eq_bytes<S: crate::store::ObjectSource + ?Sized>(
470    store: &S,
471    object: &Hash,
472    bytes: &[u8],
473) -> Result<bool, crate::store::StoreError> {
474    let mut cursor = ContentCursor::load(store, object)?;
475    let mut offset = 0usize;
476    let mut equal = true;
477    loop {
478        let chunk = cursor.remaining(store)?;
479        if chunk.is_empty() {
480            return Ok(equal && offset == bytes.len());
481        }
482        let end = offset
483            .checked_add(chunk.len())
484            .ok_or(crate::store::StoreError::ObjectTooLarge)?;
485        equal &= bytes.get(offset..end) == Some(chunk);
486        offset = end;
487        cursor.offset += chunk.len();
488    }
489}
490
491struct ContentCursor {
492    data: Vec<u8>,
493    offset: usize,
494    chunks: std::vec::IntoIter<Hash>,
495    expected: u64,
496    loaded: u64,
497}
498
499impl ContentCursor {
500    fn load<S: crate::store::ObjectSource + ?Sized>(
501        store: &S,
502        hash: &Hash,
503    ) -> Result<Self, crate::store::StoreError> {
504        Self::from_object(store.read_object(hash)?)
505    }
506
507    /// Build a cursor from an already-read top-level object, saving the
508    /// caller a second `read_object` when it needed the object anyway
509    /// (e.g. [`content_eq`] deciding whether the chunked fast path
510    /// applies).
511    fn from_object(object: Object) -> Result<Self, crate::store::StoreError> {
512        let (data, chunks, expected) = match object {
513            Object::Blob(b) => {
514                let size = b.data.len() as u64;
515                (b.data, Vec::new(), size)
516            }
517            Object::ChunkedBlob(m) => (Vec::new(), m.chunks, m.total_size),
518            _ => {
519                return Err(crate::store::StoreError::Io(io::Error::other(
520                    "content object is not a Blob or ChunkedBlob",
521                )));
522            }
523        };
524        let loaded = data.len() as u64;
525        Ok(Self {
526            data,
527            offset: 0,
528            chunks: chunks.into_iter(),
529            expected,
530            loaded,
531        })
532    }
533
534    fn remaining<S: crate::store::ObjectSource + ?Sized>(
535        &mut self,
536        store: &S,
537    ) -> Result<&[u8], crate::store::StoreError> {
538        while self.offset == self.data.len() {
539            let Some(hash) = self.chunks.next() else {
540                if self.loaded != self.expected {
541                    return Err(crate::object::MkitError::ChunkedBlobSizeMismatch {
542                        expected: self.expected,
543                        actual: self.loaded,
544                    }
545                    .into());
546                }
547                return Ok(&[]);
548            };
549            let data = read_blob_chunk(store, &hash)?;
550            self.loaded = self
551                .loaded
552                .checked_add(data.len() as u64)
553                .ok_or(crate::store::StoreError::ObjectTooLarge)?;
554            self.data = data;
555            self.offset = 0;
556        }
557        Ok(&self.data[self.offset..])
558    }
559}
560
561#[cfg(test)]
562mod equality_tests {
563    use super::*;
564    use crate::{layout::RepoLayout, object::Blob, serialize, store::ObjectStore};
565
566    fn put(store: &ObjectStore, object: &Object) -> Hash {
567        store.write(&serialize::serialize(object).unwrap()).unwrap()
568    }
569
570    #[test]
571    fn large_inline_fixed_and_cdc_content_agree() {
572        let dir = tempfile::tempdir().unwrap();
573        let store = ObjectStore::init(&RepoLayout::single(dir.path())).unwrap();
574        let data = vec![7; usize::try_from(super::super::CHUNK_THRESHOLD + 17).unwrap()];
575        let inline = put(&store, &Object::Blob(Blob { data: data.clone() }));
576        let cdc = super::super::store_file_object(&store, &data).unwrap();
577        let chunks = data
578            .chunks(65_536)
579            .map(|b| put(&store, &Object::Blob(Blob { data: b.to_vec() })))
580            .collect();
581        let fixed = put(
582            &store,
583            &Object::ChunkedBlob(ChunkedBlob {
584                total_size: data.len() as u64,
585                chunk_size: 65_536,
586                chunks,
587            }),
588        );
589        assert_ne!(inline, cdc);
590        assert_ne!(fixed, cdc);
591        assert!(content_eq(&store, &inline, &fixed).unwrap());
592        assert!(content_eq(&store, &fixed, &cdc).unwrap());
593        assert!(content_eq_bytes(&store, &fixed, &data).unwrap());
594        assert_eq!(
595            content_fingerprint(&store, &inline).unwrap(),
596            content_fingerprint(&store, &cdc).unwrap()
597        );
598        let mut changed = data;
599        changed[65_536] = 8;
600        assert!(!content_eq_bytes(&store, &fixed, &changed).unwrap());
601    }
602
603    #[test]
604    fn invalid_chunks_are_errors_even_after_content_differs() {
605        let dir = tempfile::tempdir().unwrap();
606        let store = ObjectStore::init(&RepoLayout::single(dir.path())).unwrap();
607        let a = put(
608            &store,
609            &Object::Blob(Blob {
610                data: b"a".to_vec(),
611            }),
612        );
613        let b = put(
614            &store,
615            &Object::Blob(Blob {
616                data: b"b".to_vec(),
617            }),
618        );
619        for (total_size, chunks) in [(2, vec![a]), (2, vec![a, [42; 32]])] {
620            let bad = put(
621                &store,
622                &Object::ChunkedBlob(ChunkedBlob {
623                    total_size,
624                    chunk_size: 0,
625                    chunks,
626                }),
627            );
628            assert!(content_eq(&store, &bad, &b).is_err());
629            assert!(content_eq_bytes(&store, &bad, b"b").is_err());
630            assert!(content_fingerprint(&store, &bad).is_err());
631        }
632    }
633
634    fn chunk(store: &ObjectStore, data: &[u8]) -> Hash {
635        put(
636            store,
637            &Object::Blob(Blob {
638                data: data.to_vec(),
639            }),
640        )
641    }
642
643    fn manifest(store: &ObjectStore, parts: &[&[u8]]) -> ChunkedBlob {
644        let total_size: u64 = parts.iter().map(|p| p.len() as u64).sum();
645        ChunkedBlob {
646            total_size,
647            chunk_size: 0,
648            chunks: parts.iter().map(|p| chunk(store, p)).collect(),
649        }
650    }
651
652    fn fresh_store() -> (tempfile::TempDir, ObjectStore) {
653        let dir = tempfile::tempdir().unwrap();
654        let store = ObjectStore::init(&RepoLayout::single(dir.path())).unwrap();
655        (dir, store)
656    }
657
658    // Direct tests of `chunked_content_eq`'s merge logic against small,
659    // hand-built manifests — fast and deterministic, independent of real
660    // FastCDC boundaries, covering the id-skip fast path, the
661    // read-and-resync fallback, and the documented total_size trust gap.
662
663    #[test]
664    fn chunked_fast_path_all_chunks_match_by_id() {
665        let (_dir, store) = fresh_store();
666        let x = chunk(&store, b"hello ");
667        let y = chunk(&store, b"world");
668        let ma = ChunkedBlob {
669            total_size: 11,
670            chunk_size: 0,
671            chunks: vec![x, y],
672        };
673        let mb = ma.clone();
674        assert!(super::chunked_content_eq(&store, &ma, &mb).unwrap());
675    }
676
677    #[test]
678    fn chunked_fully_misaligned_but_equal_content() {
679        let (_dir, store) = fresh_store();
680        // "ab"+"cd" vs "a"+"bcd" — no chunk hash ever matches, so every
681        // byte is read and compared through the resync fallback, yet the
682        // reassembled content is identical.
683        let ma = manifest(&store, &[b"ab", b"cd"]);
684        let mb = manifest(&store, &[b"a", b"bcd"]);
685        assert!(super::chunked_content_eq(&store, &ma, &mb).unwrap());
686        assert!(super::chunked_content_eq(&store, &mb, &ma).unwrap());
687    }
688
689    #[test]
690    fn chunked_shared_prefix_then_misaligned_equal_suffix() {
691        let (_dir, store) = fresh_store();
692        let x = chunk(&store, b"shared-prefix-");
693        let ma = ChunkedBlob {
694            total_size: 14 + 4,
695            chunk_size: 0,
696            chunks: [x]
697                .into_iter()
698                .chain(manifest(&store, &[b"ab", b"cd"]).chunks)
699                .collect(),
700        };
701        let mb = ChunkedBlob {
702            total_size: 14 + 4,
703            chunk_size: 0,
704            chunks: [x]
705                .into_iter()
706                .chain(manifest(&store, &[b"a", b"bcd"]).chunks)
707                .collect(),
708        };
709        // `x` is skipped by id; "ab"+"cd" vs "a"+"bcd" is read and resynced.
710        assert!(super::chunked_content_eq(&store, &ma, &mb).unwrap());
711    }
712
713    #[test]
714    fn chunked_append_differs_via_total_size_without_reading() {
715        let (_dir, store) = fresh_store();
716        let x = chunk(&store, b"shared");
717        let ma = ChunkedBlob {
718            total_size: 6,
719            chunk_size: 0,
720            chunks: vec![x],
721        };
722        let y = chunk(&store, b"-more");
723        let mb = ChunkedBlob {
724            total_size: 11,
725            chunk_size: 0,
726            chunks: vec![x, y],
727        };
728        assert!(!super::chunked_content_eq(&store, &ma, &mb).unwrap());
729        assert!(!super::chunked_content_eq(&store, &mb, &ma).unwrap());
730    }
731
732    #[test]
733    fn chunked_same_total_size_different_content_after_shared_prefix() {
734        let (_dir, store) = fresh_store();
735        let x = chunk(&store, b"shared-");
736        let ma = ChunkedBlob {
737            total_size: 7 + 2,
738            chunk_size: 0,
739            chunks: vec![x, chunk(&store, b"ab")],
740        };
741        let mb = ChunkedBlob {
742            total_size: 7 + 2,
743            chunk_size: 0,
744            chunks: vec![x, chunk(&store, b"ba")],
745        };
746        assert!(!super::chunked_content_eq(&store, &ma, &mb).unwrap());
747    }
748
749    #[test]
750    fn chunked_split_differently_but_actually_different_content() {
751        let (_dir, store) = fresh_store();
752        let x = chunk(&store, b"shared-");
753        let ma = ChunkedBlob {
754            total_size: 7 + 4,
755            chunk_size: 0,
756            chunks: vec![x, chunk(&store, b"abcd")],
757        };
758        let mb = ChunkedBlob {
759            total_size: 7 + 4,
760            chunk_size: 0,
761            chunks: vec![x, chunk(&store, b"ab"), chunk(&store, b"cX")],
762        };
763        assert!(!super::chunked_content_eq(&store, &ma, &mb).unwrap());
764    }
765
766    #[test]
767    fn chunked_empty_manifests_are_equal() {
768        let (_dir, store) = fresh_store();
769        let ma = ChunkedBlob {
770            total_size: 0,
771            chunk_size: 0,
772            chunks: vec![],
773        };
774        let mb = ma.clone();
775        assert!(super::chunked_content_eq(&store, &ma, &mb).unwrap());
776    }
777
778    /// Documents the accepted trust gap: `total_size` is not re-summed
779    /// from actual chunk bytes on the chunked fast path (mirroring
780    /// `LoadedBlob::len`'s existing trust). Two manifests whose chunk
781    /// sequence fully matches by id, but whose shared `total_size`
782    /// field is wrong for that sequence, are reported equal without
783    /// ever reading a chunk to notice. A manifest this malformed cannot
784    /// be produced by any of mkit's own writers (`check_reassembled_size`
785    /// gates every one); this can only arise from an object constructed
786    /// directly, bypassing them, as this test does.
787    #[test]
788    fn chunked_wrong_shared_total_size_is_not_detected_when_ids_fully_match() {
789        let (_dir, store) = fresh_store();
790        let x = chunk(&store, b"ab"); // 2 real bytes
791        let ma = ChunkedBlob {
792            total_size: 999, // wrong on both sides, identically
793            chunk_size: 0,
794            chunks: vec![x],
795        };
796        let mb = ma.clone();
797        assert!(super::chunked_content_eq(&store, &ma, &mb).unwrap());
798    }
799
800    /// The case the fast path must *not* let slip past: no chunk on
801    /// either side is ever skipped by id (every hash differs), so both
802    /// sides get fully read — and a wrong `total_size` on a side that
803    /// was fully read must still surface as
804    /// `ChunkedBlobSizeMismatch`, exactly like the old byte-cursor walk.
805    /// Regression test for a gap an independent review found: an
806    /// earlier version of `chunked_content_eq` only ever compared the
807    /// two manifests' declared `total_size` fields against *each
808    /// other*, never against either side's own real chunk bytes, so two
809    /// fully-diverging (no id ever matches) manifests that happened to
810    /// declare the same wrong `total_size` were reported merely
811    /// "unequal" instead of erroring.
812    #[test]
813    fn chunked_wrong_total_size_is_detected_when_no_chunk_is_skipped() {
814        let (_dir, store) = fresh_store();
815        // Real chunk bytes: 90 bytes on each side ("a" x90 vs "b" x90),
816        // so no chunk hash ever matches — nothing is ever skipped.
817        let a90 = vec![b'a'; 90];
818        let b90 = vec![b'b'; 90];
819        let ma = ChunkedBlob {
820            total_size: 100, // wrong: real sum is 90
821            chunk_size: 0,
822            chunks: vec![chunk(&store, &a90)],
823        };
824        let mb = ChunkedBlob {
825            total_size: 100, // also wrong, and equal to ma's
826            chunk_size: 0,
827            chunks: vec![chunk(&store, &b90)],
828        };
829        let err = super::chunked_content_eq(&store, &ma, &mb).unwrap_err();
830        assert!(
831            matches!(
832                err,
833                crate::store::StoreError::Decode(
834                    crate::object::MkitError::ChunkedBlobSizeMismatch { .. }
835                )
836            ),
837            "expected ChunkedBlobSizeMismatch, got {err:?}"
838        );
839    }
840
841    /// The same wrong-`total_size` gap, but on only one side: the other
842    /// side's declared size is accurate for its own real chunk bytes.
843    /// Content is unequal either way (lengths differ once both are read
844    /// out to their real end), but the malformed side must still error
845    /// rather than silently compare as "not equal".
846    #[test]
847    fn chunked_wrong_total_size_on_one_side_only_is_detected() {
848        let (_dir, store) = fresh_store();
849        let a80 = vec![b'a'; 80];
850        let b100 = vec![b'b'; 100];
851        let ma = ChunkedBlob {
852            total_size: 100, // wrong: real sum is 80
853            chunk_size: 0,
854            chunks: vec![chunk(&store, &a80)],
855        };
856        let mb = ChunkedBlob {
857            total_size: 100, // correct
858            chunk_size: 0,
859            chunks: vec![chunk(&store, &b100)],
860        };
861        let err = super::chunked_content_eq(&store, &ma, &mb).unwrap_err();
862        assert!(
863            matches!(
864                err,
865                crate::store::StoreError::Decode(
866                    crate::object::MkitError::ChunkedBlobSizeMismatch { .. }
867                )
868            ),
869            "expected ChunkedBlobSizeMismatch, got {err:?}"
870        );
871    }
872
873    /// End-to-end sanity check through the public `content_eq` entry
874    /// point (not the private merge function directly) over real
875    /// `FastCDC`-chunked content: an append, a single-byte in-place edit,
876    /// and a truncation each produce a different manifest and must all
877    /// compare unequal, exactly matching a ground-truth byte comparison.
878    #[test]
879    fn content_eq_real_chunked_mutations_match_ground_truth() {
880        let (_dir, store) = fresh_store();
881        let threshold = usize::try_from(super::super::CHUNK_THRESHOLD).unwrap();
882        // Deterministic pseudo-random bytes so FastCDC sees real cut
883        // points instead of one run-length-maxed chunk.
884        let mut data = Vec::with_capacity(threshold * 3);
885        let mut state: u64 = 0x1234_5678_9abc_def0;
886        for _ in 0..threshold * 3 {
887            state = state
888                .wrapping_mul(6_364_136_223_846_793_005)
889                .wrapping_add(1);
890            data.push((state >> 56) as u8);
891        }
892
893        let base = super::super::store_file_object(&store, &data).unwrap();
894
895        let mut appended = data.clone();
896        appended.extend_from_slice(b"appended tail bytes");
897        let appended_hash = super::super::store_file_object(&store, &appended).unwrap();
898
899        let mut edited = data.clone();
900        let mid = edited.len() / 2;
901        edited[mid] ^= 0xFF;
902        let edited_hash = super::super::store_file_object(&store, &edited).unwrap();
903
904        let truncated = &data[..data.len() - 500];
905        let truncated_hash = super::super::store_file_object(&store, truncated).unwrap();
906
907        let unchanged_hash = super::super::store_file_object(&store, &data).unwrap();
908
909        assert_eq!(base, unchanged_hash, "identical content dedups to one id");
910        assert!(content_eq(&store, &base, &unchanged_hash).unwrap());
911
912        for (name, other, other_bytes) in [
913            ("append", appended_hash, appended.as_slice()),
914            ("edit", edited_hash, edited.as_slice()),
915            ("truncate", truncated_hash, truncated),
916        ] {
917            assert_ne!(base, other, "{name}: expected a different object id");
918            assert_eq!(
919                content_eq(&store, &base, &other).unwrap(),
920                data == other_bytes,
921                "{name}: content_eq must match ground truth"
922            );
923            assert!(
924                !content_eq(&store, &base, &other).unwrap(),
925                "{name}: bytes differ"
926            );
927        }
928    }
929
930    /// A mid-file insertion against real `FastCDC` output: unlike
931    /// `content_eq_real_chunked_mutations_match_ground_truth`'s
932    /// single-byte edit (which may or may not shift a chunk boundary),
933    /// inserting new bytes shifts every downstream offset, which
934    /// reliably forces `FastCDC` to re-cut several chunks around the
935    /// insertion point before content-defined chunking resyncs on the
936    /// unchanged bytes further on. This is the scenario
937    /// `chunked_content_eq`'s doc comment describes ("a change confined
938    /// to part of a large file... resyncs by hash again") and, per an
939    /// independent review, the only other resync coverage exercised
940    /// hand-built or fixed-vs-CDC manifests, never two independently
941    /// `FastCDC`-chunked real files. Confirms at the manifest level
942    /// that a real divergence-then-resync actually occurred (shared
943    /// first and last chunk hashes, a different chunk in between) before
944    /// checking `content_eq` against ground truth.
945    #[test]
946    fn content_eq_real_chunked_insertion_forces_boundary_resync() {
947        let (_dir, store) = fresh_store();
948        let threshold = usize::try_from(super::super::CHUNK_THRESHOLD).unwrap();
949        let mut data = Vec::with_capacity(threshold * 4);
950        let mut state: u64 = 0x0BAD_C0DE_F00D_CAFE;
951        for _ in 0..threshold * 4 {
952            state = state
953                .wrapping_mul(6_364_136_223_846_793_005)
954                .wrapping_add(1);
955            data.push((state >> 56) as u8);
956        }
957
958        let mut inserted = data.clone();
959        let at = data.len() / 2;
960        let mut new_bytes = vec![0u8; 4096];
961        let mut s: u64 = 0xFACE_FEED_1234_5678;
962        for b in &mut new_bytes {
963            s = s.wrapping_mul(6_364_136_223_846_793_005).wrapping_add(1);
964            *b = (s >> 56) as u8;
965        }
966        inserted.splice(at..at, new_bytes.iter().copied());
967
968        let base_hash = super::super::store_file_object(&store, &data).unwrap();
969        let inserted_hash = super::super::store_file_object(&store, &inserted).unwrap();
970        assert_ne!(base_hash, inserted_hash);
971
972        let Object::ChunkedBlob(base_manifest) = store.read_object(&base_hash).unwrap() else {
973            panic!("expected base to be chunked (data.len() > CHUNK_THRESHOLD)");
974        };
975        let Object::ChunkedBlob(inserted_manifest) = store.read_object(&inserted_hash).unwrap()
976        else {
977            panic!("expected inserted to be chunked");
978        };
979        assert!(
980            base_manifest.chunks.len() > 2 && inserted_manifest.chunks.len() > 2,
981            "fixture too small to exercise multiple chunks"
982        );
983        assert_eq!(
984            base_manifest.chunks.first(),
985            inserted_manifest.chunks.first(),
986            "the unaffected prefix must still share its leading chunk by id"
987        );
988        assert_eq!(
989            base_manifest.chunks.last(),
990            inserted_manifest.chunks.last(),
991            "content-defined chunking must resync on the unaffected suffix"
992        );
993        assert_ne!(
994            base_manifest.chunks, inserted_manifest.chunks,
995            "the insertion must actually shift chunk boundaries somewhere in the middle"
996        );
997
998        assert!(
999            !content_eq(&store, &base_hash, &inserted_hash).unwrap(),
1000            "content genuinely differs after the insertion"
1001        );
1002    }
1003}