mkit_core/worktree/blob.rs
1//! Blob-content I/O: reading back what the worktree walker stored.
2//!
3//! A file's content lives either in a single inline [`Blob`](crate::object::Blob)
4//! or behind a [`ChunkedBlob`](crate::object::ChunkedBlob) manifest. [`LoadedBlob`]
5//! is the one-read-per-object view over both shapes; [`read_blob`] is the one-shot
6//! full-content convenience on top of it. Extracted from `worktree.rs` (#633) —
7//! this cluster reconstructs content and has nothing to do with walking a worktree.
8
9use std::borrow::Cow;
10use std::io;
11
12use crate::hash::Hash;
13use crate::object::{ChunkedBlob, Object};
14
15use super::{WorktreeError, WorktreeResult};
16
17/// Reassemble the full byte content of a `Blob` or `ChunkedBlob` object
18/// addressed by `hash`.
19///
20/// A plain [`Blob`](crate::object::Blob) returns its bytes directly. A
21/// [`ChunkedBlob`] manifest is reassembled
22/// by concatenating each referenced chunk (every chunk must itself be a
23/// `Blob`). This is the shared counterpart to [`store_file_object`](super::store_file_object) and
24/// backs `mkit cat`, `mkit diff`, conflict rendering, and blame so they
25/// all reconstruct large-file content the same way.
26///
27/// # Errors
28/// - [`WorktreeError::Store`] if `hash` or any chunk is missing.
29/// - [`WorktreeError::Io`] if `hash` (or a chunk) resolves to an object
30/// that is neither a `Blob` nor a `ChunkedBlob` of `Blob`s.
31/// - [`WorktreeError::Object`] if the concatenated chunks do not total
32/// the manifest's `total_size` (SPEC-OBJECTS §7).
33pub fn read_blob<S: crate::store::ObjectSource + ?Sized>(
34 store: &S,
35 hash: &Hash,
36) -> WorktreeResult<Vec<u8>> {
37 LoadedBlob::load(store, hash)?.into_content(store)
38}
39
40/// A blob's top-level object, read from the store exactly once and held so
41/// a caller needing several views of the same blob — byte length, bounded
42/// prefix, full content — pays a single top-level `read_object` instead of
43/// one per view. `diff --stat` is the motivating caller: its text-vs-binary
44/// sniff needs a prefix, then either the length (binary row) or the full
45/// content (text row), and taking each view through a separate store-level
46/// read re-read (and re-hash-verified) the same object two times per
47/// changed file, a per-entry cost that dominates a many-small-files
48/// diffstat (#624).
49///
50/// Chunk objects are still read on demand: [`Self::len`] reads none,
51/// [`Self::prefix`] reads only leading chunks, [`Self::into_content`]
52/// reads them all.
53///
54/// One read stays undeduped by design: [`Self::prefix`] on a chunked blob
55/// reads the leading chunk(s) to sniff, and [`Self::into_content`] reads
56/// every chunk (including that same leading one) to reassemble — a caller
57/// doing both, as `diff --stat` does, reads the first chunk twice. Caching
58/// it would need either interior mutability or a consuming-prefix API, to
59/// save an O(1) read on an O(n-chunks) path; not worth it (#624).
60#[derive(Debug)]
61pub enum LoadedBlob {
62 /// An inline [`Blob`](crate::object::Blob): its full content, already
63 /// in hand.
64 Inline(Vec<u8>),
65 /// A [`ChunkedBlob`] manifest: content lives in chunk objects, read
66 /// only when a view needs them.
67 Chunked(ChunkedBlob),
68}
69
70impl LoadedBlob {
71 /// Read the top-level object addressed by `hash` — one store read.
72 ///
73 /// # Errors
74 /// - [`WorktreeError::Store`] if `hash` is missing.
75 /// - [`WorktreeError::Io`] if `hash` resolves to an object that is
76 /// neither a `Blob` nor a `ChunkedBlob`.
77 pub fn load<S: crate::store::ObjectSource + ?Sized>(
78 store: &S,
79 hash: &Hash,
80 ) -> WorktreeResult<Self> {
81 match store.read_object(hash)? {
82 Object::Blob(b) => Ok(Self::Inline(b.data)),
83 Object::ChunkedBlob(manifest) => Ok(Self::Chunked(manifest)),
84 other => Err(not_a_blob("object", hash, &other)),
85 }
86 }
87
88 /// Content byte length from what is already in hand — an inline blob's
89 /// data length, a chunked blob's manifest `total_size` — no chunk
90 /// reads. `total_size` is trustworthy without re-verifying against the
91 /// chunks: every reassembly path enforces it via
92 /// [`ChunkedBlob::check_reassembled_size`], so a manifest with a wrong
93 /// `total_size` cannot have been durably written (#550).
94 #[must_use]
95 pub fn len(&self) -> u64 {
96 match self {
97 Self::Inline(data) => data.len() as u64,
98 Self::Chunked(manifest) => manifest.total_size,
99 }
100 }
101
102 /// Whether the content is zero bytes long.
103 #[must_use]
104 pub fn is_empty(&self) -> bool {
105 self.len() == 0
106 }
107
108 /// Up to `max_len` leading content bytes. Inline data is borrowed (no
109 /// copy, no reads); a chunked blob reads only as many leading chunks
110 /// as it takes to cover `max_len`, then stops — one chunk in practice,
111 /// since every chunk but possibly the last is at least
112 /// [`crate::chunker::MIN_SIZE`] bytes, well above the sniff windows
113 /// callers use (e.g. [`crate::ops::diff::BINARY_SNIFF_LEN`]).
114 ///
115 /// # Errors
116 /// - [`WorktreeError::Store`] if a needed chunk is missing.
117 /// - [`WorktreeError::Io`] if a needed chunk is not a `Blob`.
118 pub fn prefix<S: crate::store::ObjectSource + ?Sized>(
119 &self,
120 store: &S,
121 max_len: usize,
122 ) -> WorktreeResult<Cow<'_, [u8]>> {
123 match self {
124 Self::Inline(data) => Ok(Cow::Borrowed(&data[..data.len().min(max_len)])),
125 Self::Chunked(manifest) => {
126 let cap = usize::try_from(manifest.total_size)
127 .unwrap_or(max_len)
128 .min(max_len);
129 let mut data = Vec::with_capacity(cap);
130 for chunk in &manifest.chunks {
131 if data.len() >= max_len {
132 break;
133 }
134 data.extend_from_slice(&read_chunk(store, chunk)?);
135 }
136 data.truncate(max_len);
137 Ok(Cow::Owned(data))
138 }
139 }
140 }
141
142 /// The full content: inline data as-is (no further reads), a chunked
143 /// blob reassembled by reading every chunk, with the result length
144 /// enforced against the manifest's `total_size` (SPEC-OBJECTS §7).
145 ///
146 /// # Errors
147 /// - [`WorktreeError::Store`] if a chunk is missing.
148 /// - [`WorktreeError::Io`] if a chunk is not a `Blob`.
149 /// - [`WorktreeError::Object`] if the concatenated chunks do not total
150 /// the manifest's `total_size`.
151 pub fn into_content<S: crate::store::ObjectSource + ?Sized>(
152 self,
153 store: &S,
154 ) -> WorktreeResult<Vec<u8>> {
155 match self {
156 Self::Inline(data) => Ok(data),
157 Self::Chunked(manifest) => {
158 let mut data =
159 Vec::with_capacity(usize::try_from(manifest.total_size).unwrap_or(0));
160 for chunk in &manifest.chunks {
161 data.extend_from_slice(&read_chunk(store, chunk)?);
162 }
163 manifest.check_reassembled_size(data.len())?;
164 Ok(data)
165 }
166 }
167 }
168
169 /// The empty blob: what a diff side with no object (an add's old side,
170 /// a delete's new side) loads as. Zero length, empty prefix, empty
171 /// content — no store reads.
172 #[must_use]
173 pub fn empty() -> Self {
174 Self::Inline(Vec::new())
175 }
176}
177
178/// One chunk of a [`ChunkedBlob`]: must deserialize to an inline `Blob`.
179fn read_chunk<S: crate::store::ObjectSource + ?Sized>(
180 store: &S,
181 chunk: &Hash,
182) -> WorktreeResult<Vec<u8>> {
183 match store.read_object(chunk)? {
184 Object::Blob(b) => Ok(b.data),
185 other => Err(not_a_blob("chunk", chunk, &other)),
186 }
187}
188
189/// Read one manifest chunk's raw bytes, requiring a `Blob`, in
190/// [`crate::store::StoreError`]'s domain rather than [`WorktreeError`]'s
191/// (see [`read_chunk`] for the `WorktreeError` counterpart, kept separate
192/// since its error also carries the hash and the historical `"chunk ..."`
193/// wording). The single "read a manifest chunk" primitive shared by every
194/// `StoreError`-domain caller — [`content_fingerprint`],
195/// [`ContentCursor::remaining`], and [`chunked_content_eq`] — that
196/// previously each hand-rolled the same match.
197fn read_blob_chunk<S: crate::store::ObjectSource + ?Sized>(
198 store: &S,
199 chunk: &Hash,
200) -> Result<Vec<u8>, crate::store::StoreError> {
201 match store.read_object(chunk)? {
202 Object::Blob(b) => Ok(b.data),
203 _ => Err(crate::store::StoreError::Io(io::Error::other(
204 "manifest chunk is not a Blob",
205 ))),
206 }
207}
208
209/// The "expected a blob, found something else" error shared by every
210/// [`LoadedBlob`] read path; `what` is `"object"` for a top-level hash and
211/// `"chunk"` for a manifest chunk, preserving the historical wording of
212/// both messages.
213fn not_a_blob(what: &str, hash: &Hash, got: &Object) -> WorktreeError {
214 WorktreeError::Io(io::Error::other(format!(
215 "{what} {} is not a blob (got {})",
216 crate::hash::to_hex(hash),
217 got.object_type().name()
218 )))
219}
220
221/// Representation-independent content identity: byte length and raw BLAKE3.
222/// Reads each chunk once and verifies the complete manifest length. Memory is
223/// bounded by the manifest and one chunk, independent of reassembled size.
224///
225/// # Errors
226/// Missing, corrupt, wrong-type chunks and inconsistent lengths are errors.
227pub fn content_fingerprint<S: crate::store::ObjectSource + ?Sized>(
228 store: &S,
229 hash: &Hash,
230) -> Result<(u64, Hash), crate::store::StoreError> {
231 use crate::store::StoreError;
232 let mut hasher = crate::hash::Hasher::new();
233 match store.read_object(hash)? {
234 Object::Blob(b) => {
235 hasher.update(&b.data);
236 Ok((b.data.len() as u64, hasher.finalize()))
237 }
238 Object::ChunkedBlob(manifest) => {
239 let mut size = 0usize;
240 for chunk in &manifest.chunks {
241 let data = read_blob_chunk(store, chunk)?;
242 size = size
243 .checked_add(data.len())
244 .ok_or(StoreError::ObjectTooLarge)?;
245 hasher.update(&data);
246 }
247 manifest.check_reassembled_size(size)?;
248 Ok((size as u64, hasher.finalize()))
249 }
250 _ => Err(StoreError::Io(io::Error::other(
251 "content object is not a Blob or ChunkedBlob",
252 ))),
253 }
254}
255
256/// Compare file content independently of inline/chunked storage layout.
257/// Equal object IDs are a fast path; different IDs require verified content.
258///
259/// A `ChunkedBlob`-vs-`ChunkedBlob` pair (the large-file case) takes a
260/// further internal fast path (`chunked_content_eq`, private to this
261/// module), which skips reading any chunk both sides reference by the
262/// same hash instead of reassembling and byte-comparing every chunk of
263/// both blobs — see that function's doc for what it trusts and what it
264/// still fully verifies. Any other pairing
265/// (inline vs inline, or a mixed inline/chunked pair) keeps the exact
266/// byte-cursor walk this function has always used.
267///
268/// The chunked fast path's "same hash means same bytes, skip the read"
269/// trust holds only as far as `store: &S`'s reads are actually
270/// hash-verified — true of every `ObjectSource` in this crate today
271/// except [`crate::store::DisplaySource`] (used for read-only diff/show
272/// rendering, and not passed to `content_eq` by any current caller). A
273/// future caller doing so would trade that verification away for this
274/// fast path exactly as much as it already does for every other generic
275/// `ObjectSource` consumer.
276///
277/// # Errors
278/// Propagates errors from [`content_fingerprint`].
279pub fn content_eq<S: crate::store::ObjectSource + ?Sized>(
280 store: &S,
281 a: &Hash,
282 b: &Hash,
283) -> Result<bool, crate::store::StoreError> {
284 if a == b {
285 return Ok(true);
286 }
287 let obj_a = store.read_object(a)?;
288 let obj_b = store.read_object(b)?;
289 if let (Object::ChunkedBlob(ma), Object::ChunkedBlob(mb)) = (&obj_a, &obj_b) {
290 return chunked_content_eq(store, ma, mb);
291 }
292 let mut left = ContentCursor::from_object(obj_a)?;
293 let mut right = ContentCursor::from_object(obj_b)?;
294 let mut equal = true;
295 loop {
296 let a = left.remaining(store)?;
297 let b = right.remaining(store)?;
298 if a.is_empty() && b.is_empty() {
299 return Ok(equal);
300 }
301 if a.is_empty() || b.is_empty() {
302 equal = false;
303 let alen = a.len();
304 let blen = b.len();
305 left.offset += alen;
306 right.offset += blen;
307 } else {
308 let count = a.len().min(b.len());
309 equal &= a[..count] == b[..count];
310 left.offset += count;
311 right.offset += count;
312 }
313 }
314}
315
316/// Compare two [`ChunkedBlob`] manifests for content equality, skipping
317/// any chunk both sides reference by the same content-addressed hash:
318/// identical hash means identical bytes by construction — the same trust
319/// [`content_eq`]'s own `a == b` whole-object fast path already relies
320/// on, just applied per chunk instead of per object. Only chunks that
321/// diverge (a different hash at the same aligned position) are actually
322/// read and byte-compared, so a change confined to part of a large file
323/// costs only that part on both sides, not the whole file. An in-place
324/// edit of unchanged length walks straight past the shared, unaffected
325/// tail with no reads once the edited region resyncs by hash again.
326///
327/// `total_size` is trusted without re-summing actual chunk bytes only for
328/// a side whose chunks were *partly* skipped by id — the same trust
329/// [`LoadedBlob::len`] already documents: every reassembly path enforces
330/// it via [`ChunkedBlob::check_reassembled_size`], so a manifest with a
331/// wrong `total_size` cannot have been durably written through mkit's own
332/// writers (#550). A side every one of whose chunks gets actually read
333/// (no id ever matched a boundary on that side — the case an append or a
334/// wholly-rewritten file hits) has its real byte count checked against
335/// its declared `total_size` before returning, exactly like the old
336/// byte-cursor walk did, and errors the same way
337/// ([`crate::object::MkitError::ChunkedBlobSizeMismatch`]) if they
338/// disagree. Only a side that *did* skip at least one chunk by id skips
339/// this check for that side, since the skipped chunks' real lengths were
340/// never read to sum: two manifests whose declared `total_size` fields
341/// agree with each other but not with their own chunks, and whose chunk
342/// sequences also line up entirely (or partly, from the first divergence
343/// on) by hash, can still slip through unnoticed on that account. An
344/// append or truncation, meanwhile, needs zero chunk reads at all before
345/// even reaching this check: the two sides' declared sizes differ, so
346/// `content_eq` returns `Ok(false)` above before the merge walk starts.
347///
348/// # Errors
349/// [`crate::store::StoreError`] if a chunk that must be read (either
350/// side has no matching chunk to skip against) is missing, corrupt, or
351/// not a `Blob`; [`crate::object::MkitError::ChunkedBlobSizeMismatch`]
352/// if a side with no skipped chunks has a `total_size` that disagrees
353/// with its chunks' real byte sum.
354fn chunked_content_eq<S: crate::store::ObjectSource + ?Sized>(
355 store: &S,
356 ma: &ChunkedBlob,
357 mb: &ChunkedBlob,
358) -> Result<bool, crate::store::StoreError> {
359 if ma.total_size != mb.total_size {
360 return Ok(false);
361 }
362
363 let mut ia = 0usize;
364 let mut ib = 0usize;
365 let mut buf_a: Vec<u8> = Vec::new();
366 let mut buf_b: Vec<u8> = Vec::new();
367 let mut pos_a = 0usize;
368 let mut pos_b = 0usize;
369 let mut equal = true;
370 // Real bytes actually read (never chunks skipped by id) per side,
371 // and whether any chunk on that side WAS skipped by id — see the
372 // doc comment above for what these guard.
373 let mut read_a: u64 = 0;
374 let mut read_b: u64 = 0;
375 let mut skipped_a = false;
376 let mut skipped_b = false;
377
378 loop {
379 // Both cursors sit at a chunk boundary: skip a run of chunks
380 // that match by id, with no read on either side.
381 while pos_a == buf_a.len()
382 && pos_b == buf_b.len()
383 && ia < ma.chunks.len()
384 && ib < mb.chunks.len()
385 && ma.chunks[ia] == mb.chunks[ib]
386 {
387 ia += 1;
388 ib += 1;
389 skipped_a = true;
390 skipped_b = true;
391 }
392
393 if pos_a == buf_a.len() {
394 buf_a = match ma.chunks.get(ia) {
395 Some(h) => {
396 ia += 1;
397 let data = read_blob_chunk(store, h)?;
398 read_a = read_a
399 .checked_add(data.len() as u64)
400 .ok_or(crate::store::StoreError::ObjectTooLarge)?;
401 data
402 }
403 None => Vec::new(),
404 };
405 pos_a = 0;
406 }
407 if pos_b == buf_b.len() {
408 buf_b = match mb.chunks.get(ib) {
409 Some(h) => {
410 ib += 1;
411 let data = read_blob_chunk(store, h)?;
412 read_b = read_b
413 .checked_add(data.len() as u64)
414 .ok_or(crate::store::StoreError::ObjectTooLarge)?;
415 data
416 }
417 None => Vec::new(),
418 };
419 pos_b = 0;
420 }
421
422 let a_rest = &buf_a[pos_a..];
423 let b_rest = &buf_b[pos_b..];
424 if a_rest.is_empty() && b_rest.is_empty() {
425 if ia >= ma.chunks.len() && ib >= mb.chunks.len() {
426 if !skipped_a && read_a != ma.total_size {
427 return Err(crate::object::MkitError::ChunkedBlobSizeMismatch {
428 expected: ma.total_size,
429 actual: read_a,
430 }
431 .into());
432 }
433 if !skipped_b && read_b != mb.total_size {
434 return Err(crate::object::MkitError::ChunkedBlobSizeMismatch {
435 expected: mb.total_size,
436 actual: read_b,
437 }
438 .into());
439 }
440 return Ok(equal);
441 }
442 // A zero-length trailing chunk on one or both sides; loop
443 // again to pull the next one via the boundary checks above.
444 continue;
445 }
446
447 let count = a_rest.len().min(b_rest.len());
448 if count == 0 {
449 // One side has run out of chunks for good; drain the other
450 // (matching the byte-cursor path's behavior) so a read
451 // error later in its remaining chunks still surfaces.
452 equal = false;
453 pos_a = buf_a.len();
454 pos_b = buf_b.len();
455 continue;
456 }
457 if a_rest[..count] != b_rest[..count] {
458 equal = false;
459 }
460 pos_a += count;
461 pos_b += count;
462 }
463}
464
465/// Compare stored content to bytes without reassembling chunked storage.
466///
467/// # Errors
468/// Propagates errors from [`content_fingerprint`].
469pub fn content_eq_bytes<S: crate::store::ObjectSource + ?Sized>(
470 store: &S,
471 object: &Hash,
472 bytes: &[u8],
473) -> Result<bool, crate::store::StoreError> {
474 let mut cursor = ContentCursor::load(store, object)?;
475 let mut offset = 0usize;
476 let mut equal = true;
477 loop {
478 let chunk = cursor.remaining(store)?;
479 if chunk.is_empty() {
480 return Ok(equal && offset == bytes.len());
481 }
482 let end = offset
483 .checked_add(chunk.len())
484 .ok_or(crate::store::StoreError::ObjectTooLarge)?;
485 equal &= bytes.get(offset..end) == Some(chunk);
486 offset = end;
487 cursor.offset += chunk.len();
488 }
489}
490
491struct ContentCursor {
492 data: Vec<u8>,
493 offset: usize,
494 chunks: std::vec::IntoIter<Hash>,
495 expected: u64,
496 loaded: u64,
497}
498
499impl ContentCursor {
500 fn load<S: crate::store::ObjectSource + ?Sized>(
501 store: &S,
502 hash: &Hash,
503 ) -> Result<Self, crate::store::StoreError> {
504 Self::from_object(store.read_object(hash)?)
505 }
506
507 /// Build a cursor from an already-read top-level object, saving the
508 /// caller a second `read_object` when it needed the object anyway
509 /// (e.g. [`content_eq`] deciding whether the chunked fast path
510 /// applies).
511 fn from_object(object: Object) -> Result<Self, crate::store::StoreError> {
512 let (data, chunks, expected) = match object {
513 Object::Blob(b) => {
514 let size = b.data.len() as u64;
515 (b.data, Vec::new(), size)
516 }
517 Object::ChunkedBlob(m) => (Vec::new(), m.chunks, m.total_size),
518 _ => {
519 return Err(crate::store::StoreError::Io(io::Error::other(
520 "content object is not a Blob or ChunkedBlob",
521 )));
522 }
523 };
524 let loaded = data.len() as u64;
525 Ok(Self {
526 data,
527 offset: 0,
528 chunks: chunks.into_iter(),
529 expected,
530 loaded,
531 })
532 }
533
534 fn remaining<S: crate::store::ObjectSource + ?Sized>(
535 &mut self,
536 store: &S,
537 ) -> Result<&[u8], crate::store::StoreError> {
538 while self.offset == self.data.len() {
539 let Some(hash) = self.chunks.next() else {
540 if self.loaded != self.expected {
541 return Err(crate::object::MkitError::ChunkedBlobSizeMismatch {
542 expected: self.expected,
543 actual: self.loaded,
544 }
545 .into());
546 }
547 return Ok(&[]);
548 };
549 let data = read_blob_chunk(store, &hash)?;
550 self.loaded = self
551 .loaded
552 .checked_add(data.len() as u64)
553 .ok_or(crate::store::StoreError::ObjectTooLarge)?;
554 self.data = data;
555 self.offset = 0;
556 }
557 Ok(&self.data[self.offset..])
558 }
559}
560
561#[cfg(test)]
562mod equality_tests {
563 use super::*;
564 use crate::{layout::RepoLayout, object::Blob, serialize, store::ObjectStore};
565
566 fn put(store: &ObjectStore, object: &Object) -> Hash {
567 store.write(&serialize::serialize(object).unwrap()).unwrap()
568 }
569
570 #[test]
571 fn large_inline_fixed_and_cdc_content_agree() {
572 let dir = tempfile::tempdir().unwrap();
573 let store = ObjectStore::init(&RepoLayout::single(dir.path())).unwrap();
574 let data = vec![7; usize::try_from(super::super::CHUNK_THRESHOLD + 17).unwrap()];
575 let inline = put(&store, &Object::Blob(Blob { data: data.clone() }));
576 let cdc = super::super::store_file_object(&store, &data).unwrap();
577 let chunks = data
578 .chunks(65_536)
579 .map(|b| put(&store, &Object::Blob(Blob { data: b.to_vec() })))
580 .collect();
581 let fixed = put(
582 &store,
583 &Object::ChunkedBlob(ChunkedBlob {
584 total_size: data.len() as u64,
585 chunk_size: 65_536,
586 chunks,
587 }),
588 );
589 assert_ne!(inline, cdc);
590 assert_ne!(fixed, cdc);
591 assert!(content_eq(&store, &inline, &fixed).unwrap());
592 assert!(content_eq(&store, &fixed, &cdc).unwrap());
593 assert!(content_eq_bytes(&store, &fixed, &data).unwrap());
594 assert_eq!(
595 content_fingerprint(&store, &inline).unwrap(),
596 content_fingerprint(&store, &cdc).unwrap()
597 );
598 let mut changed = data;
599 changed[65_536] = 8;
600 assert!(!content_eq_bytes(&store, &fixed, &changed).unwrap());
601 }
602
603 #[test]
604 fn invalid_chunks_are_errors_even_after_content_differs() {
605 let dir = tempfile::tempdir().unwrap();
606 let store = ObjectStore::init(&RepoLayout::single(dir.path())).unwrap();
607 let a = put(
608 &store,
609 &Object::Blob(Blob {
610 data: b"a".to_vec(),
611 }),
612 );
613 let b = put(
614 &store,
615 &Object::Blob(Blob {
616 data: b"b".to_vec(),
617 }),
618 );
619 for (total_size, chunks) in [(2, vec![a]), (2, vec![a, [42; 32]])] {
620 let bad = put(
621 &store,
622 &Object::ChunkedBlob(ChunkedBlob {
623 total_size,
624 chunk_size: 0,
625 chunks,
626 }),
627 );
628 assert!(content_eq(&store, &bad, &b).is_err());
629 assert!(content_eq_bytes(&store, &bad, b"b").is_err());
630 assert!(content_fingerprint(&store, &bad).is_err());
631 }
632 }
633
634 fn chunk(store: &ObjectStore, data: &[u8]) -> Hash {
635 put(
636 store,
637 &Object::Blob(Blob {
638 data: data.to_vec(),
639 }),
640 )
641 }
642
643 fn manifest(store: &ObjectStore, parts: &[&[u8]]) -> ChunkedBlob {
644 let total_size: u64 = parts.iter().map(|p| p.len() as u64).sum();
645 ChunkedBlob {
646 total_size,
647 chunk_size: 0,
648 chunks: parts.iter().map(|p| chunk(store, p)).collect(),
649 }
650 }
651
652 fn fresh_store() -> (tempfile::TempDir, ObjectStore) {
653 let dir = tempfile::tempdir().unwrap();
654 let store = ObjectStore::init(&RepoLayout::single(dir.path())).unwrap();
655 (dir, store)
656 }
657
658 // Direct tests of `chunked_content_eq`'s merge logic against small,
659 // hand-built manifests — fast and deterministic, independent of real
660 // FastCDC boundaries, covering the id-skip fast path, the
661 // read-and-resync fallback, and the documented total_size trust gap.
662
663 #[test]
664 fn chunked_fast_path_all_chunks_match_by_id() {
665 let (_dir, store) = fresh_store();
666 let x = chunk(&store, b"hello ");
667 let y = chunk(&store, b"world");
668 let ma = ChunkedBlob {
669 total_size: 11,
670 chunk_size: 0,
671 chunks: vec![x, y],
672 };
673 let mb = ma.clone();
674 assert!(super::chunked_content_eq(&store, &ma, &mb).unwrap());
675 }
676
677 #[test]
678 fn chunked_fully_misaligned_but_equal_content() {
679 let (_dir, store) = fresh_store();
680 // "ab"+"cd" vs "a"+"bcd" — no chunk hash ever matches, so every
681 // byte is read and compared through the resync fallback, yet the
682 // reassembled content is identical.
683 let ma = manifest(&store, &[b"ab", b"cd"]);
684 let mb = manifest(&store, &[b"a", b"bcd"]);
685 assert!(super::chunked_content_eq(&store, &ma, &mb).unwrap());
686 assert!(super::chunked_content_eq(&store, &mb, &ma).unwrap());
687 }
688
689 #[test]
690 fn chunked_shared_prefix_then_misaligned_equal_suffix() {
691 let (_dir, store) = fresh_store();
692 let x = chunk(&store, b"shared-prefix-");
693 let ma = ChunkedBlob {
694 total_size: 14 + 4,
695 chunk_size: 0,
696 chunks: [x]
697 .into_iter()
698 .chain(manifest(&store, &[b"ab", b"cd"]).chunks)
699 .collect(),
700 };
701 let mb = ChunkedBlob {
702 total_size: 14 + 4,
703 chunk_size: 0,
704 chunks: [x]
705 .into_iter()
706 .chain(manifest(&store, &[b"a", b"bcd"]).chunks)
707 .collect(),
708 };
709 // `x` is skipped by id; "ab"+"cd" vs "a"+"bcd" is read and resynced.
710 assert!(super::chunked_content_eq(&store, &ma, &mb).unwrap());
711 }
712
713 #[test]
714 fn chunked_append_differs_via_total_size_without_reading() {
715 let (_dir, store) = fresh_store();
716 let x = chunk(&store, b"shared");
717 let ma = ChunkedBlob {
718 total_size: 6,
719 chunk_size: 0,
720 chunks: vec![x],
721 };
722 let y = chunk(&store, b"-more");
723 let mb = ChunkedBlob {
724 total_size: 11,
725 chunk_size: 0,
726 chunks: vec![x, y],
727 };
728 assert!(!super::chunked_content_eq(&store, &ma, &mb).unwrap());
729 assert!(!super::chunked_content_eq(&store, &mb, &ma).unwrap());
730 }
731
732 #[test]
733 fn chunked_same_total_size_different_content_after_shared_prefix() {
734 let (_dir, store) = fresh_store();
735 let x = chunk(&store, b"shared-");
736 let ma = ChunkedBlob {
737 total_size: 7 + 2,
738 chunk_size: 0,
739 chunks: vec![x, chunk(&store, b"ab")],
740 };
741 let mb = ChunkedBlob {
742 total_size: 7 + 2,
743 chunk_size: 0,
744 chunks: vec![x, chunk(&store, b"ba")],
745 };
746 assert!(!super::chunked_content_eq(&store, &ma, &mb).unwrap());
747 }
748
749 #[test]
750 fn chunked_split_differently_but_actually_different_content() {
751 let (_dir, store) = fresh_store();
752 let x = chunk(&store, b"shared-");
753 let ma = ChunkedBlob {
754 total_size: 7 + 4,
755 chunk_size: 0,
756 chunks: vec![x, chunk(&store, b"abcd")],
757 };
758 let mb = ChunkedBlob {
759 total_size: 7 + 4,
760 chunk_size: 0,
761 chunks: vec![x, chunk(&store, b"ab"), chunk(&store, b"cX")],
762 };
763 assert!(!super::chunked_content_eq(&store, &ma, &mb).unwrap());
764 }
765
766 #[test]
767 fn chunked_empty_manifests_are_equal() {
768 let (_dir, store) = fresh_store();
769 let ma = ChunkedBlob {
770 total_size: 0,
771 chunk_size: 0,
772 chunks: vec![],
773 };
774 let mb = ma.clone();
775 assert!(super::chunked_content_eq(&store, &ma, &mb).unwrap());
776 }
777
778 /// Documents the accepted trust gap: `total_size` is not re-summed
779 /// from actual chunk bytes on the chunked fast path (mirroring
780 /// `LoadedBlob::len`'s existing trust). Two manifests whose chunk
781 /// sequence fully matches by id, but whose shared `total_size`
782 /// field is wrong for that sequence, are reported equal without
783 /// ever reading a chunk to notice. A manifest this malformed cannot
784 /// be produced by any of mkit's own writers (`check_reassembled_size`
785 /// gates every one); this can only arise from an object constructed
786 /// directly, bypassing them, as this test does.
787 #[test]
788 fn chunked_wrong_shared_total_size_is_not_detected_when_ids_fully_match() {
789 let (_dir, store) = fresh_store();
790 let x = chunk(&store, b"ab"); // 2 real bytes
791 let ma = ChunkedBlob {
792 total_size: 999, // wrong on both sides, identically
793 chunk_size: 0,
794 chunks: vec![x],
795 };
796 let mb = ma.clone();
797 assert!(super::chunked_content_eq(&store, &ma, &mb).unwrap());
798 }
799
800 /// The case the fast path must *not* let slip past: no chunk on
801 /// either side is ever skipped by id (every hash differs), so both
802 /// sides get fully read — and a wrong `total_size` on a side that
803 /// was fully read must still surface as
804 /// `ChunkedBlobSizeMismatch`, exactly like the old byte-cursor walk.
805 /// Regression test for a gap an independent review found: an
806 /// earlier version of `chunked_content_eq` only ever compared the
807 /// two manifests' declared `total_size` fields against *each
808 /// other*, never against either side's own real chunk bytes, so two
809 /// fully-diverging (no id ever matches) manifests that happened to
810 /// declare the same wrong `total_size` were reported merely
811 /// "unequal" instead of erroring.
812 #[test]
813 fn chunked_wrong_total_size_is_detected_when_no_chunk_is_skipped() {
814 let (_dir, store) = fresh_store();
815 // Real chunk bytes: 90 bytes on each side ("a" x90 vs "b" x90),
816 // so no chunk hash ever matches — nothing is ever skipped.
817 let a90 = vec![b'a'; 90];
818 let b90 = vec![b'b'; 90];
819 let ma = ChunkedBlob {
820 total_size: 100, // wrong: real sum is 90
821 chunk_size: 0,
822 chunks: vec![chunk(&store, &a90)],
823 };
824 let mb = ChunkedBlob {
825 total_size: 100, // also wrong, and equal to ma's
826 chunk_size: 0,
827 chunks: vec![chunk(&store, &b90)],
828 };
829 let err = super::chunked_content_eq(&store, &ma, &mb).unwrap_err();
830 assert!(
831 matches!(
832 err,
833 crate::store::StoreError::Decode(
834 crate::object::MkitError::ChunkedBlobSizeMismatch { .. }
835 )
836 ),
837 "expected ChunkedBlobSizeMismatch, got {err:?}"
838 );
839 }
840
841 /// The same wrong-`total_size` gap, but on only one side: the other
842 /// side's declared size is accurate for its own real chunk bytes.
843 /// Content is unequal either way (lengths differ once both are read
844 /// out to their real end), but the malformed side must still error
845 /// rather than silently compare as "not equal".
846 #[test]
847 fn chunked_wrong_total_size_on_one_side_only_is_detected() {
848 let (_dir, store) = fresh_store();
849 let a80 = vec![b'a'; 80];
850 let b100 = vec![b'b'; 100];
851 let ma = ChunkedBlob {
852 total_size: 100, // wrong: real sum is 80
853 chunk_size: 0,
854 chunks: vec![chunk(&store, &a80)],
855 };
856 let mb = ChunkedBlob {
857 total_size: 100, // correct
858 chunk_size: 0,
859 chunks: vec![chunk(&store, &b100)],
860 };
861 let err = super::chunked_content_eq(&store, &ma, &mb).unwrap_err();
862 assert!(
863 matches!(
864 err,
865 crate::store::StoreError::Decode(
866 crate::object::MkitError::ChunkedBlobSizeMismatch { .. }
867 )
868 ),
869 "expected ChunkedBlobSizeMismatch, got {err:?}"
870 );
871 }
872
873 /// End-to-end sanity check through the public `content_eq` entry
874 /// point (not the private merge function directly) over real
875 /// `FastCDC`-chunked content: an append, a single-byte in-place edit,
876 /// and a truncation each produce a different manifest and must all
877 /// compare unequal, exactly matching a ground-truth byte comparison.
878 #[test]
879 fn content_eq_real_chunked_mutations_match_ground_truth() {
880 let (_dir, store) = fresh_store();
881 let threshold = usize::try_from(super::super::CHUNK_THRESHOLD).unwrap();
882 // Deterministic pseudo-random bytes so FastCDC sees real cut
883 // points instead of one run-length-maxed chunk.
884 let mut data = Vec::with_capacity(threshold * 3);
885 let mut state: u64 = 0x1234_5678_9abc_def0;
886 for _ in 0..threshold * 3 {
887 state = state
888 .wrapping_mul(6_364_136_223_846_793_005)
889 .wrapping_add(1);
890 data.push((state >> 56) as u8);
891 }
892
893 let base = super::super::store_file_object(&store, &data).unwrap();
894
895 let mut appended = data.clone();
896 appended.extend_from_slice(b"appended tail bytes");
897 let appended_hash = super::super::store_file_object(&store, &appended).unwrap();
898
899 let mut edited = data.clone();
900 let mid = edited.len() / 2;
901 edited[mid] ^= 0xFF;
902 let edited_hash = super::super::store_file_object(&store, &edited).unwrap();
903
904 let truncated = &data[..data.len() - 500];
905 let truncated_hash = super::super::store_file_object(&store, truncated).unwrap();
906
907 let unchanged_hash = super::super::store_file_object(&store, &data).unwrap();
908
909 assert_eq!(base, unchanged_hash, "identical content dedups to one id");
910 assert!(content_eq(&store, &base, &unchanged_hash).unwrap());
911
912 for (name, other, other_bytes) in [
913 ("append", appended_hash, appended.as_slice()),
914 ("edit", edited_hash, edited.as_slice()),
915 ("truncate", truncated_hash, truncated),
916 ] {
917 assert_ne!(base, other, "{name}: expected a different object id");
918 assert_eq!(
919 content_eq(&store, &base, &other).unwrap(),
920 data == other_bytes,
921 "{name}: content_eq must match ground truth"
922 );
923 assert!(
924 !content_eq(&store, &base, &other).unwrap(),
925 "{name}: bytes differ"
926 );
927 }
928 }
929
930 /// A mid-file insertion against real `FastCDC` output: unlike
931 /// `content_eq_real_chunked_mutations_match_ground_truth`'s
932 /// single-byte edit (which may or may not shift a chunk boundary),
933 /// inserting new bytes shifts every downstream offset, which
934 /// reliably forces `FastCDC` to re-cut several chunks around the
935 /// insertion point before content-defined chunking resyncs on the
936 /// unchanged bytes further on. This is the scenario
937 /// `chunked_content_eq`'s doc comment describes ("a change confined
938 /// to part of a large file... resyncs by hash again") and, per an
939 /// independent review, the only other resync coverage exercised
940 /// hand-built or fixed-vs-CDC manifests, never two independently
941 /// `FastCDC`-chunked real files. Confirms at the manifest level
942 /// that a real divergence-then-resync actually occurred (shared
943 /// first and last chunk hashes, a different chunk in between) before
944 /// checking `content_eq` against ground truth.
945 #[test]
946 fn content_eq_real_chunked_insertion_forces_boundary_resync() {
947 let (_dir, store) = fresh_store();
948 let threshold = usize::try_from(super::super::CHUNK_THRESHOLD).unwrap();
949 let mut data = Vec::with_capacity(threshold * 4);
950 let mut state: u64 = 0x0BAD_C0DE_F00D_CAFE;
951 for _ in 0..threshold * 4 {
952 state = state
953 .wrapping_mul(6_364_136_223_846_793_005)
954 .wrapping_add(1);
955 data.push((state >> 56) as u8);
956 }
957
958 let mut inserted = data.clone();
959 let at = data.len() / 2;
960 let mut new_bytes = vec![0u8; 4096];
961 let mut s: u64 = 0xFACE_FEED_1234_5678;
962 for b in &mut new_bytes {
963 s = s.wrapping_mul(6_364_136_223_846_793_005).wrapping_add(1);
964 *b = (s >> 56) as u8;
965 }
966 inserted.splice(at..at, new_bytes.iter().copied());
967
968 let base_hash = super::super::store_file_object(&store, &data).unwrap();
969 let inserted_hash = super::super::store_file_object(&store, &inserted).unwrap();
970 assert_ne!(base_hash, inserted_hash);
971
972 let Object::ChunkedBlob(base_manifest) = store.read_object(&base_hash).unwrap() else {
973 panic!("expected base to be chunked (data.len() > CHUNK_THRESHOLD)");
974 };
975 let Object::ChunkedBlob(inserted_manifest) = store.read_object(&inserted_hash).unwrap()
976 else {
977 panic!("expected inserted to be chunked");
978 };
979 assert!(
980 base_manifest.chunks.len() > 2 && inserted_manifest.chunks.len() > 2,
981 "fixture too small to exercise multiple chunks"
982 );
983 assert_eq!(
984 base_manifest.chunks.first(),
985 inserted_manifest.chunks.first(),
986 "the unaffected prefix must still share its leading chunk by id"
987 );
988 assert_eq!(
989 base_manifest.chunks.last(),
990 inserted_manifest.chunks.last(),
991 "content-defined chunking must resync on the unaffected suffix"
992 );
993 assert_ne!(
994 base_manifest.chunks, inserted_manifest.chunks,
995 "the insertion must actually shift chunk boundaries somewhere in the middle"
996 );
997
998 assert!(
999 !content_eq(&store, &base_hash, &inserted_hash).unwrap(),
1000 "content genuinely differs after the insertion"
1001 );
1002 }
1003}