mkit-core 0.5.0

Content-addressed VCS primitives for mkit: BLAKE3 hashing, canonical objects, refs, packs, and transport traits
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
//! Blob-content I/O: reading back what the worktree walker stored.
//!
//! A file's content lives either in a single inline [`Blob`](crate::object::Blob)
//! or behind a [`ChunkedBlob`](crate::object::ChunkedBlob) manifest. [`LoadedBlob`]
//! is the one-read-per-object view over both shapes; [`read_blob`] is the one-shot
//! full-content convenience on top of it. Extracted from `worktree.rs` (#633) —
//! this cluster reconstructs content and has nothing to do with walking a worktree.

use std::borrow::Cow;
use std::io;

use crate::hash::Hash;
use crate::object::{ChunkedBlob, Object};

use super::{WorktreeError, WorktreeResult};

/// Reassemble the full byte content of a `Blob` or `ChunkedBlob` object
/// addressed by `hash`.
///
/// A plain [`Blob`](crate::object::Blob) returns its bytes directly. A
/// [`ChunkedBlob`] manifest is reassembled
/// by concatenating each referenced chunk (every chunk must itself be a
/// `Blob`). This is the shared counterpart to [`store_file_object`](super::store_file_object) and
/// backs `mkit cat`, `mkit diff`, conflict rendering, and blame so they
/// all reconstruct large-file content the same way.
///
/// # Errors
/// - [`WorktreeError::Store`] if `hash` or any chunk is missing.
/// - [`WorktreeError::Io`] if `hash` (or a chunk) resolves to an object
///   that is neither a `Blob` nor a `ChunkedBlob` of `Blob`s.
/// - [`WorktreeError::Object`] if the concatenated chunks do not total
///   the manifest's `total_size` (SPEC-OBJECTS §7).
pub fn read_blob<S: crate::store::ObjectSource + ?Sized>(
    store: &S,
    hash: &Hash,
) -> WorktreeResult<Vec<u8>> {
    LoadedBlob::load(store, hash)?.into_content(store)
}

/// A blob's top-level object, read from the store exactly once and held so
/// a caller needing several views of the same blob — byte length, bounded
/// prefix, full content — pays a single top-level `read_object` instead of
/// one per view. `diff --stat` is the motivating caller: its text-vs-binary
/// sniff needs a prefix, then either the length (binary row) or the full
/// content (text row), and taking each view through a separate store-level
/// read re-read (and re-hash-verified) the same object two times per
/// changed file, a per-entry cost that dominates a many-small-files
/// diffstat (#624).
///
/// Chunk objects are still read on demand: [`Self::len`] reads none,
/// [`Self::prefix`] reads only leading chunks, [`Self::into_content`]
/// reads them all.
///
/// One read stays undeduped by design: [`Self::prefix`] on a chunked blob
/// reads the leading chunk(s) to sniff, and [`Self::into_content`] reads
/// every chunk (including that same leading one) to reassemble — a caller
/// doing both, as `diff --stat` does, reads the first chunk twice. Caching
/// it would need either interior mutability or a consuming-prefix API, to
/// save an O(1) read on an O(n-chunks) path; not worth it (#624).
#[derive(Debug)]
pub enum LoadedBlob {
    /// An inline [`Blob`](crate::object::Blob): its full content, already
    /// in hand.
    Inline(Vec<u8>),
    /// A [`ChunkedBlob`] manifest: content lives in chunk objects, read
    /// only when a view needs them.
    Chunked(ChunkedBlob),
}

impl LoadedBlob {
    /// Read the top-level object addressed by `hash` — one store read.
    ///
    /// # Errors
    /// - [`WorktreeError::Store`] if `hash` is missing.
    /// - [`WorktreeError::Io`] if `hash` resolves to an object that is
    ///   neither a `Blob` nor a `ChunkedBlob`.
    pub fn load<S: crate::store::ObjectSource + ?Sized>(
        store: &S,
        hash: &Hash,
    ) -> WorktreeResult<Self> {
        match store.read_object(hash)? {
            Object::Blob(b) => Ok(Self::Inline(b.data)),
            Object::ChunkedBlob(manifest) => Ok(Self::Chunked(manifest)),
            other => Err(not_a_blob("object", hash, &other)),
        }
    }

    /// Content byte length from what is already in hand — an inline blob's
    /// data length, a chunked blob's manifest `total_size` — no chunk
    /// reads. `total_size` is trustworthy without re-verifying against the
    /// chunks: every reassembly path enforces it via
    /// [`ChunkedBlob::check_reassembled_size`], so a manifest with a wrong
    /// `total_size` cannot have been durably written (#550).
    #[must_use]
    pub fn len(&self) -> u64 {
        match self {
            Self::Inline(data) => data.len() as u64,
            Self::Chunked(manifest) => manifest.total_size,
        }
    }

    /// Whether the content is zero bytes long.
    #[must_use]
    pub fn is_empty(&self) -> bool {
        self.len() == 0
    }

    /// Up to `max_len` leading content bytes. Inline data is borrowed (no
    /// copy, no reads); a chunked blob reads only as many leading chunks
    /// as it takes to cover `max_len`, then stops — one chunk in practice,
    /// since every chunk but possibly the last is at least
    /// [`crate::chunker::MIN_SIZE`] bytes, well above the sniff windows
    /// callers use (e.g. [`crate::ops::diff::BINARY_SNIFF_LEN`]).
    ///
    /// # Errors
    /// - [`WorktreeError::Store`] if a needed chunk is missing.
    /// - [`WorktreeError::Io`] if a needed chunk is not a `Blob`.
    pub fn prefix<S: crate::store::ObjectSource + ?Sized>(
        &self,
        store: &S,
        max_len: usize,
    ) -> WorktreeResult<Cow<'_, [u8]>> {
        match self {
            Self::Inline(data) => Ok(Cow::Borrowed(&data[..data.len().min(max_len)])),
            Self::Chunked(manifest) => {
                let cap = usize::try_from(manifest.total_size)
                    .unwrap_or(max_len)
                    .min(max_len);
                let mut data = Vec::with_capacity(cap);
                for chunk in &manifest.chunks {
                    if data.len() >= max_len {
                        break;
                    }
                    data.extend_from_slice(&read_chunk(store, chunk)?);
                }
                data.truncate(max_len);
                Ok(Cow::Owned(data))
            }
        }
    }

    /// The full content: inline data as-is (no further reads), a chunked
    /// blob reassembled by reading every chunk, with the result length
    /// enforced against the manifest's `total_size` (SPEC-OBJECTS §7).
    ///
    /// # Errors
    /// - [`WorktreeError::Store`] if a chunk is missing.
    /// - [`WorktreeError::Io`] if a chunk is not a `Blob`.
    /// - [`WorktreeError::Object`] if the concatenated chunks do not total
    ///   the manifest's `total_size`.
    pub fn into_content<S: crate::store::ObjectSource + ?Sized>(
        self,
        store: &S,
    ) -> WorktreeResult<Vec<u8>> {
        match self {
            Self::Inline(data) => Ok(data),
            Self::Chunked(manifest) => {
                let mut data =
                    Vec::with_capacity(usize::try_from(manifest.total_size).unwrap_or(0));
                for chunk in &manifest.chunks {
                    data.extend_from_slice(&read_chunk(store, chunk)?);
                }
                manifest.check_reassembled_size(data.len())?;
                Ok(data)
            }
        }
    }

    /// The empty blob: what a diff side with no object (an add's old side,
    /// a delete's new side) loads as. Zero length, empty prefix, empty
    /// content — no store reads.
    #[must_use]
    pub fn empty() -> Self {
        Self::Inline(Vec::new())
    }
}

/// One chunk of a [`ChunkedBlob`]: must deserialize to an inline `Blob`.
fn read_chunk<S: crate::store::ObjectSource + ?Sized>(
    store: &S,
    chunk: &Hash,
) -> WorktreeResult<Vec<u8>> {
    match store.read_object(chunk)? {
        Object::Blob(b) => Ok(b.data),
        other => Err(not_a_blob("chunk", chunk, &other)),
    }
}

/// Read one manifest chunk's raw bytes, requiring a `Blob`, in
/// [`crate::store::StoreError`]'s domain rather than [`WorktreeError`]'s
/// (see [`read_chunk`] for the `WorktreeError` counterpart, kept separate
/// since its error also carries the hash and the historical `"chunk ..."`
/// wording). The single "read a manifest chunk" primitive shared by every
/// `StoreError`-domain caller — [`content_fingerprint`],
/// [`ContentCursor::remaining`], and [`chunked_content_eq`] — that
/// previously each hand-rolled the same match.
fn read_blob_chunk<S: crate::store::ObjectSource + ?Sized>(
    store: &S,
    chunk: &Hash,
) -> Result<Vec<u8>, crate::store::StoreError> {
    match store.read_object(chunk)? {
        Object::Blob(b) => Ok(b.data),
        _ => Err(crate::store::StoreError::Io(io::Error::other(
            "manifest chunk is not a Blob",
        ))),
    }
}

/// The "expected a blob, found something else" error shared by every
/// [`LoadedBlob`] read path; `what` is `"object"` for a top-level hash and
/// `"chunk"` for a manifest chunk, preserving the historical wording of
/// both messages.
fn not_a_blob(what: &str, hash: &Hash, got: &Object) -> WorktreeError {
    WorktreeError::Io(io::Error::other(format!(
        "{what} {} is not a blob (got {})",
        crate::hash::to_hex(hash),
        got.object_type().name()
    )))
}

/// Representation-independent content identity: byte length and raw BLAKE3.
/// Reads each chunk once and verifies the complete manifest length. Memory is
/// bounded by the manifest and one chunk, independent of reassembled size.
///
/// # Errors
/// Missing, corrupt, wrong-type chunks and inconsistent lengths are errors.
pub fn content_fingerprint<S: crate::store::ObjectSource + ?Sized>(
    store: &S,
    hash: &Hash,
) -> Result<(u64, Hash), crate::store::StoreError> {
    use crate::store::StoreError;
    let mut hasher = crate::hash::Hasher::new();
    match store.read_object(hash)? {
        Object::Blob(b) => {
            hasher.update(&b.data);
            Ok((b.data.len() as u64, hasher.finalize()))
        }
        Object::ChunkedBlob(manifest) => {
            let mut size = 0usize;
            for chunk in &manifest.chunks {
                let data = read_blob_chunk(store, chunk)?;
                size = size
                    .checked_add(data.len())
                    .ok_or(StoreError::ObjectTooLarge)?;
                hasher.update(&data);
            }
            manifest.check_reassembled_size(size)?;
            Ok((size as u64, hasher.finalize()))
        }
        _ => Err(StoreError::Io(io::Error::other(
            "content object is not a Blob or ChunkedBlob",
        ))),
    }
}

/// Compare file content independently of inline/chunked storage layout.
/// Equal object IDs are a fast path; different IDs require verified content.
///
/// A `ChunkedBlob`-vs-`ChunkedBlob` pair (the large-file case) takes a
/// further internal fast path (`chunked_content_eq`, private to this
/// module), which skips reading any chunk both sides reference by the
/// same hash instead of reassembling and byte-comparing every chunk of
/// both blobs — see that function's doc for what it trusts and what it
/// still fully verifies. Any other pairing
/// (inline vs inline, or a mixed inline/chunked pair) keeps the exact
/// byte-cursor walk this function has always used.
///
/// The chunked fast path's "same hash means same bytes, skip the read"
/// trust holds only as far as `store: &S`'s reads are actually
/// hash-verified — true of every `ObjectSource` in this crate today
/// except [`crate::store::DisplaySource`] (used for read-only diff/show
/// rendering, and not passed to `content_eq` by any current caller). A
/// future caller doing so would trade that verification away for this
/// fast path exactly as much as it already does for every other generic
/// `ObjectSource` consumer.
///
/// # Errors
/// Propagates errors from [`content_fingerprint`].
pub fn content_eq<S: crate::store::ObjectSource + ?Sized>(
    store: &S,
    a: &Hash,
    b: &Hash,
) -> Result<bool, crate::store::StoreError> {
    if a == b {
        return Ok(true);
    }
    let obj_a = store.read_object(a)?;
    let obj_b = store.read_object(b)?;
    if let (Object::ChunkedBlob(ma), Object::ChunkedBlob(mb)) = (&obj_a, &obj_b) {
        return chunked_content_eq(store, ma, mb);
    }
    let mut left = ContentCursor::from_object(obj_a)?;
    let mut right = ContentCursor::from_object(obj_b)?;
    let mut equal = true;
    loop {
        let a = left.remaining(store)?;
        let b = right.remaining(store)?;
        if a.is_empty() && b.is_empty() {
            return Ok(equal);
        }
        if a.is_empty() || b.is_empty() {
            equal = false;
            let alen = a.len();
            let blen = b.len();
            left.offset += alen;
            right.offset += blen;
        } else {
            let count = a.len().min(b.len());
            equal &= a[..count] == b[..count];
            left.offset += count;
            right.offset += count;
        }
    }
}

/// Compare two [`ChunkedBlob`] manifests for content equality, skipping
/// any chunk both sides reference by the same content-addressed hash:
/// identical hash means identical bytes by construction — the same trust
/// [`content_eq`]'s own `a == b` whole-object fast path already relies
/// on, just applied per chunk instead of per object. Only chunks that
/// diverge (a different hash at the same aligned position) are actually
/// read and byte-compared, so a change confined to part of a large file
/// costs only that part on both sides, not the whole file. An in-place
/// edit of unchanged length walks straight past the shared, unaffected
/// tail with no reads once the edited region resyncs by hash again.
///
/// `total_size` is trusted without re-summing actual chunk bytes only for
/// a side whose chunks were *partly* skipped by id — the same trust
/// [`LoadedBlob::len`] already documents: every reassembly path enforces
/// it via [`ChunkedBlob::check_reassembled_size`], so a manifest with a
/// wrong `total_size` cannot have been durably written through mkit's own
/// writers (#550). A side every one of whose chunks gets actually read
/// (no id ever matched a boundary on that side — the case an append or a
/// wholly-rewritten file hits) has its real byte count checked against
/// its declared `total_size` before returning, exactly like the old
/// byte-cursor walk did, and errors the same way
/// ([`crate::object::MkitError::ChunkedBlobSizeMismatch`]) if they
/// disagree. Only a side that *did* skip at least one chunk by id skips
/// this check for that side, since the skipped chunks' real lengths were
/// never read to sum: two manifests whose declared `total_size` fields
/// agree with each other but not with their own chunks, and whose chunk
/// sequences also line up entirely (or partly, from the first divergence
/// on) by hash, can still slip through unnoticed on that account. An
/// append or truncation, meanwhile, needs zero chunk reads at all before
/// even reaching this check: the two sides' declared sizes differ, so
/// `content_eq` returns `Ok(false)` above before the merge walk starts.
///
/// # Errors
/// [`crate::store::StoreError`] if a chunk that must be read (either
/// side has no matching chunk to skip against) is missing, corrupt, or
/// not a `Blob`; [`crate::object::MkitError::ChunkedBlobSizeMismatch`]
/// if a side with no skipped chunks has a `total_size` that disagrees
/// with its chunks' real byte sum.
fn chunked_content_eq<S: crate::store::ObjectSource + ?Sized>(
    store: &S,
    ma: &ChunkedBlob,
    mb: &ChunkedBlob,
) -> Result<bool, crate::store::StoreError> {
    if ma.total_size != mb.total_size {
        return Ok(false);
    }

    let mut ia = 0usize;
    let mut ib = 0usize;
    let mut buf_a: Vec<u8> = Vec::new();
    let mut buf_b: Vec<u8> = Vec::new();
    let mut pos_a = 0usize;
    let mut pos_b = 0usize;
    let mut equal = true;
    // Real bytes actually read (never chunks skipped by id) per side,
    // and whether any chunk on that side WAS skipped by id — see the
    // doc comment above for what these guard.
    let mut read_a: u64 = 0;
    let mut read_b: u64 = 0;
    let mut skipped_a = false;
    let mut skipped_b = false;

    loop {
        // Both cursors sit at a chunk boundary: skip a run of chunks
        // that match by id, with no read on either side.
        while pos_a == buf_a.len()
            && pos_b == buf_b.len()
            && ia < ma.chunks.len()
            && ib < mb.chunks.len()
            && ma.chunks[ia] == mb.chunks[ib]
        {
            ia += 1;
            ib += 1;
            skipped_a = true;
            skipped_b = true;
        }

        if pos_a == buf_a.len() {
            buf_a = match ma.chunks.get(ia) {
                Some(h) => {
                    ia += 1;
                    let data = read_blob_chunk(store, h)?;
                    read_a = read_a
                        .checked_add(data.len() as u64)
                        .ok_or(crate::store::StoreError::ObjectTooLarge)?;
                    data
                }
                None => Vec::new(),
            };
            pos_a = 0;
        }
        if pos_b == buf_b.len() {
            buf_b = match mb.chunks.get(ib) {
                Some(h) => {
                    ib += 1;
                    let data = read_blob_chunk(store, h)?;
                    read_b = read_b
                        .checked_add(data.len() as u64)
                        .ok_or(crate::store::StoreError::ObjectTooLarge)?;
                    data
                }
                None => Vec::new(),
            };
            pos_b = 0;
        }

        let a_rest = &buf_a[pos_a..];
        let b_rest = &buf_b[pos_b..];
        if a_rest.is_empty() && b_rest.is_empty() {
            if ia >= ma.chunks.len() && ib >= mb.chunks.len() {
                if !skipped_a && read_a != ma.total_size {
                    return Err(crate::object::MkitError::ChunkedBlobSizeMismatch {
                        expected: ma.total_size,
                        actual: read_a,
                    }
                    .into());
                }
                if !skipped_b && read_b != mb.total_size {
                    return Err(crate::object::MkitError::ChunkedBlobSizeMismatch {
                        expected: mb.total_size,
                        actual: read_b,
                    }
                    .into());
                }
                return Ok(equal);
            }
            // A zero-length trailing chunk on one or both sides; loop
            // again to pull the next one via the boundary checks above.
            continue;
        }

        let count = a_rest.len().min(b_rest.len());
        if count == 0 {
            // One side has run out of chunks for good; drain the other
            // (matching the byte-cursor path's behavior) so a read
            // error later in its remaining chunks still surfaces.
            equal = false;
            pos_a = buf_a.len();
            pos_b = buf_b.len();
            continue;
        }
        if a_rest[..count] != b_rest[..count] {
            equal = false;
        }
        pos_a += count;
        pos_b += count;
    }
}

/// Compare stored content to bytes without reassembling chunked storage.
///
/// # Errors
/// Propagates errors from [`content_fingerprint`].
pub fn content_eq_bytes<S: crate::store::ObjectSource + ?Sized>(
    store: &S,
    object: &Hash,
    bytes: &[u8],
) -> Result<bool, crate::store::StoreError> {
    let mut cursor = ContentCursor::load(store, object)?;
    let mut offset = 0usize;
    let mut equal = true;
    loop {
        let chunk = cursor.remaining(store)?;
        if chunk.is_empty() {
            return Ok(equal && offset == bytes.len());
        }
        let end = offset
            .checked_add(chunk.len())
            .ok_or(crate::store::StoreError::ObjectTooLarge)?;
        equal &= bytes.get(offset..end) == Some(chunk);
        offset = end;
        cursor.offset += chunk.len();
    }
}

struct ContentCursor {
    data: Vec<u8>,
    offset: usize,
    chunks: std::vec::IntoIter<Hash>,
    expected: u64,
    loaded: u64,
}

impl ContentCursor {
    fn load<S: crate::store::ObjectSource + ?Sized>(
        store: &S,
        hash: &Hash,
    ) -> Result<Self, crate::store::StoreError> {
        Self::from_object(store.read_object(hash)?)
    }

    /// Build a cursor from an already-read top-level object, saving the
    /// caller a second `read_object` when it needed the object anyway
    /// (e.g. [`content_eq`] deciding whether the chunked fast path
    /// applies).
    fn from_object(object: Object) -> Result<Self, crate::store::StoreError> {
        let (data, chunks, expected) = match object {
            Object::Blob(b) => {
                let size = b.data.len() as u64;
                (b.data, Vec::new(), size)
            }
            Object::ChunkedBlob(m) => (Vec::new(), m.chunks, m.total_size),
            _ => {
                return Err(crate::store::StoreError::Io(io::Error::other(
                    "content object is not a Blob or ChunkedBlob",
                )));
            }
        };
        let loaded = data.len() as u64;
        Ok(Self {
            data,
            offset: 0,
            chunks: chunks.into_iter(),
            expected,
            loaded,
        })
    }

    fn remaining<S: crate::store::ObjectSource + ?Sized>(
        &mut self,
        store: &S,
    ) -> Result<&[u8], crate::store::StoreError> {
        while self.offset == self.data.len() {
            let Some(hash) = self.chunks.next() else {
                if self.loaded != self.expected {
                    return Err(crate::object::MkitError::ChunkedBlobSizeMismatch {
                        expected: self.expected,
                        actual: self.loaded,
                    }
                    .into());
                }
                return Ok(&[]);
            };
            let data = read_blob_chunk(store, &hash)?;
            self.loaded = self
                .loaded
                .checked_add(data.len() as u64)
                .ok_or(crate::store::StoreError::ObjectTooLarge)?;
            self.data = data;
            self.offset = 0;
        }
        Ok(&self.data[self.offset..])
    }
}

#[cfg(test)]
mod equality_tests {
    use super::*;
    use crate::{layout::RepoLayout, object::Blob, serialize, store::ObjectStore};

    fn put(store: &ObjectStore, object: &Object) -> Hash {
        store.write(&serialize::serialize(object).unwrap()).unwrap()
    }

    #[test]
    fn large_inline_fixed_and_cdc_content_agree() {
        let dir = tempfile::tempdir().unwrap();
        let store = ObjectStore::init(&RepoLayout::single(dir.path())).unwrap();
        let data = vec![7; usize::try_from(super::super::CHUNK_THRESHOLD + 17).unwrap()];
        let inline = put(&store, &Object::Blob(Blob { data: data.clone() }));
        let cdc = super::super::store_file_object(&store, &data).unwrap();
        let chunks = data
            .chunks(65_536)
            .map(|b| put(&store, &Object::Blob(Blob { data: b.to_vec() })))
            .collect();
        let fixed = put(
            &store,
            &Object::ChunkedBlob(ChunkedBlob {
                total_size: data.len() as u64,
                chunk_size: 65_536,
                chunks,
            }),
        );
        assert_ne!(inline, cdc);
        assert_ne!(fixed, cdc);
        assert!(content_eq(&store, &inline, &fixed).unwrap());
        assert!(content_eq(&store, &fixed, &cdc).unwrap());
        assert!(content_eq_bytes(&store, &fixed, &data).unwrap());
        assert_eq!(
            content_fingerprint(&store, &inline).unwrap(),
            content_fingerprint(&store, &cdc).unwrap()
        );
        let mut changed = data;
        changed[65_536] = 8;
        assert!(!content_eq_bytes(&store, &fixed, &changed).unwrap());
    }

    #[test]
    fn invalid_chunks_are_errors_even_after_content_differs() {
        let dir = tempfile::tempdir().unwrap();
        let store = ObjectStore::init(&RepoLayout::single(dir.path())).unwrap();
        let a = put(
            &store,
            &Object::Blob(Blob {
                data: b"a".to_vec(),
            }),
        );
        let b = put(
            &store,
            &Object::Blob(Blob {
                data: b"b".to_vec(),
            }),
        );
        for (total_size, chunks) in [(2, vec![a]), (2, vec![a, [42; 32]])] {
            let bad = put(
                &store,
                &Object::ChunkedBlob(ChunkedBlob {
                    total_size,
                    chunk_size: 0,
                    chunks,
                }),
            );
            assert!(content_eq(&store, &bad, &b).is_err());
            assert!(content_eq_bytes(&store, &bad, b"b").is_err());
            assert!(content_fingerprint(&store, &bad).is_err());
        }
    }

    fn chunk(store: &ObjectStore, data: &[u8]) -> Hash {
        put(
            store,
            &Object::Blob(Blob {
                data: data.to_vec(),
            }),
        )
    }

    fn manifest(store: &ObjectStore, parts: &[&[u8]]) -> ChunkedBlob {
        let total_size: u64 = parts.iter().map(|p| p.len() as u64).sum();
        ChunkedBlob {
            total_size,
            chunk_size: 0,
            chunks: parts.iter().map(|p| chunk(store, p)).collect(),
        }
    }

    fn fresh_store() -> (tempfile::TempDir, ObjectStore) {
        let dir = tempfile::tempdir().unwrap();
        let store = ObjectStore::init(&RepoLayout::single(dir.path())).unwrap();
        (dir, store)
    }

    // Direct tests of `chunked_content_eq`'s merge logic against small,
    // hand-built manifests — fast and deterministic, independent of real
    // FastCDC boundaries, covering the id-skip fast path, the
    // read-and-resync fallback, and the documented total_size trust gap.

    #[test]
    fn chunked_fast_path_all_chunks_match_by_id() {
        let (_dir, store) = fresh_store();
        let x = chunk(&store, b"hello ");
        let y = chunk(&store, b"world");
        let ma = ChunkedBlob {
            total_size: 11,
            chunk_size: 0,
            chunks: vec![x, y],
        };
        let mb = ma.clone();
        assert!(super::chunked_content_eq(&store, &ma, &mb).unwrap());
    }

    #[test]
    fn chunked_fully_misaligned_but_equal_content() {
        let (_dir, store) = fresh_store();
        // "ab"+"cd" vs "a"+"bcd" — no chunk hash ever matches, so every
        // byte is read and compared through the resync fallback, yet the
        // reassembled content is identical.
        let ma = manifest(&store, &[b"ab", b"cd"]);
        let mb = manifest(&store, &[b"a", b"bcd"]);
        assert!(super::chunked_content_eq(&store, &ma, &mb).unwrap());
        assert!(super::chunked_content_eq(&store, &mb, &ma).unwrap());
    }

    #[test]
    fn chunked_shared_prefix_then_misaligned_equal_suffix() {
        let (_dir, store) = fresh_store();
        let x = chunk(&store, b"shared-prefix-");
        let ma = ChunkedBlob {
            total_size: 14 + 4,
            chunk_size: 0,
            chunks: [x]
                .into_iter()
                .chain(manifest(&store, &[b"ab", b"cd"]).chunks)
                .collect(),
        };
        let mb = ChunkedBlob {
            total_size: 14 + 4,
            chunk_size: 0,
            chunks: [x]
                .into_iter()
                .chain(manifest(&store, &[b"a", b"bcd"]).chunks)
                .collect(),
        };
        // `x` is skipped by id; "ab"+"cd" vs "a"+"bcd" is read and resynced.
        assert!(super::chunked_content_eq(&store, &ma, &mb).unwrap());
    }

    #[test]
    fn chunked_append_differs_via_total_size_without_reading() {
        let (_dir, store) = fresh_store();
        let x = chunk(&store, b"shared");
        let ma = ChunkedBlob {
            total_size: 6,
            chunk_size: 0,
            chunks: vec![x],
        };
        let y = chunk(&store, b"-more");
        let mb = ChunkedBlob {
            total_size: 11,
            chunk_size: 0,
            chunks: vec![x, y],
        };
        assert!(!super::chunked_content_eq(&store, &ma, &mb).unwrap());
        assert!(!super::chunked_content_eq(&store, &mb, &ma).unwrap());
    }

    #[test]
    fn chunked_same_total_size_different_content_after_shared_prefix() {
        let (_dir, store) = fresh_store();
        let x = chunk(&store, b"shared-");
        let ma = ChunkedBlob {
            total_size: 7 + 2,
            chunk_size: 0,
            chunks: vec![x, chunk(&store, b"ab")],
        };
        let mb = ChunkedBlob {
            total_size: 7 + 2,
            chunk_size: 0,
            chunks: vec![x, chunk(&store, b"ba")],
        };
        assert!(!super::chunked_content_eq(&store, &ma, &mb).unwrap());
    }

    #[test]
    fn chunked_split_differently_but_actually_different_content() {
        let (_dir, store) = fresh_store();
        let x = chunk(&store, b"shared-");
        let ma = ChunkedBlob {
            total_size: 7 + 4,
            chunk_size: 0,
            chunks: vec![x, chunk(&store, b"abcd")],
        };
        let mb = ChunkedBlob {
            total_size: 7 + 4,
            chunk_size: 0,
            chunks: vec![x, chunk(&store, b"ab"), chunk(&store, b"cX")],
        };
        assert!(!super::chunked_content_eq(&store, &ma, &mb).unwrap());
    }

    #[test]
    fn chunked_empty_manifests_are_equal() {
        let (_dir, store) = fresh_store();
        let ma = ChunkedBlob {
            total_size: 0,
            chunk_size: 0,
            chunks: vec![],
        };
        let mb = ma.clone();
        assert!(super::chunked_content_eq(&store, &ma, &mb).unwrap());
    }

    /// Documents the accepted trust gap: `total_size` is not re-summed
    /// from actual chunk bytes on the chunked fast path (mirroring
    /// `LoadedBlob::len`'s existing trust). Two manifests whose chunk
    /// sequence fully matches by id, but whose shared `total_size`
    /// field is wrong for that sequence, are reported equal without
    /// ever reading a chunk to notice. A manifest this malformed cannot
    /// be produced by any of mkit's own writers (`check_reassembled_size`
    /// gates every one); this can only arise from an object constructed
    /// directly, bypassing them, as this test does.
    #[test]
    fn chunked_wrong_shared_total_size_is_not_detected_when_ids_fully_match() {
        let (_dir, store) = fresh_store();
        let x = chunk(&store, b"ab"); // 2 real bytes
        let ma = ChunkedBlob {
            total_size: 999, // wrong on both sides, identically
            chunk_size: 0,
            chunks: vec![x],
        };
        let mb = ma.clone();
        assert!(super::chunked_content_eq(&store, &ma, &mb).unwrap());
    }

    /// The case the fast path must *not* let slip past: no chunk on
    /// either side is ever skipped by id (every hash differs), so both
    /// sides get fully read — and a wrong `total_size` on a side that
    /// was fully read must still surface as
    /// `ChunkedBlobSizeMismatch`, exactly like the old byte-cursor walk.
    /// Regression test for a gap an independent review found: an
    /// earlier version of `chunked_content_eq` only ever compared the
    /// two manifests' declared `total_size` fields against *each
    /// other*, never against either side's own real chunk bytes, so two
    /// fully-diverging (no id ever matches) manifests that happened to
    /// declare the same wrong `total_size` were reported merely
    /// "unequal" instead of erroring.
    #[test]
    fn chunked_wrong_total_size_is_detected_when_no_chunk_is_skipped() {
        let (_dir, store) = fresh_store();
        // Real chunk bytes: 90 bytes on each side ("a" x90 vs "b" x90),
        // so no chunk hash ever matches — nothing is ever skipped.
        let a90 = vec![b'a'; 90];
        let b90 = vec![b'b'; 90];
        let ma = ChunkedBlob {
            total_size: 100, // wrong: real sum is 90
            chunk_size: 0,
            chunks: vec![chunk(&store, &a90)],
        };
        let mb = ChunkedBlob {
            total_size: 100, // also wrong, and equal to ma's
            chunk_size: 0,
            chunks: vec![chunk(&store, &b90)],
        };
        let err = super::chunked_content_eq(&store, &ma, &mb).unwrap_err();
        assert!(
            matches!(
                err,
                crate::store::StoreError::Decode(
                    crate::object::MkitError::ChunkedBlobSizeMismatch { .. }
                )
            ),
            "expected ChunkedBlobSizeMismatch, got {err:?}"
        );
    }

    /// The same wrong-`total_size` gap, but on only one side: the other
    /// side's declared size is accurate for its own real chunk bytes.
    /// Content is unequal either way (lengths differ once both are read
    /// out to their real end), but the malformed side must still error
    /// rather than silently compare as "not equal".
    #[test]
    fn chunked_wrong_total_size_on_one_side_only_is_detected() {
        let (_dir, store) = fresh_store();
        let a80 = vec![b'a'; 80];
        let b100 = vec![b'b'; 100];
        let ma = ChunkedBlob {
            total_size: 100, // wrong: real sum is 80
            chunk_size: 0,
            chunks: vec![chunk(&store, &a80)],
        };
        let mb = ChunkedBlob {
            total_size: 100, // correct
            chunk_size: 0,
            chunks: vec![chunk(&store, &b100)],
        };
        let err = super::chunked_content_eq(&store, &ma, &mb).unwrap_err();
        assert!(
            matches!(
                err,
                crate::store::StoreError::Decode(
                    crate::object::MkitError::ChunkedBlobSizeMismatch { .. }
                )
            ),
            "expected ChunkedBlobSizeMismatch, got {err:?}"
        );
    }

    /// End-to-end sanity check through the public `content_eq` entry
    /// point (not the private merge function directly) over real
    /// `FastCDC`-chunked content: an append, a single-byte in-place edit,
    /// and a truncation each produce a different manifest and must all
    /// compare unequal, exactly matching a ground-truth byte comparison.
    #[test]
    fn content_eq_real_chunked_mutations_match_ground_truth() {
        let (_dir, store) = fresh_store();
        let threshold = usize::try_from(super::super::CHUNK_THRESHOLD).unwrap();
        // Deterministic pseudo-random bytes so FastCDC sees real cut
        // points instead of one run-length-maxed chunk.
        let mut data = Vec::with_capacity(threshold * 3);
        let mut state: u64 = 0x1234_5678_9abc_def0;
        for _ in 0..threshold * 3 {
            state = state
                .wrapping_mul(6_364_136_223_846_793_005)
                .wrapping_add(1);
            data.push((state >> 56) as u8);
        }

        let base = super::super::store_file_object(&store, &data).unwrap();

        let mut appended = data.clone();
        appended.extend_from_slice(b"appended tail bytes");
        let appended_hash = super::super::store_file_object(&store, &appended).unwrap();

        let mut edited = data.clone();
        let mid = edited.len() / 2;
        edited[mid] ^= 0xFF;
        let edited_hash = super::super::store_file_object(&store, &edited).unwrap();

        let truncated = &data[..data.len() - 500];
        let truncated_hash = super::super::store_file_object(&store, truncated).unwrap();

        let unchanged_hash = super::super::store_file_object(&store, &data).unwrap();

        assert_eq!(base, unchanged_hash, "identical content dedups to one id");
        assert!(content_eq(&store, &base, &unchanged_hash).unwrap());

        for (name, other, other_bytes) in [
            ("append", appended_hash, appended.as_slice()),
            ("edit", edited_hash, edited.as_slice()),
            ("truncate", truncated_hash, truncated),
        ] {
            assert_ne!(base, other, "{name}: expected a different object id");
            assert_eq!(
                content_eq(&store, &base, &other).unwrap(),
                data == other_bytes,
                "{name}: content_eq must match ground truth"
            );
            assert!(
                !content_eq(&store, &base, &other).unwrap(),
                "{name}: bytes differ"
            );
        }
    }

    /// A mid-file insertion against real `FastCDC` output: unlike
    /// `content_eq_real_chunked_mutations_match_ground_truth`'s
    /// single-byte edit (which may or may not shift a chunk boundary),
    /// inserting new bytes shifts every downstream offset, which
    /// reliably forces `FastCDC` to re-cut several chunks around the
    /// insertion point before content-defined chunking resyncs on the
    /// unchanged bytes further on. This is the scenario
    /// `chunked_content_eq`'s doc comment describes ("a change confined
    /// to part of a large file... resyncs by hash again") and, per an
    /// independent review, the only other resync coverage exercised
    /// hand-built or fixed-vs-CDC manifests, never two independently
    /// `FastCDC`-chunked real files. Confirms at the manifest level
    /// that a real divergence-then-resync actually occurred (shared
    /// first and last chunk hashes, a different chunk in between) before
    /// checking `content_eq` against ground truth.
    #[test]
    fn content_eq_real_chunked_insertion_forces_boundary_resync() {
        let (_dir, store) = fresh_store();
        let threshold = usize::try_from(super::super::CHUNK_THRESHOLD).unwrap();
        let mut data = Vec::with_capacity(threshold * 4);
        let mut state: u64 = 0x0BAD_C0DE_F00D_CAFE;
        for _ in 0..threshold * 4 {
            state = state
                .wrapping_mul(6_364_136_223_846_793_005)
                .wrapping_add(1);
            data.push((state >> 56) as u8);
        }

        let mut inserted = data.clone();
        let at = data.len() / 2;
        let mut new_bytes = vec![0u8; 4096];
        let mut s: u64 = 0xFACE_FEED_1234_5678;
        for b in &mut new_bytes {
            s = s.wrapping_mul(6_364_136_223_846_793_005).wrapping_add(1);
            *b = (s >> 56) as u8;
        }
        inserted.splice(at..at, new_bytes.iter().copied());

        let base_hash = super::super::store_file_object(&store, &data).unwrap();
        let inserted_hash = super::super::store_file_object(&store, &inserted).unwrap();
        assert_ne!(base_hash, inserted_hash);

        let Object::ChunkedBlob(base_manifest) = store.read_object(&base_hash).unwrap() else {
            panic!("expected base to be chunked (data.len() > CHUNK_THRESHOLD)");
        };
        let Object::ChunkedBlob(inserted_manifest) = store.read_object(&inserted_hash).unwrap()
        else {
            panic!("expected inserted to be chunked");
        };
        assert!(
            base_manifest.chunks.len() > 2 && inserted_manifest.chunks.len() > 2,
            "fixture too small to exercise multiple chunks"
        );
        assert_eq!(
            base_manifest.chunks.first(),
            inserted_manifest.chunks.first(),
            "the unaffected prefix must still share its leading chunk by id"
        );
        assert_eq!(
            base_manifest.chunks.last(),
            inserted_manifest.chunks.last(),
            "content-defined chunking must resync on the unaffected suffix"
        );
        assert_ne!(
            base_manifest.chunks, inserted_manifest.chunks,
            "the insertion must actually shift chunk boundaries somewhere in the middle"
        );

        assert!(
            !content_eq(&store, &base_hash, &inserted_hash).unwrap(),
            "content genuinely differs after the insertion"
        );
    }
}