objects/store/mod.rs
1// SPDX-License-Identifier: Apache-2.0
2//! Backend-neutral object storage abstractions and concrete implementations.
3
4use std::path::PathBuf;
5
6use crate::object::{
7 Action, ActionId, AnnotatedTag, Blob, ContentHash, OpenedTreeBody, PartialTree, State,
8 StateAttachment, StateAttachmentId, StateId, Tree, TreeEntry, TreeEntryReader,
9 TreeResumeCursor, is_redacted_tree, is_streamable_tree,
10};
11
12pub mod codec;
13#[cfg(feature = "fs")]
14mod delta_source;
15#[cfg(feature = "fs")]
16pub mod fs;
17pub mod liveness;
18#[cfg(any(test, feature = "memory-backend"))]
19pub mod memory;
20#[cfg(test)]
21mod partial_tree_tests;
22pub use heddle_pack::store::pack;
23#[cfg(feature = "fs")]
24pub mod shallow;
25#[cfg(feature = "fs")]
26mod snapshot_commit;
27pub mod source;
28pub mod store_compliance;
29#[cfg(feature = "fs")]
30pub mod writer_lease;
31
32#[cfg(feature = "fs")]
33pub use fs::{
34 DEFAULT_PACK_INSTALL_INTENT_TTL_SECS, FsRepackOperation, FsStore, PackInstallIntent,
35 PackInstallMetricsSnapshot, PackInstallPhase, PackInstallRecoverReport,
36 install_pack_bytes_journaled, pack_install_metrics_reset, pack_install_metrics_snapshot,
37 recover_pack_install_intents, recover_pack_install_intents_with_ttl,
38};
39pub use heddle_format::compression::{CompressionConfig, CompressionError, compress, decompress};
40pub use liveness::{
41 AGENT_LEASE_DURATION, Liveness, current_boot_id, process_alive, process_birth,
42 reservation_liveness_at,
43};
44#[cfg(any(test, feature = "memory-backend"))]
45pub use memory::InMemoryStore;
46pub use pack::{
47 CancellationToken as RepackCancellationToken, LoadMonitor as RepackLoadMonitor, PackBuilder,
48 PackObjectId, PackReader, PackStats, RepackContext, RepackError, RepackHandle, RepackInventory,
49 RepackOperation, RepackOutcome, RepackPolicy, RepackReason, RepackReport, RepackResourceLimits,
50 RepackSchedule, RepackScheduler, StreamingPackBuilder, SyncData,
51};
52#[cfg(feature = "fs")]
53pub use shallow::ShallowInfo;
54#[cfg(feature = "fs")]
55#[doc(hidden)]
56pub use snapshot_commit::{
57 SNAPSHOT_COMMIT_ARTIFACT_SCHEMA, SnapshotCommitArtifact, SnapshotCommitDescriptor,
58 SnapshotPackManager,
59};
60#[cfg(feature = "async-source")]
61pub use source::AsyncObjectSource;
62pub use source::ObjectSource;
63#[cfg(feature = "fs")]
64pub use writer_lease::{
65 WriterLease, WriterLeaseAuthOutcome, WriterLeaseDraft, WriterLeaseGrant,
66 WriterLeaseReserveOutcome, WriterLeaseStatus, WriterLeaseStore, checkout_writer_lock,
67 generate_writer_lease_id, generate_writer_lease_token,
68};
69
70/// A newly-authored tree plus its immediate parent, when capture already knows
71/// that relationship. Stores may use the hint for bounded HDC1 encoding; it
72/// never changes the tree's semantic content hash.
73#[derive(Clone, Debug)]
74pub struct TreeWrite {
75 pub tree: Tree,
76 pub parent: Option<ContentHash>,
77}
78
79impl TreeWrite {
80 pub fn anchor(tree: Tree) -> Self {
81 Self { tree, parent: None }
82 }
83
84 pub fn descendant(tree: Tree, parent: ContentHash) -> Self {
85 Self {
86 tree,
87 parent: Some(parent),
88 }
89 }
90}
91
92/// Read-only objects whose authoritative representation lives outside the
93/// native Heddle object directory. Git-overlay repositories use this seam to
94/// translate objects directly from `.git` without importing a second copy.
95pub trait ExternalObjectSource: Send + Sync {
96 fn get_blob(&self, hash: &ContentHash) -> Result<Option<Blob>>;
97 fn get_tree(&self, hash: &ContentHash) -> Result<Option<Tree>>;
98 fn get_state(&self, id: &StateId) -> Result<Option<State>>;
99 fn list_states(&self) -> Result<Vec<StateId>>;
100}
101
102/// Explicit cache control for benchmarks and diagnostic tools.
103///
104/// Cache invalidation is not part of durable object storage semantics. Keeping
105/// it separate prevents remote stores such as Weft's backend from having to
106/// pretend they expose process-local cache controls merely to implement
107/// [`ObjectStore`].
108pub trait ObjectCacheControl: Send + Sync {
109 /// Drop process-local decoded-object caches, if this implementation has
110 /// any. The next read should observe the implementation's cold path.
111 fn clear_recent_caches(&self);
112}
113
114pub use crate::error::{HeddleError as StoreError, HeddleError, Result};
115
116/// Sidecar records that live outside the content-addressed object graph —
117/// signed redactions and state-visibility tiers. They never ride native packs
118/// and are transferred out-of-band. Backends that do not model them can use
119/// the default methods, while native stores override the relevant operations.
120pub trait SidecarStore: Send + Sync {
121 /// Whether the store holds any redaction record for the given blob.
122 ///
123 /// Redactions live in a sidecar (`<heddle_dir>/redactions/`) that is
124 /// structurally outside the content-addressed object graph so GC
125 /// can't reach them. The wire layer needs a cheap probe to decide
126 /// whether to ship a redaction for a blob in the closure, so this
127 /// is a separate method rather than a `get_*` + null check.
128 ///
129 /// Default impl returns `Ok(false)` — stores that don't model
130 /// redactions silently report "no redactions," which is the
131 /// correct behaviour for purely in-memory or remote-shim stores.
132 fn has_redactions_for_blob(&self, _blob: &ContentHash) -> Result<bool> {
133 Ok(false)
134 }
135
136 /// Return the raw rmp-encoded `RedactionsBlob` bytes for the given
137 /// blob, or `Ok(None)` if no redaction record exists. The bytes
138 /// are byte-identical to what was written by `put_redactions_bytes_for_blob`
139 /// (or by `Repository::put_redaction`); this is the wire-transfer
140 /// payload, not a re-serialized view.
141 ///
142 /// Default impl returns `Ok(None)`.
143 fn get_redactions_bytes_for_blob(&self, _blob: &ContentHash) -> Result<Option<Vec<u8>>> {
144 Ok(None)
145 }
146
147 /// Persist the rmp-encoded `RedactionsBlob` bytes for the given
148 /// blob. Receiver-side replay calls this after signature
149 /// verification so the bytes land in the same sidecar that the
150 /// sender's `Repository::put_redaction` writes to.
151 ///
152 /// Default impl returns an "unsupported" error — stores that don't
153 /// model redactions (e.g. read-only shims) refuse rather than
154 /// silently dropping the record.
155 fn put_redactions_bytes_for_blob(&self, _blob: &ContentHash, _bytes: &[u8]) -> Result<()> {
156 Err(HeddleError::InvalidObject(
157 "this object store does not support persisting redactions".to_string(),
158 ))
159 }
160
161 /// List every blob that has at least one redaction record. Used by
162 /// the GC pin guard and by sync to enumerate redactions for the
163 /// state closure. Order is unspecified; callers that need stable
164 /// ordering should sort.
165 ///
166 /// Default impl returns `Ok(vec![])`.
167 fn list_blobs_with_redactions(&self) -> Result<Vec<ContentHash>> {
168 Ok(Vec::new())
169 }
170
171 /// Whether the store holds any state-visibility record for `state`.
172 ///
173 /// Like redactions, state-visibility records live in a sidecar outside
174 /// the content-addressed object graph and cannot ride native packs.
175 /// Sync uses this probe while enumerating a state closure so a non-public
176 /// state can advertise the sidecar that must travel out-of-pack.
177 ///
178 /// Default impl returns `Ok(false)` for stores that do not model this
179 /// sidecar.
180 fn has_state_visibility_for_state(&self, _state: &StateId) -> Result<bool> {
181 Ok(false)
182 }
183
184 /// Return the raw rmp-encoded `StateVisibilityBlob` bytes for `state`,
185 /// or `Ok(None)` if no sidecar exists. The bytes are the wire-transfer
186 /// payload for state visibility.
187 ///
188 /// Default impl returns `Ok(None)`.
189 fn get_state_visibility_bytes_for_state(&self, _state: &StateId) -> Result<Option<Vec<u8>>> {
190 Ok(None)
191 }
192
193 /// Persist raw `StateVisibilityBlob` bytes for `state`.
194 ///
195 /// Default impl returns an "unsupported" error so stores that do not
196 /// model the sidecar refuse instead of dropping it.
197 fn put_state_visibility_bytes_for_state(&self, _state: &StateId, _bytes: &[u8]) -> Result<()> {
198 Err(HeddleError::InvalidObject(
199 "this object store does not support persisting state visibility".to_string(),
200 ))
201 }
202
203 /// List every state with at least one state-visibility record.
204 ///
205 /// Default impl returns `Ok(vec![])`.
206 fn list_states_with_visibility(&self) -> Result<Vec<StateId>> {
207 Ok(Vec::new())
208 }
209}
210
211/// The result of resolving a tree hash that may be held either as the full
212/// canonical object or only as a redacted partial projection (HRT1).
213///
214/// A full tree SUPERSEDES a partial for the same hash. `Partial` is distinct
215/// from `Absent` on purpose: a partial clone that withholds an entry must be
216/// distinguishable from a repository that is missing or corrupt, so callers can
217/// present "withheld" rather than "gone".
218#[derive(Clone, Debug)]
219pub enum TreeRead {
220 /// The full canonical tree is held.
221 Full(Tree),
222 /// Only a redacted partial projection is held: its visible entries carry
223 /// their preimage and its withheld entries are marked opaque. Verifies
224 /// against the declared `State.tree` via `PartialTree::reconstruct_root`.
225 Partial(PartialTree),
226 /// Neither a full tree nor a partial projection is held for this hash.
227 Absent,
228}
229
230/// The outcome of storing a redacted partial projection under the monotone
231/// discipline enforced by [`ObjectStore::put_partial_tree`].
232#[derive(Clone, Copy, Debug, PartialEq, Eq)]
233pub enum PartialTreeWrite {
234 /// The projection was written to the partial slot.
235 Stored {
236 /// Number of withheld (redacted) leaves.
237 redacted: usize,
238 /// Number of visible leaves.
239 visible: usize,
240 },
241 /// A full canonical tree for this hash is already held, so the projection
242 /// was dropped rather than stored — a full tree supersedes a partial, and a
243 /// partial never overwrites a full (monotone).
244 SupersededByFull,
245}
246
247/// Trait for object storage backends.
248///
249/// Sidecars remain a separate implementation seam, but every object store
250/// exposes that seam. This preserves object-safe `dyn ObjectStore` consumers
251/// such as Weft's local filesystem backend without coupling its S3 backend to
252/// the native store implementation.
253pub trait ObjectStore: SidecarStore + Send + Sync {
254 fn get_annotated_tag(&self, _hash: &ContentHash) -> Result<Option<AnnotatedTag>> {
255 Ok(None)
256 }
257 fn put_annotated_tag(&self, _tag: &AnnotatedTag) -> Result<ContentHash> {
258 Err(HeddleError::InvalidObject(
259 "object store does not support annotated tags".to_string(),
260 ))
261 }
262 fn list_annotated_tags(&self) -> Result<Vec<ContentHash>> {
263 Ok(Vec::new())
264 }
265 fn get_blob(&self, hash: &ContentHash) -> Result<Option<Blob>>;
266 fn put_blob(&self, blob: &Blob) -> Result<ContentHash>;
267
268 /// Zero-copy variant of `get_blob`. Returns a [`bytes::Bytes`]
269 /// view of the blob's content, which for `FsStore` reads is a
270 /// slice into the pack file's mmap when the entry is non-delta
271 /// and uncompressed — no allocation, no memcpy.
272 ///
273 /// Default impl wraps `get_blob`'s `Vec<u8>` in a `Bytes` (one
274 /// Arc allocation, no body copy) so backends without a native
275 /// fast path still satisfy the contract. The mount's hot read
276 /// path goes through this method instead of `get_blob` so the
277 /// pack-mmap fast path lights up automatically.
278 fn get_blob_bytes(&self, hash: &ContentHash) -> Result<Option<bytes::Bytes>> {
279 Ok(self
280 .get_blob(hash)?
281 .map(|blob| bytes::Bytes::from(blob.into_content())))
282 }
283
284 /// Return the *uncompressed* byte length of the blob identified by
285 /// `hash`, or `Ok(None)` when the blob is not in the store.
286 ///
287 /// The contract is "size without paying for content": backends are
288 /// expected to honour this with a header read or index lookup
289 /// rather than a full decompression. This is the hot path for
290 /// directory listings (`ls -l` over a thread mount) where loading
291 /// every blob just to learn its size would dominate.
292 ///
293 /// The default implementation falls back to `get_blob` so backends
294 /// without a cheap size accessor still satisfy the contract; native
295 /// stores (`FsStore`, `InMemoryStore`) override this with a
296 /// header- or hashmap-only path.
297 fn blob_size(&self, hash: &ContentHash) -> Result<Option<u64>> {
298 Ok(self.get_blob(hash)?.map(|blob| blob.content().len() as u64))
299 }
300
301 /// Filesystem path of the loose blob whose on-disk bytes are
302 /// byte-identical to the blob's *uncompressed* content, suitable
303 /// for `hard_link`/`clonefile` materialization without going
304 /// through `get_blob`.
305 ///
306 /// Returns `None` when the blob is missing, is only available via
307 /// a packfile, is stored compressed (the on-disk bytes wouldn't
308 /// match what a worktree consumer needs to read), or the backend
309 /// doesn't expose stable filesystem paths (e.g. `InMemoryStore`). The
310 /// default impl returns `None` so non-`FsStore` backends silently fall
311 /// through to the bytes path.
312 fn loose_blob_path(&self, _hash: &ContentHash) -> Option<PathBuf> {
313 None
314 }
315
316 /// Ensure the blob identified by `hash` is materialized as an
317 /// uncompressed loose file at the canonical loose path so that
318 /// `loose_blob_path` returns `Some(path)` on a subsequent call.
319 ///
320 /// This is the "warm canonical store" path that lets the
321 /// hardlink-first materializer keep its 5–10× wall-clock and
322 /// storage-allocation wins after `pack_objects + prune_loose_objects`
323 /// has moved everything into a packfile. Without this, the lazy
324 /// hardlink path silently degrades to `fs::write(decompressed)` on
325 /// every materialize, because `loose_blob_path` returns `None` for
326 /// pack-only and compressed-loose blobs.
327 ///
328 /// Cost-amortization: the first promotion of a blob pays
329 /// `decompress + atomic write`. Every subsequent materialize of
330 /// the same blob — into the same worktree on `goto`, or into a
331 /// sibling worktree on `delegate` — is a single `link(2)`. Net
332 /// win for any N > 1 materializations; break-even at N == 1.
333 ///
334 /// Pack invariants are preserved: this method does not remove the
335 /// pack-resident copy. The blob lives in both pack and loose-
336 /// uncompressed until the next `prune_loose_objects` cycle, at
337 /// which point the loose mirror is discarded and a future
338 /// materialize re-promotes on demand.
339 ///
340 /// Idempotent: a blob that's already loose-and-uncompressed is a
341 /// no-op fast path. A blob that's loose-but-compressed is
342 /// rewritten in place (atomically) with the uncompressed bytes.
343 /// A blob that's pack-resident is decompressed out of the pack
344 /// and written loose without touching the pack.
345 ///
346 /// Returns `Ok(true)` when the call did real work (a write
347 /// happened), `Ok(false)` when it was a no-op (blob was already
348 /// loose+uncompressed), and `Err` when the blob isn't in the
349 /// store at all. The default impl returns `Ok(false)` for
350 /// backends that don't expose loose paths (`InMemoryStore`), since the
351 /// hardlink path is fundamentally inapplicable there.
352 fn promote_to_loose_uncompressed(&self, _hash: &ContentHash) -> Result<bool> {
353 Ok(false)
354 }
355
356 fn put_blob_with_hash(&self, blob: &Blob, hash: ContentHash) -> Result<ContentHash> {
357 if blob.hash() != hash {
358 return Err(HeddleError::InvalidObject("blob hash mismatch".to_string()));
359 }
360 self.put_blob(blob)
361 }
362
363 fn has_blob(&self, hash: &ContentHash) -> Result<bool>;
364 /// Return whether the blob is owned by this store, excluding any configured
365 /// read-through source. Snapshot builders use this to ensure a new native
366 /// state owns its complete object closure.
367 fn has_blob_locally(&self, hash: &ContentHash) -> Result<bool> {
368 self.has_blob(hash)
369 }
370 fn get_tree(&self, hash: &ContentHash) -> Result<Option<Tree>>;
371 /// Resolve one named tree entry. Pack-capable stores override this so a
372 /// lookup can use a restartable packed record instead of materializing the
373 /// complete tree.
374 fn get_tree_entry(&self, hash: &ContentHash, name: &str) -> Result<Option<TreeEntry>> {
375 Ok(self
376 .get_tree(hash)?
377 .and_then(|tree| tree.get(name).cloned()))
378 }
379 fn put_tree(&self, tree: &Tree) -> Result<ContentHash>;
380 fn has_tree(&self, hash: &ContentHash) -> Result<bool>;
381 /// Return whether the tree is owned by this store, excluding any configured
382 /// read-through source.
383 fn has_tree_locally(&self, hash: &ContentHash) -> Result<bool> {
384 self.has_tree(hash)
385 }
386
387 // ── Redacted partial-tree projections (HRT1) ──────────────────────
388 //
389 // A partial projection lives in a slot keyed by the canonical tree hash it
390 // projects, DISTINCT from the full-tree object slot. The full-tree methods
391 // above (`get_tree`/`has_tree`) never see or return a projection; the
392 // methods below are the only way in and out of the partial slot. Backends
393 // that do not model projections inherit the default impls (report "none" /
394 // refuse the write), mirroring the sidecar seam.
395
396 /// Whether the store holds a redacted partial projection keyed by `hash`.
397 /// Independent of [`ObjectStore::has_tree`], which reports the full object.
398 fn has_partial_tree(&self, _hash: &ContentHash) -> Result<bool> {
399 Ok(false)
400 }
401
402 /// Raw HRT1 partial-projection bytes for `hash`, or `Ok(None)`. These are
403 /// byte-identical to what [`ObjectStore::put_partial_tree_bytes`] wrote —
404 /// the wire-transfer payload, not a re-serialized view.
405 fn get_partial_tree_bytes(&self, _hash: &ContentHash) -> Result<Option<Vec<u8>>> {
406 Ok(None)
407 }
408
409 /// Persist raw HRT1 partial-projection bytes keyed by `hash`. This is the
410 /// low-level slot; callers should prefer [`ObjectStore::put_partial_tree`],
411 /// which verifies the projection and enforces the monotone discipline.
412 fn put_partial_tree_bytes(&self, _hash: &ContentHash, _bytes: &[u8]) -> Result<()> {
413 Err(HeddleError::InvalidObject(
414 "this object store does not support partial tree projections".to_string(),
415 ))
416 }
417
418 /// List every tree hash for which a partial projection is held. Order is
419 /// unspecified; callers that need stable ordering should sort.
420 fn list_partial_trees(&self) -> Result<Vec<ContentHash>> {
421 Ok(Vec::new())
422 }
423
424 /// Drop any partial projection held for `hash`. Idempotent — a no-op when
425 /// none is held. Used to reclaim the partial slot once the full tree is
426 /// backfilled.
427 fn remove_partial_tree(&self, _hash: &ContentHash) -> Result<()> {
428 Ok(())
429 }
430
431 /// Store a redacted partial projection under monotone discipline.
432 ///
433 /// `hrt1` must be an HRT1 body whose visible preimages + withheld leaf
434 /// hashes reconstruct `expected` (verified through
435 /// [`codec::decode_partial_tree`], i.e. Leg 1's `reconstruct_root`).
436 /// Monotone: a partial NEVER overwrites a full tree. If the full canonical
437 /// tree for `expected` is already held, the projection is dropped and
438 /// [`PartialTreeWrite::SupersededByFull`] is returned; otherwise the raw
439 /// bytes land in the partial slot.
440 fn put_partial_tree(&self, expected: &ContentHash, hrt1: &[u8]) -> Result<PartialTreeWrite> {
441 let partial = codec::decode_partial_tree(hrt1, *expected)?;
442 if self.has_tree(expected)? {
443 return Ok(PartialTreeWrite::SupersededByFull);
444 }
445 self.put_partial_tree_bytes(expected, hrt1)?;
446 let redacted = partial.redacted_count();
447 Ok(PartialTreeWrite::Stored {
448 redacted,
449 visible: partial.leaves().len() - redacted,
450 })
451 }
452
453 /// Read the tree for `hash` as the full canonical tree, a redacted partial
454 /// projection, or absent.
455 ///
456 /// A full tree SUPERSEDES a partial: when the full object is held it is
457 /// returned even if a partial projection also exists for the same hash.
458 /// [`ObjectStore::get_tree`] by contrast returns the full tree or `None`
459 /// and NEVER the partial, so a caller that needs the complete tree cannot
460 /// be silently handed a projection.
461 fn read_tree(&self, hash: &ContentHash) -> Result<TreeRead> {
462 if let Some(full) = self.get_tree(hash)? {
463 return Ok(TreeRead::Full(full));
464 }
465 match self.get_partial_tree_bytes(hash)? {
466 Some(bytes) => Ok(TreeRead::Partial(codec::decode_partial_tree(
467 &bytes, *hash,
468 )?)),
469 None => Ok(TreeRead::Absent),
470 }
471 }
472 /// Open a streamable HTR4 tree body. Store backends use sequential
473 /// verify: resume at ordinal > 0 is refused until the bytes are hashed.
474 fn open_tree(
475 &self,
476 tree_id: &ContentHash,
477 cursor: Option<&TreeResumeCursor>,
478 ) -> Result<Option<TreeEntryReader<OpenedTreeBody>>> {
479 let Some(body) = self.get_tree_serialized(tree_id)? else {
480 return Ok(None);
481 };
482 let body = if is_streamable_tree(&body) {
483 body
484 } else {
485 let tree = self
486 .get_tree(tree_id)?
487 .ok_or_else(|| HeddleError::NotFound(format!("tree {tree_id}")))?;
488 tree.encode_lean()?
489 };
490 Ok(Some(TreeEntryReader::open(
491 OpenedTreeBody::Bytes(crate::object::BytesTreeSource::sequential_verify(body)),
492 *tree_id,
493 cursor,
494 )?))
495 }
496 fn get_state(&self, id: &StateId) -> Result<Option<State>>;
497 fn put_state(&self, state: &State) -> Result<()>;
498 fn has_state(&self, id: &StateId) -> Result<bool>;
499 fn list_states(&self) -> Result<Vec<StateId>>;
500 fn get_state_attachment(
501 &self,
502 _state: &StateId,
503 _id: &StateAttachmentId,
504 ) -> Result<Option<StateAttachment>> {
505 Ok(None)
506 }
507 fn put_state_attachment(&self, _attachment: &StateAttachment) -> Result<StateAttachmentId> {
508 Err(HeddleError::InvalidObject(
509 "object store does not support state attachments".to_string(),
510 ))
511 }
512 fn list_state_attachments(&self, _state: &StateId) -> Result<Vec<StateAttachment>> {
513 Ok(Vec::new())
514 }
515 fn get_action(&self, id: &ActionId) -> Result<Option<Action>>;
516 fn put_action(&self, action: &mut Action) -> Result<ActionId>;
517 fn list_actions(&self) -> Result<Vec<ActionId>>;
518 fn list_blobs(&self) -> Result<Vec<ContentHash>>;
519 fn list_trees(&self) -> Result<Vec<ContentHash>>;
520
521 fn put_blob_bytes_with_hash(&self, data: &[u8], hash: ContentHash) -> Result<ContentHash> {
522 self.put_blob_with_hash(&Blob::from_slice(data), hash)
523 }
524
525 /// Return the stored tree body for `hash`, without requiring HTR4.
526 ///
527 /// This is a migration seam, not a runtime compatibility reader: callers
528 /// that need current tree semantics should use [`ObjectStore::get_tree`].
529 /// Loose and packed backends must return the raw stored bytes so one-shot
530 /// migrations can canonicalize older encodings without a current-decoder
531 /// gate. Default impls that only have `get_tree` re-encode current trees.
532 fn get_tree_serialized(&self, hash: &ContentHash) -> Result<Option<Vec<u8>>> {
533 self.get_tree(hash)?
534 .map(|tree| tree.encode_canonical().map_err(HeddleError::from))
535 .transpose()
536 }
537
538 fn put_tree_serialized(&self, data: &[u8], hash: ContentHash) -> Result<ContentHash> {
539 // An HRT1 redacted projection is not a full tree: route it to the
540 // partial slot (monotone) instead of hard-refusing in the full-tree
541 // decoder.
542 if is_redacted_tree(data) {
543 self.put_partial_tree(&hash, data)?;
544 return Ok(hash);
545 }
546 let tree = codec::decode_tree_serialized_with_key(data, hash, None)?;
547 self.put_tree(&tree)
548 }
549
550 fn put_state_serialized(&self, data: &[u8], id: StateId) -> Result<()> {
551 let state = State::decode_current_msgpack(data)?;
552 if !state.accepts_stored_id(&id) {
553 return Err(HeddleError::InvalidObject(format!(
554 "state id mismatch: expected {id}, computed {}",
555 state.id()
556 )));
557 }
558 self.put_state(&state)
559 }
560
561 fn put_action_serialized(&self, data: &[u8], id: ActionId) -> Result<()> {
562 let mut action: Action = rmp_serde::from_slice(data)?;
563 let found_id = action.compute_id();
564 if found_id != id {
565 return Err(HeddleError::InvalidObject(format!(
566 "action id mismatch: expected {}, found {}",
567 id, found_id
568 )));
569 }
570 let stored_id = self.put_action(&mut action)?;
571 if stored_id != id {
572 return Err(HeddleError::InvalidObject(format!(
573 "action id mismatch after write: expected {}, found {}",
574 id, stored_id
575 )));
576 }
577 Ok(())
578 }
579
580 fn get_pack_object(
581 &self,
582 id: &pack::PackObjectId,
583 ) -> Result<Option<(pack::ObjectType, Vec<u8>)>> {
584 match id {
585 pack::PackObjectId::AnnotatedTag(hash) => Ok(self
586 .get_annotated_tag(hash)?
587 .map(|tag| (pack::ObjectType::AnnotatedTag, tag.encode_current_msgpack()))),
588 pack::PackObjectId::Hash(hash) => {
589 if let Some(blob) = self.get_blob(hash)? {
590 return Ok(Some((pack::ObjectType::Blob, blob.content().to_vec())));
591 }
592 if let Some(tree) = self.get_tree(hash)? {
593 return Ok(Some((pack::ObjectType::Tree, tree.encode_canonical()?)));
594 }
595 if let Some(action) = self.get_action(&ActionId::from_hash(*hash))? {
596 return Ok(Some((
597 pack::ObjectType::Action,
598 rmp_serde::to_vec_named(&action)?,
599 )));
600 }
601 Ok(None)
602 }
603 pack::PackObjectId::StateId(change_id) => {
604 if let Some(state) = self.get_state(change_id)? {
605 Ok(Some((
606 pack::ObjectType::State,
607 state.encode_current_msgpack()?,
608 )))
609 } else {
610 Ok(None)
611 }
612 }
613 }
614 }
615
616 /// Bulk-write a batch of blobs as a single durable unit. The default
617 /// implementation falls back to per-blob writes; backends that
618 /// support packfiles (i.e. `FsStore`) override this to install one
619 /// packfile + index — two fsyncs total instead of N. Used by the
620 /// snapshot hot path so writing 1000 small files takes ~one fsync,
621 /// not 1000.
622 ///
623 /// Blobs already present in the store are skipped on the way in
624 /// (the caller would otherwise duplicate them in the pack).
625 fn put_blobs_packed(&self, blobs: Vec<(ContentHash, Vec<u8>)>) -> Result<()> {
626 for (hash, data) in blobs {
627 if !self.has_blob(&hash)? {
628 self.put_blob_bytes_with_hash(&data, hash)?;
629 }
630 }
631 Ok(())
632 }
633
634 /// Durably install a snapshot's newly-authored immutable object closure as
635 /// one storage batch. Pack-capable backends override this to share one pack
636 /// installation across blobs, the root tree, and the state; other backends
637 /// preserve the same ordering through their ordinary object methods.
638 fn put_snapshot_objects_packed(
639 &self,
640 blobs: Vec<(ContentHash, Vec<u8>)>,
641 tree: &Tree,
642 state: &State,
643 ) -> Result<()> {
644 self.put_blobs_packed(blobs)?;
645 self.put_tree(tree)?;
646 self.put_state(state)
647 }
648
649 /// Snapshot closure variant that also durably installs immutable authored
650 /// attachments. The separate method preserves the existing backend API;
651 /// pack-capable stores override it to share the snapshot pack barrier.
652 fn put_snapshot_objects_and_attachments_packed(
653 &self,
654 blobs: Vec<(ContentHash, Vec<u8>)>,
655 tree: &Tree,
656 state: &State,
657 attachments: Vec<StateAttachment>,
658 ) -> Result<()> {
659 self.put_snapshot_objects_packed(blobs, tree, state)?;
660 for attachment in attachments {
661 self.put_state_attachment(&attachment)?;
662 }
663 Ok(())
664 }
665 fn install_pack(&self, pack_data: &[u8], index_data: &[u8]) -> Result<Vec<pack::PackObjectId>> {
666 let reader = pack::PackReader::from_slice(pack_data, index_data)?;
667 let ids = reader.list_ids()?;
668 for id in &ids {
669 let Some((obj_type, data)) = reader.get_object(id)? else {
670 continue;
671 };
672 match (id, obj_type) {
673 (pack::PackObjectId::Hash(hash), pack::ObjectType::Blob) => {
674 self.put_blob_bytes_with_hash(&data, *hash)?;
675 }
676 (pack::PackObjectId::AnnotatedTag(hash), pack::ObjectType::AnnotatedTag) => {
677 let tag = AnnotatedTag::decode_current_msgpack(&data)
678 .map_err(|error| HeddleError::InvalidObject(error.to_string()))?;
679 if tag.hash() != *hash {
680 return Err(HeddleError::InvalidObject(
681 "annotated tag hash mismatch".to_string(),
682 ));
683 }
684 self.put_annotated_tag(&tag)?;
685 }
686 (pack::PackObjectId::Hash(hash), pack::ObjectType::Tree) => {
687 self.put_tree_serialized(&data, *hash)?;
688 }
689 (pack::PackObjectId::Hash(hash), pack::ObjectType::Action) => {
690 self.put_action_serialized(&data, ActionId::from_hash(*hash))?;
691 }
692 (pack::PackObjectId::StateId(change_id), pack::ObjectType::State) => {
693 self.put_state_serialized(&data, *change_id)?;
694 }
695 (_, pack::ObjectType::TimelineOperation) => {
696 return Err(HeddleError::InvalidObject(
697 "timeline operations belong in the timeline pack store".to_string(),
698 ));
699 }
700 _ => {
701 return Err(HeddleError::InvalidObject(format!(
702 "unsupported native pack object: {:?} {:?}",
703 id, obj_type
704 )));
705 }
706 }
707 }
708 Ok(ids)
709 }
710
711 /// Install a pack and its index from on-disk files
712 /// (typically produced by `StreamingPackBuilder`). The default
713 /// impl reads both files fully and delegates to `install_pack`,
714 /// so any backend that doesn't override this still works (at the
715 /// cost of giving back the bounded-memory promise). Real fs-
716 /// backed stores override this to `rename(2)` both files into the
717 /// pack directory without ever loading them.
718 ///
719 /// On success, the source files at `pack_path`/`index_path` may
720 /// have been moved or removed depending on the backend; callers
721 /// shouldn't continue to rely on them.
722 ///
723 /// Returns the ids of the installed objects — the same set
724 /// `install_pack` reports for the equivalent byte-buffer install,
725 /// so callers (e.g. native sync) read the installed ids off the
726 /// install result instead of tracking them out-of-band.
727 fn install_pack_streaming(
728 &self,
729 pack_path: &std::path::Path,
730 index_path: &std::path::Path,
731 ) -> Result<Vec<pack::PackObjectId>> {
732 let pack_data = std::fs::read(pack_path).map_err(StoreError::from)?;
733 let index_data = std::fs::read(index_path).map_err(StoreError::from)?;
734 let ids = self.install_pack(&pack_data, &index_data)?;
735 // Default impl: clean up the staged files. Override
736 // implementations that move/rename should not call super and
737 // should manage the file lifecycle themselves.
738 let _ = std::fs::remove_file(pack_path);
739 let _ = std::fs::remove_file(index_path);
740 Ok(ids)
741 }
742
743 fn begin_snapshot_write_batch(&self) -> Result<()> {
744 Ok(())
745 }
746
747 fn flush_snapshot_write_batch(&self) -> Result<()> {
748 Ok(())
749 }
750
751 fn abort_snapshot_write_batch(&self) {}
752}