Skip to main content

objects/store/fs/
fs_store.rs

1// SPDX-License-Identifier: Apache-2.0
2//! Core FsStore structure.
3
4#[cfg(test)]
5use std::sync::atomic::AtomicUsize;
6use std::{
7    collections::{BTreeSet, HashMap, VecDeque},
8    hash::Hash,
9    path::{Path, PathBuf},
10    sync::{
11        Arc, Mutex, RwLock,
12        atomic::{AtomicBool, Ordering},
13    },
14};
15
16use heddle_format::compression::CompressionConfig;
17
18use super::{
19    fs_io::{AtomicWriteMode, write_atomic},
20    fs_paths::{
21        actions_dir, blobs_dir, packs_dir, partial_trees_dir, states_dir, tree_lineage_dir,
22        trees_dir,
23    },
24    npk1::Npk1Manager,
25};
26use crate::{
27    fs_atomic::sync_directory,
28    object::{Blob, ContentHash, State, StateId, Tree},
29    store::{Result, SnapshotPackManager, pack::PackObjectId},
30};
31
32const RECENT_BLOB_CACHE_CAPACITY: usize = 2_048;
33const RECENT_TREE_CACHE_CAPACITY: usize = 1_024;
34/// Soft cap on the in-process loose-blob verification cache. Each
35/// entry is one `ContentHash` (~32 bytes) so this is ≈2 MB of memory
36/// for the upper bound, and clock eviction is bounded by hash
37/// hits rather than store size. 65k entries covers the typical hot
38/// working set for million-blob monorepos; a daemon that materialises
39/// dozens of unrelated trees won't drift toward unbounded growth.
40const VERIFIED_LOOSE_BLOB_CACHE_CAPACITY: usize = 65_536;
41/// Blobs larger than this are not stored in `recent_blobs` so a single
42/// multi-MB read cannot thrash the hot working set. 4 MiB matches the
43/// typical "large file" boundary used elsewhere in the object path.
44pub(super) const RECENT_BLOB_CACHE_MAX_BYTES: usize = 4 * 1024 * 1024;
45/// Total-byte budget for `recent_blobs`. Without it, populate-on-read
46/// could retain `RECENT_BLOB_CACHE_CAPACITY` (2048) × the 4 MiB
47/// per-entry gate ≈ 8 GiB of deep-cloned blob bytes for a read-only
48/// workload (mount / `heddled`) that streams many cold blobs. 256 MiB
49/// caps the resident blob-cache footprint while still holding a deep
50/// hot working set of small objects (the common case).
51pub(super) const RECENT_BLOB_CACHE_MAX_TOTAL_BYTES: usize = 256 * 1024 * 1024;
52
53thread_local! {
54    static SNAPSHOT_WRITE_BATCH_DEPTHS: std::cell::RefCell<HashMap<PathBuf, usize>> =
55        std::cell::RefCell::new(HashMap::new());
56}
57
58#[derive(Clone, Copy, Debug, Eq, PartialEq)]
59pub enum LooseObjectWriteMode {
60    Durable,
61    BatchDirectorySync,
62}
63
64/// Bounded in-process object cache with second-chance clock eviction.
65///
66/// Two independent caps are enforced on every [`insert`](Self::insert):
67///
68/// * `capacity` — the maximum entry *count*.
69/// * `byte_budget` — a soft cap on the cumulative *bytes* of the
70///   cached values, sized by the per-entry `sizer` closure. `None`
71///   disables the byte cap (caches whose values are effectively
72///   fixed-size, e.g. the `()`-valued verified-loose cache).
73///
74/// The byte budget is what keeps populate-on-read bounded: a read-only
75/// workload (mount / `heddled`) that streams many multi-MB blobs
76/// through `get_blob` can otherwise retain `capacity × max-entry-bytes`
77/// of deep-cloned `Vec`s. With the budget, inserting a new large blob
78/// advances the second-chance clock until the total fits.
79///
80/// [`get`](Self::get) marks the entry recently used through an atomic bit, so
81/// cache hits need only a shared map lock. Eviction advances a clock queue and
82/// gives marked entries one additional chance before removal. Both hits and
83/// amortized eviction stay O(1), including for the 65k-entry verification
84/// cache.
85#[derive(Debug)]
86pub(super) struct RecentObjectCache<K, V> {
87    entries: HashMap<K, RecentObjectCacheEntry<V>>,
88    eviction_clock: VecDeque<K>,
89    capacity: usize,
90    /// Soft cap on cumulative cached bytes; `None` = count-only.
91    byte_budget: Option<usize>,
92    /// `sizer(value)` in bytes. Only consulted when `byte_budget`
93    /// is `Some`.
94    sizer: fn(&V) -> usize,
95    /// Running sum of `sizer(v)` over all `entries`.
96    cached_bytes: usize,
97}
98
99#[derive(Debug)]
100struct RecentObjectCacheEntry<V> {
101    value: V,
102    recently_accessed: AtomicBool,
103}
104
105impl<K, V> RecentObjectCache<K, V>
106where
107    K: Copy + Eq + Hash,
108{
109    /// Count-capped cache with no byte budget. Used for caches whose
110    /// values are effectively fixed-size (e.g. the verified-loose
111    /// marker cache).
112    pub(super) fn with_capacity(capacity: usize) -> Self {
113        Self {
114            entries: HashMap::new(),
115            eviction_clock: VecDeque::new(),
116            capacity,
117            byte_budget: None,
118            sizer: |_| 0,
119            cached_bytes: 0,
120        }
121    }
122
123    /// Cache capped by *both* entry count and cumulative bytes.
124    /// `sizer` reports each value's heap-ish footprint; the cache
125    /// advances the second-chance clock until both caps hold.
126    pub(super) fn with_byte_budget(
127        capacity: usize,
128        byte_budget: usize,
129        sizer: fn(&V) -> usize,
130    ) -> Self {
131        Self {
132            entries: HashMap::new(),
133            eviction_clock: VecDeque::new(),
134            capacity,
135            byte_budget: Some(byte_budget),
136            sizer,
137            cached_bytes: 0,
138        }
139    }
140
141    /// Lookup with lock-free second-chance promotion inside an already-held
142    /// shared map lock. Only insertion and eviction require exclusive access.
143    pub(super) fn get(&self, key: &K) -> Option<&V> {
144        let entry = self.entries.get(key)?;
145        entry.recently_accessed.store(true, Ordering::Relaxed);
146        Some(&entry.value)
147    }
148
149    /// Presence check without promotion. Cheap enough to run under a
150    /// read lock — used both by verified-loose probes and by `has_*`
151    /// existence checks that must not serialize concurrent readers on
152    /// the exclusive write lock a promoting `get` would need.
153    pub(super) fn contains(&self, key: &K) -> bool {
154        self.entries.contains_key(key)
155    }
156
157    /// Drop `key` from the cache entirely. Returns the evicted value if
158    /// present. Targeted counterpart to the redaction-`purge` cache
159    /// drop: a purged blob's bytes must not linger in `recent_blobs`
160    /// where a long-lived process would keep serving (or reporting
161    /// present) the destroyed content. The production purge path drops
162    /// the whole cache via `clear_recent_caches` (it crosses the
163    /// generic `ObjectStore` seam); this per-key variant backs the
164    /// store-level `evict_recent_blob` used in tests.
165    #[cfg(test)]
166    pub(super) fn remove(&mut self, key: &K) -> Option<V> {
167        let removed = self.entries.remove(key)?.value;
168        self.cached_bytes = self.cached_bytes.saturating_sub((self.sizer)(&removed));
169        Some(removed)
170    }
171
172    pub(super) fn insert(&mut self, key: K, value: V) {
173        if self.capacity == 0 {
174            return;
175        }
176        let new_bytes = self.byte_budget.map(|_| (self.sizer)(&value)).unwrap_or(0);
177        let entry = RecentObjectCacheEntry {
178            value,
179            recently_accessed: AtomicBool::new(false),
180        };
181        if let Some(old) = self.entries.insert(key, entry) {
182            self.cached_bytes = self.cached_bytes.saturating_sub(
183                self.byte_budget
184                    .map(|_| (self.sizer)(&old.value))
185                    .unwrap_or(0),
186            );
187        } else {
188            self.eviction_clock.push_back(key);
189        }
190        self.cached_bytes += new_bytes;
191        self.evict_to_fit(key);
192    }
193
194    /// Advance the second-chance clock until both the count cap and the byte
195    /// budget hold. A recently read entry is marked cold and moved to the back
196    /// once before it can be evicted. The freshly inserted entry starts at the
197    /// back, so it is not the first target (a single entry larger than the
198    /// whole budget is kept — the budget is a soft cap, not a hard per-entry
199    /// gate; the per-entry `RECENT_BLOB_CACHE_MAX_BYTES` gate already bounds
200    /// the largest thing that reaches here).
201    fn evict_to_fit(&mut self, admitted_key: K) {
202        loop {
203            let over_count = self.entries.len() > self.capacity;
204            let over_bytes = self
205                .byte_budget
206                .is_some_and(|budget| self.cached_bytes > budget && self.entries.len() > 1);
207            if !over_count && !over_bytes {
208                break;
209            }
210            let Some(candidate) = self.eviction_clock.pop_front() else {
211                break;
212            };
213            let Some(entry) = self.entries.get(&candidate) else {
214                continue;
215            };
216            // The value that triggered this pass has not had an opportunity to
217            // serve a read yet. Keep it for this pass when another victim
218            // exists; otherwise an all-hot full cache would cycle through the
219            // residents and evict the new external object immediately.
220            if candidate == admitted_key && self.entries.len() > 1 {
221                self.eviction_clock.push_back(candidate);
222                continue;
223            }
224            if entry.recently_accessed.swap(false, Ordering::Relaxed) {
225                self.eviction_clock.push_back(candidate);
226                continue;
227            }
228            if let Some(evicted) = self.entries.remove(&candidate) {
229                self.cached_bytes = self.cached_bytes.saturating_sub(
230                    self.byte_budget
231                        .map(|_| (self.sizer)(&evicted.value))
232                        .unwrap_or(0),
233                );
234            }
235        }
236    }
237}
238
239/// Filesystem-based storage for Heddle objects.
240///
241/// Layout:
242/// ```text
243/// .heddle/
244///   objects/
245///     blobs/
246///       ab/
247///         cdef1234...
248///     trees/
249///       ab/
250///         cdef1234...
251///     states/
252///       <state_id>.state
253///   actions/
254///     <action_id>.action
255///   packs/
256///     <hash>.pack
257///     <hash>.idx
258/// ```
259pub struct FsStore {
260    pub(super) root: PathBuf,
261    pub(super) compression: CompressionConfig,
262    pub(super) snapshot_delta_search: bool,
263    pack_manager: RwLock<SnapshotPackManager>,
264    npk1_manager: RwLock<Npk1Manager>,
265    pub(super) recent_blobs: RwLock<RecentObjectCache<ContentHash, Blob>>,
266    pub(super) recent_trees: RwLock<RecentObjectCache<ContentHash, Tree>>,
267    pub(super) recent_states: RwLock<RecentObjectCache<StateId, State>>,
268    pub(super) external_source: Option<Arc<dyn super::super::ExternalObjectSource>>,
269    loose_object_write_mode: LooseObjectWriteMode,
270    pending_directory_syncs: Mutex<BTreeSet<PathBuf>>,
271    #[cfg(test)]
272    snapshot_batch_flushes: AtomicUsize,
273    /// In-process trust cache for loose-blob cache mirrors. A hash
274    /// enters this bounded clock cache when this process either (a) wrote the blob
275    /// itself via `promote_to_loose_uncompressed` or (b) successfully
276    /// hash-verified it on first read. Bytes-on-disk for any entry
277    /// in this cache can be trusted without a re-hash by subsequent
278    /// `loose_blob_path` calls within the same process.
279    ///
280    /// Capped at [`VERIFIED_LOOSE_BLOB_CACHE_CAPACITY`] entries so a
281    /// long-lived process (`heddled`) materialising many unrelated
282    /// trees doesn't drift into unbounded memory growth. Second-chance
283    /// eviction; an evicted hash pays one extra BLAKE3 on its next
284    /// read (cost-of-evict ≈ working-set-size BLAKE3 ops). Stored as
285    /// `RecentObjectCache<…, ()>` to share the clock-eviction
286    /// machinery with the other on-store caches; the unit value is
287    /// a marker that the corresponding loose mirror was verified.
288    ///
289    /// Pairs with `AtomicWriteMode::NoSync` on the write side: a
290    /// crashed promote leaves a torn cache-mirror file, but its
291    /// hash won't match on the next process's first-read verify,
292    /// so the reader falls through to a fresh promote off the pack.
293    pub(super) verified_loose_blobs: RwLock<RecentObjectCache<ContentHash, ()>>,
294}
295
296impl Clone for FsStore {
297    fn clone(&self) -> Self {
298        let mut cloned = Self::with_compression(&self.root, self.compression);
299        cloned.snapshot_delta_search = self.snapshot_delta_search;
300        cloned.loose_object_write_mode = self.loose_object_write_mode;
301        cloned.external_source = self.external_source.clone();
302        cloned
303    }
304}
305
306impl FsStore {
307    /// Create a new filesystem store rooted at the given path.
308    ///
309    /// The path should be the `.heddle` directory.
310    pub fn new(root: impl AsRef<Path>) -> Self {
311        let root = root.as_ref().to_path_buf();
312        let pack_manager = SnapshotPackManager::new(packs_dir(&root));
313        let npk1_manager = Npk1Manager::new(packs_dir(&root));
314        Self {
315            root,
316            compression: CompressionConfig::default(),
317            snapshot_delta_search: false,
318            pack_manager: RwLock::new(pack_manager),
319            npk1_manager: RwLock::new(npk1_manager),
320            recent_blobs: RwLock::new(RecentObjectCache::with_byte_budget(
321                RECENT_BLOB_CACHE_CAPACITY,
322                RECENT_BLOB_CACHE_MAX_TOTAL_BYTES,
323                |blob: &Blob| blob.content().len(),
324            )),
325            recent_trees: RwLock::new(RecentObjectCache::with_capacity(RECENT_TREE_CACHE_CAPACITY)),
326            recent_states: RwLock::new(RecentObjectCache::with_capacity(
327                RECENT_TREE_CACHE_CAPACITY,
328            )),
329            external_source: None,
330            loose_object_write_mode: LooseObjectWriteMode::Durable,
331            pending_directory_syncs: Mutex::new(BTreeSet::new()),
332            #[cfg(test)]
333            snapshot_batch_flushes: AtomicUsize::new(0),
334            verified_loose_blobs: RwLock::new(RecentObjectCache::with_capacity(
335                VERIFIED_LOOSE_BLOB_CACHE_CAPACITY,
336            )),
337        }
338    }
339
340    /// Create a new filesystem store with custom compression settings.
341    pub fn with_compression(root: impl AsRef<Path>, compression: CompressionConfig) -> Self {
342        let root = root.as_ref().to_path_buf();
343        let pack_manager = SnapshotPackManager::new(packs_dir(&root));
344        let npk1_manager = Npk1Manager::new(packs_dir(&root));
345        Self {
346            root,
347            compression,
348            snapshot_delta_search: false,
349            pack_manager: RwLock::new(pack_manager),
350            npk1_manager: RwLock::new(npk1_manager),
351            recent_blobs: RwLock::new(RecentObjectCache::with_byte_budget(
352                RECENT_BLOB_CACHE_CAPACITY,
353                RECENT_BLOB_CACHE_MAX_TOTAL_BYTES,
354                |blob: &Blob| blob.content().len(),
355            )),
356            recent_trees: RwLock::new(RecentObjectCache::with_capacity(RECENT_TREE_CACHE_CAPACITY)),
357            recent_states: RwLock::new(RecentObjectCache::with_capacity(
358                RECENT_TREE_CACHE_CAPACITY,
359            )),
360            external_source: None,
361            loose_object_write_mode: LooseObjectWriteMode::Durable,
362            pending_directory_syncs: Mutex::new(BTreeSet::new()),
363            #[cfg(test)]
364            snapshot_batch_flushes: AtomicUsize::new(0),
365            verified_loose_blobs: RwLock::new(RecentObjectCache::with_capacity(
366                VERIFIED_LOOSE_BLOB_CACHE_CAPACITY,
367            )),
368        }
369    }
370
371    /// Initialize the directory structure.
372    pub fn init(&self) -> Result<()> {
373        // Durable create so the object-store layout dirs survive crash
374        // between mkdir and first object write (L6 residual migration).
375        crate::fs_atomic::create_dir_all_durable(&blobs_dir(&self.root))?;
376        crate::fs_atomic::create_dir_all_durable(&trees_dir(&self.root))?;
377        crate::fs_atomic::create_dir_all_durable(&partial_trees_dir(&self.root))?;
378        crate::fs_atomic::create_dir_all_durable(&tree_lineage_dir(&self.root))?;
379        crate::fs_atomic::create_dir_all_durable(&states_dir(&self.root))?;
380        crate::fs_atomic::create_dir_all_durable(&actions_dir(&self.root))?;
381        crate::fs_atomic::create_dir_all_durable(&packs_dir(&self.root))?;
382        Ok(())
383    }
384
385    /// Get the root path.
386    pub fn root(&self) -> &Path {
387        &self.root
388    }
389
390    /// Get the compression configuration.
391    pub fn compression(&self) -> CompressionConfig {
392        self.compression
393    }
394
395    /// Set the compression configuration.
396    pub fn set_compression(&mut self, compression: CompressionConfig) {
397        self.compression = compression;
398    }
399
400    /// Enable or disable sliding-window delta search for snapshot packs.
401    pub fn set_snapshot_delta_search(&mut self, enabled: bool) {
402        self.snapshot_delta_search = enabled;
403    }
404
405    pub fn loose_object_write_mode(&self) -> LooseObjectWriteMode {
406        self.loose_object_write_mode
407    }
408
409    pub fn set_loose_object_write_mode(&mut self, mode: LooseObjectWriteMode) {
410        self.loose_object_write_mode = mode;
411    }
412
413    /// Configure a read-through source for objects not present in the native
414    /// store. Writes always remain native.
415    pub fn set_external_source(&mut self, source: Arc<dyn super::super::ExternalObjectSource>) {
416        self.external_source = Some(source);
417    }
418
419    fn flush_pending_directory_syncs(&self) -> Result<usize> {
420        let pending_dirs = {
421            let mut guard = self.pending_directory_syncs.lock().map_err(|_| {
422                crate::store::HeddleError::Config(
423                    "Failed to acquire pending directory sync lock".to_string(),
424                )
425            })?;
426            if guard.is_empty() {
427                return Ok(0);
428            }
429            let dirs = guard.iter().cloned().collect::<Vec<_>>();
430            guard.clear();
431            dirs
432        };
433
434        for (index, dir) in pending_dirs.iter().enumerate() {
435            if let Err(error) = sync_directory(dir) {
436                if let Ok(mut guard) = self.pending_directory_syncs.lock() {
437                    guard.extend(pending_dirs[index..].iter().cloned());
438                }
439                return Err(error.into());
440            }
441        }
442
443        Ok(pending_dirs.len())
444    }
445
446    /// Reload pack files from disk.
447    ///
448    /// Runs L8 install-intent recovery first so crash windows between pack
449    /// and index publish are finished or aborted before packs are loaded.
450    /// Uses the default intent TTL so abandoned staging is swept.
451    pub fn reload_packs(&self) -> Result<()> {
452        let packs = packs_dir(&self.root);
453        let _ = super::pack_install_journal::recover_pack_install_intents_with_ttl(
454            &packs,
455            Some(super::pack_install_journal::DEFAULT_PACK_INSTALL_INTENT_TTL_SECS),
456        )?;
457        // Option D backstop: remove any legacy unpaired packs without intent.
458        let _ = super::fs_pack::prune_unpaired_pack_files(&packs)?;
459        let mut manager = self.pack_manager.write().map_err(|_| {
460            crate::store::HeddleError::Config("Failed to acquire pack manager lock".to_string())
461        })?;
462        manager.reload()?;
463        drop(manager);
464        let mut npk1 = self.npk1_manager.write().map_err(|_| {
465            crate::store::HeddleError::Config("Failed to acquire NPK1 manager lock".to_string())
466        })?;
467        npk1.reload()
468    }
469
470    /// Reload pack files only if the immutable pack set changed on disk.
471    /// Cheap discovery when nothing changed; full reload when a sibling
472    /// `FsStore` installed a pack or atomically replaced a generation.
473    ///
474    /// Returns `true` when a reload happened. Used by `get_*` and
475    /// `has_*` paths after an in-memory miss to recover from the
476    /// "two FsStores backing the same `.heddle/` directory" case
477    /// (typical for lightweight thread worktrees).
478    ///
479    /// Double-checked locking: the read-lock fast path means a
480    /// thundering herd of concurrent misses doesn't serialize on
481    /// the write lock; only the first thread that observes a stale
482    /// view escalates and does the reload.
483    pub(super) fn reload_packs_if_stale(&self) -> Result<bool> {
484        // Fast path: read-lock and bail out if the disk snapshot still matches.
485        let generic_stale = {
486            let manager = self.pack_manager.read().map_err(|_| {
487                crate::store::HeddleError::Config("Failed to acquire pack manager lock".to_string())
488            })?;
489            manager.needs_reload()?
490        };
491        let npk1_stale = {
492            let manager = self.npk1_manager.read().map_err(|_| {
493                crate::store::HeddleError::Config("Failed to acquire NPK1 manager lock".to_string())
494            })?;
495            manager.needs_reload()?
496        };
497        if !generic_stale && !npk1_stale {
498            return Ok(false);
499        }
500        // Slow path: take the write lock and re-check (another
501        // thread may have already reloaded between our drop and
502        // re-acquire).
503        let mut manager = self.pack_manager.write().map_err(|_| {
504            crate::store::HeddleError::Config("Failed to acquire pack manager lock".to_string())
505        })?;
506        let generic_reloaded = manager.reload_if_stale()?;
507        drop(manager);
508        let mut npk1 = self.npk1_manager.write().map_err(|_| {
509            crate::store::HeddleError::Config("Failed to acquire NPK1 manager lock".to_string())
510        })?;
511        let npk1_reloaded = if npk1.needs_reload()? {
512            npk1.reload()?;
513            true
514        } else {
515            false
516        };
517        Ok(generic_reloaded || npk1_reloaded)
518    }
519
520    /// Get the pack manager for pack operations.
521    pub fn pack_manager(&self) -> &RwLock<SnapshotPackManager> {
522        &self.pack_manager
523    }
524
525    pub(super) fn npk1_manager(&self) -> &RwLock<Npk1Manager> {
526        &self.npk1_manager
527    }
528
529    pub fn clear_recent_object_caches(&self) {
530        if let Ok(mut blobs) = self.recent_blobs.write() {
531            *blobs = RecentObjectCache::with_byte_budget(
532                RECENT_BLOB_CACHE_CAPACITY,
533                RECENT_BLOB_CACHE_MAX_TOTAL_BYTES,
534                |blob: &Blob| blob.content().len(),
535            );
536        }
537        if let Ok(mut trees) = self.recent_trees.write() {
538            *trees = RecentObjectCache::with_capacity(RECENT_TREE_CACHE_CAPACITY);
539        }
540        if let Ok(mut states) = self.recent_states.write() {
541            *states = RecentObjectCache::with_capacity(RECENT_TREE_CACHE_CAPACITY);
542        }
543    }
544
545    /// Drop a single blob hash from the in-process `recent_blobs`
546    /// cache. Targeted counterpart to the redaction-`purge` cache drop:
547    /// after the loose bytes are physically deleted, a long-lived
548    /// process must not keep serving (or reporting present) the purged
549    /// content from cache. Idempotent — a miss is a no-op. Test-only:
550    /// the production purge path crosses the generic `ObjectStore` seam
551    /// and drops the whole cache via `clear_recent_caches`.
552    #[cfg(test)]
553    pub(super) fn evict_recent_blob(&self, hash: &ContentHash) {
554        if let Ok(mut cache) = self.recent_blobs.write() {
555            cache.remove(hash);
556        }
557    }
558
559    pub fn pack_ids(&self) -> Result<Vec<PackObjectId>> {
560        let manager = self.pack_manager.read().map_err(|_| {
561            crate::store::HeddleError::Config("Failed to acquire pack manager lock".to_string())
562        })?;
563        let mut ids = manager.list_all_ids()?;
564        drop(manager);
565        let npk1 = self.npk1_manager.read().map_err(|_| {
566            crate::store::HeddleError::Config("Failed to acquire NPK1 manager lock".to_string())
567        })?;
568        ids.extend(npk1.list_ids()?.into_iter().map(PackObjectId::Hash));
569        ids.sort();
570        ids.dedup();
571        Ok(ids)
572    }
573
574    pub(super) fn write_loose_object_atomic(&self, path: &Path, data: &[u8]) -> Result<()> {
575        let batch_active = SNAPSHOT_WRITE_BATCH_DEPTHS
576            .with(|depths| depths.borrow().get(&self.root).copied().unwrap_or_default() > 0);
577        let configured_mode = if batch_active {
578            LooseObjectWriteMode::BatchDirectorySync
579        } else {
580            self.loose_object_write_mode
581        };
582
583        let mode = match configured_mode {
584            LooseObjectWriteMode::Durable => AtomicWriteMode::Durable,
585            LooseObjectWriteMode::BatchDirectorySync => AtomicWriteMode::BatchDirectorySync,
586        };
587        write_atomic(path, data, mode, Some(&self.pending_directory_syncs))
588    }
589
590    /// Durable atomic write for pack/index bytes when not going through the
591    /// L8 journal (tests / rare call sites). Prefer
592    /// [`super::pack_install_journal::install_pack_bytes_journaled`].
593    #[allow(dead_code)]
594    pub(super) fn write_pack_atomic(&self, path: &Path, data: &[u8]) -> Result<()> {
595        write_atomic(path, data, AtomicWriteMode::Durable, None)
596    }
597
598    /// Atomic write tuned for *cache-mirror* loose objects: no fsync
599    /// at any level. The authoritative copy lives in a pack; if a
600    /// crash leaves the cache mirror torn, the read-side hash check
601    /// catches it and `promote_to_loose_uncompressed` rebuilds it
602    /// from the pack on the next access.
603    ///
604    /// On macOS APFS, `sync_data` alone costs ~5 ms per call (it
605    /// behaves like `F_FULLFSYNC` for tiny writes), and the parent
606    /// directory fsync is ~3-10 ms on top. For 1k blobs, that's
607    /// 5-15 seconds of pure fsync wallclock — the dominant cost in
608    /// the cold materialize path. Dropping both pays back ~30× on
609    /// raw create+rename throughput (measured: 200/s with sync_data
610    /// vs 5500/s without).
611    ///
612    /// Safety contract: this is only valid for files whose authority
613    /// lives elsewhere. Used by `promote_to_loose_uncompressed`; the
614    /// matching `loose_blob_path` reader hash-verifies before
615    /// trusting the bytes. Do *not* use for `put_blob` / `put_tree`
616    /// / `put_state` — those are the authoritative copy and must
617    /// survive a crash.
618    pub(super) fn write_loose_object_cache(&self, path: &Path, data: &[u8]) -> Result<()> {
619        self.write_reconstructible_cache(path, data)
620    }
621
622    /// Atomically publish reconstructible cache bytes without a durability
623    /// barrier. The caller must be able to rebuild the file from an
624    /// authoritative object after a crash.
625    pub(super) fn write_reconstructible_cache(&self, path: &Path, data: &[u8]) -> Result<()> {
626        write_atomic(path, data, AtomicWriteMode::NoSync, None)
627    }
628
629    pub(super) fn begin_snapshot_write_batch_impl(&self) -> Result<()> {
630        SNAPSHOT_WRITE_BATCH_DEPTHS.with(|depths| {
631            *depths.borrow_mut().entry(self.root.clone()).or_default() += 1;
632        });
633        Ok(())
634    }
635
636    pub(super) fn flush_snapshot_write_batch_impl(&self) -> Result<()> {
637        let had_batch = SNAPSHOT_WRITE_BATCH_DEPTHS.with(|depths| {
638            let mut depths = depths.borrow_mut();
639            let Some(depth) = depths.get_mut(&self.root) else {
640                return false;
641            };
642            *depth -= 1;
643            if *depth == 0 {
644                depths.remove(&self.root);
645            }
646            true
647        });
648        if !had_batch {
649            return Ok(());
650        }
651
652        #[cfg(test)]
653        self.snapshot_batch_flushes.fetch_add(1, Ordering::Relaxed);
654
655        // Batches may overlap across snapshot preparers. Each successful
656        // preparer must establish durability for its own writes before it can
657        // publish an oplog edge, even while another batch remains active.
658        // Draining the shared set is safe: entries taken by another flush are
659        // already durable, and every write from this batch was queued before
660        // this call acquired the set.
661        let _ = self.flush_pending_directory_syncs()?;
662        Ok(())
663    }
664
665    pub(super) fn abort_snapshot_write_batch_impl(&self) {
666        let should_flush = SNAPSHOT_WRITE_BATCH_DEPTHS.with(|depths| {
667            let mut depths = depths.borrow_mut();
668            let Some(depth) = depths.get_mut(&self.root) else {
669                // A preceding flush may have removed the thread-local batch
670                // before its directory sync failed. Preserve abort's
671                // conservative retry of those pending syncs.
672                return true;
673            };
674            *depth -= 1;
675            if *depth == 0 {
676                depths.remove(&self.root);
677                true
678            } else {
679                false
680            }
681        });
682        // Immutable objects staged by a failed snapshot are harmless orphans.
683        // Never clear another concurrent preparation's pending directory syncs;
684        // when this was the last batch, conservatively make every staged rename
685        // durable before returning.
686        if should_flush {
687            let _ = self.flush_pending_directory_syncs();
688        }
689    }
690
691    #[cfg(test)]
692    pub(super) fn pending_directory_sync_count(&self) -> usize {
693        self.pending_directory_syncs
694            .lock()
695            .map(|pending| pending.len())
696            .unwrap_or(0)
697    }
698
699    #[cfg(test)]
700    pub(super) fn snapshot_batch_flush_count(&self) -> usize {
701        self.snapshot_batch_flushes.load(Ordering::Relaxed)
702    }
703}