Skip to main content

heddle_fs_prims/
fs_atomic.rs

1// SPDX-License-Identifier: Apache-2.0
2use std::{
3    fs::{self, File, OpenOptions},
4    io::{self, Write},
5    path::{Path, PathBuf},
6    sync::{
7        Arc, Mutex, OnceLock,
8        atomic::{AtomicU64, Ordering},
9    },
10    time::{SystemTime, UNIX_EPOCH},
11};
12
13#[derive(Default)]
14struct CloneDurabilityStats {
15    barriers: AtomicU64,
16    skipped: AtomicU64,
17}
18
19#[derive(Clone)]
20struct CloneDurabilityEntry {
21    root: PathBuf,
22    stats: Arc<CloneDurabilityStats>,
23}
24
25fn clone_durability_entries() -> &'static Mutex<Vec<CloneDurabilityEntry>> {
26    static ENTRIES: OnceLock<Mutex<Vec<CloneDurabilityEntry>>> = OnceLock::new();
27    ENTRIES.get_or_init(|| Mutex::new(Vec::new()))
28}
29
30fn deferred_clone_stats(path: &Path) -> Option<Arc<CloneDurabilityStats>> {
31    clone_durability_entries()
32        .lock()
33        .ok()?
34        .iter()
35        .rev()
36        .find(|entry| path.starts_with(&entry.root))
37        .map(|entry| Arc::clone(&entry.stats))
38}
39
40pub fn clone_write_is_deferred(path: &Path) -> bool {
41    deferred_clone_stats(path).is_some()
42}
43
44pub fn record_deferred_clone_barrier(path: &Path) {
45    if let Some(stats) = deferred_clone_stats(path) {
46        stats.skipped.fetch_add(1, Ordering::Relaxed);
47    }
48}
49
50/// Path-scoped durability suppression for reconstructible clone writes.
51///
52/// The caller must first persist a clone-intent marker outside this scope and
53/// must call [`commit`](Self::commit) only after hash-verifying the fetched
54/// closure. Writes outside `root` retain their ordinary per-operation fsyncs.
55pub struct CloneDurabilityBatch {
56    root: PathBuf,
57    stats: Arc<CloneDurabilityStats>,
58}
59
60impl CloneDurabilityBatch {
61    pub fn begin(root: impl AsRef<Path>) -> Self {
62        let root = root.as_ref().to_path_buf();
63        let stats = Arc::new(CloneDurabilityStats::default());
64        clone_durability_entries()
65            .lock()
66            .unwrap_or_else(std::sync::PoisonError::into_inner)
67            .push(CloneDurabilityEntry {
68                root: root.clone(),
69                stats: Arc::clone(&stats),
70            });
71        Self { root, stats }
72    }
73
74    /// Flush every dirty file and directory on the destination filesystem in
75    /// one kernel barrier. Refs remain unpublished while this runs.
76    pub fn commit(&self) -> io::Result<()> {
77        sync_filesystem(&self.root)?;
78        self.stats.barriers.fetch_add(1, Ordering::Relaxed);
79        Ok(())
80    }
81
82    pub fn barrier_count(&self) -> u64 {
83        self.stats.barriers.load(Ordering::Relaxed)
84    }
85
86    pub fn skipped_barrier_count(&self) -> u64 {
87        self.stats.skipped.load(Ordering::Relaxed)
88    }
89}
90
91impl Drop for CloneDurabilityBatch {
92    fn drop(&mut self) {
93        let mut entries = clone_durability_entries()
94            .lock()
95            .unwrap_or_else(std::sync::PoisonError::into_inner);
96        if let Some(index) = entries
97            .iter()
98            .rposition(|entry| entry.root == self.root && Arc::ptr_eq(&entry.stats, &self.stats))
99        {
100            entries.remove(index);
101        }
102    }
103}
104
105#[derive(Clone, Copy)]
106enum AtomicWriteKind {
107    Normal,
108    Secret,
109}
110
111impl AtomicWriteKind {
112    fn open_tmp(self, tmp: &Path) -> io::Result<File> {
113        let mut options = OpenOptions::new();
114        options.create_new(true).write(true);
115
116        #[cfg(unix)]
117        if matches!(self, Self::Secret) {
118            use std::os::unix::fs::OpenOptionsExt;
119            options.mode(0o600);
120        }
121
122        options.open(tmp)
123    }
124
125    fn enforce_before_write(self, file: &File) -> io::Result<()> {
126        match self {
127            Self::Normal => Ok(()),
128            Self::Secret => enforce_secret_permissions_before_write(file),
129        }
130    }
131}
132
133#[cfg(unix)]
134fn enforce_secret_permissions_before_write(file: &File) -> io::Result<()> {
135    use std::os::unix::fs::PermissionsExt;
136
137    file.set_permissions(fs::Permissions::from_mode(0o600))?;
138    let mode = file.metadata()?.permissions().mode() & 0o777;
139    if mode != 0o600 {
140        return Err(io::Error::new(
141            io::ErrorKind::PermissionDenied,
142            format!("secret temp file permissions are {mode:o}, expected 600"),
143        ));
144    }
145    Ok(())
146}
147
148#[cfg(not(unix))]
149fn enforce_secret_permissions_before_write(_file: &File) -> io::Result<()> {
150    // Non-Unix platforms do not expose POSIX mode bits through
151    // OpenOptions. The secret variant still uses the same create-new,
152    // write-fsync-rename discipline, but cannot verify a 0600 mode.
153    Ok(())
154}
155
156static TEMP_PATH_COUNTER: AtomicU64 = AtomicU64::new(0);
157
158/// POSIX `ENOSPC`. Identical on Linux and macOS. Windows surfaces disk-full
159/// as `ERROR_DISK_FULL` (112) or `ERROR_HANDLE_DISK_FULL` (39); we cover
160/// those by also checking `ErrorKind::StorageFull` (stable as of 1.83) and
161/// the older `ErrorKind::Other` "no space" message text as a fallback.
162const ENOSPC: i32 = 28;
163
164/// POSIX `ENOTEMPTY`. Linux=39, macOS/BSD=66. Windows surfaces this as
165/// `ERROR_DIR_NOT_EMPTY` (145). `ErrorKind::DirectoryNotEmpty` covers the
166/// portable case, but the raw codes are the canonical signal — Rust may
167/// still surface raw OS errors for paths the kernel reports unusually.
168const ENOTEMPTY_LINUX: i32 = 39;
169const ENOTEMPTY_MACOS: i32 = 66;
170const ENOTEMPTY_WINDOWS: i32 = 145;
171
172/// POSIX `EACCES`. Same code on Linux and macOS. `ErrorKind::PermissionDenied`
173/// covers Windows `ERROR_ACCESS_DENIED` (5) too.
174const EACCES: i32 = 13;
175
176/// POSIX `ENOENT`. Same code on Linux and macOS. `ErrorKind::NotFound` covers
177/// Windows `ERROR_FILE_NOT_FOUND` (2) and `ERROR_PATH_NOT_FOUND` (3).
178const ENOENT: i32 = 2;
179
180/// POSIX `EROFS`. Linux=30, macOS=30. `ErrorKind::ReadOnlyFilesystem` is
181/// the portable variant (stable as of 1.83).
182const EROFS: i32 = 30;
183
184/// POSIX `EXDEV` ("cross-device link"). Linux=18, macOS=18.
185/// `ErrorKind::CrossesDevices` is the portable variant (stable as of 1.83).
186const EXDEV: i32 = 18;
187
188/// Returns true when an `io::Error` indicates the filesystem is out of
189/// space. Centralised here because it's the same predicate used by
190/// `write_file_atomic` (the inner helper) and by the higher-level
191/// `cmd_snapshot` recovery path that prints the actionable message.
192pub fn is_out_of_space(err: &io::Error) -> bool {
193    if err.raw_os_error() == Some(ENOSPC) {
194        return true;
195    }
196    // `ErrorKind::StorageFull` is the portable kind. It maps to ENOSPC
197    // on Unix and the Windows disk-full codes. Available since Rust
198    // 1.83; the workspace MSRV is well past that.
199    if err.kind() == io::ErrorKind::StorageFull {
200        return true;
201    }
202    // `write_all` translates a short write into `WriteZero`. On a full
203    // disk, kernel can return a short write rather than ENOSPC outright
204    // (especially over network filesystems), so a `WriteZero` we couldn't
205    // otherwise classify is treated as out-of-space — overly inclusive
206    // here is safer than missing the signal.
207    if err.kind() == io::ErrorKind::WriteZero {
208        return true;
209    }
210    false
211}
212
213/// Returns true when an `io::Error` indicates a directory could not be
214/// removed because it still contained entries. The apply planner only removes
215/// tracked descendants; when tracked content is removed and the parent
216/// directory still holds untracked or explicitly ignored siblings, `remove_dir`
217/// returns this signal. We need both `ErrorKind::DirectoryNotEmpty` and the raw
218/// codes — Linux=39, macOS/BSD=66, Windows=145 — because Rust does not
219/// always translate every kernel surface into the portable `ErrorKind`.
220pub fn is_directory_not_empty(err: &io::Error) -> bool {
221    if err.kind() == io::ErrorKind::DirectoryNotEmpty {
222        return true;
223    }
224    matches!(
225        err.raw_os_error(),
226        Some(ENOTEMPTY_LINUX) | Some(ENOTEMPTY_MACOS) | Some(ENOTEMPTY_WINDOWS)
227    )
228}
229
230/// Returns true when an `io::Error` indicates the operation was denied
231/// for permissions reasons (`EACCES` on Unix, `ERROR_ACCESS_DENIED` on
232/// Windows). The portable `ErrorKind::PermissionDenied` covers most
233/// surfaces; the raw `EACCES` check handles oddball platforms that
234/// surface the OS code without translating to the portable kind.
235pub fn is_permission_denied(err: &io::Error) -> bool {
236    if err.kind() == io::ErrorKind::PermissionDenied {
237        return true;
238    }
239    err.raw_os_error() == Some(EACCES)
240}
241
242/// Returns true when an `io::Error` indicates the path referenced by an
243/// operation does not exist (`ENOENT` on Unix, `ERROR_FILE_NOT_FOUND` /
244/// `ERROR_PATH_NOT_FOUND` on Windows). Use this *only* at call sites
245/// where the operation expected the path to exist — the predicate alone
246/// can't distinguish "I expected this" from "I checked optionally".
247pub fn is_not_found(err: &io::Error) -> bool {
248    if err.kind() == io::ErrorKind::NotFound {
249        return true;
250    }
251    err.raw_os_error() == Some(ENOENT)
252}
253
254/// Returns true when an `io::Error` indicates the underlying filesystem
255/// is mounted read-only (`EROFS` on Unix). The portable
256/// `ErrorKind::ReadOnlyFilesystem` is preferred when present; we also
257/// match the raw OS code because some platforms (notably older macOS
258/// surfaces and certain remote filesystems) do not always translate.
259pub fn is_read_only_filesystem(err: &io::Error) -> bool {
260    if err.kind() == io::ErrorKind::ReadOnlyFilesystem {
261        return true;
262    }
263    err.raw_os_error() == Some(EROFS)
264}
265
266/// Returns true when an `io::Error` indicates a `rename` (or other
267/// link-style operation) attempted to bridge two filesystems (`EXDEV`).
268/// This is what trips when `temp_path` lands on a different mount than
269/// the destination — typically because `TMPDIR` is on a different volume,
270/// or the parent directory itself is a bind mount. We match both the
271/// portable `ErrorKind::CrossesDevices` and the raw `EXDEV` code.
272pub fn is_cross_device_link(err: &io::Error) -> bool {
273    if err.kind() == io::ErrorKind::CrossesDevices {
274        return true;
275    }
276    err.raw_os_error() == Some(EXDEV)
277}
278
279pub fn temp_path(path: &Path) -> PathBuf {
280    let parent = path.parent().unwrap_or_else(|| Path::new("."));
281    let file_name = path
282        .file_name()
283        .and_then(|s| s.to_str())
284        .filter(|s| !s.is_empty())
285        .unwrap_or("heddle-tmp");
286    let unique = SystemTime::now()
287        .duration_since(UNIX_EPOCH)
288        .map(|d| d.as_nanos())
289        .unwrap_or(0);
290    let counter = TEMP_PATH_COUNTER.fetch_add(1, Ordering::Relaxed);
291    let pid = std::process::id();
292    parent.join(format!(".{file_name}.tmp-{pid}-{unique}-{counter}"))
293}
294
295/// Kick a file's dirty page cache into background writeback WITHOUT waiting
296/// for it or issuing a device flush. Best-effort: any error is ignored, since
297/// the caller's subsequent `fsync` is what actually guarantees durability —
298/// this only *starts* the I/O early so many files' writeback overlaps instead
299/// of each `fsync` flushing its file synchronously from scratch.
300///
301/// Linux-only (`sync_file_range`); a no-op elsewhere, where the batched-fsync
302/// pass in [`stage_temp_files_durable`] simply runs without the overlap.
303#[cfg(target_os = "linux")]
304fn kick_writeback(file: &File) {
305    use std::os::unix::io::AsRawFd;
306    // SYNC_FILE_RANGE_WRITE = 2: initiate writeback of dirty pages in the
307    // given range (0..0 = whole file) without blocking. No barrier, no error
308    // path — a failure just means the later `sync_all` does the work.
309    const SYNC_FILE_RANGE_WRITE: libc::c_uint = 2;
310    unsafe {
311        libc::sync_file_range(file.as_raw_fd(), 0, 0, SYNC_FILE_RANGE_WRITE);
312    }
313}
314
315#[cfg(not(target_os = "linux"))]
316fn kick_writeback(_file: &File) {}
317
318/// Write many temp files with a single overlapped-writeback durability pass.
319///
320/// For each `(temp_path, bytes)`: create the temp file and write its contents,
321/// then start its page-cache writeback in the background ([`kick_writeback`]).
322/// After every file is written, `fsync` each one. On return, every temp file's
323/// data is on stable storage — the SAME guarantee as writing + `fsync`-ing each
324/// file individually — but the writeback I/O overlaps instead of serializing
325/// one synchronous `fsync` barrier per file.
326///
327/// This is the bulk-ref hot path (`heddle import local` of N branches publishes N ref
328/// files in one batch): the per-file `write → fsync` loop paid ~N serial fsync
329/// barriers (~2.3s for 800 refs on a local SSD); overlapping the writeback
330/// collapses that to ~0.1s with no change to the durability contract. Callers
331/// still `rename` each temp into place and `fsync` the parent directory to make
332/// the renames durable.
333///
334/// The temp files' parent directories must already exist. On the first write
335/// error the partial temp files are left for the caller's rollback/cleanup to
336/// remove (they are uniquely named and never renamed into place).
337pub fn stage_temp_files_durable(files: &[(PathBuf, Vec<u8>)]) -> io::Result<()> {
338    let mut handles: Vec<File> = Vec::with_capacity(files.len());
339    for (temp_path, bytes) in files {
340        let mut file = File::create(temp_path).map_err(|err| enrich_write_error(temp_path, err))?;
341        file.write_all(bytes)
342            .map_err(|err| enrich_write_error(temp_path, err))?;
343        kick_writeback(&file);
344        handles.push(file);
345    }
346    // Barrier pass: by now most files' writeback is already in flight (or done),
347    // so each `sync_all` blocks only on the tail, not a cold synchronous flush.
348    for (file, (temp_path, _)) in handles.iter().zip(files) {
349        sync_file(file, temp_path).map_err(|err| enrich_write_error(temp_path, err))?;
350    }
351    Ok(())
352}
353
354/// Unix directory barrier. Windows cannot flush a directory handle without
355/// privileges: this compatibility helper has no durability guarantee there.
356/// Durable file publication must use `durable_rename`, which uses a non-privileged
357/// write-through move on Windows. Installation journals must never use this
358/// helper as a Windows durability barrier.
359#[cfg(windows)]
360pub fn sync_directory(_path: &Path) -> io::Result<()> {
361    Ok(())
362}
363
364#[cfg(not(windows))]
365pub fn sync_directory(path: &Path) -> io::Result<()> {
366    if let Some(stats) = deferred_clone_stats(path) {
367        stats.skipped.fetch_add(1, Ordering::Relaxed);
368        return Ok(());
369    }
370    let dir = OpenOptions::new().read(true).open(path)?;
371    dir.sync_all()
372}
373
374/// Publish flushed file data and durably update the source/destination entries.
375/// Windows uses MoveFileExW write-through; Unix syncs both rename directories.
376/// Existing reconstructible clone scopes can defer Unix barriers; installation
377/// uses `Directory::durable_rename` to bypass that explicit deferral.
378pub fn durable_rename(source: &Path, destination: &Path) -> io::Result<()> {
379    #[cfg(windows)]
380    {
381        use std::os::windows::ffi::OsStrExt;
382
383        use windows_sys::Win32::Storage::FileSystem::{
384            MOVEFILE_REPLACE_EXISTING, MOVEFILE_WRITE_THROUGH, MoveFileExW,
385        };
386        fn wide(path: &Path) -> io::Result<Vec<u16>> {
387            let value: Vec<_> = std::path::absolute(path)?
388                .as_os_str()
389                .encode_wide()
390                .collect();
391            if value.contains(&0) {
392                return Err(io::Error::new(io::ErrorKind::InvalidInput, "NUL in path"));
393            }
394            Ok(value.into_iter().chain(std::iter::once(0)).collect())
395        }
396        let source = wide(source)?;
397        let destination = wide(destination)?;
398        // SAFETY: live NUL-terminated UTF-16 paths; no delayed or cross-volume move.
399        if unsafe {
400            MoveFileExW(
401                source.as_ptr(),
402                destination.as_ptr(),
403                MOVEFILE_REPLACE_EXISTING | MOVEFILE_WRITE_THROUGH,
404            )
405        } == 0
406        {
407            return Err(io::Error::last_os_error());
408        }
409        Ok(())
410    }
411    #[cfg(not(windows))]
412    {
413        fs::rename(source, destination)?;
414        let destination_parent = destination.parent().unwrap_or_else(|| Path::new("."));
415        sync_directory(destination_parent)?;
416        let source_parent = source.parent().unwrap_or_else(|| Path::new("."));
417        if source_parent != destination_parent {
418            sync_directory(source_parent)?;
419        }
420        Ok(())
421    }
422}
423
424/// Sync one file unless it belongs to an active clone durability batch.
425pub fn sync_file(file: &File, path: &Path) -> io::Result<()> {
426    if let Some(stats) = deferred_clone_stats(path) {
427        stats.skipped.fetch_add(1, Ordering::Relaxed);
428        return Ok(());
429    }
430    file.sync_all()
431}
432
433pub fn sync_file_data(file: &File, path: &Path) -> io::Result<()> {
434    if let Some(stats) = deferred_clone_stats(path) {
435        stats.skipped.fetch_add(1, Ordering::Relaxed);
436        return Ok(());
437    }
438    file.sync_data()
439}
440
441#[cfg(any(target_os = "linux", target_os = "android"))]
442fn sync_filesystem(path: &Path) -> io::Result<()> {
443    use std::os::fd::AsRawFd;
444
445    let file = OpenOptions::new().read(true).open(path)?;
446    // SAFETY: `file` owns a live descriptor for the duration of the call.
447    if unsafe { libc::syncfs(file.as_raw_fd()) } == 0 {
448        Ok(())
449    } else {
450        Err(io::Error::last_os_error())
451    }
452}
453
454#[cfg(all(unix, not(any(target_os = "linux", target_os = "android"))))]
455fn sync_filesystem(_path: &Path) -> io::Result<()> {
456    // `syncfs(2)` is Linux-specific. `sync(2)` is the closest portable Unix
457    // whole-filesystem commit primitive and avoids a barrier per clone object.
458    unsafe { libc::sync() };
459    Ok(())
460}
461
462#[cfg(windows)]
463fn sync_filesystem(path: &Path) -> io::Result<()> {
464    // Windows exposes no non-privileged `syncfs` equivalent. Keep one logical
465    // clone commit phase, flushing every clone file only after verification;
466    // directory metadata is covered by NTFS journaling (see `sync_directory`).
467    for entry in fs::read_dir(path)? {
468        let entry = entry?;
469        let file_type = entry.file_type()?;
470        if file_type.is_dir() {
471            sync_filesystem(&entry.path())?;
472        } else if file_type.is_file() {
473            OpenOptions::new()
474                .read(true)
475                .open(entry.path())?
476                .sync_all()?;
477        }
478    }
479    Ok(())
480}
481
482/// Collect missing path components (deepest-first) and the deepest pre-existing
483/// parent that will hold the first new dirent. Used by durable dir creators so
484/// post-create fsync covers every new link without weakening create semantics.
485fn plan_missing_dirs(path: &Path) -> (Vec<PathBuf>, Option<PathBuf>) {
486    // Walk from `path` upward until we hit an existing directory (or run out of
487    // parents). `missing[0]` is the leaf; `missing.last()` is the shallowest new dir.
488    let mut missing: Vec<PathBuf> = Vec::new();
489    {
490        let mut cur = path;
491        loop {
492            match fs::metadata(cur) {
493                Ok(meta) if meta.is_dir() => break,
494                Ok(_) => {
495                    // Exists but is not a directory. Fall through to the create
496                    // call so the error matches the platform/create helper.
497                    break;
498                }
499                Err(e) if e.kind() == io::ErrorKind::NotFound => {
500                    missing.push(cur.to_path_buf());
501                    match cur.parent() {
502                        // `Path::new("a").parent()` is `Some("")` for a single
503                        // relative component — treat empty as cwd (`.`).
504                        Some(parent) if parent.as_os_str().is_empty() => break,
505                        // Root is its own parent (`"/".parent() == Some("/")`).
506                        Some(parent) if parent != cur => cur = parent,
507                        _ => break,
508                    }
509                }
510                // Permission / IO errors walking ancestors: let the create
511                // helper surface a consistent failure for the full path.
512                Err(_) => break,
513            }
514        }
515    }
516
517    let deepest_existing = missing
518        .last()
519        .and_then(|shallowest| match shallowest.parent() {
520            Some(parent) if parent.as_os_str().is_empty() => Some(PathBuf::from(".")),
521            Some(parent) => Some(parent.to_path_buf()),
522            None => None,
523        });
524
525    (missing, deepest_existing)
526}
527
528/// Fsync newly created directories deepest-first, then the deepest pre-existing
529/// parent so each new child dirent is durable. No-op when nothing was created.
530fn sync_new_dirents(missing: &[PathBuf], deepest_existing: Option<&Path>) -> io::Result<()> {
531    if missing.is_empty() {
532        return Ok(());
533    }
534    for dir in missing {
535        sync_directory(dir)?;
536    }
537    // Fsync the deepest pre-existing parent so the first new child dirent
538    // (the grandparent→shard link in the classic `blobs/ab/` case) is durable.
539    if let Some(existing) = deepest_existing {
540        sync_directory(existing)?;
541    }
542    Ok(())
543}
544
545/// Create a directory and any missing ancestors, making new dirents crash-durable.
546///
547/// Bare [`fs::create_dir_all`] only ensures the directories exist in the live
548/// filesystem view. After the first write into a newly created shard
549/// (e.g. `blobs/ab/…`), [`write_file_atomic`] fsyncs the shard directory so
550/// the *file* dirent is durable — but the grandparent that holds the new
551/// shard dirent may never be fsynced. A crash can then drop the entire new
552/// shard tree despite per-file durability (GAP_MAP L6).
553///
554/// This helper:
555/// 1. creates missing ancestor directories (same end state as `create_dir_all`);
556/// 2. fsyncs each newly created directory, deepest-first;
557/// 3. fsyncs the deepest pre-existing parent so the new child dirent is durable.
558///
559/// On Windows, directory fsync is a no-op (see [`sync_directory`]); creation
560/// still proceeds. Cost is once per new path segment (typically once per
561/// object-store shard).
562pub fn create_dir_all_durable(path: &Path) -> io::Result<()> {
563    let (missing, deepest_existing) = plan_missing_dirs(path);
564    fs::create_dir_all(path)?;
565    sync_new_dirents(&missing, deepest_existing.as_deref())
566}
567
568/// Wrap an `io::Error` raised while writing `path` so that ENOSPC carries
569/// an actionable message naming the path. Non-ENOSPC errors pass through
570/// unchanged. The wrapped error's `raw_os_error()` still returns 28, and
571/// [`is_out_of_space`] still detects it — callers (e.g. `cmd_snapshot`)
572/// rely on this for stable exit-code mapping.
573///
574/// Thin wrapper over [`enrich_fs_error`] for the historical "writing"
575/// call sites. New code should prefer `enrich_fs_error(path, "writing", err)`
576/// directly so the operation name is explicit at the call site.
577fn enrich_write_error(path: &Path, err: io::Error) -> io::Error {
578    enrich_fs_error(path, "writing", err)
579}
580
581/// Wrap an `io::Error` produced by a filesystem operation against `path`
582/// with a heddle-context message naming both the operation and the path.
583///
584/// The mapping covers the cases users actually hit and the messages we
585/// promise from heddle's CLI surface:
586/// - **ENOTEMPTY** — usually `remove_dir` against a directory that still
587///   holds untracked or explicitly ignored content, such as build output.
588///   The high-level fix is to leave the directory in place, but when the
589///   error does surface (e.g. a path the planner *did* expect to remove),
590///   the message names the path so the user can investigate.
591/// - **EACCES** — naming the path and the action ("removing", "writing",
592///   "renaming") is enough for the user to inspect mode bits.
593/// - **ENOENT** — caller-driven: only enriched when the operation
594///   expected the path to exist (so optional reads like a missing index
595///   pass through unchanged via the `is_not_found` predicate).
596/// - **EROFS** — points the user at the filesystem mount, not at heddle.
597/// - **EXDEV** — points the user at the temp path / mount mismatch.
598/// - **ENOSPC** — same actionable disk-full message the snapshot path
599///   already relies on.
600///
601/// `op` is a verb in the present-progressive ("writing", "removing",
602/// "renaming", "creating") so the resulting message reads naturally:
603///   `"could not remove `<path>` because it contains content..."`.
604///
605/// The wrapped error preserves `raw_os_error()` (callers still classify
606/// disk-full via [`is_out_of_space`]) and exposes the original `io::Error`
607/// through the `Error::source` chain (so `RUST_BACKTRACE=1` and
608/// `anyhow`'s chain printer still surface the OS error).
609pub fn enrich_fs_error(path: &Path, op: &'static str, err: io::Error) -> io::Error {
610    if is_out_of_space(&err) {
611        let msg = format!(
612            "out of disk space {op} {}: free disk space and re-run the command — your working tree is unchanged",
613            path.display()
614        );
615        return io::Error::new(
616            io::ErrorKind::StorageFull,
617            EnrichedFsError { msg, source: err },
618        );
619    }
620    if is_directory_not_empty(&err) {
621        let msg = format!(
622            "could not remove directory `{}` because it contains content (heddle-ignored or otherwise) — leaving in place",
623            path.display()
624        );
625        return io::Error::new(
626            io::ErrorKind::DirectoryNotEmpty,
627            EnrichedFsError { msg, source: err },
628        );
629    }
630    if is_read_only_filesystem(&err) {
631        let msg = format!(
632            "filesystem is read-only — `{}` cannot be modified",
633            path.display()
634        );
635        return io::Error::new(
636            io::ErrorKind::ReadOnlyFilesystem,
637            EnrichedFsError { msg, source: err },
638        );
639    }
640    if is_permission_denied(&err) {
641        let msg = format!(
642            "permission denied {op} `{}` — check filesystem permissions",
643            path.display()
644        );
645        return io::Error::new(
646            io::ErrorKind::PermissionDenied,
647            EnrichedFsError { msg, source: err },
648        );
649    }
650    if is_not_found(&err) {
651        let msg = format!("could not find `{}` for {op}", path.display());
652        return io::Error::new(
653            io::ErrorKind::NotFound,
654            EnrichedFsError { msg, source: err },
655        );
656    }
657    if is_cross_device_link(&err) {
658        let msg = format!(
659            "cannot rename across filesystems — temp file for `{}` lives on a different mount; set TMPDIR to the same filesystem as the destination",
660            path.display()
661        );
662        return io::Error::new(
663            io::ErrorKind::CrossesDevices,
664            EnrichedFsError { msg, source: err },
665        );
666    }
667    err
668}
669
670/// Wrap an `EXDEV` error from `fs::rename` with both the source temp path
671/// and the destination — the user needs both to understand which mount
672/// boundary the rename tripped on. Other error kinds delegate to
673/// [`enrich_fs_error`] using the destination as the principal path.
674pub fn enrich_rename_error(src: &Path, dst: &Path, err: io::Error) -> io::Error {
675    if is_cross_device_link(&err) {
676        let msg = format!(
677            "cannot rename across filesystems — temp file at `{}` cannot be renamed to `{}`; set TMPDIR to the same filesystem as the destination",
678            src.display(),
679            dst.display()
680        );
681        return io::Error::new(
682            io::ErrorKind::CrossesDevices,
683            EnrichedFsError { msg, source: err },
684        );
685    }
686    enrich_fs_error(dst, "renaming", err)
687}
688
689#[derive(Debug)]
690struct EnrichedFsError {
691    msg: String,
692    source: io::Error,
693}
694
695impl std::fmt::Display for EnrichedFsError {
696    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
697        f.write_str(&self.msg)
698    }
699}
700
701impl std::error::Error for EnrichedFsError {
702    fn source(&self) -> Option<&(dyn std::error::Error + 'static)> {
703        Some(&self.source)
704    }
705}
706
707pub struct StagedAtomicWrite {
708    path: PathBuf,
709    tmp: PathBuf,
710    pending: bool,
711}
712
713impl StagedAtomicWrite {
714    pub fn publish(mut self) -> io::Result<()> {
715        durable_rename(&self.tmp, &self.path)
716            .map_err(|error| enrich_rename_error(&self.tmp, &self.path, error))?;
717        self.pending = false;
718        Ok(())
719    }
720}
721
722impl Drop for StagedAtomicWrite {
723    fn drop(&mut self) {
724        if self.pending {
725            let _ = fs::remove_file(&self.tmp);
726        }
727    }
728}
729
730fn stage_file_atomic_impl(
731    path: &Path,
732    bytes: &[u8],
733    kind: AtomicWriteKind,
734    before_write: impl FnOnce(&File, &Path) -> io::Result<()>,
735) -> io::Result<StagedAtomicWrite> {
736    let parent = path.parent().unwrap_or_else(|| Path::new("."));
737    create_dir_all_durable(parent).map_err(|e| enrich_fs_error(parent, "creating", e))?;
738
739    let tmp = temp_path(path);
740    let inner = (|| -> io::Result<()> {
741        let mut file = kind.open_tmp(&tmp)?;
742        kind.enforce_before_write(&file)?;
743        before_write(&file, &tmp)?;
744        file.write_all(bytes)?;
745        sync_file(&file, &tmp)?;
746        Ok(())
747    })();
748
749    if let Err(err) = inner {
750        // Best-effort cleanup. On ENOSPC the tempfile may itself be the
751        // cause of the disk pressure; removing it gives the user back
752        // some slack before they re-run.
753        let _ = fs::remove_file(&tmp);
754        return Err(enrich_write_error(path, err));
755    }
756
757    Ok(StagedAtomicWrite {
758        path: path.to_path_buf(),
759        tmp,
760        pending: true,
761    })
762}
763
764fn write_file_atomic_impl(
765    path: &Path,
766    bytes: &[u8],
767    kind: AtomicWriteKind,
768    before_write: impl FnOnce(&File, &Path) -> io::Result<()>,
769) -> io::Result<()> {
770    stage_file_atomic_impl(path, bytes, kind, before_write)?.publish()
771}
772
773pub fn write_file_atomic(path: &Path, bytes: &[u8]) -> io::Result<()> {
774    write_file_atomic_impl(path, bytes, AtomicWriteKind::Normal, |_, _| Ok(()))
775}
776
777/// Atomically publish reconstructible bytes without forcing them to stable
778/// storage. The final path is never observed partially written, but a crash may
779/// lose this write. Callers must have an independent durable source from which
780/// the file can be rebuilt.
781pub fn write_file_atomic_reconstructible(path: &Path, bytes: &[u8]) -> io::Result<()> {
782    let parent = path.parent().unwrap_or_else(|| Path::new("."));
783    fs::create_dir_all(parent).map_err(|error| enrich_fs_error(parent, "creating", error))?;
784    let tmp = temp_path(path);
785    let result = (|| -> io::Result<()> {
786        let mut file = AtomicWriteKind::Normal.open_tmp(&tmp)?;
787        file.write_all(bytes)?;
788        file.flush()?;
789        drop(file);
790        fs::rename(&tmp, path).map_err(|error| enrich_rename_error(&tmp, path, error))
791    })();
792    if let Err(error) = result {
793        let _ = fs::remove_file(&tmp);
794        return Err(enrich_write_error(path, error));
795    }
796    Ok(())
797}
798
799/// Create a directory tree with owner-only permissions on Unix (`0o700`),
800/// making newly created dirents crash-durable (same fsync chain as
801/// [`create_dir_all_durable`]).
802///
803/// Used for `.heddle` / `~/.heddle` trees that hold credentials, keys, and
804/// repository secrets. On Unix, missing ancestors are created with mode
805/// `0o700` and then fsynced deepest-first, plus the deepest pre-existing
806/// parent. On non-Unix platforms this falls back to durable
807/// [`create_dir_all_durable`] (no portable POSIX mode API). Existing
808/// directories are left as-is (creation-time privacy; callers that need to
809/// tighten existing modes should do so explicitly).
810pub fn create_private_dir_all(path: &Path) -> io::Result<()> {
811    #[cfg(unix)]
812    {
813        use std::os::unix::fs::DirBuilderExt;
814        let (missing, deepest_existing) = plan_missing_dirs(path);
815        let mut builder = fs::DirBuilder::new();
816        builder.recursive(true).mode(0o700);
817        match builder.create(path) {
818            Ok(()) => {}
819            Err(e) if e.kind() == io::ErrorKind::AlreadyExists => {}
820            Err(e) => return Err(e),
821        }
822        sync_new_dirents(&missing, deepest_existing.as_deref())
823    }
824    #[cfg(not(unix))]
825    {
826        // No portable POSIX mode API — same durable create chain as public dirs.
827        create_dir_all_durable(path)
828    }
829}
830
831/// Atomically write secret material without ever creating a group/world
832/// readable temporary file.
833///
834/// On Unix the temp inode is created with `OpenOptions::mode(0o600)` before
835/// any bytes are written, then the open file descriptor is enforced to exact
836/// `0600` before the payload is written. Permission failures are hard errors
837/// and the temp file is removed best-effort. On non-Unix platforms there is no
838/// portable POSIX mode API, so this uses the normal create-new temp file,
839/// fsync, and rename sequence.
840pub fn write_file_atomic_secret(path: &Path, bytes: &[u8]) -> io::Result<()> {
841    write_file_atomic_impl(path, bytes, AtomicWriteKind::Secret, |_, _| Ok(()))
842}
843
844pub fn stage_file_atomic_secret(path: &Path, bytes: &[u8]) -> io::Result<StagedAtomicWrite> {
845    stage_file_atomic_impl(path, bytes, AtomicWriteKind::Secret, |_, _| Ok(()))
846}
847
848/// Publish an existing on-disk file at `src` to `dst` with the same
849/// crash-consistency contract as [`write_file_atomic`]:
850///
851/// 1. `fsync` the source so its data blocks are stable before any directory
852///    entry is updated (`rename` moves a dirent; it does not re-write bytes).
853/// 2. `rename(src, dst)` when both paths share a filesystem — atomic dirent
854///    publish.
855/// 3. On `EXDEV`, stream-copy into a *same-directory* temp, `fsync` the temp,
856///    `rename` over `dst`, then remove `src`. Never write the final path in
857///    place: a crash mid-copy must not leave a torn content-addressed object
858///    under its final name.
859/// 4. `fsync` the destination parent so the new dirent is durable.
860///
861/// If `dst` already exists and rename reports `AlreadyExists` (Windows;
862/// POSIX `rename` replaces files), the source is removed and `Ok(())` is
863/// returned — content-addressed install idempotency.
864///
865/// Non-`EXDEV` rename failures propagate. Callers must not silently fall
866/// through to a raw in-place copy on unrelated errors (the previous
867/// streaming-pack install path did exactly that).
868/// Fsync an existing regular file's data blocks.
869///
870/// On Windows, `FlushFileBuffers` requires write access — a read-only
871/// `File::open` + `sync_all` returns `ERROR_ACCESS_DENIED` (code 5). Open
872/// with write so pack install / L8 journal publish works under Windows
873/// tempdirs (projfs smoke fixtures).
874fn fsync_file_data(path: &Path) -> io::Result<()> {
875    let file = OpenOptions::new()
876        .read(true)
877        .write(true)
878        .open(path)
879        .map_err(|e| enrich_fs_error(path, "opening", e))?;
880    sync_file(&file, path).map_err(|e| enrich_fs_error(path, "syncing", e))
881}
882
883pub fn publish_file_durable(src: &Path, dst: &Path) -> io::Result<()> {
884    let parent = dst.parent().unwrap_or_else(|| Path::new("."));
885    create_dir_all_durable(parent).map_err(|e| enrich_fs_error(parent, "creating", e))?;
886
887    // Data-block durability before publishing the dirent. Required even on
888    // the same-filesystem rename path: StreamingPackBuilder (and similar
889    // staged writers) only `flush` buffered writers; without this fsync a
890    // crash after rename can lose the published object.
891    fsync_file_data(src)?;
892
893    match durable_rename(src, dst) {
894        Ok(()) => {}
895        Err(e) if e.kind() == io::ErrorKind::AlreadyExists => {
896            // Content-addressed install: destination already present.
897            let _ = fs::remove_file(src);
898        }
899        Err(e) if is_cross_device_link(&e) => {
900            publish_file_via_copy_durable(src, dst)?;
901        }
902        Err(e) => return Err(enrich_rename_error(src, dst, e)),
903    }
904
905    Ok(())
906}
907
908/// Cross-device publish path: copy to a same-dir temp, fsync, rename over
909/// `dst`. Exposed to unit tests so the no-torn-final-path contract is
910/// exercised without needing a real multi-mount layout.
911fn publish_file_via_copy_durable(src: &Path, dst: &Path) -> io::Result<()> {
912    let parent = dst.parent().unwrap_or_else(|| Path::new("."));
913    create_dir_all_durable(parent).map_err(|e| enrich_fs_error(parent, "creating", e))?;
914
915    let tmp = temp_path(dst);
916    let result = (|| -> io::Result<()> {
917        fs::copy(src, &tmp).map_err(|e| enrich_fs_error(&tmp, "writing", e))?;
918        fsync_file_data(&tmp)?;
919        durable_rename(&tmp, dst).map_err(|e| enrich_rename_error(&tmp, dst, e))?;
920        let _ = fs::remove_file(src);
921        Ok(())
922    })();
923    if result.is_err() {
924        let _ = fs::remove_file(&tmp);
925    }
926    result
927}
928
929#[cfg(test)]
930mod tests {
931    use super::*;
932
933    fn enospc_io_error() -> io::Error {
934        io::Error::from_raw_os_error(ENOSPC)
935    }
936
937    #[test]
938    fn is_out_of_space_detects_enospc_raw() {
939        assert!(is_out_of_space(&enospc_io_error()));
940    }
941
942    #[test]
943    fn is_out_of_space_detects_storage_full_kind() {
944        let err = io::Error::new(io::ErrorKind::StorageFull, "mock disk full");
945        assert!(is_out_of_space(&err));
946    }
947
948    #[test]
949    fn is_out_of_space_detects_write_zero() {
950        let err = io::Error::new(io::ErrorKind::WriteZero, "short write");
951        assert!(is_out_of_space(&err));
952    }
953
954    #[test]
955    fn is_out_of_space_rejects_unrelated_errors() {
956        assert!(!is_out_of_space(&io::Error::new(
957            io::ErrorKind::NotFound,
958            "missing"
959        )));
960        assert!(!is_out_of_space(&io::Error::new(
961            io::ErrorKind::PermissionDenied,
962            "nope"
963        )));
964        assert!(!is_out_of_space(&io::Error::other("generic")));
965    }
966
967    #[test]
968    fn is_directory_not_empty_detects_kind() {
969        let err = io::Error::new(io::ErrorKind::DirectoryNotEmpty, "still has children");
970        assert!(is_directory_not_empty(&err));
971    }
972
973    #[test]
974    fn is_directory_not_empty_detects_raw_codes() {
975        for code in [ENOTEMPTY_LINUX, ENOTEMPTY_MACOS, ENOTEMPTY_WINDOWS] {
976            assert!(
977                is_directory_not_empty(&io::Error::from_raw_os_error(code)),
978                "expected raw OS error {code} to classify as ENOTEMPTY"
979            );
980        }
981    }
982
983    #[test]
984    fn is_directory_not_empty_rejects_unrelated() {
985        assert!(!is_directory_not_empty(&io::Error::new(
986            io::ErrorKind::NotFound,
987            "missing"
988        )));
989        assert!(!is_directory_not_empty(&enospc_io_error()));
990    }
991
992    #[test]
993    fn is_permission_denied_detects_kind_and_raw() {
994        assert!(is_permission_denied(&io::Error::new(
995            io::ErrorKind::PermissionDenied,
996            "nope"
997        )));
998        assert!(is_permission_denied(&io::Error::from_raw_os_error(EACCES)));
999    }
1000
1001    #[test]
1002    fn is_not_found_detects_kind_and_raw() {
1003        assert!(is_not_found(&io::Error::new(
1004            io::ErrorKind::NotFound,
1005            "missing"
1006        )));
1007        assert!(is_not_found(&io::Error::from_raw_os_error(ENOENT)));
1008    }
1009
1010    #[test]
1011    fn is_read_only_filesystem_detects_raw() {
1012        assert!(is_read_only_filesystem(&io::Error::from_raw_os_error(
1013            EROFS
1014        )));
1015    }
1016
1017    #[test]
1018    fn is_cross_device_link_detects_raw() {
1019        assert!(is_cross_device_link(&io::Error::from_raw_os_error(EXDEV)));
1020    }
1021
1022    #[test]
1023    fn enrich_fs_error_passes_through_unclassified() {
1024        let path = Path::new("/tmp/example");
1025        let original = io::Error::other("weird");
1026        let wrapped = enrich_fs_error(path, "writing", original);
1027        // Unclassified errors are returned untouched.
1028        assert_eq!(wrapped.kind(), io::ErrorKind::Other);
1029        assert_eq!(wrapped.to_string(), "weird");
1030    }
1031
1032    #[test]
1033    fn enrich_fs_error_wraps_enospc_with_path_and_recovery_hint() {
1034        let path = Path::new("/repo/.heddle/state/abc.bin");
1035        let wrapped = enrich_fs_error(path, "writing", enospc_io_error());
1036
1037        // Stable kind so the CLI exit-code mapper finds it.
1038        assert_eq!(wrapped.kind(), io::ErrorKind::StorageFull);
1039        // Message names the failure, the path, and the recovery.
1040        let msg = wrapped.to_string();
1041        assert!(
1042            msg.contains("out of disk space"),
1043            "missing failure name: {msg}"
1044        );
1045        assert!(
1046            msg.contains("/repo/.heddle/state/abc.bin"),
1047            "missing path: {msg}"
1048        );
1049        assert!(
1050            msg.contains("free disk space") && msg.contains("re-run"),
1051            "missing recovery hint: {msg}"
1052        );
1053        assert!(
1054            msg.contains("working tree is unchanged"),
1055            "missing reassurance: {msg}"
1056        );
1057        // Source chain preserved so callers that walk `source()` (e.g.
1058        // anyhow's chain printer) can still see the original ENOSPC.
1059        let src = std::error::Error::source(&wrapped as &dyn std::error::Error)
1060            .or_else(|| wrapped.get_ref().and_then(|e| e.source()))
1061            .expect("source preserved");
1062        assert!(src.to_string().to_lowercase().contains("space"));
1063    }
1064
1065    #[test]
1066    fn enrich_fs_error_wraps_enotempty_with_directory_message() {
1067        let path = Path::new("/repo/web");
1068        let wrapped = enrich_fs_error(
1069            path,
1070            "removing",
1071            io::Error::from_raw_os_error(ENOTEMPTY_MACOS),
1072        );
1073        assert_eq!(wrapped.kind(), io::ErrorKind::DirectoryNotEmpty);
1074        let msg = wrapped.to_string();
1075        assert!(
1076            msg.contains("could not remove directory"),
1077            "missing action: {msg}"
1078        );
1079        assert!(msg.contains("/repo/web"), "missing path: {msg}");
1080        assert!(
1081            msg.contains("heddle-ignored"),
1082            "missing heddle-ignored hint: {msg}"
1083        );
1084        assert!(
1085            msg.contains("leaving in place"),
1086            "missing reassurance: {msg}"
1087        );
1088        // raw_os_error() does NOT round-trip — `io::Error::new(kind, source)`
1089        // synthesizes a new error whose `raw_os_error()` is None — but the
1090        // source chain still exposes the original OS code for callers that
1091        // walk it.
1092        let src = wrapped.get_ref().and_then(|e| e.source()).expect("source");
1093        let original = src
1094            .downcast_ref::<io::Error>()
1095            .expect("original io::Error preserved");
1096        assert_eq!(original.raw_os_error(), Some(ENOTEMPTY_MACOS));
1097    }
1098
1099    #[test]
1100    fn enrich_fs_error_wraps_eacces_with_op_and_path() {
1101        let path = Path::new("/repo/.heddle/state/index.bin");
1102        let wrapped = enrich_fs_error(path, "writing", io::Error::from_raw_os_error(EACCES));
1103        assert_eq!(wrapped.kind(), io::ErrorKind::PermissionDenied);
1104        let msg = wrapped.to_string();
1105        assert!(msg.starts_with("permission denied writing"), "msg: {msg}");
1106        assert!(msg.contains("/repo/.heddle/state/index.bin"), "msg: {msg}");
1107        assert!(msg.contains("check filesystem permissions"), "msg: {msg}");
1108    }
1109
1110    #[test]
1111    fn enrich_fs_error_wraps_enoent_with_op_and_path() {
1112        let path = Path::new("/repo/.heddle");
1113        let wrapped = enrich_fs_error(path, "opening", io::Error::from_raw_os_error(ENOENT));
1114        assert_eq!(wrapped.kind(), io::ErrorKind::NotFound);
1115        let msg = wrapped.to_string();
1116        assert!(msg.contains("could not find"), "missing action: {msg}");
1117        assert!(msg.contains("/repo/.heddle"), "missing path: {msg}");
1118        assert!(msg.contains("for opening"), "missing op: {msg}");
1119    }
1120
1121    #[test]
1122    fn enrich_fs_error_wraps_erofs_with_path() {
1123        let path = Path::new("/mnt/readonly/.heddle/state/index.bin");
1124        let wrapped = enrich_fs_error(path, "writing", io::Error::from_raw_os_error(EROFS));
1125        assert_eq!(wrapped.kind(), io::ErrorKind::ReadOnlyFilesystem);
1126        let msg = wrapped.to_string();
1127        assert!(msg.contains("filesystem is read-only"), "msg: {msg}");
1128        assert!(
1129            msg.contains("/mnt/readonly/.heddle/state/index.bin"),
1130            "msg: {msg}"
1131        );
1132        assert!(msg.contains("cannot be modified"), "msg: {msg}");
1133    }
1134
1135    #[test]
1136    fn enrich_rename_error_wraps_exdev_with_src_and_dst() {
1137        let src = Path::new("/tmp-mount/.x.tmp-1234");
1138        let dst = Path::new("/repo/.heddle/state/index.bin");
1139        let wrapped = enrich_rename_error(src, dst, io::Error::from_raw_os_error(EXDEV));
1140        assert_eq!(wrapped.kind(), io::ErrorKind::CrossesDevices);
1141        let msg = wrapped.to_string();
1142        assert!(
1143            msg.contains("cannot rename across filesystems"),
1144            "msg: {msg}"
1145        );
1146        assert!(msg.contains("/tmp-mount/.x.tmp-1234"), "missing src: {msg}");
1147        assert!(
1148            msg.contains("/repo/.heddle/state/index.bin"),
1149            "missing dst: {msg}"
1150        );
1151        assert!(msg.contains("TMPDIR"), "missing recovery hint: {msg}");
1152    }
1153
1154    #[test]
1155    fn enrich_rename_error_falls_through_to_generic_for_other_kinds() {
1156        let src = Path::new("/tmp/.x.tmp");
1157        let dst = Path::new("/repo/file");
1158        let wrapped = enrich_rename_error(src, dst, io::Error::from_raw_os_error(EACCES));
1159        // Non-EXDEV rename failures get the generic `enrich_fs_error`
1160        // treatment, which preserves the dst path and the "renaming" op.
1161        assert_eq!(wrapped.kind(), io::ErrorKind::PermissionDenied);
1162        let msg = wrapped.to_string();
1163        assert!(msg.starts_with("permission denied renaming"), "msg: {msg}");
1164        assert!(msg.contains("/repo/file"), "missing dst: {msg}");
1165    }
1166
1167    #[test]
1168    fn enrich_write_error_passes_through_non_enospc_unclassified() {
1169        // The historical helper now delegates to `enrich_fs_error`, so a
1170        // generic Other error still passes through unchanged.
1171        let path = Path::new("/tmp/example");
1172        let original = io::Error::other("weird");
1173        let wrapped = enrich_write_error(path, original);
1174        assert_eq!(wrapped.kind(), io::ErrorKind::Other);
1175        assert_eq!(wrapped.to_string(), "weird");
1176    }
1177
1178    #[test]
1179    fn write_file_atomic_round_trip() {
1180        let dir = tempfile::TempDir::new().unwrap();
1181        let target = dir.path().join("nested/under/here/file.bin");
1182        write_file_atomic(&target, b"hello").unwrap();
1183        assert_eq!(fs::read(&target).unwrap(), b"hello");
1184    }
1185
1186    #[test]
1187    fn stage_temp_files_durable_writes_every_file_verbatim() {
1188        // The bulk-ref hot path stages N temp files in one overlapped-writeback
1189        // pass. Every file must land with its exact bytes — the batching is a
1190        // durability/perf optimization, never a content one.
1191        let dir = tempfile::TempDir::new().unwrap();
1192        let files: Vec<(PathBuf, Vec<u8>)> = (0..50)
1193            .map(|i| {
1194                (
1195                    dir.path().join(format!("ref-{i}.tmp")),
1196                    format!("change-id-{i}\n").into_bytes(),
1197                )
1198            })
1199            .collect();
1200
1201        stage_temp_files_durable(&files).unwrap();
1202
1203        for (path, bytes) in &files {
1204            assert_eq!(&fs::read(path).unwrap(), bytes, "mismatch at {path:?}");
1205        }
1206    }
1207
1208    #[test]
1209    fn stage_temp_files_durable_empty_batch_is_ok() {
1210        // A publish with no new-content plans (e.g. a pure delete batch) hands
1211        // an empty slice; it must be a clean no-op, not an error.
1212        stage_temp_files_durable(&[]).unwrap();
1213    }
1214
1215    #[test]
1216    fn stage_temp_files_durable_errors_when_parent_missing() {
1217        // The helper does NOT create parent directories (callers pre-create
1218        // them via `alloc_temp_path`); a missing parent surfaces as an error
1219        // rather than silently dropping the write.
1220        let dir = tempfile::TempDir::new().unwrap();
1221        let files = vec![(dir.path().join("does/not/exist/ref.tmp"), b"x".to_vec())];
1222        assert!(stage_temp_files_durable(&files).is_err());
1223    }
1224
1225    #[cfg(unix)]
1226    #[test]
1227    fn create_private_dir_all_sets_0700() {
1228        use std::os::unix::fs::PermissionsExt;
1229
1230        let dir = tempfile::TempDir::new().unwrap();
1231        let target = dir.path().join("nested/private");
1232        create_private_dir_all(&target).expect("create private dir");
1233        let mode = fs::metadata(&target).unwrap().permissions().mode() & 0o777;
1234        assert_eq!(mode, 0o700, "new private dir must be 0700, got {mode:o}");
1235        // Intermediate ancestors created by the recursive private create must
1236        // also be owner-only (DirBuilder mode applies to each new segment).
1237        let mid_mode = fs::metadata(dir.path().join("nested"))
1238            .unwrap()
1239            .permissions()
1240            .mode()
1241            & 0o777;
1242        assert_eq!(
1243            mid_mode, 0o700,
1244            "intermediate private ancestor must be 0700"
1245        );
1246        // Idempotent after durable create: re-run is success and modes stick.
1247        create_private_dir_all(&target).expect("idempotent private create");
1248        let mode_again = fs::metadata(&target).unwrap().permissions().mode() & 0o777;
1249        assert_eq!(mode_again, 0o700);
1250    }
1251
1252    #[cfg(unix)]
1253    #[test]
1254    fn write_file_atomic_secret_is_0600_before_write_and_after_rename() {
1255        use std::os::unix::fs::PermissionsExt;
1256
1257        let dir = tempfile::TempDir::new().unwrap();
1258        let target = dir.path().join("nested/secret.txt");
1259        let mut observed_tmp_mode = None;
1260
1261        write_file_atomic_impl(&target, b"secret", AtomicWriteKind::Secret, |file, tmp| {
1262            let fd_mode = file.metadata()?.permissions().mode() & 0o777;
1263            let path_mode = fs::metadata(tmp)?.permissions().mode() & 0o777;
1264            observed_tmp_mode = Some((fd_mode, path_mode));
1265            Ok(())
1266        })
1267        .unwrap();
1268
1269        assert_eq!(observed_tmp_mode, Some((0o600, 0o600)));
1270        let final_mode = fs::metadata(&target).unwrap().permissions().mode() & 0o777;
1271        assert_eq!(final_mode, 0o600);
1272        assert_eq!(fs::read(&target).unwrap(), b"secret");
1273    }
1274
1275    #[test]
1276    fn write_file_atomic_secret_cleans_up_when_pre_write_check_fails() {
1277        let dir = tempfile::TempDir::new().unwrap();
1278        let target = dir.path().join("secret.txt");
1279        let mut tmp_path = None;
1280
1281        let err = write_file_atomic_impl(&target, b"secret", AtomicWriteKind::Secret, |_, tmp| {
1282            tmp_path = Some(tmp.to_path_buf());
1283            Err(io::Error::new(
1284                io::ErrorKind::PermissionDenied,
1285                "injected permission failure",
1286            ))
1287        })
1288        .expect_err("permission failure should propagate");
1289
1290        assert!(is_permission_denied(&err), "unexpected error: {err}");
1291        assert!(!target.exists(), "secret write must not publish target");
1292        let tmp = tmp_path.expect("pre-write hook observed temp path");
1293        assert!(!tmp.exists(), "failed secret write should remove temp file");
1294    }
1295
1296    #[test]
1297    fn staged_secret_is_unpublished_until_publish() {
1298        let dir = tempfile::TempDir::new().unwrap();
1299        let target = dir.path().join("secret.txt");
1300        let staged = stage_file_atomic_secret(&target, b"secret").unwrap();
1301
1302        assert!(!target.exists());
1303        staged.publish().unwrap();
1304        assert_eq!(fs::read(target).unwrap(), b"secret");
1305    }
1306
1307    #[test]
1308    fn dropping_staged_secret_removes_temporary_file() {
1309        let dir = tempfile::TempDir::new().unwrap();
1310        let target = dir.path().join("secret.txt");
1311        let staged = stage_file_atomic_secret(&target, b"secret").unwrap();
1312        drop(staged);
1313
1314        assert!(!target.exists());
1315        assert_eq!(fs::read_dir(dir.path()).unwrap().count(), 0);
1316    }
1317
1318    /// Regression for heddle#105: `sync_directory` must succeed on any
1319    /// writable directory. The original implementation called
1320    /// `OpenOptions::new().read(true).open(dir)` + `sync_all()`, which
1321    /// fails on Windows with `ERROR_ACCESS_DENIED` (5) because Windows
1322    /// directory handles require `FILE_FLAG_BACKUP_SEMANTICS` and
1323    /// `FlushFileBuffers` on a directory handle is not a supported
1324    /// operation. The failure cascaded through `write_file_atomic` into
1325    /// `Repository::init_default`, breaking `heddle init` on Windows.
1326    #[test]
1327    fn sync_directory_succeeds_on_writable_tempdir() {
1328        let dir = tempfile::TempDir::new().unwrap();
1329        sync_directory(dir.path()).expect("sync_directory on writable tempdir");
1330    }
1331
1332    /// Regression for heddle#105: full `write_file_atomic` round-trip
1333    /// against a freshly-created nested directory must not surface
1334    /// `PermissionDenied`. The previous failure mode was the
1335    /// `sync_directory(parent)` call at the end of `write_file_atomic`.
1336    #[test]
1337    fn write_file_atomic_does_not_permission_deny_on_parent_sync() {
1338        let dir = tempfile::TempDir::new().unwrap();
1339        let target = dir.path().join("oplog/oplog.bin");
1340        let result = write_file_atomic(&target, b"hello");
1341        if let Err(e) = &result {
1342            assert!(
1343                !is_permission_denied(e),
1344                "write_file_atomic surfaced PermissionDenied on a writable \
1345                 tempdir (heddle#105): {e}"
1346            );
1347        }
1348        result.expect("write_file_atomic");
1349    }
1350
1351    #[test]
1352    fn publish_file_durable_renames_and_removes_source() {
1353        let dir = tempfile::TempDir::new().unwrap();
1354        let src = dir.path().join("staged.pack");
1355        let dst = dir.path().join("objects/packs/final.pack");
1356        fs::write(&src, b"pack-bytes").unwrap();
1357
1358        publish_file_durable(&src, &dst).unwrap();
1359
1360        assert!(!src.exists(), "source must be consumed by publish");
1361        assert_eq!(fs::read(&dst).unwrap(), b"pack-bytes");
1362    }
1363
1364    /// Windows: FlushFileBuffers needs write access; read-only open +
1365    /// sync_all fails with ERROR_ACCESS_DENIED and broke L8 pack install /
1366    /// projfs fixture setup under tempdirs.
1367    #[test]
1368    fn publish_file_durable_syncs_source_without_permission_deny() {
1369        let dir = tempfile::TempDir::new().unwrap();
1370        let src = dir.path().join("staged.bin");
1371        let dst = dir.path().join("final.bin");
1372        fs::write(&src, b"need-fsync-before-rename").unwrap();
1373        let result = publish_file_durable(&src, &dst);
1374        if let Err(e) = &result {
1375            assert!(
1376                !is_permission_denied(e),
1377                "publish_file_durable PermissionDenied on source fsync: {e}"
1378            );
1379        }
1380        result.expect("publish_file_durable");
1381        assert_eq!(fs::read(&dst).unwrap(), b"need-fsync-before-rename");
1382    }
1383
1384    #[test]
1385    fn publish_file_via_copy_durable_never_writes_final_path_directly() {
1386        // Regression for streaming pack install: the EXDEV fallback used
1387        // `fs::copy(src, dst)` straight into the content-addressed final
1388        // path. A crash mid-copy left a torn pack under its BLAKE3 name
1389        // (readers treat that name as authoritative). The durable path
1390        // must land bytes at a temp sibling first, then rename.
1391        let dir = tempfile::TempDir::new().unwrap();
1392        let src = dir.path().join("staged.pack");
1393        let dst = dir.path().join("final.pack");
1394        // Pre-existing destination simulates a previous torn install that
1395        // a naive in-place copy would non-atomically overwrite.
1396        fs::write(&dst, b"TORN-OLD-CONTENT!!!!!!!!!!!!!").unwrap();
1397        fs::write(&src, b"complete-new-pack-bytes").unwrap();
1398
1399        publish_file_via_copy_durable(&src, &dst).unwrap();
1400
1401        assert!(!src.exists(), "source must be removed after copy publish");
1402        assert_eq!(fs::read(&dst).unwrap(), b"complete-new-pack-bytes");
1403        // No leftover temps in the destination directory.
1404        let leftovers: Vec<_> = fs::read_dir(dir.path())
1405            .unwrap()
1406            .filter_map(|e| e.ok())
1407            .map(|e| e.file_name().to_string_lossy().into_owned())
1408            .filter(|name| name.contains(".tmp-"))
1409            .collect();
1410        assert!(
1411            leftovers.is_empty(),
1412            "durable copy must not leave temp siblings: {leftovers:?}"
1413        );
1414    }
1415
1416    #[test]
1417    fn publish_file_via_copy_durable_cleans_temp_when_rename_cannot_publish() {
1418        // If the final rename cannot complete, the temp sibling must be
1419        // removed so a crash/retry path doesn't accumulate junk — and the
1420        // pre-existing destination must be left untouched (atomic replace
1421        // failed → old bytes still authoritative).
1422        let dir = tempfile::TempDir::new().unwrap();
1423        let src = dir.path().join("staged.pack");
1424        let dst_dir = dir.path().join("final.pack");
1425        fs::write(&src, b"new-bytes").unwrap();
1426        // Make `dst` a directory so `rename(temp, dst)` fails (EISDIR /
1427        // ERROR_ACCESS_DENIED class). The copy-into-temp step succeeds;
1428        // only the publish rename fails.
1429        fs::create_dir(&dst_dir).unwrap();
1430
1431        let err = publish_file_via_copy_durable(&src, &dst_dir).expect_err("rename over dir");
1432        assert!(
1433            err.kind() == io::ErrorKind::AlreadyExists
1434                || err.raw_os_error().is_some()
1435                || is_permission_denied(&err)
1436                || err.kind() == io::ErrorKind::Other
1437                || err.kind() == io::ErrorKind::IsADirectory
1438                || err.kind() == io::ErrorKind::DirectoryNotEmpty,
1439            "unexpected error kind for rename-over-dir: {err:?}"
1440        );
1441        assert!(src.exists(), "failed publish must leave source intact");
1442        assert!(dst_dir.is_dir(), "destination directory must be untouched");
1443        let leftovers: Vec<_> = fs::read_dir(dir.path())
1444            .unwrap()
1445            .filter_map(|e| e.ok())
1446            .map(|e| e.file_name().to_string_lossy().into_owned())
1447            .filter(|name| name.contains(".tmp-"))
1448            .collect();
1449        assert!(
1450            leftovers.is_empty(),
1451            "failed publish must clean temp siblings: {leftovers:?}"
1452        );
1453    }
1454
1455    #[test]
1456    fn publish_file_durable_propagates_non_exdev_rename_failures() {
1457        // The previous install_pack_files_streaming path treated *any*
1458        // rename failure as "try fs::copy into the final path". A
1459        // permission / type error must surface, not be laundered into a
1460        // second write attempt against the content-addressed name.
1461        let dir = tempfile::TempDir::new().unwrap();
1462        let src = dir.path().join("staged.pack");
1463        let dst = dir.path().join("final.pack");
1464        fs::write(&src, b"pack-bytes").unwrap();
1465        fs::create_dir(&dst).unwrap();
1466
1467        let err = publish_file_durable(&src, &dst).expect_err("rename over directory");
1468        assert!(
1469            !is_cross_device_link(&err),
1470            "failure must not be misclassified as EXDEV: {err}"
1471        );
1472        // Source remains for the caller to retry / clean up.
1473        assert!(src.exists());
1474    }
1475
1476    /// GAP_MAP L6: nested shard directories must be creatable via the durable
1477    /// helper. We cannot observe fsync from userspace, but we can assert the
1478    /// end state matches `create_dir_all` (full nested path exists as dirs).
1479    #[test]
1480    fn create_dir_all_durable_creates_nested_path() {
1481        let dir = tempfile::TempDir::new().unwrap();
1482        // Classic object-store shard layout: grandparent holds the new shard
1483        // dirent (`ab`), parent is the shard itself.
1484        let shard = dir.path().join("blobs/ab");
1485        create_dir_all_durable(&shard).expect("create nested shard path");
1486        assert!(shard.is_dir(), "leaf shard directory must exist");
1487        assert!(
1488            dir.path().join("blobs").is_dir(),
1489            "intermediate grandparent must exist"
1490        );
1491        // Idempotent: re-running against an existing tree is a no-op success.
1492        create_dir_all_durable(&shard).expect("idempotent durable create");
1493        assert!(shard.is_dir());
1494    }
1495
1496    /// GAP_MAP L6: `write_file_atomic` must still round-trip when the full
1497    /// parent chain is missing — it now goes through `create_dir_all_durable`
1498    /// instead of bare `create_dir_all`.
1499    #[test]
1500    fn write_file_atomic_creates_missing_shard_parents() {
1501        let dir = tempfile::TempDir::new().unwrap();
1502        let target = dir.path().join("blobs/ab/object.bin");
1503        write_file_atomic(&target, b"shard-bytes").unwrap();
1504        assert_eq!(fs::read(&target).unwrap(), b"shard-bytes");
1505        assert!(dir.path().join("blobs/ab").is_dir());
1506    }
1507
1508    /// Existing parent chain: durable create must not fail or alter contents.
1509    #[test]
1510    fn create_dir_all_durable_ok_when_path_already_exists() {
1511        let dir = tempfile::TempDir::new().unwrap();
1512        let nested = dir.path().join("already/there");
1513        fs::create_dir_all(&nested).unwrap();
1514        create_dir_all_durable(&nested).expect("existing dir");
1515        assert!(nested.is_dir());
1516    }
1517}