fp-dotfiles-manager 0.2.5

Minimal, zero-dependency Chezmoi-based dotfiles manager
//! A self-healing directory lock for the sync engine.
//!
//! `create_dir` is atomic, so it is the right primitive for mutual exclusion,
//! but a directory that is only removed by `Drop` leaks whenever the process
//! exits abnormally: `std::process::exit` skips `Drop`, `SIGKILL` cannot be
//! caught, and the release profile uses `panic = "abort"` so a panic skips it
//! too. A leaked lock would then block every future sync until logout.
//!
//! So the lock records its owner's pid, and a lock whose owner is gone is
//! reclaimed rather than trusted.

use crate::logger::*;
use crate::paths::concat_paths;

/// A lock dir with no `pid` file is only ever seen in the microseconds between
/// `create_dir` and writing the pid, or when left by an older build. Wait this
/// long before assuming the latter.
const PID_GRACE_SECS: u64 = 30;

/// Backstop against pid reuse: a lock older than this is stale even if its pid
/// still happens to be alive.
const MAX_AGE_SECS: u64 = 6 * 60 * 60;

/// Releases the lock directory when the guard goes out of scope.
pub struct LockGuard {
    path: String,
    released: bool,
}

impl LockGuard {
    /// Releases the lock immediately.
    ///
    /// `Drop` does not run for `std::process::exit`, so any exit path that
    /// bypasses unwinding must call this first or the lock leaks.
    pub fn release(&mut self) {
        if !self.released {
            let _ = std::fs::remove_dir_all(&self.path);
            self.released = true;
        }
    }
}

impl Drop for LockGuard {
    fn drop(&mut self) {
        self.release();
    }
}

/// Whether `pid` still names a running process.
pub fn process_is_alive(pid: u32) -> bool {
    if pid == 0 {
        return false;
    }
    #[cfg(target_os = "linux")]
    {
        std::path::Path::new(&format!("/proc/{pid}")).exists()
    }
    #[cfg(not(target_os = "linux"))]
    {
        // No portable zero-dependency liveness probe; `break_stale_lock` falls
        // back to the age heuristics there.
        let _ = pid;
        false
    }
}

/// Whether the lock directory was last modified at least `secs` ago.
fn is_older_than(path: &str, secs: u64) -> bool {
    std::fs::metadata(path)
        .and_then(|m| m.modified())
        .ok()
        .and_then(|t| std::time::SystemTime::now().duration_since(t).ok())
        .map(|age| age.as_secs() >= secs)
        .unwrap_or(false)
}

/// Reads the pid recorded in a lock directory, if any.
fn owner_pid(path: &str) -> Option<u32> {
    std::fs::read_to_string(concat_paths(path, "pid"))
        .ok()
        .and_then(|s| s.trim().parse::<u32>().ok())
}

/// Decides whether an existing lock may be broken, removing it if so.
///
/// Returns true when the lock was stale and is now gone, false when a live
/// process still owns it.
fn break_stale_lock(path: &str) -> bool {
    if is_older_than(path, MAX_AGE_SECS) {
        let _ = std::fs::remove_dir_all(path);
        return true;
    }

    match owner_pid(path) {
        Some(pid) => {
            if process_is_alive(pid) {
                return false;
            }
        }
        // No pid recorded: a live process writes one immediately, so give the
        // owner a grace period before assuming the lock was abandoned.
        None => {
            if !is_older_than(path, PID_GRACE_SECS) {
                return false;
            }
        }
    }

    let _ = std::fs::remove_dir_all(path);
    true
}

/// Takes the lock at `path`, recovering it if a previous run died holding it.
///
/// `create_dir` is the atomic primitive, so the loser of a race always finds the
/// directory already present. Returns `None` when a live sync owns the lock.
pub fn acquire(path: &str) -> Option<LockGuard> {
    let take = || -> Option<LockGuard> {
        std::fs::create_dir(path).ok()?;
        // Best effort: a missing pid only makes the lock age-gated, not broken.
        let _ = std::fs::write(concat_paths(path, "pid"), std::process::id().to_string());
        Some(LockGuard {
            path: path.to_string(),
            released: false,
        })
    };

    if let Some(guard) = take() {
        return Some(guard);
    }
    // Someone else holds it. Only reclaim it if they are gone.
    if break_stale_lock(path) {
        return take();
    }
    None
}

/// Takes the lock, reporting the contention to the user when one is live.
pub fn acquire_or_report(path: &str) -> Option<LockGuard> {
    if let Some(guard) = acquire(path) {
        return Some(guard);
    }
    log_warn("Another dotfiles sync is already running. Skipping duplicate run.");
    None
}

#[cfg(test)]
mod tests {
    use super::*;

    /// A pid that is almost certainly not running.
    fn dead_pid() -> u32 {
        // PIDs are allocated monotonically up to /proc/sys/kernel/pid_max, so
        // one past the current maximum cannot be live.
        std::fs::read_to_string("/proc/sys/kernel/pid_max")
            .ok()
            .and_then(|s| s.trim().parse::<u32>().ok())
            .unwrap_or(4_194_304)
    }

    /// A lock path that no run currently owns.
    fn free_lock_path(dir: &tempfile::TempDir) -> String {
        dir.path().join("test.lock").to_string_lossy().into_owned()
    }

    /// An *already held* lock path, as if a previous run had taken it.
    fn held_lock_path(dir: &tempfile::TempDir) -> String {
        let path = free_lock_path(dir);
        std::fs::create_dir(&path).unwrap();
        path
    }

    fn set_pid(path: &str, pid: u32) {
        std::fs::write(concat_paths(path, "pid"), pid.to_string()).unwrap();
    }

    fn backdate(path: &str, secs: u64) {
        let then = std::time::SystemTime::now() - std::time::Duration::from_secs(secs);
        // Backdate the dir itself, since that is what `is_older_than` stats.
        std::fs::File::open(path)
            .unwrap()
            .set_modified(then)
            .unwrap();
        // And the pid file, if the scenario created one.
        let pid_path = std::path::Path::new(path).join("pid");
        if pid_path.exists() {
            std::fs::File::options()
                .write(true)
                .open(&pid_path)
                .unwrap()
                .set_modified(then)
                .unwrap();
        }
    }

    #[test]
    fn acquires_and_releases() {
        let dir = tempfile::tempdir().unwrap();
        let path = free_lock_path(&dir);

        let mut guard = acquire(&path).expect("uncontended lock should be free");
        assert!(std::path::Path::new(&path).exists());
        assert_eq!(
            std::fs::read_to_string(concat_paths(&path, "pid")).unwrap(),
            std::process::id().to_string()
        );
        guard.release();
        assert!(!std::path::Path::new(&path).exists());
    }

    #[test]
    fn release_is_idempotent() {
        let dir = tempfile::tempdir().unwrap();
        let path = free_lock_path(&dir);
        let mut guard = acquire(&path).unwrap();
        guard.release();
        // A second release, and the Drop that follows, must not error or panic.
        guard.release();
        drop(guard);
        assert!(!std::path::Path::new(&path).exists());
    }

    #[test]
    fn guard_removes_lock_on_drop() {
        let dir = tempfile::tempdir().unwrap();
        let path = free_lock_path(&dir);
        drop(acquire(&path).unwrap());
        assert!(!std::path::Path::new(&path).exists());
    }

    #[test]
    fn live_owner_is_not_disrupted() {
        let dir = tempfile::tempdir().unwrap();
        let path = held_lock_path(&dir);
        set_pid(&path, std::process::id());

        assert!(acquire(&path).is_none(), "stole a live lock");
        assert!(std::path::Path::new(&path).exists(), "removed a live lock");
    }

    #[test]
    fn dead_owner_is_reclaimed() {
        let dir = tempfile::tempdir().unwrap();
        let path = held_lock_path(&dir);
        set_pid(&path, dead_pid());

        let guard = acquire(&path).expect("dead owner should be reclaimable");
        // Reclaimed by us, so the pid now belongs to this process.
        assert_eq!(
            std::fs::read_to_string(concat_paths(&path, "pid")).unwrap(),
            std::process::id().to_string()
        );
        drop(guard);
    }

    #[test]
    fn fresh_lock_without_pid_is_respected() {
        // Covers the window between create_dir and writing the pid.
        let dir = tempfile::tempdir().unwrap();
        let path = held_lock_path(&dir);

        assert!(acquire(&path).is_none(), "stole a lock mid-acquisition");
    }

    #[test]
    fn abandoned_lock_without_pid_is_reclaimed_after_grace() {
        let dir = tempfile::tempdir().unwrap();
        let path = held_lock_path(&dir);
        // No pid file at all, as left by an older build.
        backdate(&path, PID_GRACE_SECS + 5);

        let guard = acquire(&path).expect("abandoned lock should be reclaimable");
        assert!(std::path::Path::new(&path).exists());
        drop(guard);
    }

    #[test]
    fn ancient_lock_is_reclaimed_even_with_a_live_pid() {
        // Pid reuse must not pin the lock forever.
        let dir = tempfile::tempdir().unwrap();
        let path = held_lock_path(&dir);
        set_pid(&path, std::process::id());
        backdate(&path, MAX_AGE_SECS + 60);

        let guard = acquire(&path).expect("ancient lock should be reclaimable");
        assert!(std::path::Path::new(&path).exists());
        drop(guard);
    }

    #[test]
    fn malformed_pid_is_treated_as_unknown() {
        let dir = tempfile::tempdir().unwrap();
        let path = held_lock_path(&dir);
        std::fs::write(concat_paths(&path, "pid"), "not-a-pid").unwrap();
        backdate(&path, PID_GRACE_SECS + 5);

        let guard = acquire(&path).expect("malformed pid should fall back to age");
        assert!(std::path::Path::new(&path).exists());
        drop(guard);
    }

    #[test]
    fn pid_zero_is_not_alive() {
        assert!(!process_is_alive(0));
        assert!(process_is_alive(std::process::id()));
    }

    #[test]
    fn pid_file_does_not_block_removal() {
        // The old Drop used remove_dir, which fails on a non-empty directory and
        // would leak the lock. remove_dir_all is required now that the lock
        // holds a pid file.
        let dir = tempfile::tempdir().unwrap();
        let path = free_lock_path(&dir);
        let guard = acquire(&path).unwrap();
        assert!(std::path::Path::new(&path).join("pid").exists());
        drop(guard);
        assert!(!std::path::Path::new(&path).exists());
    }

    #[test]
    fn a_second_acquire_is_refused_while_held() {
        let dir = tempfile::tempdir().unwrap();
        let path = free_lock_path(&dir);
        let _first = acquire(&path).unwrap();
        assert!(acquire(&path).is_none(), "double-acquired a live lock");
    }
}