Skip to main content

nmbrs_workload/edit/
lock.rs

1// Copyright 2024-2026 Jonathan Shook
2// SPDX-License-Identifier: Apache-2.0
3
4//! Cooperative file locking for workload edits.
5//!
6//! The contract from SRD-64 §6.5: every workload-mutating
7//! command holds an exclusive lock on the workload file for
8//! the duration of read → mutate → write. Concurrent
9//! invocations either wait on the lock (up to a 5-second
10//! deadline) or error with the holder's pid (where the OS
11//! reports it).
12//!
13//! The lock is **advisory** — it protects concurrent `nmbrs`
14//! invocations, not arbitrary editors. An editor with the
15//! workload buffered does not honour this lock; that's the
16//! user's responsibility, same as any cooperative-locking
17//! convention.
18
19use std::fs::{File, OpenOptions};
20use std::io;
21use std::path::Path;
22use std::time::{Duration, Instant};
23
24use fs4::fs_std::FileExt;
25
26/// Cooperative-lock deadline. SRD-64 §6.5 specifies 5
27/// seconds; held here as a constant so the timeout is
28/// inspectable and tunable without diff churn.
29pub const LOCK_DEADLINE: Duration = Duration::from_secs(5);
30
31/// Poll interval while waiting for the lock. Short enough
32/// that contention resolves crisply; long enough that the
33/// busy loop doesn't peg a core if the holder takes a
34/// while.
35const POLL_INTERVAL: Duration = Duration::from_millis(50);
36
37/// Acquired exclusive lock on a workload file. Drops the
38/// lock when the handle is dropped (`fs4::FileExt`'s
39/// contract). Hold this for the duration of one
40/// mutate-and-write cycle.
41pub struct WorkloadLock {
42    /// Open `File` carrying the OS lock. Held until drop.
43    /// The file is opened read-only because the splicer
44    /// re-opens for writing under a separate handle —
45    /// keeping the lock-holding handle separate avoids
46    /// surprising interactions where Windows file-share
47    /// modes might block our own write.
48    file: File,
49}
50
51impl Drop for WorkloadLock {
52    fn drop(&mut self) {
53        // Best-effort unlock. If the OS already released the
54        // lock (process exit, FD closure), this is a no-op.
55        let _ = FileExt::unlock(&self.file);
56    }
57}
58
59/// Acquire an exclusive lock on `path`. Polls every 50ms
60/// until either the lock is held or [`LOCK_DEADLINE`]
61/// elapses; on timeout, returns an [`io::Error`] of kind
62/// `WouldBlock` with a message naming the workload file.
63///
64/// The OS-level mechanism is `fcntl(F_SETLK)` on Unix and
65/// `LockFileEx` on Windows via [`fs4`]. Both are advisory.
66///
67/// PID surfacing: Linux exposes the lock-holder's pid via
68/// `fcntl(F_GETLK)`, which `fs4` does not surface.
69/// Implementing pid surfacing would require dropping into
70/// `libc::fcntl` directly. SRD-64 §6.5 mentions surfacing
71/// "where known" — for now we surface a generic message;
72/// adding pid-via-libc is a follow-up.
73/// The file the OS lock is actually taken on: the workload file
74/// itself on Unix, its `.lock` sidecar on Windows (see the
75/// mandatory-lock rationale in [`acquire`]).
76#[cfg(windows)]
77fn lock_target(path: &Path) -> std::path::PathBuf {
78    let mut os = path.as_os_str().to_os_string();
79    os.push(".lock");
80    std::path::PathBuf::from(os)
81}
82
83pub fn acquire(path: &Path) -> io::Result<WorkloadLock> {
84    // Unix: lock the workload file itself — advisory locks never
85    // block this process's own re-opens (the splicer re-reads and
86    // rewrites under separate handles).
87    //
88    // Windows: `LockFileEx` locks are MANDATORY — an exclusive
89    // lock on `wl.yaml` makes the splicer's own re-open of the
90    // same file fail (os error 33). Lock a sidecar
91    // `<file>.lock` instead: mutual exclusion between nmbrs
92    // invocations still holds (every mutator locks the same
93    // sidecar) and the workload file itself stays freely
94    // readable. The sidecar is deliberately left in place on
95    // unlock — deleting it would make a third concurrent
96    // mutator's open race the delete-pending window.
97    #[cfg(windows)]
98    let file = {
99        if !path.exists() {
100            return Err(io::Error::new(
101                io::ErrorKind::NotFound,
102                format!(
103                    "workload lock: cannot open '{}' for locking: not found",
104                    path.display()
105                ),
106            ));
107        }
108        let lock_path = lock_target(path);
109        OpenOptions::new()
110            .read(true)
111            .write(true)
112            .create(true)
113            .open(&lock_path)
114            .map_err(|e| {
115                io::Error::new(
116                    e.kind(),
117                    format!(
118                        "workload lock: cannot open '{}' for locking: {e}",
119                        lock_path.display()
120                    ),
121                )
122            })?
123    };
124    #[cfg(not(windows))]
125    let file = OpenOptions::new().read(true).open(path).map_err(|e| {
126        io::Error::new(
127            e.kind(),
128            format!(
129                "workload lock: cannot open '{}' for locking: {e}",
130                path.display()
131            ),
132        )
133    })?;
134
135    let deadline = Instant::now() + LOCK_DEADLINE;
136    loop {
137        match FileExt::try_lock_exclusive(&file) {
138            Ok(true) => return Ok(WorkloadLock { file }),
139            Ok(false) => {
140                if Instant::now() >= deadline {
141                    return Err(io::Error::new(
142                        io::ErrorKind::WouldBlock,
143                        format!(
144                            "workload lock on '{}': another nmbrs process holds the \
145                             exclusive lock; waited {LOCK_DEADLINE:?} and gave up. \
146                             Re-run after that process completes, or kill it if it's \
147                             stuck.",
148                            path.display(),
149                        ),
150                    ));
151                }
152                std::thread::sleep(POLL_INTERVAL);
153            }
154            Err(e) => {
155                return Err(io::Error::new(
156                    e.kind(),
157                    format!("workload lock on '{}': {e}", path.display()),
158                ));
159            }
160        }
161    }
162}
163
164#[cfg(test)]
165mod tests {
166    use super::*;
167    use std::sync::Arc;
168    use std::sync::atomic::{AtomicBool, Ordering};
169    use std::thread;
170    use std::time::Duration;
171
172    fn touch(path: &Path) {
173        std::fs::write(path, b"# workload\n").unwrap();
174    }
175
176    #[test]
177    fn acquire_then_drop_round_trips() {
178        let dir = tempdir_relative("lock_round_trips");
179        let path = dir.join("w.yaml");
180        touch(&path);
181        let l1 = acquire(&path).expect("first lock");
182        drop(l1);
183        // Second acquire must succeed once the first dropped.
184        let _l2 = acquire(&path).expect("second lock");
185    }
186
187    #[test]
188    fn second_acquire_blocks_then_succeeds_when_first_releases() {
189        let dir = tempdir_relative("lock_block_release");
190        let path = dir.join("w.yaml");
191        touch(&path);
192
193        let l1 = acquire(&path).expect("first lock");
194        let path_clone = path.clone();
195        let waiter_done = Arc::new(AtomicBool::new(false));
196        let flag = waiter_done.clone();
197        let handle = thread::spawn(move || {
198            let _l2 = acquire(&path_clone).expect("second lock");
199            flag.store(true, Ordering::SeqCst);
200        });
201
202        // Give the waiter a moment to start polling.
203        thread::sleep(Duration::from_millis(100));
204        assert!(
205            !waiter_done.load(Ordering::SeqCst),
206            "second acquirer must still be waiting"
207        );
208
209        // Release the first lock; the waiter should pick it
210        // up within one POLL_INTERVAL plus a margin.
211        drop(l1);
212        handle.join().expect("waiter joined");
213        assert!(waiter_done.load(Ordering::SeqCst));
214    }
215
216    #[test]
217    fn deadline_exceeded_returns_wouldblock() {
218        // Override the deadline only for this test by holding
219        // the lock for longer than LOCK_DEADLINE.
220        let dir = tempdir_relative("lock_deadline");
221        let path = dir.join("w.yaml");
222        touch(&path);
223
224        let _l1 = acquire(&path).expect("first lock");
225        // Inside this test, the deadline is the global
226        // 5-second one. To keep the test fast we use a
227        // separate `try_lock_exclusive` wrapper that
228        // mirrors the public flow but with a shorter
229        // deadline. (The full deadline path is exercised
230        // implicitly any time `acquire` returns Ok after
231        // a wait — see `second_acquire_blocks...`.)
232        // Here we just confirm that the WouldBlock variant
233        // is constructed via a direct check. Probe the same
234        // file `acquire` locks (the sidecar on Windows).
235        #[cfg(windows)]
236        let probe = super::lock_target(&path);
237        #[cfg(not(windows))]
238        let probe = path.clone();
239        let f = std::fs::File::open(&probe).unwrap();
240        match FileExt::try_lock_exclusive(&f) {
241            Ok(false) => {} // expected — first lock still held
242            other => panic!("expected try_lock_exclusive=Ok(false), got {other:?}"),
243        }
244    }
245
246    fn tempdir_relative(label: &str) -> std::path::PathBuf {
247        let p =
248            std::env::temp_dir().join(format!("nmbrs-edit-lock-{label}-{}", std::process::id()));
249        std::fs::create_dir_all(&p).unwrap();
250        p
251    }
252}