Skip to main content

fs_core/
file_device.rs

1//! File-backed `BlockDevice`. Used for disk images, raw `/dev/diskN` reads,
2//! anything that std::fs::File can address.
3
4use crate::block::{BlockDevice, BlockRead};
5use crate::error::{Error, Result};
6use std::fs::{File, OpenOptions};
7use std::io::{Seek, SeekFrom, Write};
8use std::path::Path;
9use std::sync::atomic::{AtomicU64, Ordering};
10use std::sync::RwLock;
11
12/// A file opened as a block device.
13///
14/// # Readers share the lock; writers take it alone
15///
16/// A read was once `seek` then `read` under a plain mutex, which made
17/// the file's cursor shared state: two threads reading different offsets
18/// had to take turns, not because the device could not serve them at
19/// once but because one would have moved the other's cursor.
20///
21/// On Unix the cursor is not involved at all — `pread` takes the offset
22/// as an argument — so readers hold the lock *shared* and genuinely
23/// overlap. On Windows the equivalent (`seek_read`) *does* move the file
24/// pointer, so readers there take it exclusively and only that platform
25/// pays for the cursor.
26///
27/// Writers take it exclusively on both, because `write_at` is `seek`
28/// plus `write_all` and neither another writer's seek nor a reader may
29/// land in the middle of it.
30///
31/// # THE LOCK IS NOT AN OPTIMISATION, IT IS THE READ/WRITE CONTRACT
32///
33/// Reads briefly took no lock at all, which read as a natural
34/// consequence of positioned reads needing no cursor. It was not: it
35/// silently dropped the exclusion between readers and writers that the
36/// single mutex had provided, so a read overlapping a `write_at` could
37/// observe part of it. `write_all` is permitted to become several
38/// `write` calls, and a read is a loop of positioned reads — either
39/// split is a window, and the second one does not need the first.
40///
41/// So a reader holds the lock for the whole of [`FileDevice::read_at`],
42/// not for each positioned read inside it. Per-read guards would leave
43/// exactly the same hole one level down, and a rarer tear is worse than
44/// a common one because nobody can reproduce it.
45///
46/// # A known limit, stated rather than fixed
47///
48/// `std::sync::RwLock` does not promise writer preference on every
49/// platform, and the read path here is the hot one. A device under
50/// sustained parallel reads can therefore make a writer wait longer than
51/// a fair queue would. That is a throughput property, not a correctness
52/// one, and it is left alone rather than solved with a hand-rolled queue
53/// nobody would be able to audit.
54pub struct FileDevice {
55    file: File,
56    /// Shared by readers, exclusive to writers — see the type's own
57    /// note. `()` rather than the file, because it orders access rather
58    /// than owning the handle: positioned reads need no cursor, so
59    /// putting the `File` in here would reintroduce the serialisation
60    /// the shared guard exists to avoid.
61    io_lock: RwLock<()>,
62    /// ATOMIC BECAUSE [`BlockDevice::set_len`] MOVES IT, AND
63    /// `size_bytes` TAKES NO LOCK.
64    ///
65    /// It was a plain `u64`, which was right while nothing could change
66    /// it. `set_len` takes `&self` -- every method on these traits does,
67    /// because the devices are held behind `Arc<dyn _>` -- so the number
68    /// needs interior mutability, and the cheapest correct one is an
69    /// atomic rather than putting it under `io_lock`.
70    ///
71    /// Not under `io_lock` DELIBERATELY: `size_bytes` is called on the
72    /// hot path of every wrapper in this crate ([`crate::CachingDevice`]
73    /// asks it twice per read), and routing it through a lock a writer
74    /// holds exclusively would make an ordinary read contend with an
75    /// ordinary write to learn a number that fits in a register.
76    ///
77    /// `set_len` still takes `io_lock` exclusively -- it has real I/O to
78    /// exclude -- and publishes this afterwards. See its own note for
79    /// which of the two it moves first.
80    size: AtomicU64,
81    writable: bool,
82    /// Whether `set_len` can do anything: a writable handle on a REGULAR
83    /// FILE.
84    ///
85    /// Both halves are needed and neither implies the other. A read-only
86    /// handle obviously cannot truncate. A BLOCK DEVICE NODE is the case
87    /// that is easy to miss: `/dev/sdX` opened read-write is writable,
88    /// `write_at` works on it, and its length is the kernel's rather than
89    /// ours -- `ftruncate` on it is not a resize, and answering `true`
90    /// here would promise an image writer room it can never get.
91    ///
92    /// Decided once, at open, from the same `metadata` call that
93    /// `measure_size` is about to make: asking on every `can_grow` would
94    /// turn a capability question into a syscall, and the file type of an
95    /// already-open descriptor does not change.
96    growable: bool,
97    /// Test-only witness that a thread reached `io_lock` — see
98    /// [`LockArrivals`]. Absent from every non-test build, and the
99    /// `arriving` it feeds compiles to nothing there.
100    #[cfg(test)]
101    arrivals: LockArrivals,
102}
103
104/// How many threads are inside an `io_lock` acquisition and have not
105/// yet been granted the guard.
106///
107/// # WHY A TEST NEEDS THIS AND CANNOT DO WITHOUT IT
108///
109/// The exclusion this type promises can only be asserted negatively —
110/// an operation that must *not* proceed — and "did not finish within
111/// 300ms" is not that assertion. It is satisfied just as well by a
112/// worker the OS has not scheduled, so on a loaded runner a build with
113/// no lock at all passes. That is rust-fs-core#104, and it applied to
114/// all four of the tests that are the only evidence #77's fix works.
115///
116/// Signalling readiness from the top of the worker closure does not
117/// fix it: the gap between the signal and the call under test has no
118/// synchronisation in it, so the signal proves the thread ran once,
119/// not that it reached the lock.
120///
121/// A counter incremented immediately before the acquisition and
122/// decremented immediately after it is granted turns that into a
123/// positive observation the test can wait for. Seeing it non-zero
124/// while the test itself holds the lock means: this thread is in the
125/// acquisition, and it cannot leave until we let go. Remove the lock
126/// from the operation and the counter is never touched, so the wait
127/// times out and names what failed rather than passing.
128///
129/// Measured on `a_read_cannot_proceed_while_a_write_holds_the_lock`
130/// with `read_at`'s guard deleted, one commit, this machine:
131///
132/// | test shape                                | result   | time  |
133/// |-------------------------------------------|----------|-------|
134/// | ready signal, 400ms deschedule in the gap | **ok**   | 0.41s |
135/// | ready signal, no deschedule               | FAILED   | 0.33s |
136/// | this counter                              | FAILED   | 10.0s |
137///
138/// The first row is the defect: a build with no lock at all, passing.
139/// The gap has no upper bound on it, so 400ms is an illustration of a
140/// class rather than a threshold.
141///
142/// THE DECREMENT IS WHAT KEEPS THIS HONEST. If it went missing the
143/// counter would stick above zero, every wait would return
144/// immediately, and all four exclusion tests would pass without a
145/// worker ever reaching the lock — the same unwitnessed shape one
146/// level up. See
147/// `an_uncontended_operation_leaves_no_thread_waiting_at_the_lock`.
148#[cfg(test)]
149#[derive(Default)]
150struct LockArrivals {
151    waiting: std::sync::atomic::AtomicUsize,
152}
153
154/// Decrements the moment the guard is granted, whatever happens.
155#[cfg(test)]
156struct Arrival<'a>(&'a LockArrivals);
157
158#[cfg(test)]
159impl Drop for Arrival<'_> {
160    fn drop(&mut self) {
161        self.0
162            .waiting
163            .fetch_sub(1, std::sync::atomic::Ordering::SeqCst);
164    }
165}
166
167impl FileDevice {
168    /// Open read-only.
169    pub fn open<P: AsRef<Path>>(path: P) -> Result<Self> {
170        let file = File::open(path)?;
171        let size = measure_size(&file)?;
172        Ok(Self {
173            file,
174            io_lock: RwLock::new(()),
175            size: AtomicU64::new(size),
176            writable: false,
177            // A read-only handle cannot change any length, whatever it
178            // is open on.
179            growable: false,
180            #[cfg(test)]
181            arrivals: LockArrivals::default(),
182        })
183    }
184
185    /// Open read-write. Errors if the path is not writable.
186    pub fn open_rw<P: AsRef<Path>>(path: P) -> Result<Self> {
187        let file = OpenOptions::new().read(true).write(true).open(path)?;
188        let size = measure_size(&file)?;
189        let growable = is_regular_file(&file);
190        Ok(Self {
191            file,
192            io_lock: RwLock::new(()),
193            size: AtomicU64::new(size),
194            writable: true,
195            growable,
196            #[cfg(test)]
197            arrivals: LockArrivals::default(),
198        })
199    }
200
201    /// Open read-write if possible, fall back to read-only otherwise.
202    pub fn open_best_effort<P: AsRef<Path>>(path: P) -> Result<Self> {
203        let p = path.as_ref();
204        match Self::open_rw(p) {
205            Ok(d) => Ok(d),
206            Err(_) => Self::open(p),
207        }
208    }
209}
210
211/// How many bytes the opened handle addresses.
212///
213/// For a **regular file** this is the metadata length, and 0 is a
214/// legitimate answer: an empty file is empty.
215///
216/// For a **device node** it is not an answer at all. `st_size` is 0 for
217/// every block and character device on the platforms this crate ships
218/// to, so the size has to be asked for directly. Measured against an
219/// 8 MiB backing store:
220///
221/// | source                    | `metadata().len()` | `lseek(SEEK_END)` | ioctl     |
222/// |---------------------------|--------------------|-------------------|-----------|
223/// | macOS `/dev/disk9` (blk)  | 0                  | 0                 | 8388608   |
224/// | macOS `/dev/rdisk9` (chr) | 0                  | 0                 | 8388608   |
225/// | Linux `/dev/loop0` (blk)  | 0                  | 8388608           | 8388608   |
226///
227/// **`lseek(SEEK_END)` is not the portable fallback it looks like.** It
228/// answers 0 on macOS for both node types, so a device opened there
229/// would still report itself empty. The ioctl is the only mechanism that
230/// answered on both platforms, and it happens to avoid the objection to
231/// the seek as well: it does not move the file cursor. That matters
232/// here, because `read_once` uses `pread` specifically so reads need no
233/// lock -- see this type's own note.
234///
235/// When the size cannot be measured this returns an error rather than 0,
236/// because **a device reporting 0 is not inert, it is invisible.**
237/// `read_at` keeps serving real bytes, while
238/// [`BlockReadStreamer::read`] returns `Ok(0)` on its first call,
239/// [`CachingDevice`] treats every read as past-the-end and caches
240/// nothing, and every slice cut from it inherits a parent claiming to be
241/// empty. Each of those four failures is silent, which is the one
242/// outcome worth refusing outright.
243///
244/// [`BlockReadStreamer::read`]: crate::BlockReadStreamer
245/// [`CachingDevice`]: crate::CachingDevice
246#[cfg(unix)]
247fn measure_size(file: &File) -> Result<u64> {
248    use std::os::unix::fs::FileTypeExt;
249    let meta = file.metadata()?;
250    let ft = meta.file_type();
251    if !ft.is_block_device() && !ft.is_char_device() {
252        return Ok(meta.len());
253    }
254    device_size_bytes(file)
255}
256
257/// Windows keeps the metadata length, because no equivalent measurement
258/// has been made there. `\\.\PhysicalDriveN` is therefore still subject
259/// to the defect this function exists to fix; saying so is better than
260/// shipping an untested `DeviceIoControl` and implying otherwise.
261#[cfg(not(unix))]
262fn measure_size(file: &File) -> Result<u64> {
263    Ok(file.metadata()?.len())
264}
265
266// The crate has no dependencies and this is not worth acquiring one for:
267// `ioctl` is in libc, which is already linked into every std target.
268#[cfg(any(target_os = "macos", target_os = "ios", target_os = "linux"))]
269unsafe extern "C" {
270    fn ioctl(fd: std::os::raw::c_int, request: std::os::raw::c_ulong, ...) -> std::os::raw::c_int;
271}
272
273/// macOS: `<sys/disk.h>` gives the block size and the block count
274/// separately, and neither alone is the answer.
275#[cfg(any(target_os = "macos", target_os = "ios"))]
276fn device_size_bytes(file: &File) -> Result<u64> {
277    use std::io;
278    use std::os::fd::AsRawFd;
279    // _IOR('d', 24, u32) and _IOR('d', 25, u64).
280    const DKIOCGETBLOCKSIZE: std::os::raw::c_ulong = 0x4004_6418;
281    const DKIOCGETBLOCKCOUNT: std::os::raw::c_ulong = 0x4008_6419;
282
283    let fd = file.as_raw_fd();
284    let mut block_size: u32 = 0;
285    let mut block_count: u64 = 0;
286    // SAFETY: `fd` is open for as long as `file` is borrowed, and each
287    // request writes exactly the width its own encoding names into a
288    // local of precisely that type.
289    unsafe {
290        if ioctl(fd, DKIOCGETBLOCKSIZE, &raw mut block_size) < 0 {
291            return Err(io::Error::last_os_error().into());
292        }
293        if ioctl(fd, DKIOCGETBLOCKCOUNT, &raw mut block_count) < 0 {
294            return Err(io::Error::last_os_error().into());
295        }
296    }
297    block_count
298        .checked_mul(u64::from(block_size))
299        .ok_or_else(|| {
300            Error::Io(io::Error::other(format!(
301                "device reports {block_count} blocks of {block_size} bytes, \
302                 whose product does not fit in u64"
303            )))
304        })
305}
306
307/// `BLKGETSIZE64` as `_IOR(0x12, 114, size_t)`, for a given pointer
308/// width.
309///
310/// THE SIZE FIELD OF AN IOCTL REQUEST IS `sizeof(size_t)` -- the
311/// USERSPACE POINTER WIDTH, not the width of the value the kernel
312/// writes back. The payload is a `u64` on both, but the request NUMBER
313/// differs: `0x8008_1272` where a pointer is 8 bytes, `0x8004_1272`
314/// where it is 4. A literal for one width is rejected by the kernel on
315/// the other.
316///
317/// That is a regression this change would have INTRODUCED rather than
318/// inherited. Before it, an unmeasurable device node gave `Ok` with a
319/// size of 0; now it is an error, so a 64-bit-only constant would make
320/// [`FileDevice::open`] fail for EVERY block device on a 32-bit target.
321///
322/// Taking the width as a parameter is what makes it testable: nothing
323/// available here runs 32-bit, and `cargo check --target` compiles a
324/// wrong literal perfectly happily, so the only witness possible is to
325/// compute both encodings and compare them with the two numbers the
326/// kernel headers actually define.
327///
328/// Available in every TEST build rather than on Linux alone, because a
329/// test that cannot run is not a witness: gated to Linux it would be
330/// compiled and never executed on the machine the work is done on, and
331/// the arithmetic is the same arithmetic everywhere.
332#[cfg(any(target_os = "linux", test))]
333const fn blkgetsize64_for(pointer_width: usize) -> std::os::raw::c_ulong {
334    /// `_IOC_READ << _IOC_DIRSHIFT`.
335    const READ: std::os::raw::c_ulong = 0x8000_0000;
336    /// `_IOC_TYPESHIFT` is 8, `_IOC_SIZESHIFT` is 16.
337    const TYPE: std::os::raw::c_ulong = 0x12;
338    const NR: std::os::raw::c_ulong = 114;
339    READ | ((pointer_width as std::os::raw::c_ulong) << 16) | (TYPE << 8) | NR
340}
341
342/// Linux: `<linux/fs.h>` answers in bytes in one call.
343#[cfg(target_os = "linux")]
344fn device_size_bytes(file: &File) -> Result<u64> {
345    use std::io;
346    use std::os::fd::AsRawFd;
347    // _IOR(0x12, 114, size_t), encoded for THIS target -- see
348    // `blkgetsize64_for`. Identical to the familiar 0x8008_1272 on a
349    // 64-bit target and correct on a 32-bit one.
350    const BLKGETSIZE64: std::os::raw::c_ulong = blkgetsize64_for(std::mem::size_of::<usize>());
351
352    let mut size: u64 = 0;
353    // SAFETY: as above -- one call, writing one u64 into a u64.
354    let rc = unsafe { ioctl(file.as_raw_fd(), BLKGETSIZE64, &raw mut size) };
355    if rc < 0 {
356        return Err(io::Error::last_os_error().into());
357    }
358    Ok(size)
359}
360
361/// Every other Unix: refuse rather than guess. `lseek` is measured wrong
362/// on one of the two platforms tested, so extending it here on the
363/// strength of that would be picking the silent failure.
364#[cfg(all(
365    unix,
366    not(any(target_os = "macos", target_os = "ios", target_os = "linux"))
367))]
368fn device_size_bytes(_file: &File) -> Result<u64> {
369    use std::io;
370    Err(Error::Io(io::Error::other(
371        "no measured way to read a device node's size on this platform; \
372         open the backing image file rather than the device node",
373    )))
374}
375
376impl FileDevice {
377    /// Marks this thread as having reached `io_lock` and not yet been
378    /// granted it. The returned value decrements on drop — which, at
379    /// the two call sites below, is after the acquisition returns.
380    ///
381    /// Compiles to nothing outside a test build: the field it counts
382    /// does not exist there. See [`LockArrivals`] for why a test cannot
383    /// establish the same thing from outside.
384    #[cfg(test)]
385    fn arriving(&self) -> Arrival<'_> {
386        self.arrivals
387            .waiting
388            .fetch_add(1, std::sync::atomic::Ordering::SeqCst);
389        Arrival(&self.arrivals)
390    }
391
392    /// THE ONLY PLACES `io_lock` IS ACQUIRED — two on Unix, one on
393    /// Windows.
394    ///
395    /// Not a wrapper for its own sake: the arrival counter has to sit
396    /// immediately before the acquisition to mean anything, and one
397    /// pair of methods is what stops a third call site being added
398    /// without it. `_arrival` outlives the tail expression and is
399    /// dropped once the guard has been granted — which is what makes
400    /// a non-zero count mean "waiting" rather than "has waited".
401    ///
402    /// The `cfg` is on the statement rather than on two bodies of
403    /// `arriving`, so a non-test build contains the acquisition and
404    /// nothing else.
405    ///
406    /// # `shared_guard` IS UNIX-ONLY, AND THAT IS THE DESIGN RATHER
407    /// THAN TIDINESS
408    ///
409    /// Nothing on Windows takes `io_lock` shared. `read_guard` there is
410    /// `exclusive_guard`, deliberately, because `seek_read` moves the
411    /// file pointer and a reader must exclude other readers as well as
412    /// writers — see `read_guard`. So on Windows this method is not
413    /// merely unused, it MUST NOT BE CALLED: a future caller reaching
414    /// for the cheaper guard would reintroduce the cursor race that the
415    /// exclusive read guard exists to prevent.
416    ///
417    /// `cargo clippy --target x86_64-pc-windows-msvc --all-targets
418    /// -- -D warnings` reported it as `method shared_guard is never
419    /// used`, and `#[allow(dead_code)]` would have been the wrong
420    /// answer: it silences the compiler on a platform where the right
421    /// statement is that the method does not exist. Compiling it out
422    /// makes a call site that should not exist fail to build.
423    #[cfg(unix)]
424    fn shared_guard(&self) -> std::sync::RwLockReadGuard<'_, ()> {
425        #[cfg(test)]
426        let _arrival = self.arriving();
427        self.io_lock.read().unwrap()
428    }
429
430    fn exclusive_guard(&self) -> std::sync::RwLockWriteGuard<'_, ()> {
431        #[cfg(test)]
432        let _arrival = self.arriving();
433        self.io_lock.write().unwrap()
434    }
435
436    /// The guard a read holds for the whole of `read_at`.
437    ///
438    /// Unix takes it SHARED: `pread` carries its own offset, so readers
439    /// do not disturb each other and only need to be kept apart from
440    /// writers.
441    ///
442    /// Windows takes it EXCLUSIVE, because `seek_read` moves the file
443    /// pointer — there, one reader really can spoil another's offset, so
444    /// readers must exclude readers as well as writers. Same lock, same
445    /// call site, and only that platform pays for the cursor.
446    #[cfg(unix)]
447    fn read_guard(&self) -> std::sync::RwLockReadGuard<'_, ()> {
448        self.shared_guard()
449    }
450
451    #[cfg(windows)]
452    fn read_guard(&self) -> std::sync::RwLockWriteGuard<'_, ()> {
453        self.exclusive_guard()
454    }
455
456    /// One positioned read, returning what it got.
457    ///
458    /// TAKES NO LOCK ON EITHER PLATFORM. The caller holds `read_guard`
459    /// for the whole read; acquiring anything here would be a second,
460    /// non-reentrant acquisition of the same lock — on Windows, where
461    /// that guard is exclusive, an immediate self-deadlock.
462    #[cfg(unix)]
463    fn read_once(&self, offset: u64, buf: &mut [u8]) -> Result<usize> {
464        use std::os::unix::fs::FileExt;
465        Ok(self.file.read_at(buf, offset)?)
466    }
467
468    /// Windows: `seek_read` DOES move the file pointer. The exclusion
469    /// that needs is held by the caller's guard, not taken here — see
470    /// `read_guard`.
471    #[cfg(windows)]
472    fn read_once(&self, offset: u64, buf: &mut [u8]) -> Result<usize> {
473        use std::os::windows::fs::FileExt;
474        Ok(self.file.seek_read(buf, offset)?)
475    }
476}
477
478impl BlockRead for FileDevice {
479    fn read_at(&self, offset: u64, buf: &mut [u8]) -> Result<()> {
480        // HELD ACROSS THE WHOLE LOOP, NOT AROUND EACH POSITIONED READ.
481        //
482        // The loop below can issue several reads for one call, and a
483        // guard taken inside it would let a `write_at` land between two
484        // of them: the caller would get some bytes from before the write
485        // and some from after, which is the tear this exists to prevent
486        // moved one level down and made rarer. Rarer is worse — nobody
487        // can reproduce it.
488        //
489        // NO TEST PINS THIS PLACEMENT, and the reason is worth writing
490        // down because the obvious one is wrong. A regular file does
491        // return short reads — at EOF — and this loop retries rather
492        // than failing on one, so a read spanning EOF really does run
493        // twice and the guard really would be released in between. That
494        // discriminator was built: a 4096-byte file, a reader asking
495        // 4090..4102, a writer parked on the lock growing the file at
496        // 4090, where interleaved old-and-new bytes are reachable no
497        // other way. 200 trials with the guard moved inside the loop
498        // produced 0 tears. The window between releasing at the end of
499        // one iteration and retaking at the start of the next is a few
500        // instructions, and a writer already waiting never won it.
501        //
502        // So this is unwitnessed, not inert: the placement is correct
503        // and the race it prevents is simply too narrow to enter on
504        // demand. Measured, not argued.
505        let _guard = self.read_guard();
506
507        // A SHORT READ IS AN ERROR NAMING WHAT WAS ASKED FOR AND WHAT
508        // ARRIVED, not a smaller answer: a caller that asked for a block
509        // and got half of one cannot tell the difference from bytes.
510        let mut total = 0usize;
511        while total < buf.len() {
512            let n = self.read_once(offset + total as u64, &mut buf[total..])?;
513            if n == 0 {
514                return Err(Error::ShortRead {
515                    offset,
516                    want: buf.len(),
517                    got: total,
518                });
519            }
520            total += n;
521        }
522        Ok(())
523    }
524
525    fn size_bytes(&self) -> u64 {
526        // `Acquire` pairs with the `Release` store in `set_len`: a
527        // thread that observes a grown size must also observe the
528        // `ftruncate` that produced it.
529        self.size.load(Ordering::Acquire)
530    }
531}
532
533impl BlockDevice for FileDevice {
534    /// A write past the end is refused, not an extension.
535    ///
536    /// `write_all` at a seeked offset EXTENDS a file, and this method
537    /// had no bound of its own, so a write straddling the end grew the
538    /// backing store while `size_bytes` went on reporting the length
539    /// taken at construction -- measured on a 4096-byte file:
540    /// `write_at(4094, 8 bytes)` returned `Ok`, the file became 4102
541    /// bytes, `size_bytes()` stayed 4096, and `read_at(4096, 6)` then
542    /// handed those bytes back. The two halves of one device disagreed
543    /// about where it ended, and a caller bounding its reads by
544    /// `size_bytes` -- which is what [`crate::CachingDevice`] does,
545    /// clamping every block it fetches -- could never reach them.
546    ///
547    /// `RwBytes` in this crate's own test devices already refuses the
548    /// same operation, commenting "a device is not a `Vec`", and the
549    /// slice adapters in [`crate::slice`] clamp their window
550    /// specifically because this method did not:
551    /// `slice_rw_length_is_clamped_and_a_write_past_it_does_not_grow_the_image`
552    /// names the file's length on disk as its oracle. This is the same
553    /// rule one layer down, where it was missing.
554    ///
555    /// The alternative -- letting the size move and reopening -- is
556    /// what [`BlockRead::size_bytes`]'s contract forbids. See
557    /// rust-fs-core#70.
558    fn write_at(&self, offset: u64, buf: &[u8]) -> Result<()> {
559        if !self.writable {
560            return Err(Error::ReadOnly);
561        }
562        // READ ONCE, COMPARED AND REPORTED FROM THE SAME VALUE. `size`
563        // is atomic now that `set_len` moves it, and loading it twice
564        // could bound the write against one length and name another in
565        // the error -- an `OutOfBounds` whose `size` field does not
566        // explain its own refusal.
567        let size = self.size_bytes();
568        // `checked_add` because a caller-supplied offset near `u64::MAX`
569        // would otherwise wrap and land back inside the device.
570        let end = offset.checked_add(buf.len() as u64);
571        if end.is_none_or(|end| end > size) {
572            return Err(Error::OutOfBounds {
573                offset,
574                len: buf.len() as u64,
575                size,
576            });
577        }
578        // EXCLUSIVE: excludes other writers' seeks and every reader.
579        let _guard = self.exclusive_guard();
580        let mut f = &self.file;
581        f.seek(SeekFrom::Start(offset))?;
582        f.write_all(buf)?;
583        Ok(())
584    }
585
586    fn flush(&self) -> Result<()> {
587        if !self.writable {
588            return Ok(());
589        }
590        // EXCLUSIVE for the same reason as `write_at`: this pushes
591        // buffered bytes at the file and must not interleave with a
592        // write or a read.
593        let _guard = self.exclusive_guard();
594        let mut f = &self.file;
595        f.flush()?;
596        self.file.sync_data()?;
597        Ok(())
598    }
599
600    fn is_writable(&self) -> bool {
601        self.writable
602    }
603
604    /// Set the file's length, and the length this device reports, as one
605    /// operation.
606    ///
607    /// # THE POINT IS THAT THE TWO MOVE TOGETHER
608    ///
609    /// #75 refused a write past the end because `write_all` at a seeked
610    /// offset grew the FILE while `size_bytes` went on reporting the
611    /// length taken at construction, so the two halves of one device
612    /// disagreed about where it ended and a caller bounding its reads by
613    /// `size_bytes` could never reach what it had written
614    /// (rust-fs-core#70). That bound stays. This method is the other
615    /// half: growth that says so, and moves the number with it.
616    ///
617    /// An implementation that called `File::set_len` and left `self.size`
618    /// alone would be #70 with a nicer name on it.
619    ///
620    /// # THE NARROWER LENGTH IS PUBLISHED FIRST, IN BOTH DIRECTIONS
621    ///
622    /// `ftruncate` and the store are two steps, and one of the two
623    /// orderings has a window in it. Take a shrink done store-then-
624    /// publish: between the truncate and the store, this device declares
625    /// 8192 bytes over a 4096-byte file, so a concurrent read inside the
626    /// declared device falls off the end of the real one and comes back
627    /// `ShortRead`. The other direction is harmless — a device that
628    /// briefly declares 4096 bytes over an 8192-byte file is only
629    /// under-reporting, which is the state every `FileDevice` is in
630    /// whenever something else appends to its file.
631    ///
632    /// So: a GROW truncates and then stores, and a SHRINK stores and then
633    /// truncates. The invariant is one sentence — THE DECLARED SIZE NEVER
634    /// EXCEEDS THE FILE'S REAL LENGTH — and it holds at every instant
635    /// rather than only at the ends.
636    ///
637    /// # `io_lock` EXCLUSIVELY, LIKE A WRITE
638    ///
639    /// For the same reason `write_at` and `flush` take it: this changes
640    /// the file underneath every reader, and a read must not observe
641    /// half of it. `size_bytes` deliberately does NOT take the lock —
642    /// see the field — so the ordering above is what keeps a reader that
643    /// asked the size mid-call from being misled, not the lock.
644    ///
645    /// # WHAT IT REFUSES
646    ///
647    /// [`Error::ReadOnly`] on a handle opened with [`FileDevice::open`],
648    /// before touching the file. [`Error::Custom`] on a handle that is
649    /// writable but not a regular file — a block device node, whose
650    /// length belongs to the kernel — naming that reason rather than
651    /// letting `ftruncate`'s `EINVAL` stand in for it. See
652    /// [`FileDevice::can_grow`], which is the question to ask instead of
653    /// discovering either of these.
654    fn set_len(&self, new_len: u64) -> Result<()> {
655        if !self.writable {
656            return Err(Error::ReadOnly);
657        }
658        if !self.growable {
659            return Err(Error::Custom(
660                "this FileDevice is open on something that is not a regular file \
661                 -- a device node's length is the kernel's, not ours, and cannot \
662                 be set through this handle"
663                    .to_string(),
664            ));
665        }
666
667        // EXCLUSIVE: excludes every reader and every other writer, the
668        // same as `write_at`.
669        let _guard = self.exclusive_guard();
670
671        let old = self.size.load(Ordering::Acquire);
672        if new_len < old {
673            // Narrow the declared device BEFORE the bytes go. See above.
674            self.size.store(new_len, Ordering::Release);
675        }
676        if let Err(e) = self.file.set_len(new_len) {
677            // A FAILED SHRINK HAS ALREADY NARROWED THE DECLARATION, and
678            // leaving it there would make the device deny bytes it still
679            // holds -- #70's defect, reached by the error path. Re-measure
680            // rather than restoring `old`: whether `ftruncate` did nothing
681            // or something is not knowable from its error, and the file
682            // itself is the only honest answer.
683            if let Ok(actual) = measure_size(&self.file) {
684                self.size.store(actual, Ordering::Release);
685            }
686            return Err(e.into());
687        }
688        self.size.store(new_len, Ordering::Release);
689        Ok(())
690    }
691
692    fn can_grow(&self) -> bool {
693        self.growable
694    }
695}
696
697/// Is this handle open on a regular file?
698///
699/// The other half of `can_grow`, and the half a reader is most likely to
700/// assume. A `FileDevice` is opened just as often on `/dev/sdX` as on an
701/// image: `measure_size` has a whole `ioctl` path for exactly that case.
702/// Such a handle is writable, `write_at` works on it, and its length is
703/// fixed by the kernel -- `ftruncate` on a block device is not a resize.
704///
705/// A failed `metadata` call answers `false`. The question is "may this
706/// device promise it can grow", and a promise nobody could verify is not
707/// one to make.
708fn is_regular_file(file: &File) -> bool {
709    file.metadata().is_ok_and(|m| m.file_type().is_file())
710}
711
712#[cfg(test)]
713mod tests {
714    /// THE TWO NUMBERS THE KERNEL HEADERS DEFINE, and the pointer
715    /// widths they belong to. External knowledge, not a restatement of
716    /// the formula -- a test that recomputed the encoding on both sides
717    /// would agree with itself whatever the encoding was.
718    ///
719    /// This cannot prove the ioctl works on 32-bit; nothing available
720    /// here runs 32-bit, and `cargo check --target` compiles a wrong
721    /// literal happily. What it does is stop the encoding quietly
722    /// reverting to a single hardcoded number, which is the way this
723    /// defect arrived.
724    #[test]
725    fn blkgetsize64_is_encoded_for_the_pointer_width() {
726        const KNOWN: &[(usize, u64)] = &[(4, 0x8004_1272), (8, 0x8008_1272)];
727        for (width, want) in KNOWN {
728            assert_eq!(
729                super::blkgetsize64_for(*width) as u64,
730                *want,
731                "_IOR(0x12, 114, size_t) with a {width}-byte size_t is {want:#010x}"
732            );
733        }
734    }
735
736    /// And the one this build will actually issue is the one for THIS
737    /// target, rather than whichever happened to be written down.
738    #[test]
739    fn this_target_issues_its_own_encoding() {
740        let width = std::mem::size_of::<usize>();
741        let expected = if width == 8 {
742            0x8008_1272u64
743        } else {
744            0x8004_1272u64
745        };
746        assert_eq!(super::blkgetsize64_for(width) as u64, expected);
747    }
748
749    use super::*;
750
751    /// Concurrent readers do not serialise, and none of them sees
752    /// another's offset.
753    ///
754    /// THE BUG THIS REPLACES: reads were `seek` then `read` under one
755    /// mutex, so the file cursor was shared state. Two threads reading
756    /// different parts of the same image took turns for no reason the
757    /// device imposed. Worse, the shape was one edit away from being
758    /// wrong rather than merely slow -- drop the lock without moving to
759    /// positioned reads and every reader corrupts every other reader's
760    /// offset.
761    ///
762    /// The assertion is on the BYTES rather than on timing: a test that
763    /// measured overlap would be a flake on a loaded machine, while a
764    /// reader that got another's offset returns the wrong bytes every
765    /// time.
766    #[test]
767    fn many_threads_reading_different_offsets_each_get_their_own_bytes() {
768        let path = temp_path("parallel_reads");
769        let _c = Cleanup(path.clone());
770        // Each 256-byte page filled with its own page number, so a read
771        // that landed at the wrong offset is obvious from one byte.
772        let mut bytes = Vec::with_capacity(64 * 256);
773        for page in 0..64u8 {
774            bytes.extend(std::iter::repeat_n(page, 256));
775        }
776        std::fs::write(&path, &bytes).expect("write the image");
777
778        let dev = std::sync::Arc::new(FileDevice::open(&path).expect("open"));
779        let mut handles = Vec::new();
780        for page in 0..64u8 {
781            let dev = dev.clone();
782            handles.push(std::thread::spawn(move || {
783                // Several times each, so a thread that raced would have
784                // many chances to read somebody else's page.
785                for _ in 0..50 {
786                    let mut buf = [0u8; 256];
787                    dev.read_at(u64::from(page) * 256, &mut buf).expect("read");
788                    assert!(
789                        buf.iter().all(|b| *b == page),
790                        "page {page} came back holding another page's bytes"
791                    );
792                }
793            }));
794        }
795        for h in handles {
796            h.join().expect("a reader panicked");
797        }
798    }
799
800    use std::sync::atomic::{AtomicU64, Ordering};
801
802    /// Unique temp path under the system temp dir (no extra dev-deps).
803    fn temp_path(tag: &str) -> std::path::PathBuf {
804        static N: AtomicU64 = AtomicU64::new(0);
805        let n = N.fetch_add(1, Ordering::Relaxed);
806        let pid = std::process::id();
807        std::env::temp_dir().join(format!("fs_core_{tag}_{pid}_{n}.bin"))
808    }
809
810    struct Cleanup(std::path::PathBuf);
811    impl Drop for Cleanup {
812        fn drop(&mut self) {
813            let _ = std::fs::remove_file(&self.0);
814        }
815    }
816
817    #[test]
818    fn open_rw_round_trips_write_then_read() {
819        let path = temp_path("rw");
820        let _g = Cleanup(path.clone());
821        std::fs::write(&path, vec![0u8; 32]).unwrap();
822
823        let dev = FileDevice::open_rw(&path).unwrap();
824        assert!(dev.is_writable());
825        assert_eq!(dev.size_bytes(), 32);
826
827        dev.write_at(8, &[0xAA, 0xBB, 0xCC, 0xDD]).unwrap();
828        dev.flush().unwrap();
829
830        let mut buf = [0u8; 4];
831        dev.read_at(8, &mut buf).unwrap();
832        assert_eq!(buf, [0xAA, 0xBB, 0xCC, 0xDD]);
833    }
834
835    #[test]
836    fn open_rw_errors_on_missing_path() {
837        let path = temp_path("missing");
838        assert!(FileDevice::open_rw(&path).is_err());
839    }
840
841    #[test]
842    fn open_best_effort_uses_rw_when_writable() {
843        let path = temp_path("best_rw");
844        let _g = Cleanup(path.clone());
845        std::fs::write(&path, vec![0u8; 16]).unwrap();
846
847        let dev = FileDevice::open_best_effort(&path).unwrap();
848        assert!(dev.is_writable());
849        dev.write_at(0, &[0x11; 4]).unwrap();
850    }
851
852    #[test]
853    #[cfg(unix)]
854    fn open_best_effort_falls_back_to_read_only() {
855        use std::os::unix::fs::PermissionsExt;
856
857        let path = temp_path("best_ro");
858        let _g = Cleanup(path.clone());
859        std::fs::write(&path, vec![0xEFu8; 16]).unwrap();
860        // Read-only permissions force `open_rw` to fail; fall back to `open`.
861        std::fs::set_permissions(&path, std::fs::Permissions::from_mode(0o444)).unwrap();
862
863        let dev = FileDevice::open_best_effort(&path).unwrap();
864        assert!(!dev.is_writable());
865        // Writes are rejected at the read-only layer.
866        assert!(matches!(dev.write_at(0, &[0u8; 4]), Err(Error::ReadOnly)));
867        // Read still works.
868        let mut buf = [0u8; 4];
869        dev.read_at(0, &mut buf).unwrap();
870        assert_eq!(buf, [0xEF; 4]);
871        // Flush on a read-only device is a no-op success.
872        dev.flush().unwrap();
873    }
874
875    use std::sync::mpsc;
876    use std::time::Duration;
877
878    /// How long an operation that must finish is given. Generous on
879    /// purpose: a loaded machine makes it slower, not flakier.
880    ///
881    /// EVERY DEADLINE IN THE EXCLUSION TESTS IS THIS ONE. There used to
882    /// be a second, short one — 300ms, after which a worker that had
883    /// not finished was taken to have been blocked. That is the defect
884    /// rust-fs-core#104 describes: an unscheduled worker is
885    /// indistinguishable from a blocked one, so the assertion passed
886    /// for a reason unrelated to the lock and would have kept passing
887    /// with the lock removed. Blocking is now established by
888    /// [`await_arrival`] instead, which fails when nothing arrives
889    /// rather than passing when nothing happens.
890    const UNBLOCKED_WITHIN: Duration = Duration::from_secs(10);
891
892    /// A 4 KiB image of one repeated byte, opened read-write.
893    fn rw_image(tag: &str, fill: u8) -> (std::sync::Arc<FileDevice>, Cleanup) {
894        let path = temp_path(tag);
895        let cleanup = Cleanup(path.clone());
896        std::fs::write(&path, vec![fill; 4096]).expect("write the image");
897        let dev = std::sync::Arc::new(FileDevice::open_rw(&path).expect("open rw"));
898        (dev, cleanup)
899    }
900
901    fn waiting_at_the_lock(dev: &FileDevice) -> usize {
902        dev.arrivals.waiting.load(Ordering::SeqCst)
903    }
904
905    /// Blocks until a thread is parked inside an `io_lock` acquisition,
906    /// and PANICS IF NONE EVER IS.
907    ///
908    /// This is the half that carries the evidence. Once it returns, the
909    /// worker is between the counter's increment and the guard being
910    /// granted — and since the caller holds that guard and has not let
911    /// go, the worker cannot leave. "The operation is blocked on the
912    /// lock" is then a fact about the program's state rather than an
913    /// inference from a stopwatch.
914    ///
915    /// An operation that stopped taking the lock never increments, so
916    /// this times out and says which operation never arrived. The
917    /// deadline can only produce a false FAILURE, which is the safe
918    /// direction and the opposite of what it replaced.
919    fn await_arrival(dev: &FileDevice, operation: &str) {
920        let deadline = std::time::Instant::now() + UNBLOCKED_WITHIN;
921        while waiting_at_the_lock(dev) == 0 {
922            assert!(
923                std::time::Instant::now() < deadline,
924                "{operation} never reached io_lock. It either never ran, or it \
925                 does not take the lock at all — which is the exclusion this \
926                 test exists to assert"
927            );
928            std::thread::sleep(Duration::from_micros(200));
929        }
930    }
931
932    /// Releases the holder's guard and THEN joins the worker, on every
933    /// path out of a test including a panicking assertion.
934    ///
935    /// Both halves are rust-fs-core#105. Discarding the `JoinHandle`
936    /// let a failing assertion unwind the test thread while the worker
937    /// was still inside `read_at`/`write_at`/`flush` with the file
938    /// open, so `Cleanup` removed the temp file underneath it — a race
939    /// on exactly the run where a real regression is being diagnosed,
940    /// and a leaked file per failure on a platform that will not unlink
941    /// an open file.
942    ///
943    /// THE ORDER IS NOT INCIDENTAL. Joining first would wait for a
944    /// worker this very thread is blocking, and the test would hang
945    /// instead of failing. Dropping the guard is what lets the worker
946    /// finish so the join can return.
947    ///
948    /// Declared after the `Cleanup` it protects, so it drops first and
949    /// the file still exists when the worker touches it.
950    struct ReleaseThenJoin<G> {
951        held: Option<G>,
952        worker: Option<std::thread::JoinHandle<()>>,
953    }
954
955    impl<G> ReleaseThenJoin<G> {
956        /// Let the blocked operation through, keeping the join.
957        fn release(&mut self) {
958            self.held = None;
959        }
960    }
961
962    impl<G> Drop for ReleaseThenJoin<G> {
963        fn drop(&mut self) {
964            self.held = None;
965            if let Some(worker) = self.worker.take() {
966                // Ignored on purpose: this runs while unwinding a
967                // failed assertion, and a panic in a drop during
968                // unwinding aborts the process, which would replace the
969                // test's own message with nothing.
970                let _ = worker.join();
971            }
972        }
973    }
974
975    /// THE WORKER IS JOINED BEFORE THE TEMP FILE GOES, ON THE PANIC
976    /// PATH SPECIFICALLY.
977    ///
978    /// rust-fs-core#105 is a failure-path defect, so the only way to
979    /// witness it is to fail on purpose. The four exclusion tests
980    /// discarded their `JoinHandle`, so a failing assertion unwound the
981    /// test thread while the worker was still inside the operation with
982    /// the file open, and `Cleanup` removed the file underneath it —
983    /// worst on the one run that matters, the run where a real
984    /// regression tripped one of them.
985    ///
986    /// This reproduces that shape with the file replaced by a recorder,
987    /// because the ordering is what is being asserted and a removed
988    /// file cannot say when it went. With the join dropped the recorder
989    /// sees `cleanup` first; with the release and the join in the order
990    /// [`ReleaseThenJoin`] fixes, it sees `worker` first.
991    ///
992    /// **Two panics are printed while this test runs, and both are the
993    /// test working.** The first is the deliberate one. The second is
994    /// the worker's: unwinding past a held `RwLock` guard poisons it,
995    /// and every acquisition in this module `unwrap`s, so the read the
996    /// worker was blocked on comes back `PoisonError` instead of
997    /// bytes. That is why the worker's arrival is recorded from a
998    /// `Drop` rather than after the read — the ordering claim is about
999    /// when the thread *ends*, and it must hold whether the read
1000    /// returns or unwinds.
1001    #[test]
1002    fn a_panicking_assertion_joins_the_worker_before_the_cleanup_runs() {
1003        use std::panic::AssertUnwindSafe;
1004        use std::sync::{Arc, Mutex};
1005
1006        /// Stands in for `Cleanup`: same position, same drop timing,
1007        /// but it says when it ran.
1008        struct Recorder(Arc<Mutex<Vec<&'static str>>>);
1009        impl Drop for Recorder {
1010            fn drop(&mut self) {
1011                self.0.lock().unwrap().push("cleanup");
1012            }
1013        }
1014
1015        let order: Arc<Mutex<Vec<&'static str>>> = Arc::new(Mutex::new(Vec::new()));
1016        let (dev, _c) = rw_image("panic_joins", 0x99);
1017
1018        let outcome = std::panic::catch_unwind(AssertUnwindSafe(|| {
1019            // Declared first, so it drops last — exactly where
1020            // `Cleanup` sits in the four exclusion tests.
1021            let _recorder = Recorder(Arc::clone(&order));
1022
1023            let held = dev.io_lock.write().unwrap();
1024            let worker = {
1025                let dev = Arc::clone(&dev);
1026                let order = Arc::clone(&order);
1027                std::thread::spawn(move || {
1028                    /// Records the worker ENDING, return or unwind.
1029                    struct Ended(Arc<Mutex<Vec<&'static str>>>);
1030                    impl Drop for Ended {
1031                        fn drop(&mut self) {
1032                            self.0.lock().unwrap().push("worker");
1033                        }
1034                    }
1035                    let _ended = Ended(order);
1036                    let mut buf = [0u8; 4096];
1037                    let _ = dev.read_at(0, &mut buf);
1038                })
1039            };
1040            let _lock = ReleaseThenJoin {
1041                held: Some(held),
1042                worker: Some(worker),
1043            };
1044
1045            await_arrival(&dev, "read_at");
1046            panic!("the assertion an exclusion test exists to make, failing");
1047        }));
1048
1049        // NAMED, NOT MERELY PRESENT. `await_arrival` panics too, and
1050        // an `is_err()` satisfied by that one would report a join this
1051        // test never exercised.
1052        let payload = outcome.expect_err("the deliberate panic must have unwound");
1053        let message = payload
1054            .downcast_ref::<&str>()
1055            .copied()
1056            .or_else(|| payload.downcast_ref::<String>().map(String::as_str))
1057            .unwrap_or("<panic payload was not a string>");
1058        assert!(
1059            message.contains("an exclusion test exists to make"),
1060            "the test unwound for the wrong reason, so the ordering below is \
1061             about some other failure: {message}"
1062        );
1063
1064        assert_eq!(
1065            *order.lock().unwrap(),
1066            vec!["worker", "cleanup"],
1067            "the worker was still inside read_at when cleanup ran: an unwinding \
1068             test must release the guard, join the worker, and only then let the \
1069             temp file be removed"
1070        );
1071    }
1072
1073    /// THE COUNTER COMES BACK DOWN.
1074    ///
1075    /// [`await_arrival`] is only evidence while a non-zero `waiting`
1076    /// means a thread is parked *now*. A missing decrement would leave
1077    /// it stuck above zero, every wait would return immediately, and
1078    /// all four exclusion tests would pass without a worker ever
1079    /// reaching the lock — the same unwitnessed shape they were fixed
1080    /// for, one level up. So the uncontended path is asserted too:
1081    /// every acquisition site, no contention, nothing left behind.
1082    #[test]
1083    fn an_uncontended_operation_leaves_no_thread_waiting_at_the_lock() {
1084        let (dev, _c) = rw_image("arrivals_settle", 0x0F);
1085        assert_eq!(waiting_at_the_lock(&dev), 0, "nothing has run yet");
1086
1087        let mut buf = [0u8; 16];
1088        dev.read_at(0, &mut buf).expect("read");
1089        assert_eq!(
1090            waiting_at_the_lock(&dev),
1091            0,
1092            "read_at left an arrival behind"
1093        );
1094
1095        dev.write_at(0, &[0x10u8; 16]).expect("write");
1096        assert_eq!(
1097            waiting_at_the_lock(&dev),
1098            0,
1099            "write_at left an arrival behind"
1100        );
1101
1102        dev.flush().expect("flush");
1103        assert_eq!(waiting_at_the_lock(&dev), 0, "flush left an arrival behind");
1104    }
1105
1106    /// A READ CONCURRENT WITH A WRITE MUST NOT PROCEED.
1107    ///
1108    /// This is the regression. Reads were briefly taken with no lock at
1109    /// all, which quietly removed the exclusion the original single
1110    /// mutex gave and left a read free to run through the middle of a
1111    /// `write_at` — `write_all` may become several `write` calls, and
1112    /// `read_at` is itself a loop, so either side can split.
1113    ///
1114    /// # Why the lock rather than the tear is the assertion
1115    ///
1116    /// The obvious test races a writer against readers and looks for a
1117    /// region holding bytes from both sides of the write. On a regular
1118    /// file that test cannot fail: a single `write` call is atomic
1119    /// against `pread` on both Linux and macOS, and `write_all` only
1120    /// splits above roughly 2 GiB, so the tear it looks for is
1121    /// unreachable at any size a test would use. It would pass with the
1122    /// fix reverted — an assertion whose outcome does not depend on the
1123    /// defect, which is worse than no assertion.
1124    ///
1125    /// So the exclusion itself is asserted, by holding the very guard
1126    /// `write_at` takes and requiring that a read cannot get past it.
1127    /// That is deterministic, needs no tear to be reproducible, and
1128    /// fails the moment `read_at` stops taking the lock.
1129    ///
1130    /// # And "cannot get past it" is observed, not timed
1131    ///
1132    /// See [`await_arrival`] and rust-fs-core#104: the reader is
1133    /// required to *arrive* at the lock, which a build that does not
1134    /// take the lock cannot do, rather than merely to not finish within
1135    /// a short window, which a build that does not take the lock
1136    /// manages easily on a loaded machine.
1137    #[test]
1138    fn a_read_cannot_proceed_while_a_write_holds_the_lock() {
1139        let (dev, _c) = rw_image("read_excluded_by_write", 0x5A);
1140
1141        // Stands in for a write in progress: the same exclusive guard
1142        // `write_at` holds across its seek and write. Taken directly
1143        // rather than through `exclusive_guard`, so the holder is not
1144        // itself counted as an arrival.
1145        let held = dev.io_lock.write().unwrap();
1146        assert_eq!(
1147            waiting_at_the_lock(&dev),
1148            0,
1149            "the holder must not count as a waiter, or the wait below proves nothing"
1150        );
1151
1152        let (done_tx, done_rx) = mpsc::channel();
1153        let worker = {
1154            let dev = std::sync::Arc::clone(&dev);
1155            std::thread::spawn(move || {
1156                let mut buf = [0u8; 4096];
1157                let outcome = dev.read_at(0, &mut buf);
1158                let _ = done_tx.send(outcome.map(|()| buf[0]));
1159            })
1160        };
1161        let mut lock = ReleaseThenJoin {
1162            held: Some(held),
1163            worker: Some(worker),
1164        };
1165
1166        // PROVE THE READER IS PARKED IN THE LOCK. Not that it started,
1167        // and not that it failed to finish in time: that it is inside
1168        // the acquisition this thread is holding shut.
1169        await_arrival(&dev, "read_at");
1170        assert!(
1171            matches!(done_rx.try_recv(), Err(mpsc::TryRecvError::Empty)),
1172            "a read completed while the exclusive write guard was held. Reads and \
1173             writes are not mutually excluded, so a read overlapping a write_at \
1174             can observe a partially written region"
1175        );
1176
1177        // And it is blocked rather than broken: it completes once the
1178        // writer lets go. Without this half the test would pass against
1179        // a read_at that simply never returned.
1180        lock.release();
1181        let first = done_rx
1182            .recv_timeout(UNBLOCKED_WITHIN)
1183            .expect("the read must proceed once the write guard is released")
1184            .expect("and must succeed");
1185        assert_eq!(first, 0x5A, "the read returned the wrong bytes");
1186    }
1187
1188    /// AND THE EXCLUSION HOLDS THE OTHER WAY ROUND.
1189    ///
1190    /// A write must not start while a read is in progress, or the read
1191    /// it interleaves with is the one that tears. Asserted with a shared
1192    /// guard, which is what a Unix reader holds.
1193    #[test]
1194    fn a_write_cannot_proceed_while_a_read_holds_the_lock() {
1195        let (dev, _c) = rw_image("write_excluded_by_read", 0x11);
1196
1197        // Stands in for a read in progress.
1198        let held = dev.io_lock.read().unwrap();
1199        assert_eq!(waiting_at_the_lock(&dev), 0, "the holder is not a waiter");
1200
1201        let (done_tx, done_rx) = mpsc::channel();
1202        let worker = {
1203            let dev = std::sync::Arc::clone(&dev);
1204            std::thread::spawn(move || {
1205                let _ = done_tx.send(dev.write_at(0, &[0x22u8; 4096]));
1206            })
1207        };
1208        let mut lock = ReleaseThenJoin {
1209            held: Some(held),
1210            worker: Some(worker),
1211        };
1212
1213        await_arrival(&dev, "write_at");
1214        assert!(
1215            matches!(done_rx.try_recv(), Err(mpsc::TryRecvError::Empty)),
1216            "a write completed while a read guard was held; a write_at may not \
1217             run through a read that is already in progress"
1218        );
1219
1220        lock.release();
1221        done_rx
1222            .recv_timeout(UNBLOCKED_WITHIN)
1223            .expect("the write must proceed once the read releases")
1224            .expect("and must succeed");
1225
1226        let mut buf = [0u8; 4];
1227        dev.read_at(0, &mut buf).expect("read back");
1228        assert_eq!(buf, [0x22; 4], "the write did not land");
1229    }
1230
1231    /// AND A FLUSH IS A WRITE FOR THIS PURPOSE.
1232    ///
1233    /// `flush` pushes buffered bytes at the file and calls `sync_data`,
1234    /// so it must not interleave with a read or a write any more than
1235    /// `write_at` may. The exclusive guard was here before this test
1236    /// was, and stating an invariant in a comment is not testing it:
1237    /// with the guard removed the whole suite stayed green.
1238    #[test]
1239    fn a_flush_cannot_proceed_while_a_read_holds_the_lock() {
1240        let (dev, _c) = rw_image("flush_excluded_by_read", 0x33);
1241
1242        // Stands in for a read in progress.
1243        let held = dev.io_lock.read().unwrap();
1244        assert_eq!(waiting_at_the_lock(&dev), 0, "the holder is not a waiter");
1245
1246        let (done_tx, done_rx) = mpsc::channel();
1247        let worker = {
1248            let dev = std::sync::Arc::clone(&dev);
1249            std::thread::spawn(move || {
1250                let _ = done_tx.send(dev.flush());
1251            })
1252        };
1253        let mut lock = ReleaseThenJoin {
1254            held: Some(held),
1255            worker: Some(worker),
1256        };
1257
1258        await_arrival(&dev, "flush");
1259        assert!(
1260            matches!(done_rx.try_recv(), Err(mpsc::TryRecvError::Empty)),
1261            "a flush completed while a read guard was held; flush takes the lock \
1262             exclusively for the same reason write_at does"
1263        );
1264
1265        lock.release();
1266        done_rx
1267            .recv_timeout(UNBLOCKED_WITHIN)
1268            .expect("the flush must proceed once the read releases")
1269            .expect("and must succeed");
1270    }
1271
1272    /// READERS STILL OVERLAP, WHICH IS THE POINT OF THE SHARED GUARD.
1273    ///
1274    /// THE OVER-CORRECTION THIS CATCHES: restoring read/write exclusion
1275    /// with a plain mutex, or by taking the write half of this lock on
1276    /// the read path, would pass both tests above and quietly undo the
1277    /// reader parallelism the positioned-read work existed for. Nothing
1278    /// else in the suite would notice, because every other assertion is
1279    /// about bytes and serialised readers return the right bytes.
1280    ///
1281    /// Unix only: on Windows `seek_read` moves the file pointer, so
1282    /// readers there take the guard exclusively on purpose and this
1283    /// would correctly block.
1284    #[test]
1285    #[cfg(unix)]
1286    fn a_read_does_not_exclude_another_read() {
1287        let (dev, _c) = rw_image("reads_overlap", 0x77);
1288
1289        // Stands in for another reader already inside `read_at`.
1290        let held = dev.io_lock.read().unwrap();
1291
1292        let (done_tx, done_rx) = mpsc::channel();
1293        let worker = {
1294            let dev = std::sync::Arc::clone(&dev);
1295            std::thread::spawn(move || {
1296                let mut buf = [0u8; 4096];
1297                let outcome = dev.read_at(0, &mut buf);
1298                let _ = done_tx.send(outcome.map(|()| buf[0]));
1299            })
1300        };
1301        // Holds the read guard for the whole assertion, so the read
1302        // below completes WHILE another reader holds the lock — which
1303        // is the claim. No arrival wait here and no ready signal: this
1304        // assertion is a positive one, and a worker that has not been
1305        // scheduled makes it slower, never falsely green.
1306        let lock = ReleaseThenJoin {
1307            held: Some(held),
1308            worker: Some(worker),
1309        };
1310
1311        let first = done_rx
1312            .recv_timeout(UNBLOCKED_WITHIN)
1313            .expect(
1314                "a read blocked behind another read. On Unix the guard must be \
1315                 shared -- positioned reads need no cursor, and serialising them \
1316                 undoes the parallelism the read path was rewritten for",
1317            )
1318            .expect("and the read must succeed");
1319        assert_eq!(first, 0x77);
1320        drop(lock);
1321    }
1322}