fs_core/file_device.rs
1//! File-backed `BlockDevice`. Used for disk images, raw `/dev/diskN` reads,
2//! anything that std::fs::File can address.
3
4use crate::block::{BlockDevice, BlockRead};
5use crate::error::{Error, Result};
6use std::fs::{File, OpenOptions};
7use std::io::{Seek, SeekFrom, Write};
8use std::path::Path;
9use std::sync::atomic::{AtomicU64, Ordering};
10use std::sync::RwLock;
11
12/// A file opened as a block device.
13///
14/// # Readers share the lock; writers take it alone
15///
16/// A read was once `seek` then `read` under a plain mutex, which made
17/// the file's cursor shared state: two threads reading different offsets
18/// had to take turns, not because the device could not serve them at
19/// once but because one would have moved the other's cursor.
20///
21/// On Unix the cursor is not involved at all — `pread` takes the offset
22/// as an argument — so readers hold the lock *shared* and genuinely
23/// overlap. On Windows the equivalent (`seek_read`) *does* move the file
24/// pointer, so readers there take it exclusively and only that platform
25/// pays for the cursor.
26///
27/// Writers take it exclusively on both, because `write_at` is `seek`
28/// plus `write_all` and neither another writer's seek nor a reader may
29/// land in the middle of it.
30///
31/// # THE LOCK IS NOT AN OPTIMISATION, IT IS THE READ/WRITE CONTRACT
32///
33/// Reads briefly took no lock at all, which read as a natural
34/// consequence of positioned reads needing no cursor. It was not: it
35/// silently dropped the exclusion between readers and writers that the
36/// single mutex had provided, so a read overlapping a `write_at` could
37/// observe part of it. `write_all` is permitted to become several
38/// `write` calls, and a read is a loop of positioned reads — either
39/// split is a window, and the second one does not need the first.
40///
41/// So a reader holds the lock for the whole of [`FileDevice::read_at`],
42/// not for each positioned read inside it. Per-read guards would leave
43/// exactly the same hole one level down, and a rarer tear is worse than
44/// a common one because nobody can reproduce it.
45///
46/// # A known limit, stated rather than fixed
47///
48/// `std::sync::RwLock` does not promise writer preference on every
49/// platform, and the read path here is the hot one. A device under
50/// sustained parallel reads can therefore make a writer wait longer than
51/// a fair queue would. That is a throughput property, not a correctness
52/// one, and it is left alone rather than solved with a hand-rolled queue
53/// nobody would be able to audit.
54pub struct FileDevice {
55 file: File,
56 /// Shared by readers, exclusive to writers — see the type's own
57 /// note. `()` rather than the file, because it orders access rather
58 /// than owning the handle: positioned reads need no cursor, so
59 /// putting the `File` in here would reintroduce the serialisation
60 /// the shared guard exists to avoid.
61 io_lock: RwLock<()>,
62 /// ATOMIC BECAUSE [`BlockDevice::set_len`] MOVES IT, AND
63 /// `size_bytes` TAKES NO LOCK.
64 ///
65 /// It was a plain `u64`, which was right while nothing could change
66 /// it. `set_len` takes `&self` -- every method on these traits does,
67 /// because the devices are held behind `Arc<dyn _>` -- so the number
68 /// needs interior mutability, and the cheapest correct one is an
69 /// atomic rather than putting it under `io_lock`.
70 ///
71 /// Not under `io_lock` DELIBERATELY: `size_bytes` is called on the
72 /// hot path of every wrapper in this crate ([`crate::CachingDevice`]
73 /// asks it twice per read), and routing it through a lock a writer
74 /// holds exclusively would make an ordinary read contend with an
75 /// ordinary write to learn a number that fits in a register.
76 ///
77 /// `set_len` still takes `io_lock` exclusively -- it has real I/O to
78 /// exclude -- and publishes this afterwards. See its own note for
79 /// which of the two it moves first.
80 size: AtomicU64,
81 writable: bool,
82 /// Whether `set_len` can do anything: a writable handle on a REGULAR
83 /// FILE.
84 ///
85 /// Both halves are needed and neither implies the other. A read-only
86 /// handle obviously cannot truncate. A BLOCK DEVICE NODE is the case
87 /// that is easy to miss: `/dev/sdX` opened read-write is writable,
88 /// `write_at` works on it, and its length is the kernel's rather than
89 /// ours -- `ftruncate` on it is not a resize, and answering `true`
90 /// here would promise an image writer room it can never get.
91 ///
92 /// Decided once, at open, from the same `metadata` call that
93 /// `measure_size` is about to make: asking on every `can_grow` would
94 /// turn a capability question into a syscall, and the file type of an
95 /// already-open descriptor does not change.
96 growable: bool,
97 /// Test-only witness that a thread reached `io_lock` — see
98 /// [`LockArrivals`]. Absent from every non-test build, and the
99 /// `arriving` it feeds compiles to nothing there.
100 #[cfg(test)]
101 arrivals: LockArrivals,
102}
103
104/// How many threads are inside an `io_lock` acquisition and have not
105/// yet been granted the guard.
106///
107/// # WHY A TEST NEEDS THIS AND CANNOT DO WITHOUT IT
108///
109/// The exclusion this type promises can only be asserted negatively —
110/// an operation that must *not* proceed — and "did not finish within
111/// 300ms" is not that assertion. It is satisfied just as well by a
112/// worker the OS has not scheduled, so on a loaded runner a build with
113/// no lock at all passes. That is rust-fs-core#104, and it applied to
114/// all four of the tests that are the only evidence #77's fix works.
115///
116/// Signalling readiness from the top of the worker closure does not
117/// fix it: the gap between the signal and the call under test has no
118/// synchronisation in it, so the signal proves the thread ran once,
119/// not that it reached the lock.
120///
121/// A counter incremented immediately before the acquisition and
122/// decremented immediately after it is granted turns that into a
123/// positive observation the test can wait for. Seeing it non-zero
124/// while the test itself holds the lock means: this thread is in the
125/// acquisition, and it cannot leave until we let go. Remove the lock
126/// from the operation and the counter is never touched, so the wait
127/// times out and names what failed rather than passing.
128///
129/// Measured on `a_read_cannot_proceed_while_a_write_holds_the_lock`
130/// with `read_at`'s guard deleted, one commit, this machine:
131///
132/// | test shape | result | time |
133/// |-------------------------------------------|----------|-------|
134/// | ready signal, 400ms deschedule in the gap | **ok** | 0.41s |
135/// | ready signal, no deschedule | FAILED | 0.33s |
136/// | this counter | FAILED | 10.0s |
137///
138/// The first row is the defect: a build with no lock at all, passing.
139/// The gap has no upper bound on it, so 400ms is an illustration of a
140/// class rather than a threshold.
141///
142/// THE DECREMENT IS WHAT KEEPS THIS HONEST. If it went missing the
143/// counter would stick above zero, every wait would return
144/// immediately, and all four exclusion tests would pass without a
145/// worker ever reaching the lock — the same unwitnessed shape one
146/// level up. See
147/// `an_uncontended_operation_leaves_no_thread_waiting_at_the_lock`.
148#[cfg(test)]
149#[derive(Default)]
150struct LockArrivals {
151 waiting: std::sync::atomic::AtomicUsize,
152}
153
154/// Decrements the moment the guard is granted, whatever happens.
155#[cfg(test)]
156struct Arrival<'a>(&'a LockArrivals);
157
158#[cfg(test)]
159impl Drop for Arrival<'_> {
160 fn drop(&mut self) {
161 self.0
162 .waiting
163 .fetch_sub(1, std::sync::atomic::Ordering::SeqCst);
164 }
165}
166
167impl FileDevice {
168 /// Open read-only.
169 pub fn open<P: AsRef<Path>>(path: P) -> Result<Self> {
170 let file = File::open(path)?;
171 let size = measure_size(&file)?;
172 Ok(Self {
173 file,
174 io_lock: RwLock::new(()),
175 size: AtomicU64::new(size),
176 writable: false,
177 // A read-only handle cannot change any length, whatever it
178 // is open on.
179 growable: false,
180 #[cfg(test)]
181 arrivals: LockArrivals::default(),
182 })
183 }
184
185 /// Open read-write. Errors if the path is not writable.
186 pub fn open_rw<P: AsRef<Path>>(path: P) -> Result<Self> {
187 let file = OpenOptions::new().read(true).write(true).open(path)?;
188 let size = measure_size(&file)?;
189 let growable = is_regular_file(&file);
190 Ok(Self {
191 file,
192 io_lock: RwLock::new(()),
193 size: AtomicU64::new(size),
194 writable: true,
195 growable,
196 #[cfg(test)]
197 arrivals: LockArrivals::default(),
198 })
199 }
200
201 /// Open read-write if possible, fall back to read-only otherwise.
202 pub fn open_best_effort<P: AsRef<Path>>(path: P) -> Result<Self> {
203 let p = path.as_ref();
204 match Self::open_rw(p) {
205 Ok(d) => Ok(d),
206 Err(_) => Self::open(p),
207 }
208 }
209}
210
211/// How many bytes the opened handle addresses.
212///
213/// For a **regular file** this is the metadata length, and 0 is a
214/// legitimate answer: an empty file is empty.
215///
216/// For a **device node** it is not an answer at all. `st_size` is 0 for
217/// every block and character device on the platforms this crate ships
218/// to, so the size has to be asked for directly. Measured against an
219/// 8 MiB backing store:
220///
221/// | source | `metadata().len()` | `lseek(SEEK_END)` | ioctl |
222/// |---------------------------|--------------------|-------------------|-----------|
223/// | macOS `/dev/disk9` (blk) | 0 | 0 | 8388608 |
224/// | macOS `/dev/rdisk9` (chr) | 0 | 0 | 8388608 |
225/// | Linux `/dev/loop0` (blk) | 0 | 8388608 | 8388608 |
226///
227/// **`lseek(SEEK_END)` is not the portable fallback it looks like.** It
228/// answers 0 on macOS for both node types, so a device opened there
229/// would still report itself empty. The ioctl is the only mechanism that
230/// answered on both platforms, and it happens to avoid the objection to
231/// the seek as well: it does not move the file cursor. That matters
232/// here, because `read_once` uses `pread` specifically so reads need no
233/// lock -- see this type's own note.
234///
235/// When the size cannot be measured this returns an error rather than 0,
236/// because **a device reporting 0 is not inert, it is invisible.**
237/// `read_at` keeps serving real bytes, while
238/// [`BlockReadStreamer::read`] returns `Ok(0)` on its first call,
239/// [`CachingDevice`] treats every read as past-the-end and caches
240/// nothing, and every slice cut from it inherits a parent claiming to be
241/// empty. Each of those four failures is silent, which is the one
242/// outcome worth refusing outright.
243///
244/// [`BlockReadStreamer::read`]: crate::BlockReadStreamer
245/// [`CachingDevice`]: crate::CachingDevice
246#[cfg(unix)]
247fn measure_size(file: &File) -> Result<u64> {
248 use std::os::unix::fs::FileTypeExt;
249 let meta = file.metadata()?;
250 let ft = meta.file_type();
251 if !ft.is_block_device() && !ft.is_char_device() {
252 return Ok(meta.len());
253 }
254 device_size_bytes(file)
255}
256
257/// Windows keeps the metadata length, because no equivalent measurement
258/// has been made there. `\\.\PhysicalDriveN` is therefore still subject
259/// to the defect this function exists to fix; saying so is better than
260/// shipping an untested `DeviceIoControl` and implying otherwise.
261#[cfg(not(unix))]
262fn measure_size(file: &File) -> Result<u64> {
263 Ok(file.metadata()?.len())
264}
265
266// The crate has no dependencies and this is not worth acquiring one for:
267// `ioctl` is in libc, which is already linked into every std target.
268#[cfg(any(target_os = "macos", target_os = "ios", target_os = "linux"))]
269unsafe extern "C" {
270 fn ioctl(fd: std::os::raw::c_int, request: std::os::raw::c_ulong, ...) -> std::os::raw::c_int;
271}
272
273/// macOS: `<sys/disk.h>` gives the block size and the block count
274/// separately, and neither alone is the answer.
275#[cfg(any(target_os = "macos", target_os = "ios"))]
276fn device_size_bytes(file: &File) -> Result<u64> {
277 use std::io;
278 use std::os::fd::AsRawFd;
279 // _IOR('d', 24, u32) and _IOR('d', 25, u64).
280 const DKIOCGETBLOCKSIZE: std::os::raw::c_ulong = 0x4004_6418;
281 const DKIOCGETBLOCKCOUNT: std::os::raw::c_ulong = 0x4008_6419;
282
283 let fd = file.as_raw_fd();
284 let mut block_size: u32 = 0;
285 let mut block_count: u64 = 0;
286 // SAFETY: `fd` is open for as long as `file` is borrowed, and each
287 // request writes exactly the width its own encoding names into a
288 // local of precisely that type.
289 unsafe {
290 if ioctl(fd, DKIOCGETBLOCKSIZE, &raw mut block_size) < 0 {
291 return Err(io::Error::last_os_error().into());
292 }
293 if ioctl(fd, DKIOCGETBLOCKCOUNT, &raw mut block_count) < 0 {
294 return Err(io::Error::last_os_error().into());
295 }
296 }
297 block_count
298 .checked_mul(u64::from(block_size))
299 .ok_or_else(|| {
300 Error::Io(io::Error::other(format!(
301 "device reports {block_count} blocks of {block_size} bytes, \
302 whose product does not fit in u64"
303 )))
304 })
305}
306
307/// `BLKGETSIZE64` as `_IOR(0x12, 114, size_t)`, for a given pointer
308/// width.
309///
310/// THE SIZE FIELD OF AN IOCTL REQUEST IS `sizeof(size_t)` -- the
311/// USERSPACE POINTER WIDTH, not the width of the value the kernel
312/// writes back. The payload is a `u64` on both, but the request NUMBER
313/// differs: `0x8008_1272` where a pointer is 8 bytes, `0x8004_1272`
314/// where it is 4. A literal for one width is rejected by the kernel on
315/// the other.
316///
317/// That is a regression this change would have INTRODUCED rather than
318/// inherited. Before it, an unmeasurable device node gave `Ok` with a
319/// size of 0; now it is an error, so a 64-bit-only constant would make
320/// [`FileDevice::open`] fail for EVERY block device on a 32-bit target.
321///
322/// Taking the width as a parameter is what makes it testable: nothing
323/// available here runs 32-bit, and `cargo check --target` compiles a
324/// wrong literal perfectly happily, so the only witness possible is to
325/// compute both encodings and compare them with the two numbers the
326/// kernel headers actually define.
327///
328/// Available in every TEST build rather than on Linux alone, because a
329/// test that cannot run is not a witness: gated to Linux it would be
330/// compiled and never executed on the machine the work is done on, and
331/// the arithmetic is the same arithmetic everywhere.
332#[cfg(any(target_os = "linux", test))]
333const fn blkgetsize64_for(pointer_width: usize) -> std::os::raw::c_ulong {
334 /// `_IOC_READ << _IOC_DIRSHIFT`.
335 const READ: std::os::raw::c_ulong = 0x8000_0000;
336 /// `_IOC_TYPESHIFT` is 8, `_IOC_SIZESHIFT` is 16.
337 const TYPE: std::os::raw::c_ulong = 0x12;
338 const NR: std::os::raw::c_ulong = 114;
339 READ | ((pointer_width as std::os::raw::c_ulong) << 16) | (TYPE << 8) | NR
340}
341
342/// Linux: `<linux/fs.h>` answers in bytes in one call.
343#[cfg(target_os = "linux")]
344fn device_size_bytes(file: &File) -> Result<u64> {
345 use std::io;
346 use std::os::fd::AsRawFd;
347 // _IOR(0x12, 114, size_t), encoded for THIS target -- see
348 // `blkgetsize64_for`. Identical to the familiar 0x8008_1272 on a
349 // 64-bit target and correct on a 32-bit one.
350 const BLKGETSIZE64: std::os::raw::c_ulong = blkgetsize64_for(std::mem::size_of::<usize>());
351
352 let mut size: u64 = 0;
353 // SAFETY: as above -- one call, writing one u64 into a u64.
354 let rc = unsafe { ioctl(file.as_raw_fd(), BLKGETSIZE64, &raw mut size) };
355 if rc < 0 {
356 return Err(io::Error::last_os_error().into());
357 }
358 Ok(size)
359}
360
361/// Every other Unix: refuse rather than guess. `lseek` is measured wrong
362/// on one of the two platforms tested, so extending it here on the
363/// strength of that would be picking the silent failure.
364#[cfg(all(
365 unix,
366 not(any(target_os = "macos", target_os = "ios", target_os = "linux"))
367))]
368fn device_size_bytes(_file: &File) -> Result<u64> {
369 use std::io;
370 Err(Error::Io(io::Error::other(
371 "no measured way to read a device node's size on this platform; \
372 open the backing image file rather than the device node",
373 )))
374}
375
376impl FileDevice {
377 /// Marks this thread as having reached `io_lock` and not yet been
378 /// granted it. The returned value decrements on drop — which, at
379 /// the two call sites below, is after the acquisition returns.
380 ///
381 /// Compiles to nothing outside a test build: the field it counts
382 /// does not exist there. See [`LockArrivals`] for why a test cannot
383 /// establish the same thing from outside.
384 #[cfg(test)]
385 fn arriving(&self) -> Arrival<'_> {
386 self.arrivals
387 .waiting
388 .fetch_add(1, std::sync::atomic::Ordering::SeqCst);
389 Arrival(&self.arrivals)
390 }
391
392 /// THE ONLY PLACES `io_lock` IS ACQUIRED — two on Unix, one on
393 /// Windows.
394 ///
395 /// Not a wrapper for its own sake: the arrival counter has to sit
396 /// immediately before the acquisition to mean anything, and one
397 /// pair of methods is what stops a third call site being added
398 /// without it. `_arrival` outlives the tail expression and is
399 /// dropped once the guard has been granted — which is what makes
400 /// a non-zero count mean "waiting" rather than "has waited".
401 ///
402 /// The `cfg` is on the statement rather than on two bodies of
403 /// `arriving`, so a non-test build contains the acquisition and
404 /// nothing else.
405 ///
406 /// # `shared_guard` IS UNIX-ONLY, AND THAT IS THE DESIGN RATHER
407 /// THAN TIDINESS
408 ///
409 /// Nothing on Windows takes `io_lock` shared. `read_guard` there is
410 /// `exclusive_guard`, deliberately, because `seek_read` moves the
411 /// file pointer and a reader must exclude other readers as well as
412 /// writers — see `read_guard`. So on Windows this method is not
413 /// merely unused, it MUST NOT BE CALLED: a future caller reaching
414 /// for the cheaper guard would reintroduce the cursor race that the
415 /// exclusive read guard exists to prevent.
416 ///
417 /// `cargo clippy --target x86_64-pc-windows-msvc --all-targets
418 /// -- -D warnings` reported it as `method shared_guard is never
419 /// used`, and `#[allow(dead_code)]` would have been the wrong
420 /// answer: it silences the compiler on a platform where the right
421 /// statement is that the method does not exist. Compiling it out
422 /// makes a call site that should not exist fail to build.
423 #[cfg(unix)]
424 fn shared_guard(&self) -> std::sync::RwLockReadGuard<'_, ()> {
425 #[cfg(test)]
426 let _arrival = self.arriving();
427 self.io_lock.read().unwrap()
428 }
429
430 fn exclusive_guard(&self) -> std::sync::RwLockWriteGuard<'_, ()> {
431 #[cfg(test)]
432 let _arrival = self.arriving();
433 self.io_lock.write().unwrap()
434 }
435
436 /// The guard a read holds for the whole of `read_at`.
437 ///
438 /// Unix takes it SHARED: `pread` carries its own offset, so readers
439 /// do not disturb each other and only need to be kept apart from
440 /// writers.
441 ///
442 /// Windows takes it EXCLUSIVE, because `seek_read` moves the file
443 /// pointer — there, one reader really can spoil another's offset, so
444 /// readers must exclude readers as well as writers. Same lock, same
445 /// call site, and only that platform pays for the cursor.
446 #[cfg(unix)]
447 fn read_guard(&self) -> std::sync::RwLockReadGuard<'_, ()> {
448 self.shared_guard()
449 }
450
451 #[cfg(windows)]
452 fn read_guard(&self) -> std::sync::RwLockWriteGuard<'_, ()> {
453 self.exclusive_guard()
454 }
455
456 /// One positioned read, returning what it got.
457 ///
458 /// TAKES NO LOCK ON EITHER PLATFORM. The caller holds `read_guard`
459 /// for the whole read; acquiring anything here would be a second,
460 /// non-reentrant acquisition of the same lock — on Windows, where
461 /// that guard is exclusive, an immediate self-deadlock.
462 #[cfg(unix)]
463 fn read_once(&self, offset: u64, buf: &mut [u8]) -> Result<usize> {
464 use std::os::unix::fs::FileExt;
465 Ok(self.file.read_at(buf, offset)?)
466 }
467
468 /// Windows: `seek_read` DOES move the file pointer. The exclusion
469 /// that needs is held by the caller's guard, not taken here — see
470 /// `read_guard`.
471 #[cfg(windows)]
472 fn read_once(&self, offset: u64, buf: &mut [u8]) -> Result<usize> {
473 use std::os::windows::fs::FileExt;
474 Ok(self.file.seek_read(buf, offset)?)
475 }
476}
477
478impl BlockRead for FileDevice {
479 fn read_at(&self, offset: u64, buf: &mut [u8]) -> Result<()> {
480 // HELD ACROSS THE WHOLE LOOP, NOT AROUND EACH POSITIONED READ.
481 //
482 // The loop below can issue several reads for one call, and a
483 // guard taken inside it would let a `write_at` land between two
484 // of them: the caller would get some bytes from before the write
485 // and some from after, which is the tear this exists to prevent
486 // moved one level down and made rarer. Rarer is worse — nobody
487 // can reproduce it.
488 //
489 // NO TEST PINS THIS PLACEMENT, and the reason is worth writing
490 // down because the obvious one is wrong. A regular file does
491 // return short reads — at EOF — and this loop retries rather
492 // than failing on one, so a read spanning EOF really does run
493 // twice and the guard really would be released in between. That
494 // discriminator was built: a 4096-byte file, a reader asking
495 // 4090..4102, a writer parked on the lock growing the file at
496 // 4090, where interleaved old-and-new bytes are reachable no
497 // other way. 200 trials with the guard moved inside the loop
498 // produced 0 tears. The window between releasing at the end of
499 // one iteration and retaking at the start of the next is a few
500 // instructions, and a writer already waiting never won it.
501 //
502 // So this is unwitnessed, not inert: the placement is correct
503 // and the race it prevents is simply too narrow to enter on
504 // demand. Measured, not argued.
505 let _guard = self.read_guard();
506
507 // A SHORT READ IS AN ERROR NAMING WHAT WAS ASKED FOR AND WHAT
508 // ARRIVED, not a smaller answer: a caller that asked for a block
509 // and got half of one cannot tell the difference from bytes.
510 let mut total = 0usize;
511 while total < buf.len() {
512 let n = self.read_once(offset + total as u64, &mut buf[total..])?;
513 if n == 0 {
514 return Err(Error::ShortRead {
515 offset,
516 want: buf.len(),
517 got: total,
518 });
519 }
520 total += n;
521 }
522 Ok(())
523 }
524
525 fn size_bytes(&self) -> u64 {
526 // `Acquire` pairs with the `Release` store in `set_len`: a
527 // thread that observes a grown size must also observe the
528 // `ftruncate` that produced it.
529 self.size.load(Ordering::Acquire)
530 }
531}
532
533impl BlockDevice for FileDevice {
534 /// A write past the end is refused, not an extension.
535 ///
536 /// `write_all` at a seeked offset EXTENDS a file, and this method
537 /// had no bound of its own, so a write straddling the end grew the
538 /// backing store while `size_bytes` went on reporting the length
539 /// taken at construction -- measured on a 4096-byte file:
540 /// `write_at(4094, 8 bytes)` returned `Ok`, the file became 4102
541 /// bytes, `size_bytes()` stayed 4096, and `read_at(4096, 6)` then
542 /// handed those bytes back. The two halves of one device disagreed
543 /// about where it ended, and a caller bounding its reads by
544 /// `size_bytes` -- which is what [`crate::CachingDevice`] does,
545 /// clamping every block it fetches -- could never reach them.
546 ///
547 /// `RwBytes` in this crate's own test devices already refuses the
548 /// same operation, commenting "a device is not a `Vec`", and the
549 /// slice adapters in [`crate::slice`] clamp their window
550 /// specifically because this method did not:
551 /// `slice_rw_length_is_clamped_and_a_write_past_it_does_not_grow_the_image`
552 /// names the file's length on disk as its oracle. This is the same
553 /// rule one layer down, where it was missing.
554 ///
555 /// The alternative -- letting the size move and reopening -- is
556 /// what [`BlockRead::size_bytes`]'s contract forbids. See
557 /// rust-fs-core#70.
558 fn write_at(&self, offset: u64, buf: &[u8]) -> Result<()> {
559 if !self.writable {
560 return Err(Error::ReadOnly);
561 }
562 // READ ONCE, COMPARED AND REPORTED FROM THE SAME VALUE. `size`
563 // is atomic now that `set_len` moves it, and loading it twice
564 // could bound the write against one length and name another in
565 // the error -- an `OutOfBounds` whose `size` field does not
566 // explain its own refusal.
567 let size = self.size_bytes();
568 // `checked_add` because a caller-supplied offset near `u64::MAX`
569 // would otherwise wrap and land back inside the device.
570 let end = offset.checked_add(buf.len() as u64);
571 if end.is_none_or(|end| end > size) {
572 return Err(Error::OutOfBounds {
573 offset,
574 len: buf.len() as u64,
575 size,
576 });
577 }
578 // EXCLUSIVE: excludes other writers' seeks and every reader.
579 let _guard = self.exclusive_guard();
580 let mut f = &self.file;
581 f.seek(SeekFrom::Start(offset))?;
582 f.write_all(buf)?;
583 Ok(())
584 }
585
586 fn flush(&self) -> Result<()> {
587 if !self.writable {
588 return Ok(());
589 }
590 // EXCLUSIVE for the same reason as `write_at`: this pushes
591 // buffered bytes at the file and must not interleave with a
592 // write or a read.
593 let _guard = self.exclusive_guard();
594 let mut f = &self.file;
595 f.flush()?;
596 self.file.sync_data()?;
597 Ok(())
598 }
599
600 fn is_writable(&self) -> bool {
601 self.writable
602 }
603
604 /// Set the file's length, and the length this device reports, as one
605 /// operation.
606 ///
607 /// # THE POINT IS THAT THE TWO MOVE TOGETHER
608 ///
609 /// #75 refused a write past the end because `write_all` at a seeked
610 /// offset grew the FILE while `size_bytes` went on reporting the
611 /// length taken at construction, so the two halves of one device
612 /// disagreed about where it ended and a caller bounding its reads by
613 /// `size_bytes` could never reach what it had written
614 /// (rust-fs-core#70). That bound stays. This method is the other
615 /// half: growth that says so, and moves the number with it.
616 ///
617 /// An implementation that called `File::set_len` and left `self.size`
618 /// alone would be #70 with a nicer name on it.
619 ///
620 /// # THE NARROWER LENGTH IS PUBLISHED FIRST, IN BOTH DIRECTIONS
621 ///
622 /// `ftruncate` and the store are two steps, and one of the two
623 /// orderings has a window in it. Take a shrink done store-then-
624 /// publish: between the truncate and the store, this device declares
625 /// 8192 bytes over a 4096-byte file, so a concurrent read inside the
626 /// declared device falls off the end of the real one and comes back
627 /// `ShortRead`. The other direction is harmless — a device that
628 /// briefly declares 4096 bytes over an 8192-byte file is only
629 /// under-reporting, which is the state every `FileDevice` is in
630 /// whenever something else appends to its file.
631 ///
632 /// So: a GROW truncates and then stores, and a SHRINK stores and then
633 /// truncates. The invariant is one sentence — THE DECLARED SIZE NEVER
634 /// EXCEEDS THE FILE'S REAL LENGTH — and it holds at every instant
635 /// rather than only at the ends.
636 ///
637 /// # `io_lock` EXCLUSIVELY, LIKE A WRITE
638 ///
639 /// For the same reason `write_at` and `flush` take it: this changes
640 /// the file underneath every reader, and a read must not observe
641 /// half of it. `size_bytes` deliberately does NOT take the lock —
642 /// see the field — so the ordering above is what keeps a reader that
643 /// asked the size mid-call from being misled, not the lock.
644 ///
645 /// # WHAT IT REFUSES
646 ///
647 /// [`Error::ReadOnly`] on a handle opened with [`FileDevice::open`],
648 /// before touching the file. [`Error::Custom`] on a handle that is
649 /// writable but not a regular file — a block device node, whose
650 /// length belongs to the kernel — naming that reason rather than
651 /// letting `ftruncate`'s `EINVAL` stand in for it. See
652 /// [`FileDevice::can_grow`], which is the question to ask instead of
653 /// discovering either of these.
654 fn set_len(&self, new_len: u64) -> Result<()> {
655 if !self.writable {
656 return Err(Error::ReadOnly);
657 }
658 if !self.growable {
659 return Err(Error::Custom(
660 "this FileDevice is open on something that is not a regular file \
661 -- a device node's length is the kernel's, not ours, and cannot \
662 be set through this handle"
663 .to_string(),
664 ));
665 }
666
667 // EXCLUSIVE: excludes every reader and every other writer, the
668 // same as `write_at`.
669 let _guard = self.exclusive_guard();
670
671 let old = self.size.load(Ordering::Acquire);
672 if new_len < old {
673 // Narrow the declared device BEFORE the bytes go. See above.
674 self.size.store(new_len, Ordering::Release);
675 }
676 if let Err(e) = self.file.set_len(new_len) {
677 // A FAILED SHRINK HAS ALREADY NARROWED THE DECLARATION, and
678 // leaving it there would make the device deny bytes it still
679 // holds -- #70's defect, reached by the error path. Re-measure
680 // rather than restoring `old`: whether `ftruncate` did nothing
681 // or something is not knowable from its error, and the file
682 // itself is the only honest answer.
683 if let Ok(actual) = measure_size(&self.file) {
684 self.size.store(actual, Ordering::Release);
685 }
686 return Err(e.into());
687 }
688 self.size.store(new_len, Ordering::Release);
689 Ok(())
690 }
691
692 fn can_grow(&self) -> bool {
693 self.growable
694 }
695}
696
697/// Is this handle open on a regular file?
698///
699/// The other half of `can_grow`, and the half a reader is most likely to
700/// assume. A `FileDevice` is opened just as often on `/dev/sdX` as on an
701/// image: `measure_size` has a whole `ioctl` path for exactly that case.
702/// Such a handle is writable, `write_at` works on it, and its length is
703/// fixed by the kernel -- `ftruncate` on a block device is not a resize.
704///
705/// A failed `metadata` call answers `false`. The question is "may this
706/// device promise it can grow", and a promise nobody could verify is not
707/// one to make.
708fn is_regular_file(file: &File) -> bool {
709 file.metadata().is_ok_and(|m| m.file_type().is_file())
710}
711
712#[cfg(test)]
713mod tests {
714 /// THE TWO NUMBERS THE KERNEL HEADERS DEFINE, and the pointer
715 /// widths they belong to. External knowledge, not a restatement of
716 /// the formula -- a test that recomputed the encoding on both sides
717 /// would agree with itself whatever the encoding was.
718 ///
719 /// This cannot prove the ioctl works on 32-bit; nothing available
720 /// here runs 32-bit, and `cargo check --target` compiles a wrong
721 /// literal happily. What it does is stop the encoding quietly
722 /// reverting to a single hardcoded number, which is the way this
723 /// defect arrived.
724 #[test]
725 fn blkgetsize64_is_encoded_for_the_pointer_width() {
726 const KNOWN: &[(usize, u64)] = &[(4, 0x8004_1272), (8, 0x8008_1272)];
727 for (width, want) in KNOWN {
728 assert_eq!(
729 super::blkgetsize64_for(*width) as u64,
730 *want,
731 "_IOR(0x12, 114, size_t) with a {width}-byte size_t is {want:#010x}"
732 );
733 }
734 }
735
736 /// And the one this build will actually issue is the one for THIS
737 /// target, rather than whichever happened to be written down.
738 #[test]
739 fn this_target_issues_its_own_encoding() {
740 let width = std::mem::size_of::<usize>();
741 let expected = if width == 8 {
742 0x8008_1272u64
743 } else {
744 0x8004_1272u64
745 };
746 assert_eq!(super::blkgetsize64_for(width) as u64, expected);
747 }
748
749 use super::*;
750
751 /// Concurrent readers do not serialise, and none of them sees
752 /// another's offset.
753 ///
754 /// THE BUG THIS REPLACES: reads were `seek` then `read` under one
755 /// mutex, so the file cursor was shared state. Two threads reading
756 /// different parts of the same image took turns for no reason the
757 /// device imposed. Worse, the shape was one edit away from being
758 /// wrong rather than merely slow -- drop the lock without moving to
759 /// positioned reads and every reader corrupts every other reader's
760 /// offset.
761 ///
762 /// The assertion is on the BYTES rather than on timing: a test that
763 /// measured overlap would be a flake on a loaded machine, while a
764 /// reader that got another's offset returns the wrong bytes every
765 /// time.
766 #[test]
767 fn many_threads_reading_different_offsets_each_get_their_own_bytes() {
768 let path = temp_path("parallel_reads");
769 let _c = Cleanup(path.clone());
770 // Each 256-byte page filled with its own page number, so a read
771 // that landed at the wrong offset is obvious from one byte.
772 let mut bytes = Vec::with_capacity(64 * 256);
773 for page in 0..64u8 {
774 bytes.extend(std::iter::repeat_n(page, 256));
775 }
776 std::fs::write(&path, &bytes).expect("write the image");
777
778 let dev = std::sync::Arc::new(FileDevice::open(&path).expect("open"));
779 let mut handles = Vec::new();
780 for page in 0..64u8 {
781 let dev = dev.clone();
782 handles.push(std::thread::spawn(move || {
783 // Several times each, so a thread that raced would have
784 // many chances to read somebody else's page.
785 for _ in 0..50 {
786 let mut buf = [0u8; 256];
787 dev.read_at(u64::from(page) * 256, &mut buf).expect("read");
788 assert!(
789 buf.iter().all(|b| *b == page),
790 "page {page} came back holding another page's bytes"
791 );
792 }
793 }));
794 }
795 for h in handles {
796 h.join().expect("a reader panicked");
797 }
798 }
799
800 use std::sync::atomic::{AtomicU64, Ordering};
801
802 /// Unique temp path under the system temp dir (no extra dev-deps).
803 fn temp_path(tag: &str) -> std::path::PathBuf {
804 static N: AtomicU64 = AtomicU64::new(0);
805 let n = N.fetch_add(1, Ordering::Relaxed);
806 let pid = std::process::id();
807 std::env::temp_dir().join(format!("fs_core_{tag}_{pid}_{n}.bin"))
808 }
809
810 struct Cleanup(std::path::PathBuf);
811 impl Drop for Cleanup {
812 fn drop(&mut self) {
813 let _ = std::fs::remove_file(&self.0);
814 }
815 }
816
817 #[test]
818 fn open_rw_round_trips_write_then_read() {
819 let path = temp_path("rw");
820 let _g = Cleanup(path.clone());
821 std::fs::write(&path, vec![0u8; 32]).unwrap();
822
823 let dev = FileDevice::open_rw(&path).unwrap();
824 assert!(dev.is_writable());
825 assert_eq!(dev.size_bytes(), 32);
826
827 dev.write_at(8, &[0xAA, 0xBB, 0xCC, 0xDD]).unwrap();
828 dev.flush().unwrap();
829
830 let mut buf = [0u8; 4];
831 dev.read_at(8, &mut buf).unwrap();
832 assert_eq!(buf, [0xAA, 0xBB, 0xCC, 0xDD]);
833 }
834
835 #[test]
836 fn open_rw_errors_on_missing_path() {
837 let path = temp_path("missing");
838 assert!(FileDevice::open_rw(&path).is_err());
839 }
840
841 #[test]
842 fn open_best_effort_uses_rw_when_writable() {
843 let path = temp_path("best_rw");
844 let _g = Cleanup(path.clone());
845 std::fs::write(&path, vec![0u8; 16]).unwrap();
846
847 let dev = FileDevice::open_best_effort(&path).unwrap();
848 assert!(dev.is_writable());
849 dev.write_at(0, &[0x11; 4]).unwrap();
850 }
851
852 #[test]
853 #[cfg(unix)]
854 fn open_best_effort_falls_back_to_read_only() {
855 use std::os::unix::fs::PermissionsExt;
856
857 let path = temp_path("best_ro");
858 let _g = Cleanup(path.clone());
859 std::fs::write(&path, vec![0xEFu8; 16]).unwrap();
860 // Read-only permissions force `open_rw` to fail; fall back to `open`.
861 std::fs::set_permissions(&path, std::fs::Permissions::from_mode(0o444)).unwrap();
862
863 let dev = FileDevice::open_best_effort(&path).unwrap();
864 assert!(!dev.is_writable());
865 // Writes are rejected at the read-only layer.
866 assert!(matches!(dev.write_at(0, &[0u8; 4]), Err(Error::ReadOnly)));
867 // Read still works.
868 let mut buf = [0u8; 4];
869 dev.read_at(0, &mut buf).unwrap();
870 assert_eq!(buf, [0xEF; 4]);
871 // Flush on a read-only device is a no-op success.
872 dev.flush().unwrap();
873 }
874
875 use std::sync::mpsc;
876 use std::time::Duration;
877
878 /// How long an operation that must finish is given. Generous on
879 /// purpose: a loaded machine makes it slower, not flakier.
880 ///
881 /// EVERY DEADLINE IN THE EXCLUSION TESTS IS THIS ONE. There used to
882 /// be a second, short one — 300ms, after which a worker that had
883 /// not finished was taken to have been blocked. That is the defect
884 /// rust-fs-core#104 describes: an unscheduled worker is
885 /// indistinguishable from a blocked one, so the assertion passed
886 /// for a reason unrelated to the lock and would have kept passing
887 /// with the lock removed. Blocking is now established by
888 /// [`await_arrival`] instead, which fails when nothing arrives
889 /// rather than passing when nothing happens.
890 const UNBLOCKED_WITHIN: Duration = Duration::from_secs(10);
891
892 /// A 4 KiB image of one repeated byte, opened read-write.
893 fn rw_image(tag: &str, fill: u8) -> (std::sync::Arc<FileDevice>, Cleanup) {
894 let path = temp_path(tag);
895 let cleanup = Cleanup(path.clone());
896 std::fs::write(&path, vec![fill; 4096]).expect("write the image");
897 let dev = std::sync::Arc::new(FileDevice::open_rw(&path).expect("open rw"));
898 (dev, cleanup)
899 }
900
901 fn waiting_at_the_lock(dev: &FileDevice) -> usize {
902 dev.arrivals.waiting.load(Ordering::SeqCst)
903 }
904
905 /// Blocks until a thread is parked inside an `io_lock` acquisition,
906 /// and PANICS IF NONE EVER IS.
907 ///
908 /// This is the half that carries the evidence. Once it returns, the
909 /// worker is between the counter's increment and the guard being
910 /// granted — and since the caller holds that guard and has not let
911 /// go, the worker cannot leave. "The operation is blocked on the
912 /// lock" is then a fact about the program's state rather than an
913 /// inference from a stopwatch.
914 ///
915 /// An operation that stopped taking the lock never increments, so
916 /// this times out and says which operation never arrived. The
917 /// deadline can only produce a false FAILURE, which is the safe
918 /// direction and the opposite of what it replaced.
919 fn await_arrival(dev: &FileDevice, operation: &str) {
920 let deadline = std::time::Instant::now() + UNBLOCKED_WITHIN;
921 while waiting_at_the_lock(dev) == 0 {
922 assert!(
923 std::time::Instant::now() < deadline,
924 "{operation} never reached io_lock. It either never ran, or it \
925 does not take the lock at all — which is the exclusion this \
926 test exists to assert"
927 );
928 std::thread::sleep(Duration::from_micros(200));
929 }
930 }
931
932 /// Releases the holder's guard and THEN joins the worker, on every
933 /// path out of a test including a panicking assertion.
934 ///
935 /// Both halves are rust-fs-core#105. Discarding the `JoinHandle`
936 /// let a failing assertion unwind the test thread while the worker
937 /// was still inside `read_at`/`write_at`/`flush` with the file
938 /// open, so `Cleanup` removed the temp file underneath it — a race
939 /// on exactly the run where a real regression is being diagnosed,
940 /// and a leaked file per failure on a platform that will not unlink
941 /// an open file.
942 ///
943 /// THE ORDER IS NOT INCIDENTAL. Joining first would wait for a
944 /// worker this very thread is blocking, and the test would hang
945 /// instead of failing. Dropping the guard is what lets the worker
946 /// finish so the join can return.
947 ///
948 /// Declared after the `Cleanup` it protects, so it drops first and
949 /// the file still exists when the worker touches it.
950 struct ReleaseThenJoin<G> {
951 held: Option<G>,
952 worker: Option<std::thread::JoinHandle<()>>,
953 }
954
955 impl<G> ReleaseThenJoin<G> {
956 /// Let the blocked operation through, keeping the join.
957 fn release(&mut self) {
958 self.held = None;
959 }
960 }
961
962 impl<G> Drop for ReleaseThenJoin<G> {
963 fn drop(&mut self) {
964 self.held = None;
965 if let Some(worker) = self.worker.take() {
966 // Ignored on purpose: this runs while unwinding a
967 // failed assertion, and a panic in a drop during
968 // unwinding aborts the process, which would replace the
969 // test's own message with nothing.
970 let _ = worker.join();
971 }
972 }
973 }
974
975 /// THE WORKER IS JOINED BEFORE THE TEMP FILE GOES, ON THE PANIC
976 /// PATH SPECIFICALLY.
977 ///
978 /// rust-fs-core#105 is a failure-path defect, so the only way to
979 /// witness it is to fail on purpose. The four exclusion tests
980 /// discarded their `JoinHandle`, so a failing assertion unwound the
981 /// test thread while the worker was still inside the operation with
982 /// the file open, and `Cleanup` removed the file underneath it —
983 /// worst on the one run that matters, the run where a real
984 /// regression tripped one of them.
985 ///
986 /// This reproduces that shape with the file replaced by a recorder,
987 /// because the ordering is what is being asserted and a removed
988 /// file cannot say when it went. With the join dropped the recorder
989 /// sees `cleanup` first; with the release and the join in the order
990 /// [`ReleaseThenJoin`] fixes, it sees `worker` first.
991 ///
992 /// **Two panics are printed while this test runs, and both are the
993 /// test working.** The first is the deliberate one. The second is
994 /// the worker's: unwinding past a held `RwLock` guard poisons it,
995 /// and every acquisition in this module `unwrap`s, so the read the
996 /// worker was blocked on comes back `PoisonError` instead of
997 /// bytes. That is why the worker's arrival is recorded from a
998 /// `Drop` rather than after the read — the ordering claim is about
999 /// when the thread *ends*, and it must hold whether the read
1000 /// returns or unwinds.
1001 #[test]
1002 fn a_panicking_assertion_joins_the_worker_before_the_cleanup_runs() {
1003 use std::panic::AssertUnwindSafe;
1004 use std::sync::{Arc, Mutex};
1005
1006 /// Stands in for `Cleanup`: same position, same drop timing,
1007 /// but it says when it ran.
1008 struct Recorder(Arc<Mutex<Vec<&'static str>>>);
1009 impl Drop for Recorder {
1010 fn drop(&mut self) {
1011 self.0.lock().unwrap().push("cleanup");
1012 }
1013 }
1014
1015 let order: Arc<Mutex<Vec<&'static str>>> = Arc::new(Mutex::new(Vec::new()));
1016 let (dev, _c) = rw_image("panic_joins", 0x99);
1017
1018 let outcome = std::panic::catch_unwind(AssertUnwindSafe(|| {
1019 // Declared first, so it drops last — exactly where
1020 // `Cleanup` sits in the four exclusion tests.
1021 let _recorder = Recorder(Arc::clone(&order));
1022
1023 let held = dev.io_lock.write().unwrap();
1024 let worker = {
1025 let dev = Arc::clone(&dev);
1026 let order = Arc::clone(&order);
1027 std::thread::spawn(move || {
1028 /// Records the worker ENDING, return or unwind.
1029 struct Ended(Arc<Mutex<Vec<&'static str>>>);
1030 impl Drop for Ended {
1031 fn drop(&mut self) {
1032 self.0.lock().unwrap().push("worker");
1033 }
1034 }
1035 let _ended = Ended(order);
1036 let mut buf = [0u8; 4096];
1037 let _ = dev.read_at(0, &mut buf);
1038 })
1039 };
1040 let _lock = ReleaseThenJoin {
1041 held: Some(held),
1042 worker: Some(worker),
1043 };
1044
1045 await_arrival(&dev, "read_at");
1046 panic!("the assertion an exclusion test exists to make, failing");
1047 }));
1048
1049 // NAMED, NOT MERELY PRESENT. `await_arrival` panics too, and
1050 // an `is_err()` satisfied by that one would report a join this
1051 // test never exercised.
1052 let payload = outcome.expect_err("the deliberate panic must have unwound");
1053 let message = payload
1054 .downcast_ref::<&str>()
1055 .copied()
1056 .or_else(|| payload.downcast_ref::<String>().map(String::as_str))
1057 .unwrap_or("<panic payload was not a string>");
1058 assert!(
1059 message.contains("an exclusion test exists to make"),
1060 "the test unwound for the wrong reason, so the ordering below is \
1061 about some other failure: {message}"
1062 );
1063
1064 assert_eq!(
1065 *order.lock().unwrap(),
1066 vec!["worker", "cleanup"],
1067 "the worker was still inside read_at when cleanup ran: an unwinding \
1068 test must release the guard, join the worker, and only then let the \
1069 temp file be removed"
1070 );
1071 }
1072
1073 /// THE COUNTER COMES BACK DOWN.
1074 ///
1075 /// [`await_arrival`] is only evidence while a non-zero `waiting`
1076 /// means a thread is parked *now*. A missing decrement would leave
1077 /// it stuck above zero, every wait would return immediately, and
1078 /// all four exclusion tests would pass without a worker ever
1079 /// reaching the lock — the same unwitnessed shape they were fixed
1080 /// for, one level up. So the uncontended path is asserted too:
1081 /// every acquisition site, no contention, nothing left behind.
1082 #[test]
1083 fn an_uncontended_operation_leaves_no_thread_waiting_at_the_lock() {
1084 let (dev, _c) = rw_image("arrivals_settle", 0x0F);
1085 assert_eq!(waiting_at_the_lock(&dev), 0, "nothing has run yet");
1086
1087 let mut buf = [0u8; 16];
1088 dev.read_at(0, &mut buf).expect("read");
1089 assert_eq!(
1090 waiting_at_the_lock(&dev),
1091 0,
1092 "read_at left an arrival behind"
1093 );
1094
1095 dev.write_at(0, &[0x10u8; 16]).expect("write");
1096 assert_eq!(
1097 waiting_at_the_lock(&dev),
1098 0,
1099 "write_at left an arrival behind"
1100 );
1101
1102 dev.flush().expect("flush");
1103 assert_eq!(waiting_at_the_lock(&dev), 0, "flush left an arrival behind");
1104 }
1105
1106 /// A READ CONCURRENT WITH A WRITE MUST NOT PROCEED.
1107 ///
1108 /// This is the regression. Reads were briefly taken with no lock at
1109 /// all, which quietly removed the exclusion the original single
1110 /// mutex gave and left a read free to run through the middle of a
1111 /// `write_at` — `write_all` may become several `write` calls, and
1112 /// `read_at` is itself a loop, so either side can split.
1113 ///
1114 /// # Why the lock rather than the tear is the assertion
1115 ///
1116 /// The obvious test races a writer against readers and looks for a
1117 /// region holding bytes from both sides of the write. On a regular
1118 /// file that test cannot fail: a single `write` call is atomic
1119 /// against `pread` on both Linux and macOS, and `write_all` only
1120 /// splits above roughly 2 GiB, so the tear it looks for is
1121 /// unreachable at any size a test would use. It would pass with the
1122 /// fix reverted — an assertion whose outcome does not depend on the
1123 /// defect, which is worse than no assertion.
1124 ///
1125 /// So the exclusion itself is asserted, by holding the very guard
1126 /// `write_at` takes and requiring that a read cannot get past it.
1127 /// That is deterministic, needs no tear to be reproducible, and
1128 /// fails the moment `read_at` stops taking the lock.
1129 ///
1130 /// # And "cannot get past it" is observed, not timed
1131 ///
1132 /// See [`await_arrival`] and rust-fs-core#104: the reader is
1133 /// required to *arrive* at the lock, which a build that does not
1134 /// take the lock cannot do, rather than merely to not finish within
1135 /// a short window, which a build that does not take the lock
1136 /// manages easily on a loaded machine.
1137 #[test]
1138 fn a_read_cannot_proceed_while_a_write_holds_the_lock() {
1139 let (dev, _c) = rw_image("read_excluded_by_write", 0x5A);
1140
1141 // Stands in for a write in progress: the same exclusive guard
1142 // `write_at` holds across its seek and write. Taken directly
1143 // rather than through `exclusive_guard`, so the holder is not
1144 // itself counted as an arrival.
1145 let held = dev.io_lock.write().unwrap();
1146 assert_eq!(
1147 waiting_at_the_lock(&dev),
1148 0,
1149 "the holder must not count as a waiter, or the wait below proves nothing"
1150 );
1151
1152 let (done_tx, done_rx) = mpsc::channel();
1153 let worker = {
1154 let dev = std::sync::Arc::clone(&dev);
1155 std::thread::spawn(move || {
1156 let mut buf = [0u8; 4096];
1157 let outcome = dev.read_at(0, &mut buf);
1158 let _ = done_tx.send(outcome.map(|()| buf[0]));
1159 })
1160 };
1161 let mut lock = ReleaseThenJoin {
1162 held: Some(held),
1163 worker: Some(worker),
1164 };
1165
1166 // PROVE THE READER IS PARKED IN THE LOCK. Not that it started,
1167 // and not that it failed to finish in time: that it is inside
1168 // the acquisition this thread is holding shut.
1169 await_arrival(&dev, "read_at");
1170 assert!(
1171 matches!(done_rx.try_recv(), Err(mpsc::TryRecvError::Empty)),
1172 "a read completed while the exclusive write guard was held. Reads and \
1173 writes are not mutually excluded, so a read overlapping a write_at \
1174 can observe a partially written region"
1175 );
1176
1177 // And it is blocked rather than broken: it completes once the
1178 // writer lets go. Without this half the test would pass against
1179 // a read_at that simply never returned.
1180 lock.release();
1181 let first = done_rx
1182 .recv_timeout(UNBLOCKED_WITHIN)
1183 .expect("the read must proceed once the write guard is released")
1184 .expect("and must succeed");
1185 assert_eq!(first, 0x5A, "the read returned the wrong bytes");
1186 }
1187
1188 /// AND THE EXCLUSION HOLDS THE OTHER WAY ROUND.
1189 ///
1190 /// A write must not start while a read is in progress, or the read
1191 /// it interleaves with is the one that tears. Asserted with a shared
1192 /// guard, which is what a Unix reader holds.
1193 #[test]
1194 fn a_write_cannot_proceed_while_a_read_holds_the_lock() {
1195 let (dev, _c) = rw_image("write_excluded_by_read", 0x11);
1196
1197 // Stands in for a read in progress.
1198 let held = dev.io_lock.read().unwrap();
1199 assert_eq!(waiting_at_the_lock(&dev), 0, "the holder is not a waiter");
1200
1201 let (done_tx, done_rx) = mpsc::channel();
1202 let worker = {
1203 let dev = std::sync::Arc::clone(&dev);
1204 std::thread::spawn(move || {
1205 let _ = done_tx.send(dev.write_at(0, &[0x22u8; 4096]));
1206 })
1207 };
1208 let mut lock = ReleaseThenJoin {
1209 held: Some(held),
1210 worker: Some(worker),
1211 };
1212
1213 await_arrival(&dev, "write_at");
1214 assert!(
1215 matches!(done_rx.try_recv(), Err(mpsc::TryRecvError::Empty)),
1216 "a write completed while a read guard was held; a write_at may not \
1217 run through a read that is already in progress"
1218 );
1219
1220 lock.release();
1221 done_rx
1222 .recv_timeout(UNBLOCKED_WITHIN)
1223 .expect("the write must proceed once the read releases")
1224 .expect("and must succeed");
1225
1226 let mut buf = [0u8; 4];
1227 dev.read_at(0, &mut buf).expect("read back");
1228 assert_eq!(buf, [0x22; 4], "the write did not land");
1229 }
1230
1231 /// AND A FLUSH IS A WRITE FOR THIS PURPOSE.
1232 ///
1233 /// `flush` pushes buffered bytes at the file and calls `sync_data`,
1234 /// so it must not interleave with a read or a write any more than
1235 /// `write_at` may. The exclusive guard was here before this test
1236 /// was, and stating an invariant in a comment is not testing it:
1237 /// with the guard removed the whole suite stayed green.
1238 #[test]
1239 fn a_flush_cannot_proceed_while_a_read_holds_the_lock() {
1240 let (dev, _c) = rw_image("flush_excluded_by_read", 0x33);
1241
1242 // Stands in for a read in progress.
1243 let held = dev.io_lock.read().unwrap();
1244 assert_eq!(waiting_at_the_lock(&dev), 0, "the holder is not a waiter");
1245
1246 let (done_tx, done_rx) = mpsc::channel();
1247 let worker = {
1248 let dev = std::sync::Arc::clone(&dev);
1249 std::thread::spawn(move || {
1250 let _ = done_tx.send(dev.flush());
1251 })
1252 };
1253 let mut lock = ReleaseThenJoin {
1254 held: Some(held),
1255 worker: Some(worker),
1256 };
1257
1258 await_arrival(&dev, "flush");
1259 assert!(
1260 matches!(done_rx.try_recv(), Err(mpsc::TryRecvError::Empty)),
1261 "a flush completed while a read guard was held; flush takes the lock \
1262 exclusively for the same reason write_at does"
1263 );
1264
1265 lock.release();
1266 done_rx
1267 .recv_timeout(UNBLOCKED_WITHIN)
1268 .expect("the flush must proceed once the read releases")
1269 .expect("and must succeed");
1270 }
1271
1272 /// READERS STILL OVERLAP, WHICH IS THE POINT OF THE SHARED GUARD.
1273 ///
1274 /// THE OVER-CORRECTION THIS CATCHES: restoring read/write exclusion
1275 /// with a plain mutex, or by taking the write half of this lock on
1276 /// the read path, would pass both tests above and quietly undo the
1277 /// reader parallelism the positioned-read work existed for. Nothing
1278 /// else in the suite would notice, because every other assertion is
1279 /// about bytes and serialised readers return the right bytes.
1280 ///
1281 /// Unix only: on Windows `seek_read` moves the file pointer, so
1282 /// readers there take the guard exclusively on purpose and this
1283 /// would correctly block.
1284 #[test]
1285 #[cfg(unix)]
1286 fn a_read_does_not_exclude_another_read() {
1287 let (dev, _c) = rw_image("reads_overlap", 0x77);
1288
1289 // Stands in for another reader already inside `read_at`.
1290 let held = dev.io_lock.read().unwrap();
1291
1292 let (done_tx, done_rx) = mpsc::channel();
1293 let worker = {
1294 let dev = std::sync::Arc::clone(&dev);
1295 std::thread::spawn(move || {
1296 let mut buf = [0u8; 4096];
1297 let outcome = dev.read_at(0, &mut buf);
1298 let _ = done_tx.send(outcome.map(|()| buf[0]));
1299 })
1300 };
1301 // Holds the read guard for the whole assertion, so the read
1302 // below completes WHILE another reader holds the lock — which
1303 // is the claim. No arrival wait here and no ready signal: this
1304 // assertion is a positive one, and a worker that has not been
1305 // scheduled makes it slower, never falsely green.
1306 let lock = ReleaseThenJoin {
1307 held: Some(held),
1308 worker: Some(worker),
1309 };
1310
1311 let first = done_rx
1312 .recv_timeout(UNBLOCKED_WITHIN)
1313 .expect(
1314 "a read blocked behind another read. On Unix the guard must be \
1315 shared -- positioned reads need no cursor, and serialising them \
1316 undoes the parallelism the read path was rewritten for",
1317 )
1318 .expect("and the read must succeed");
1319 assert_eq!(first, 0x77);
1320 drop(lock);
1321 }
1322}