Skip to main content

detcore/syscalls/
files.rs

1/*
2 * Copyright (c) Meta Platforms, Inc. and affiliates.
3 * All rights reserved.
4 *
5 * This source code is licensed under the BSD-style license found in the
6 * LICENSE file in the root directory of this source tree.
7 */
8
9//! System calls for dealing with the file system.
10
11use std::collections::BTreeMap;
12use std::collections::BTreeSet;
13use std::net::Ipv4Addr;
14use std::net::Ipv6Addr;
15use std::os::unix::fs::MetadataExt;
16use std::path::Path;
17use std::path::PathBuf;
18
19use nix::fcntl::AtFlags;
20use nix::fcntl::OFlag;
21use rand::RngExt as _;
22use reverie::Error;
23use reverie::Guest;
24use reverie::Stack;
25use reverie::syscalls;
26use reverie::syscalls::Addr;
27use reverie::syscalls::AddrMut;
28use reverie::syscalls::Errno;
29use reverie::syscalls::FcntlCmd::*;
30use reverie::syscalls::MapFlags;
31use reverie::syscalls::MemoryAccess;
32use reverie::syscalls::PathPtr;
33use reverie::syscalls::ProtFlags;
34use reverie::syscalls::ReadAddr;
35use reverie::syscalls::SockFlag;
36use reverie::syscalls::StatPtr;
37use reverie::syscalls::StatxMask;
38use reverie::syscalls::Syscall;
39use reverie::syscalls::SyscallInfo;
40use reverie::syscalls::Sysno;
41use reverie::syscalls::Timespec;
42use reverie::syscalls::Whence;
43use reverie::syscalls::family::StatFamily;
44use tracing::error;
45use tracing::info;
46use tracing::trace;
47use tracing::warn;
48
49use super::deterministic_stdio_inode_for_resource;
50use crate::config::SchedHeuristic;
51use crate::dirents::*;
52use crate::fd::*;
53use crate::procfs::MountInfoSnapshot;
54use crate::procfs::ProcfsFile;
55use crate::procfs::ProcfsSnapshotContext;
56use crate::record_or_replay::RecordOrReplay;
57use crate::resources::Device;
58use crate::resources::Permission;
59use crate::resources::ResourceID;
60use crate::resources::Resources;
61use crate::resources::SABRE_INTERNAL_PIPE_IO_FYI;
62use crate::scheduler::runqueue::LAST_PRIORITY;
63use crate::stat::*;
64use crate::tool_global::*;
65use crate::tool_local::CapturedDetFdInstallError;
66use crate::tool_local::Detcore;
67use crate::tool_local::finish_partial_record_or_replay_write;
68use crate::types::*;
69
70/// A conversion from SOCK_* flags to O_* flags which makes unsafe (but checked during testing) assumptions.
71fn oflag_from_sock_bits(s_bits: i32) -> OFlag {
72    // An otherwise unsafe "cast" which leans on the `linux_flags_assumptions` below.
73    OFlag::from_bits_truncate(s_bits & (libc::SOCK_CLOEXEC | libc::SOCK_NONBLOCK))
74}
75
76const UNIX_AUTOBIND_NAME_LEN: usize = 6;
77// AUTONOMOUS-BOT-IMPLEMENTED
78// TODO-HUMAN-REVIEW(PR-2150): Review timer-slack procfs parsing,
79// per-operation target checks, and scalar/vector I/O emulation.
80const TIMER_SLACK_PARSE_BYTES: usize = 66;
81#[derive(Clone, Copy)]
82struct TimerSlackIovec {
83    base: usize,
84    len: usize,
85}
86
87#[derive(Clone, Copy, Debug, PartialEq, Eq)]
88struct TimerSlackBinding {
89    target: i32,
90    device: u64,
91    inode: u64,
92}
93
94fn classify_timer_slack_binding(
95    binding: TimerSlackBinding,
96    observed_identity: Option<(u64, u64)>,
97    current_tid: i32,
98) -> Result<(), Errno> {
99    if observed_identity != Some((binding.device, binding.inode)) {
100        // The bound task exited. A missing path and a new task that recycled
101        // the same numeric TID are both ESRCH for the old open inode.
102        return Err(Errno::ESRCH);
103    }
104    if current_tid != binding.target {
105        // Cross-task CAP_SYS_NICE access is intentionally not exposed.
106        return Err(Errno::EPERM);
107    }
108    Ok(())
109}
110
111fn parse_timer_slack_write(bytes: &[u8]) -> Result<u64, Errno> {
112    // `kstrtoull_from_user` copies at most sign + 64 binary digits + newline,
113    // then accepts decimal digits with one optional leading '+' and one
114    // optional trailing newline. An embedded NUL terminates the C string.
115    let bytes = &bytes[..bytes.len().min(TIMER_SLACK_PARSE_BYTES)];
116    let end = bytes
117        .iter()
118        .position(|byte| *byte == 0)
119        .unwrap_or(bytes.len());
120    let mut value = &bytes[..end];
121    if value.first() == Some(&b'+') {
122        value = &value[1..];
123    }
124    if value.last() == Some(&b'\n') {
125        value = &value[..value.len() - 1];
126    }
127    if value.is_empty() || !value.iter().all(u8::is_ascii_digit) {
128        return Err(Errno::EINVAL);
129    }
130    value.iter().try_fold(0_u64, |parsed, digit| {
131        parsed
132            .checked_mul(10)
133            .and_then(|parsed| parsed.checked_add(u64::from(digit - b'0')))
134            .ok_or(Errno::ERANGE)
135    })
136}
137
138fn vectored_offset(low: u64, high: u64) -> i64 {
139    if std::mem::size_of::<usize>() == 8 {
140        low as i64
141    } else {
142        ((high << 32) | (low & u32::MAX as u64)) as i64
143    }
144}
145
146/// RNG vectors use either the shared stream or an independent explicit offset.
147struct RandomVectoredRead {
148    address: usize,
149    count: usize,
150    offset: Option<u64>,
151    flags: i32,
152}
153
154/// Apply Linux's checks after complete iovec import, before touching output.
155fn validate_random_vector_read(offset: Option<u64>, total: usize, flags: i32) -> Result<(), Errno> {
156    if total == 0 {
157        return Ok(());
158    }
159    if let Some(offset) = offset
160        && offset
161            .checked_add(total as u64)
162            .is_none_or(|end| end > i64::MAX as u64)
163    {
164        return Err(Errno::EINVAL);
165    }
166    // Linux 7.1's RWF_NOSIGNAL is newer than the pinned libc crate. Native
167    // random-device controls accept it, including with RWF_NOWAIT.
168    const RWF_NOSIGNAL: i32 = 0x100;
169    const KNOWN: i32 = libc::RWF_HIPRI
170        | libc::RWF_DSYNC
171        | libc::RWF_SYNC
172        | libc::RWF_NOWAIT
173        | libc::RWF_APPEND
174        | libc::RWF_NOAPPEND
175        | libc::RWF_ATOMIC
176        | libc::RWF_DONTCACHE
177        | RWF_NOSIGNAL;
178    // Unknown flags win over conflicting recognized flags, while recognized
179    // but unsupported ATOMIC/DONTCACHE are rejected after APPEND/NOAPPEND.
180    if flags & !KNOWN != 0 {
181        return Err(Errno::EOPNOTSUPP);
182    }
183    if flags & libc::RWF_APPEND != 0 && flags & libc::RWF_NOAPPEND != 0 {
184        return Err(Errno::EINVAL);
185    }
186    if flags & (libc::RWF_ATOMIC | libc::RWF_DONTCACHE) != 0 {
187        return Err(Errno::EOPNOTSUPP);
188    }
189    Ok(())
190}
191
192fn read_iovecs<M: MemoryAccess>(
193    memory: &M,
194    address: Option<Addr<libc::iovec>>,
195    count: usize,
196) -> Result<Vec<TimerSlackIovec>, Errno> {
197    if count > libc::UIO_MAXIOV as usize {
198        return Err(Errno::EINVAL);
199    }
200    if count == 0 {
201        return Ok(Vec::new());
202    }
203    let address = address.ok_or(Errno::EFAULT)?;
204    let mut iovecs = vec![
205        libc::iovec {
206            iov_base: std::ptr::null_mut(),
207            iov_len: 0,
208        };
209        count
210    ];
211    memory.read_values(address, &mut iovecs)?;
212    let total = iovecs.iter().try_fold(0_usize, |total, iovec| {
213        total.checked_add(iovec.iov_len).ok_or(Errno::EINVAL)
214    })?;
215    if total > isize::MAX as usize {
216        return Err(Errno::EINVAL);
217    }
218    Ok(iovecs
219        .into_iter()
220        .map(|iovec| TimerSlackIovec {
221            base: iovec.iov_base as usize,
222            len: iovec.iov_len,
223        })
224        .collect())
225}
226
227fn copy_timer_slack_output<M: MemoryAccess>(
228    memory: &mut M,
229    destination: Option<AddrMut<'_, u8>>,
230    bytes: &[u8],
231) -> Result<usize, Errno> {
232    if bytes.is_empty() {
233        return Ok(0);
234    }
235    let copied = memory.write(destination.ok_or(Errno::EFAULT)?, bytes)?;
236    if copied == 0 {
237        Err(Errno::EFAULT)
238    } else {
239        Ok(copied)
240    }
241}
242
243/// Capacity used for pipes that Detcore makes physically nonblocking.
244///
245/// Linux normally creates 64-KiB pipes on this platform, but silently falls back to two pages
246/// once the creating UID crosses `pipe-user-pages-soft`. That host-global accounting can change
247/// between the two executions of `hermit run --verify`, changing whether the same write succeeds
248/// immediately or enters the scheduler's `InternalIOPolling` retry path. Two pages is the
249/// pressure-mode capacity on supported x86-64 Linux hosts and, unlike 64 KiB, never requires an
250/// unprivileged capacity increase while the soft limit is active.
251///
252/// This is also the ceiling the guest is allowed to raise a pipe to, and the
253/// value `/proc/sys/fs/pipe-max-size` reports. Those three must be ONE constant:
254/// a pinned capacity, an advertised maximum and an enforced maximum that can
255/// drift apart are three copies of one rule, and a duplicated rule drifts in
256/// N-1 places while each copy looks right on its own.
257pub(crate) const DETERMINISTIC_PIPE_CAPACITY_BYTES: i32 = 8 * 1024;
258
259/// Whether a guest's `F_SETPIPE_SZ` request must be refused as a growth past
260/// the deterministic ceiling.
261///
262/// Separated from the handler so the BOUNDARY is testable without a guest. The
263/// boundary is the whole content of this rule: a request for exactly the pinned
264/// capacity must be allowed, or hermit's own pin value becomes unreachable to a
265/// guest that reads `/proc/sys/fs/pipe-max-size` and asks for precisely what it
266/// was told.
267pub(crate) fn pipe_capacity_request_exceeds_ceiling(requested: i32) -> bool {
268    requested > DETERMINISTIC_PIPE_CAPACITY_BYTES
269}
270
271/// Why the pin failed, and which descriptors Linux had already created when it did.
272///
273/// `pipe2` has already SUCCEEDED by the time the capacity is pinned, so the two descriptors
274/// exist in the guest whatever happens next. Carrying them alongside the errno is what makes
275/// releasing them possible; classifying separately from acting on it is what makes the
276/// classification unit-testable without a guest.
277#[derive(Debug, PartialEq, Eq)]
278struct PipeCapacityFailure {
279    created_fds: [i32; 2],
280    error: Errno,
281}
282
283impl PipeCapacityFailure {
284    fn close_syscalls(&self) -> [syscalls::Close; 2] {
285        self.created_fds
286            .map(|fd| syscalls::Close::new().with_fd(fd))
287    }
288}
289
290/// Classify the result of pinning a pipe's capacity.
291///
292/// The ONLY success shape is Linux returning exactly the requested capacity. `F_SETPIPE_SZ`
293/// returns the capacity it actually applied, and it may round; a rounded value is a pipe whose
294/// size we did not choose, which is the host-dependent capacity this path exists to remove. It
295/// is not a kernel errno, so it is reported as `EIO` rather than dressed up as one.
296fn pipe_capacity_failure(
297    created_fds: [i32; 2],
298    capacity_result: Result<i64, Errno>,
299) -> Option<PipeCapacityFailure> {
300    let error = match capacity_result {
301        Ok(applied) if applied == i64::from(DETERMINISTIC_PIPE_CAPACITY_BYTES) => return None,
302        Ok(_) => Errno::EIO,
303        Err(error) => error,
304    };
305    Some(PipeCapacityFailure { created_fds, error })
306}
307
308fn should_tag_sabre_internal_pipe_io(
309    discovers_live_metadata: bool,
310    fd_type: FdType,
311    physically_nonblocking: bool,
312    logically_nonblocking: bool,
313) -> bool {
314    discovers_live_metadata
315        && fd_type == FdType::Pipe
316        && physically_nonblocking
317        && !logically_nonblocking
318}
319
320fn random_device_lseek_result(status_flags: i32, whence: Whence) -> Result<i64, Errno> {
321    if status_flags & OFlag::O_PATH.bits() != 0 {
322        return Err(Errno::EBADF);
323    }
324    match whence {
325        Whence::SEEK_SET
326        | Whence::SEEK_CUR
327        | Whence::SEEK_END
328        | Whence::SEEK_DATA
329        | Whence::SEEK_HOLE => Ok(0),
330        _ => Err(Errno::EINVAL),
331    }
332}
333
334fn require_random_device_read_access(status_flags: i32) -> Result<(), Errno> {
335    if status_flags & libc::O_PATH != 0
336        || !matches!(
337            status_flags & libc::O_ACCMODE,
338            libc::O_RDONLY | libc::O_RDWR
339        )
340    {
341        Err(Errno::EBADF)
342    } else {
343        Ok(())
344    }
345}
346
347/// Inherited container output is a stream even when an outer runner stores it
348/// in a seekable file.  The backing file also carries Hermit's own diagnostics,
349/// so exposing its live offset makes tool logging guest-visible.  Preserve real
350/// file semantics after a guest replaces stdout/stderr: `dup2(file, 1)` copies
351/// the file's resource rather than this container-output resource.
352fn is_inherited_container_output(resource: Option<ResourceID>) -> bool {
353    matches!(
354        resource,
355        Some(ResourceID::Device(
356            Device::ContainerStdout | Device::ContainerStderr
357        ))
358    )
359}
360
361fn unix_autobind_addrlen() -> i32 {
362    (std::mem::offset_of!(libc::sockaddr_un, sun_path) + UNIX_AUTOBIND_NAME_LEN) as i32
363}
364
365fn unix_autobind_address(port: u16) -> libc::sockaddr_un {
366    // Linux autobind names are a leading NUL followed by five lowercase hex digits.
367    let mut address: libc::sockaddr_un = unsafe { std::mem::zeroed() };
368    address.sun_family = libc::AF_UNIX as libc::sa_family_t;
369    for (destination, source) in address.sun_path[1..UNIX_AUTOBIND_NAME_LEN]
370        .iter_mut()
371        .zip(format!("{port:05x}").bytes())
372    {
373        *destination = source as libc::c_char;
374    }
375    address
376}
377
378// TODO-HUMAN-REVIEW(PR-904): Review the TCP_INFO compatibility boundary.
379/// Retain the logical TCP state and negotiated option header while hiding all
380/// host timing, rate, packet, and byte counters.
381fn canonicalize_tcp_info(info: &mut [u8]) {
382    for (offset, byte) in info.iter_mut().enumerate() {
383        if !matches!(offset, 0 | 1 | 5 | 6) {
384            *byte = 0;
385        }
386    }
387}
388
389// Hermit exposes exactly one isolated guest network namespace.
390const DETERMINISTIC_NETNS_COOKIE: u64 = 1;
391
392// Above Linux's PID range and below the high-bit IDs used by kernel autobind.
393const DETERMINISTIC_NETLINK_PORT_ID_BASE: u32 = 0x4000_0000;
394
395/// Does the new guest descriptor `fd` reopen an anonymous pipe that the same
396/// process already holds as a scheduler-managed pipe (`managed_pipe_fds`)?
397///
398/// Both `/proc/<pid>/fd/<fd>` links read `pipe:[<inode>]` for the same pipe.
399/// A named FIFO links to its path, and on ptrace a host pipe is never in
400/// `managed_pipe_fds` (see `scheduler_managed_pipe_fds` for SaBRe), so neither
401/// matches: a host writer is outside the scheduler and must not be polled as if
402/// it were a guest
403/// (<https://github.com/rrnewton/hermit/pull/3534#issuecomment-5962369261>).
404fn reopens_scheduler_managed_pipe(pid: i32, fd: RawFd, managed_pipe_fds: &[RawFd]) -> bool {
405    let link = |fd: RawFd| std::fs::read_link(format!("/proc/{pid}/fd/{fd}")).ok();
406    let Some(opened) = link(fd) else {
407        return false;
408    };
409    if !opened
410        .to_str()
411        .is_some_and(|name| name.starts_with("pipe:["))
412    {
413        return false;
414    }
415    managed_pipe_fds
416        .iter()
417        .any(|&held| held != fd && link(held).as_ref() == Some(&opened))
418}
419
420/// The path the kernel resolved an open descriptor to, read from the guest's
421/// own `/proc/<pid>/fd/<fd>` link. This is the evidence authority for "which
422/// object was opened": it is produced by the kernel from the descriptor itself,
423/// so it is independent of the pathname spelling the guest used.
424///
425/// Used ONLY as a fallback for a pathname that does not classify on its own
426/// (see the call site): a spelling that already classifies must keep its own
427/// classification, because `/proc/self/...` and `/proc/thread-self/...` are
428/// defined by the spelling and resolve to a different numeric path.
429///
430/// `None` when the link cannot be read (the descriptor is gone, or procfs is
431/// unavailable), in which case the lexical result stands unchanged.
432fn resolved_open_path(pid: i32, fd: RawFd) -> Option<PathBuf> {
433    let link = std::fs::read_link(format!("/proc/{pid}/fd/{fd}")).ok()?;
434    // A deleted or anonymous target is not a stable object name.
435    link.is_absolute().then_some(link)
436}
437
438/// Resolve an `AT_FDCWD`-relative spelling in the guest's filesystem view.
439///
440/// Replayer chroots the guest, so the tracer-visible cwd includes the replay
441/// root. Stripping `/proc/<pid>/root` produces the same guest-absolute path in
442/// record and replay without depending on the opened descriptor (which is an
443/// eventfd placeholder during replay).
444fn resolved_at_fdcwd_path(pid: i32, path: &Path) -> Option<PathBuf> {
445    debug_assert!(!path.is_absolute());
446    let root = std::fs::read_link(format!("/proc/{pid}/root")).ok()?;
447    let cwd = std::fs::read_link(format!("/proc/{pid}/cwd")).ok()?;
448    let guest_cwd = cwd.strip_prefix(root).ok()?;
449    Some(Path::new("/").join(guest_cwd).join(path))
450}
451
452/// Writes back the guest bytes that the utimensat lookup buffer covered.
453fn restore_lookup_buffer<M: MemoryAccess>(memory: &mut M, buffer: StatPtr, saved: &[u8]) {
454    if memory.write_exact(buffer.0.cast(), saved).is_err() {
455        info!("Could not restore the guest bytes under the utimensat lookup buffer.");
456    }
457}
458
459/// Whether the utimensat lookup buffer overlaps the guest memory Linux reads
460/// for `call`: the path through its NUL and, with `guest_times`, the two
461/// timespecs. A path that cannot be read counts as overlapping, so that the
462/// kernel, not a lookup, reports the fault.
463fn utimensat_input_overlaps<M: MemoryAccess>(
464    memory: &M,
465    call: &syscalls::Utimensat,
466    guest_times: bool,
467    buffer: StatPtr,
468) -> bool {
469    use reverie::syscalls::FromToRaw;
470
471    let start = buffer.0.as_raw();
472    let end = start + std::mem::size_of::<libc::stat>();
473    let overlaps = |addr: usize, len: usize| addr < end && start < addr.saturating_add(len);
474    let path = call.path().is_some_and(|path| match path.read(memory) {
475        Ok(path) => overlaps(call.path().into_raw(), path.as_os_str().len() + 1),
476        Err(_) => true,
477    });
478    let times = guest_times
479        && call
480            .times()
481            .is_some_and(|times| overlaps(times.as_raw(), std::mem::size_of::<[Timespec; 2]>()));
482    path || times
483}
484
485impl<T: RecordOrReplay> Detcore<T> {
486    async fn observe_timer_slack_identity<G: Guest<Self>>(
487        &self,
488        guest: &mut G,
489        target: i32,
490    ) -> Result<Option<(u64, u64)>, Error> {
491        // Re-resolve the numeric proc path on every operation. Linux gives a
492        // recycled TID a different proc inode, while the original open file
493        // description remains bound to the exited task's inode.
494        let path = format!("/proc/{target}/timerslack_ns");
495        let path_bytes = path.as_bytes();
496        let mut path_buffer = [0_u8; 64];
497        assert!(path_bytes.len() < path_buffer.len());
498        path_buffer[..path_bytes.len()].copy_from_slice(path_bytes);
499
500        let mut stack = guest.stack().await;
501        let path_address = stack.push(path_buffer).cast::<libc::c_char>();
502        let statptr = StatPtr(stack.reserve());
503        let stack_guard = stack.commit()?;
504        let call = syscalls::Fstatat::new()
505            .with_dirfd(libc::AT_FDCWD)
506            .with_path(PathPtr::from_ptr(
507                path_address.as_raw() as *const libc::c_char
508            ))
509            .with_stat(Some(statptr))
510            .with_flags(AtFlags::empty());
511        let mut identity = match guest.inject_with_retry(call).await {
512            Ok(_) => {
513                let stat = statptr.read(&guest.memory())?;
514                Some((stat.st_dev, stat.st_ino))
515            }
516            Err(Errno::ENOENT) | Err(Errno::ESRCH) => None,
517            Err(error) => return Err(error.into()),
518        };
519        drop(stack_guard);
520        // Replayer runs the guest in a filesystem chroot whose `/proc` path is
521        // intentionally absent, while the tracing process remains in the same
522        // PID namespace and can resolve the task through its own proc mount.
523        // Use that equivalent view only for record/replay; other backends keep
524        // the guest-path result above as their sole authority.
525        if identity.is_none()
526            && guest.config().recordreplay_modes
527            && let Ok(metadata) = std::fs::metadata(path)
528        {
529            identity = Some((metadata.dev(), metadata.ino()));
530        }
531        Ok(identity)
532    }
533
534    async fn require_current_timer_slack_target<G: Guest<Self>>(
535        &self,
536        guest: &mut G,
537        binding: TimerSlackBinding,
538    ) -> Result<(), Error> {
539        let observed_identity = self
540            .observe_timer_slack_identity(guest, binding.target)
541            .await?;
542        let current = guest.inject(syscalls::Gettid::new()).await? as i32;
543        classify_timer_slack_binding(binding, observed_identity, current).map_err(Into::into)
544    }
545
546    fn timer_slack_binding<G: Guest<Self>>(
547        &self,
548        guest: &G,
549        fd: RawFd,
550    ) -> Result<Option<TimerSlackBinding>, Errno> {
551        guest.thread_state().with_detfd(fd, |detfd| {
552            detfd
553                .procfs_timer_slack_binding()
554                .map(|(target, device, inode)| TimerSlackBinding {
555                    target,
556                    device,
557                    inode,
558                })
559        })
560    }
561
562    fn require_timer_slack_access<G: Guest<Self>>(
563        &self,
564        guest: &G,
565        fd: RawFd,
566        write: bool,
567    ) -> Result<(), Errno> {
568        guest.thread_state().with_detfd(fd, |detfd| {
569            let flags = detfd.status_flags();
570            let mode = flags & libc::O_ACCMODE;
571            let denied = flags & libc::O_PATH != 0
572                || if write {
573                    mode == libc::O_RDONLY
574                } else {
575                    mode == libc::O_WRONLY
576                };
577            (!denied).then_some(()).ok_or(Errno::EBADF)
578        })?
579    }
580
581    fn read_timer_slack_input<G: Guest<Self>>(
582        &self,
583        guest: &G,
584        buffer: Option<Addr<u8>>,
585        count: usize,
586    ) -> Result<u64, Errno> {
587        let mut bytes = vec![0_u8; count.min(TIMER_SLACK_PARSE_BYTES)];
588        if !bytes.is_empty() {
589            guest
590                .memory()
591                .read_exact(buffer.ok_or(Errno::EFAULT)?, &mut bytes)?;
592        }
593        parse_timer_slack_write(&bytes)
594    }
595
596    async fn read_timer_slack<G: Guest<Self>>(
597        &self,
598        guest: &mut G,
599        fd: RawFd,
600        buffer: Option<AddrMut<'_, u8>>,
601        maximum: usize,
602    ) -> Result<i64, Error> {
603        self.require_timer_slack_access(guest, fd, false)?;
604        if maximum == 0 {
605            return Ok(0);
606        }
607        let binding = self
608            .timer_slack_binding(guest, fd)?
609            .expect("timer-slack read lost its procfs classification");
610        self.require_current_timer_slack_target(guest, binding)
611            .await?;
612        let value = guest.thread_state().timer_slack_ns;
613        let preview = guest
614            .thread_state()
615            .with_detfd(fd, |detfd| detfd.preview_procfs_timer_slack(value, maximum))?
616            .expect("timer-slack procfs state disappeared");
617        let copied = copy_timer_slack_output(&mut guest.memory(), buffer, &preview.bytes)?;
618        if copied != 0 {
619            guest.thread_state().with_detfd(fd, |detfd| {
620                detfd.commit_procfs_timer_slack_read(&preview, copied);
621            })?;
622        }
623        Ok(copied as i64)
624    }
625
626    async fn pread_timer_slack<G: Guest<Self>>(
627        &self,
628        guest: &mut G,
629        fd: RawFd,
630        buffer: Option<AddrMut<'_, u8>>,
631        maximum: usize,
632        offset: i64,
633    ) -> Result<i64, Error> {
634        if offset < 0 {
635            return Err(Errno::EINVAL.into());
636        }
637        self.require_timer_slack_access(guest, fd, false)?;
638        if maximum == 0 {
639            return Ok(0);
640        }
641        let binding = self
642            .timer_slack_binding(guest, fd)?
643            .expect("timer-slack pread lost its procfs classification");
644        self.require_current_timer_slack_target(guest, binding)
645            .await?;
646        let value = guest.thread_state().timer_slack_ns;
647        let bytes = guest
648            .thread_state()
649            .with_detfd(fd, |detfd| {
650                detfd.take_procfs_timer_slack_at(value, offset as usize, maximum)
651            })?
652            .expect("timer-slack procfs state disappeared");
653        Ok(copy_timer_slack_output(&mut guest.memory(), buffer, &bytes)? as i64)
654    }
655
656    async fn readv_timer_slack<G: Guest<Self>>(
657        &self,
658        guest: &mut G,
659        fd: RawFd,
660        iovecs: Vec<TimerSlackIovec>,
661        offset: Option<i64>,
662        flags: i32,
663    ) -> Result<i64, Error> {
664        self.require_timer_slack_access(guest, fd, false)?;
665        let maximum = iovecs.iter().map(|iovec| iovec.len).sum::<usize>();
666        if maximum == 0 {
667            return Ok(0);
668        }
669        if flags & !libc::RWF_HIPRI != 0 {
670            return Err(Errno::EOPNOTSUPP.into());
671        }
672        let binding = self
673            .timer_slack_binding(guest, fd)?
674            .expect("timer-slack readv lost its procfs classification");
675        self.require_current_timer_slack_target(guest, binding)
676            .await?;
677        let value = guest.thread_state().timer_slack_ns;
678        let mut positioned_offset = offset.map(|offset| offset as usize);
679        let mut total = 0_usize;
680        for iovec in iovecs {
681            if iovec.len == 0 {
682                continue;
683            }
684            let sequential_preview = if positioned_offset.is_none() {
685                guest.thread_state().with_detfd(fd, |detfd| {
686                    detfd.preview_procfs_timer_slack(value, iovec.len)
687                })?
688            } else {
689                None
690            };
691            let bytes = match (&sequential_preview, positioned_offset) {
692                (Some(preview), None) => preview.bytes.clone(),
693                (None, Some(offset)) => guest
694                    .thread_state()
695                    .with_detfd(fd, |detfd| {
696                        detfd.take_procfs_timer_slack_at(value, offset, iovec.len)
697                    })?
698                    .expect("timer-slack procfs state disappeared"),
699                _ => unreachable!("timer-slack read mode changed while reading"),
700            };
701            if bytes.is_empty() {
702                break;
703            }
704            let copied = match copy_timer_slack_output(
705                &mut guest.memory(),
706                AddrMut::from_raw(iovec.base),
707                &bytes,
708            ) {
709                Ok(copied) => copied,
710                Err(_) if total > 0 => return Ok(total as i64),
711                Err(error) => return Err(error.into()),
712            };
713            if let Some(preview) = &sequential_preview {
714                guest.thread_state().with_detfd(fd, |detfd| {
715                    detfd.commit_procfs_timer_slack_read(preview, copied);
716                })?;
717            }
718            total += copied;
719            if let Some(offset) = positioned_offset.as_mut() {
720                *offset += copied;
721            }
722            if copied != bytes.len() {
723                return Ok(total as i64);
724            }
725            if bytes.len() != iovec.len {
726                break;
727            }
728        }
729        Ok(total as i64)
730    }
731
732    async fn write_timer_slack<G: Guest<Self>>(
733        &self,
734        guest: &mut G,
735        fd: RawFd,
736        buffer: Option<Addr<'_, u8>>,
737        count: usize,
738    ) -> Result<i64, Error> {
739        self.require_timer_slack_access(guest, fd, true)?;
740        let requested = self.read_timer_slack_input(guest, buffer, count)?;
741        let binding = self
742            .timer_slack_binding(guest, fd)?
743            .expect("timer-slack write lost its procfs classification");
744        self.require_current_timer_slack_target(guest, binding)
745            .await?;
746        let state = guest.thread_state_mut();
747        state.timer_slack_ns = if requested == 0 {
748            state.default_timer_slack_ns
749        } else {
750            requested
751        };
752        i64::try_from(count).map_err(|_| Errno::EINVAL.into())
753    }
754
755    async fn writev_timer_slack<G: Guest<Self>>(
756        &self,
757        guest: &mut G,
758        fd: RawFd,
759        iovecs: Vec<TimerSlackIovec>,
760        flags: i32,
761    ) -> Result<i64, Error> {
762        self.require_timer_slack_access(guest, fd, true)?;
763        if iovecs.iter().all(|iovec| iovec.len == 0) {
764            return Ok(0);
765        }
766        if flags & !libc::RWF_HIPRI != 0 {
767            return Err(Errno::EOPNOTSUPP.into());
768        }
769        // Procfs supplies only `.write`, so Linux's writev fallback invokes it
770        // once per nonempty iovec. Preserve its partial-success behavior and
771        // let the last successful segment determine the current slack.
772        let mut total = 0_i64;
773        for iovec in iovecs {
774            if iovec.len == 0 {
775                continue;
776            }
777            let buffer = Addr::from_raw(iovec.base).ok_or(Errno::EFAULT);
778            let requested = match buffer
779                .and_then(|buffer| self.read_timer_slack_input(guest, Some(buffer), iovec.len))
780            {
781                Ok(requested) => requested,
782                Err(_error) if total > 0 => return Ok(total),
783                Err(error) => return Err(error.into()),
784            };
785            let binding = self
786                .timer_slack_binding(guest, fd)?
787                .expect("timer-slack writev lost its procfs classification");
788            if let Err(error) = self
789                .require_current_timer_slack_target(guest, binding)
790                .await
791            {
792                return if total > 0 { Ok(total) } else { Err(error) };
793            }
794            let state = guest.thread_state_mut();
795            state.timer_slack_ns = if requested == 0 {
796                state.default_timer_slack_ns
797            } else {
798                requested
799            };
800            total = total
801                .checked_add(i64::try_from(iovec.len).map_err(|_| Errno::EINVAL)?)
802                .ok_or(Errno::EINVAL)?;
803        }
804        Ok(total)
805    }
806    /// Set the kernel's O_NONBLOCK on `fd`'s open file description, keeping its
807    /// other status flags. The caller records the change with
808    /// `maybe_set_nonblocking_fd`; the guest-visible flags are unchanged.
809    async fn inject_physical_nonblocking<G: Guest<Self>>(
810        &self,
811        guest: &mut G,
812        fd: RawFd,
813    ) -> Result<(), Errno> {
814        let flags = guest
815            .inject(syscalls::Fcntl::new().with_fd(fd).with_cmd(F_GETFL))
816            .await?;
817        guest
818            .inject(
819                syscalls::Fcntl::new()
820                    .with_fd(fd)
821                    .with_cmd(F_SETFL(flags as i32 | OFlag::O_NONBLOCK.bits())),
822            )
823            .await?;
824        Ok(())
825    }
826
827    /// Inject an extra fstat to retrieve file metadata.
828    ///
829    /// The kernel needs a writable `struct stat` in the guest. It is staged
830    /// first in the guest stack scratch, which on the ptrace backend lies just
831    /// below the red zone under the guest's stack pointer and costs nothing
832    /// when the stack has room. That memory is not guaranteed to be writable:
833    /// the guest may run with its stack pointer just above a guard page (a
834    /// thread, fiber or alternate signal stack), or within a few hundred bytes
835    /// of the lowest page of the main-thread stack, which a tracer's write does
836    /// not grow. The fault then belongs to Detcore's bookkeeping, not to the
837    /// guest, so the same fstat is repeated with its buffer in a transient
838    /// private page that is unmapped before the guest resumes
839    /// (<https://github.com/rrnewton/hermit/issues/3328>).
840    pub(crate) async fn inject_fstat<G: Guest<Self>>(
841        &self,
842        guest: &mut G,
843        raw_fd: RawFd,
844    ) -> Result<libc::stat, Errno> {
845        info!(
846            "Injecting additional fstat to retrieve file metadata on fd {}.",
847            raw_fd
848        );
849        let copied = match self.inject_fstat_on_stack(guest, raw_fd).await {
850            Err(Errno::EFAULT) => {
851                info!(
852                    "Guest stack scratch cannot hold the fstat buffer for fd {}; \
853                     using a transient page instead.",
854                    raw_fd
855                );
856                self.inject_fstat_in_transient_page(guest, raw_fd).await?
857            }
858            result => result?,
859        };
860        trace!("extra fstat returned inode {}", copied.st_ino);
861        Ok(copied)
862    }
863
864    /// The fast path of [`Self::inject_fstat`]: the buffer lives in the guest
865    /// stack scratch. Returns `EFAULT` when that scratch is not writable.
866    async fn inject_fstat_on_stack<G: Guest<Self>>(
867        &self,
868        guest: &mut G,
869        raw_fd: RawFd,
870    ) -> Result<libc::stat, Errno> {
871        let mut stack = guest.stack().await;
872        let statptr: StatPtr = StatPtr(stack.reserve());
873        // Keep the guard until the buffer is no longer used. Backends whose
874        // scratch is a Tool-owned arena (DBT, SaBRe) free it when the guard
875        // drops, so dropping it before the injected fstat would let the kernel
876        // write into freed memory.
877        let _stack_guard = stack.commit()?;
878        let copied = Self::inject_fstat_into(guest, raw_fd, statptr).await?;
879        // clear stack memory used for fstat allocation
880        guest
881            .memory()
882            .write_exact(statptr.0.cast(), &[0; std::mem::size_of::<libc::stat>()])?;
883        Ok(copied)
884    }
885
886    /// The fallback of [`Self::inject_fstat`]: the buffer lives in a private
887    /// anonymous page mapped for this call only. The guest never learns its
888    /// address, and it is unmapped before the guest resumes, so the guest's
889    /// address space is the same as before the call.
890    async fn inject_fstat_in_transient_page<G: Guest<Self>>(
891        &self,
892        guest: &mut G,
893        raw_fd: RawFd,
894    ) -> Result<libc::stat, Errno> {
895        let len = std::mem::size_of::<libc::stat>();
896        let mapped = guest
897            .inject_with_retry(Syscall::Mmap(
898                syscalls::Mmap::new()
899                    .with_addr(None)
900                    .with_len(len)
901                    .with_prot(ProtFlags::PROT_READ | ProtFlags::PROT_WRITE)
902                    .with_flags(MapFlags::MAP_PRIVATE | MapFlags::MAP_ANONYMOUS)
903                    .with_fd(-1)
904                    .with_offset(0),
905            ))
906            .await?;
907        let page = usize::try_from(mapped)
908            .ok()
909            .and_then(AddrMut::<libc::stat>::from_raw)
910            .unwrap_or_else(|| panic!("transient fstat page mmap returned {mapped}"));
911        let copied = Self::inject_fstat_into(guest, raw_fd, StatPtr(page)).await;
912        if let Err(errno) = guest
913            .inject_with_retry(Syscall::Munmap(
914                syscalls::Munmap::new()
915                    .with_addr(Some(page.cast::<libc::c_void>().into()))
916                    .with_len(len),
917            ))
918            .await
919        {
920            // Not expected: the page was mapped by this call and its address
921            // never reached the guest. The metadata is still valid, so a
922            // leftover page is no reason to fail the guest's syscall.
923            warn!(
924                "[detcore] could not unmap the transient fstat page for fd {}: {}",
925                raw_fd, errno
926            );
927        }
928        copied
929    }
930
931    /// Inject `fstat(raw_fd, statptr)` and read back what the kernel wrote.
932    async fn inject_fstat_into<G: Guest<Self>>(
933        guest: &mut G,
934        raw_fd: RawFd,
935        statptr: StatPtr<'_>,
936    ) -> Result<libc::stat, Errno> {
937        // NOTE: Must retry the injection here. This could get interrupted and
938        // we don't want to rerun the entire syscall handler twice.
939        guest
940            .inject_with_retry(Syscall::Fstat(
941                syscalls::Fstat::new()
942                    .with_fd(raw_fd)
943                    .with_stat(Some(statptr)),
944            ))
945            .await?;
946        statptr.read(&guest.memory())
947    }
948
949    // helper function to track a new file descriptor.
950    pub(crate) async fn add_fd<G: Guest<Self>>(
951        &self,
952        guest: &mut G,
953        fd: RawFd,
954        flags: OFlag,
955        ty: FdType,
956    ) -> Result<(), Errno> {
957        let stat = if guest.config().virtualize_metadata {
958            match self.inject_fstat(guest, fd).await {
959                Ok(stat) => Some(stat.into()),
960                Err(errno) => {
961                    // `fd` is already open in the guest, but Detcore cannot
962                    // model it: with metadata virtualization on, a descriptor
963                    // without its stat is an invariant violation. The caller
964                    // reports this error as the syscall's result, so close
965                    // the descriptor rather than leave an untracked one open
966                    // behind that error. Not retried on EINTR: Linux releases
967                    // the descriptor even then, and a retry could close a
968                    // reused number.
969                    if let Err(close_errno) = guest.inject(syscalls::Close::new().with_fd(fd)).await
970                    {
971                        warn!(
972                            "[detcore] could not close fd {} after failing to record its \
973                             metadata ({}): {}",
974                            fd, errno, close_errno
975                        );
976                    }
977                    return Err(errno);
978                }
979            }
980        } else {
981            None
982        };
983        guest.thread_state().add_fd(fd, flags, ty, stat)
984    }
985
986    pub(crate) async fn release_port_for_open_file<G: Guest<Self>>(
987        &self,
988        guest: &mut G,
989        open_file_id: OpenFileId,
990    ) -> Option<u16> {
991        let response = send_and_update_time(guest, GlobalRequest::ReleasePort(open_file_id)).await;
992        match response.1 {
993            GlobalResponse::ReleasePort(port) => port,
994            other => panic!("unexpected release-port response: {other:?}"),
995        }
996    }
997
998    pub(crate) async fn restore_port_for_open_file<G: Guest<Self>>(
999        &self,
1000        guest: &mut G,
1001        open_file_id: OpenFileId,
1002        port: u16,
1003    ) {
1004        let response =
1005            send_and_update_time(guest, GlobalRequest::AddUsedPort(port, open_file_id)).await;
1006        match response.1 {
1007            GlobalResponse::AddUsedPort => {}
1008            other => panic!("unexpected restore-port response: {other:?}"),
1009        }
1010    }
1011
1012    /// Openat system call.
1013    pub async fn handle_openat<G: Guest<Self>>(
1014        &self,
1015        guest: &mut G,
1016        call: syscalls::Openat,
1017    ) -> Result<i64, Error> {
1018        let path = call.path().ok_or(Errno::EFAULT)?;
1019        let path: PathBuf = path.read(&guest.memory())?;
1020        // A relative spelling is not the object. `chdir("/sys/module/kvm");
1021        // open("refcnt")` and an absolute open name the SAME kernel object, so
1022        // classifying the unresolved lexical pathname lets one spelling bypass
1023        // normalization and expose the host value. Absolute paths are already
1024        // bound; a dirfd supplies its own prefix; AT_FDCWD-relative spellings
1025        // are resolved through the guest's root and cwd before the open so the
1026        // result does not depend on Replayer's placeholder descriptor.
1027        let observed_path = if path.is_absolute() {
1028            path.clone()
1029        } else if call.dirfd() == libc::AT_FDCWD {
1030            resolved_at_fdcwd_path(guest.pid().as_raw(), &path).unwrap_or_else(|| path.clone())
1031        } else {
1032            guest
1033                .thread_state()
1034                .with_detfd(call.dirfd(), |detfd| detfd.path())?
1035                .map_or_else(|| path.clone(), |directory| directory.join(&path))
1036        };
1037
1038        let resource = ResourceID::Path(path.clone());
1039        // Ask for permission to resolve this path into a file:
1040        let request = guest.thread_state().mk_request(resource, Permission::R);
1041        resource_request(guest, request).await;
1042        let res = self.record_or_replay(guest, Syscall::Openat(call)).await;
1043
1044        match res {
1045            Ok(fd) => {
1046                let fd = fd as RawFd;
1047                let fd_type = path.to_str().map_or(FdType::Regular, |fname| {
1048                    if fname == "/dev/random" || fname == "/dev/urandom" {
1049                        FdType::Rng
1050                    } else {
1051                        FdType::Regular
1052                    }
1053                });
1054                // A guest-created pipe reopened by path (`/dev/stdin`,
1055                // `/proc/self/fd/N`, bash's `< <(cmd)` as `/dev/fd/63`) is a NEW open
1056                // file description, so it carries neither the Pipe type nor the
1057                // physical O_NONBLOCK that `handle_pipe2` gave the original. Left as a
1058                // physically blocking Regular fd, a read that waits for a writer blocks
1059                // in the kernel while holding the scheduler turn, and the writer never
1060                // runs (https://github.com/rrnewton/hermit/issues/1850). Give it the
1061                // same treatment as `handle_pipe2`: Pipe type plus a Detcore-internal
1062                // physical O_NONBLOCK, which F_GETFL hides from the guest. Only a pipe
1063                // this process already holds as scheduler-managed qualifies; a host
1064                // pipe or a named FIFO keeps `deterministic_read`'s Regular handling.
1065                // Gated like the forced O_NONBLOCK in F_SETFL: Replayer's descriptor
1066                // is an eventfd placeholder, so record and replay would not classify
1067                // alike.
1068                let fd_type = if fd_type == FdType::Regular
1069                    && self.cfg.use_nonblocking_sockets()
1070                    && !self.cfg.recordreplay_modes
1071                    && !call.flags().contains(OFlag::O_PATH)
1072                    && reopens_scheduler_managed_pipe(
1073                        guest.pid().as_raw(),
1074                        fd,
1075                        &guest.thread_state().scheduler_managed_pipe_fds(),
1076                    )
1077                    && (call.flags().contains(OFlag::O_NONBLOCK)
1078                        || self.inject_physical_nonblocking(guest, fd).await.is_ok())
1079                {
1080                    FdType::Pipe
1081                } else {
1082                    fd_type
1083                };
1084                self.add_fd(guest, fd, call.flags(), fd_type).await?;
1085                if fd_type == FdType::Pipe {
1086                    self.maybe_set_nonblocking_fd(guest, fd);
1087                }
1088                // Classify the spelling the guest used FIRST. Several kinds are
1089                // defined by that spelling and MUST keep it: `/proc/self/...`,
1090                // `/proc/thread-self/...` and the mountinfo aliases all resolve
1091                // through `/proc/<pid>/fd/<fd>` to a numeric `/proc/<pid>/...`
1092                // path, which is a DIFFERENT (or absent) classification. So
1093                // resolution must never overwrite a spelling that already
1094                // classifies.
1095                //
1096                // Only when the spelling yields nothing do we ask the kernel
1097                // what the descriptor actually names. That is exactly the
1098                // AT_FDCWD/alias gap -- `chdir("/sys/module/kvm"); open("refcnt")`
1099                // classifies as nothing lexically -- and scoping it this way
1100                // makes the fallback MONOTONE: it can only add a classification
1101                // where there was none, never change one that already existed.
1102                let mut procfs = ProcfsFile::from_path(&observed_path).or_else(|| {
1103                    resolved_open_path(guest.pid().as_raw(), fd)
1104                        .filter(|resolved| resolved != &observed_path)
1105                        .and_then(|resolved| ProcfsFile::from_path(&resolved))
1106                });
1107                if procfs
1108                    .as_ref()
1109                    .is_some_and(ProcfsFile::needs_bound_thread_identity)
1110                {
1111                    // TODO-HUMAN-REVIEW(PR-964): Bind thread-self at open time,
1112                    // matching procfs inode resolution even if another thread or
1113                    // a forked process later reads the shared descriptor.
1114                    let tgid = guest.inject(syscalls::Getpid::new()).await? as i32;
1115                    let tid = guest.inject(syscalls::Gettid::new()).await? as i32;
1116                    let ppid = guest.inject(syscalls::Getppid::new()).await? as i32;
1117                    procfs
1118                        .as_mut()
1119                        .expect("thread identity request lost its procfs file")
1120                        .bind_thread_identity(tgid, tid, ppid);
1121                }
1122                if procfs
1123                    .as_ref()
1124                    .and_then(ProcfsFile::timer_slack_target)
1125                    .is_some()
1126                {
1127                    let target = procfs
1128                        .as_ref()
1129                        .and_then(ProcfsFile::timer_slack_target)
1130                        .expect("timer-slack target disappeared");
1131                    let stat = match guest.thread_state().with_detfd(fd, |detfd| detfd.stat())? {
1132                        Some(stat) => libc::stat::from(&stat),
1133                        None => self.inject_fstat(guest, fd).await?,
1134                    };
1135                    // Recorder opens a real proc inode, while Replayer reserves
1136                    // the recorded descriptor number with an anonymous eventfd.
1137                    // A real proc descriptor is the strongest open-time task
1138                    // incarnation witness. For a virtual replay descriptor,
1139                    // bind the live numeric proc path instead; every operation
1140                    // re-resolves that same path, so exit or TID reuse still
1141                    // changes the inode and returns ESRCH.
1142                    let identity = if stat.st_mode & libc::S_IFMT == libc::S_IFREG {
1143                        (stat.st_dev, stat.st_ino)
1144                    } else {
1145                        self.observe_timer_slack_identity(guest, target)
1146                            .await?
1147                            .unwrap_or((stat.st_dev, stat.st_ino))
1148                    };
1149                    procfs
1150                        .as_mut()
1151                        .expect("timer-slack classification disappeared")
1152                        .bind_timer_slack_identity(identity.0, identity.1);
1153                }
1154                guest.thread_state().with_detfd(fd, |detfd| {
1155                    detfd.set_path(&observed_path);
1156                    if let Some(procfs) = procfs.clone() {
1157                        detfd.set_procfs(procfs);
1158                    }
1159                })?;
1160                resource_release_all(guest).await;
1161                Ok(fd as i64)
1162            }
1163            // TODO: audit for error-nondeterminism:
1164            Err(e) => {
1165                resource_release_all(guest).await;
1166                Err(e.into())
1167            }
1168        }
1169    }
1170
1171    /// SYS_close system call.
1172    pub async fn handle_close<G: Guest<Self>>(
1173        &self,
1174        guest: &mut G,
1175        call: syscalls::Close,
1176    ) -> Result<i64, Error> {
1177        let fd = call.fd();
1178        let res = self.record_or_replay(guest, call).await;
1179        let fd_was_released = !matches!(res, Err(Errno::EBADF) | Err(Errno::ERESTARTSYS));
1180        if fd_was_released {
1181            if let Some(open_file_id) = guest.thread_state_mut().remove_fd(fd) {
1182                self.release_port_for_open_file(guest, open_file_id).await;
1183            }
1184            trace!("Closed {}", fd);
1185        }
1186        res.map_err(Error::from)
1187    }
1188
1189    // AUTONOMOUS-BOT-IMPLEMENTED
1190    // TODO-HUMAN-REVIEW(PR-838): Review close_range descriptor-table synchronization.
1191    /// Close a contiguous descriptor range and mirror successful closes in Detcore.
1192    ///
1193    /// The pinned Reverie revision exposes close_range as `Syscall::Other`. The
1194    /// common flags=0 operation cannot block and is deterministic for the
1195    /// process-local descriptor table. CLOSE_RANGE_UNSHARE and
1196    /// CLOSE_RANGE_CLOEXEC need separate shared-table modeling, so return ENOSYS
1197    /// for nonzero flags rather than letting strict execution silently diverge.
1198    pub async fn handle_close_range<G: Guest<Self>>(
1199        &self,
1200        guest: &mut G,
1201        call: Syscall,
1202    ) -> Result<i64, Error> {
1203        let Syscall::Other(_, args) = call else {
1204            unreachable!("close_range unexpectedly gained a typed variant")
1205        };
1206        let first = args.arg0 as u32;
1207        let last = args.arg1 as u32;
1208        let flags = args.arg2 as u32;
1209        if flags != 0 {
1210            return Err(Errno::ENOSYS.into());
1211        }
1212
1213        let result = self.record_or_replay(guest, call).await;
1214        if result.is_ok() {
1215            let released = guest.thread_state_mut().remove_fd_range(first, last);
1216            for open_file_id in released {
1217                self.release_port_for_open_file(guest, open_file_id).await;
1218            }
1219        }
1220        result.map_err(Error::from)
1221    }
1222
1223    // AUTONOMOUS-BOT-IMPLEMENTED
1224    // TODO-HUMAN-REVIEW(#2373)
1225    /// Advisory whole-file locks, forwarded to the kernel.
1226    ///
1227    /// This was previously an unconditional no-op success, justified by the
1228    /// claim that "an advisory whole-file lock is never contended within the
1229    /// serialized container". That is false: serializing guest threads stops
1230    /// them EXECUTING simultaneously, it does not stop their lock HOLD
1231    /// INTERVALS from overlapping. A holder that is descheduled -- because it
1232    /// blocked, forked, or simply used up its timeslice -- keeps holding while
1233    /// another process runs and observes the lock. Measured before this change,
1234    /// on both ptrace and DBI, two processes held the same `LOCK_EX`
1235    /// simultaneously while native correctly returned `EWOULDBLOCK`.
1236    ///
1237    /// A no-op is the wrong failure direction for a determinism tool. It is
1238    /// deterministically wrong, so double-run verification cannot see it, and it
1239    /// silently removes mutual exclusion from every guest that uses a lockfile.
1240    ///
1241    /// Forwarding is what `fcntl` already does for POSIX record locks, which is
1242    /// why those work. The guest's descriptor is a real host descriptor, so the
1243    /// kernel supplies the whole contract for free and consistently with itself:
1244    /// shared vs exclusive, `LOCK_NB`, upgrade/downgrade (which Linux performs
1245    /// non-atomically -- see below, this handler compensates), release on
1246    /// `LOCK_UN`, release when the last descriptor for the open file
1247    /// description is closed, and release on process exit.
1248    ///
1249    /// Determinism, scoped to what is actually true. When every contender is
1250    /// inside the container the outcome is a function of which guest holds the
1251    /// lock, and that is fixed by Detcore's deterministic schedule, so a given
1252    /// program and seed produce the same acquisition outcome every run.
1253    ///
1254    /// The scope is not decoration. Because this forwards to the kernel, a
1255    /// process OUTSIDE the container holding a lock on a guest-visible file
1256    /// does change the guest's result -- measured: with a host `flock -x`
1257    /// holder, a guest `LOCK_EX|LOCK_NB` returns `EWOULDBLOCK`, and acquires
1258    /// without one. That is a host-state leak, it is faithful to Linux, and it
1259    /// is the same leak `fcntl` record locks have always had here. Hermit
1260    /// already declines to make a mutating external filesystem deterministic,
1261    /// and lock state on a shared file is part of that state. Do not restate
1262    /// this as "no host state enters the decision": it does, and the previous
1263    /// bug in this very function came from writing down a determinism argument
1264    /// that was broader than the truth.
1265    ///
1266    /// Note that the no-op this replaced was not host-independent in any useful
1267    /// sense either -- it was host-independent by being wrong in all cases.
1268    ///
1269    /// # Why a blocking request is probed non-blockingly, and what that costs
1270    ///
1271    /// A guest thread parked inside a kernel `flock` is not visible to the
1272    /// deterministic scheduler as blocked, so nothing runs to release the lock
1273    /// and the whole container wedges -- measured: a four-way contention guest
1274    /// that completes natively hung indefinitely under a plain forwarding
1275    /// implementation. So a blocking operation is rewritten to `LOCK_NB` and,
1276    /// if it turns out to be contended, refused rather than hung.
1277    ///
1278    /// That rewrite is not free, and the cost is a *lock the guest already
1279    /// owns*. Linux converts an `flock` lock in place and the conversion is not
1280    /// atomic: `flock_lock_inode` deletes this open file description's existing
1281    /// lock **before** it scans for a conflict, so a contended `LOCK_SH` ->
1282    /// `LOCK_EX` conversion leaves the caller holding nothing and then reports
1283    /// `EWOULDBLOCK`. Natively the guest never observes that intermediate
1284    /// state, because a *blocking* request would sleep and eventually acquire.
1285    /// Under the rewrite it would: the guest asked to wait, got told "no", and
1286    /// silently lost the shared lock it was already relying on.
1287    ///
1288    /// So this handler restores the prior mode before refusing a *blocking*
1289    /// conversion, making the refusal side-effect-free. It deliberately does
1290    /// **not** restore when the guest itself passed `LOCK_NB`: there the drop is
1291    /// exactly what Linux does, and re-acquiring would be a divergence in the
1292    /// other direction. `DetFd::flock_mode` is what makes the two cases
1293    /// distinguishable -- it records the mode Detcore last saw the kernel grant
1294    /// for this open file description, so a first acquisition (nothing to lose)
1295    /// is not confused with a conversion (something to lose).
1296    ///
1297    /// That cache covers locks Detcore granted while it had sole knowledge of
1298    /// the open file description. State becomes permanently unknown when the
1299    /// descriptor is inherited across a process fork, discovered after tracing
1300    /// begins, or received through `SCM_RIGHTS`, because another process can
1301    /// change that shared kernel lock without updating this cache. A blocking
1302    /// conversion in unknown state is refused before the nonblocking probe, so
1303    /// the refusal cannot destroy a lock Detcore cannot restore.
1304    pub async fn handle_flock<G: Guest<Self>>(
1305        &self,
1306        guest: &mut G,
1307        call: syscalls::Flock,
1308    ) -> Result<i64, Error> {
1309        const LOCK_NB: i32 = libc::LOCK_NB;
1310        /// `LOCK_SH`/`LOCK_EX`/`LOCK_UN` with `LOCK_NB` (and any padding) masked off.
1311        const MODE_MASK: i32 = libc::LOCK_SH | libc::LOCK_EX | libc::LOCK_UN;
1312
1313        let (fd, operation) = (call.fd(), call.operation());
1314        let requested = operation & MODE_MASK;
1315        let caller_wants_nonblocking = operation & LOCK_NB != 0;
1316        let releasing = requested == libc::LOCK_UN;
1317        let valid_operation = operation & !(MODE_MASK | LOCK_NB) == 0
1318            && matches!(requested, libc::LOCK_SH | libc::LOCK_EX | libc::LOCK_UN);
1319        let dettid = guest.thread_state().dettid;
1320
1321        // Preserve kernel validation for malformed operations. In particular,
1322        // an unknown descriptor with an invalid mode must report EINVAL rather
1323        // than being mistaken for a valid blocking request and refused with
1324        // ENOLCK.
1325        if !valid_operation {
1326            return self
1327                .record_or_replay(guest, call)
1328                .await
1329                .map_err(Error::from);
1330        }
1331
1332        // The mode Detcore last saw the kernel grant this open file description.
1333        let known_held = guest
1334            .thread_state()
1335            .with_detfd(fd, |detfd| detfd.known_flock_mode())
1336            .unwrap_or(None);
1337
1338        // A nonblocking conversion is allowed to have Linux's documented
1339        // non-atomic side effect. A blocking conversion is not: if this open
1340        // file description was inherited or received and its prior mode is
1341        // unknown, probing could silently drop a lock we cannot restore.
1342        // Validate the descriptor without changing lock state, then refuse the
1343        // uncertain blocking operation before issuing flock at all.
1344        if !caller_wants_nonblocking && !releasing && known_held.is_none() {
1345            guest
1346                .inject_with_retry(Syscall::Fcntl(
1347                    syscalls::Fcntl::new().with_fd(fd).with_cmd(F_GETFD),
1348                ))
1349                .await?;
1350            error!(
1351                "[dtid {dettid}] blocking flock(fd={fd}, operation={operation:#x}) refused: \
1352                 this open file description existed before Detcore observed its lock state, \
1353                 so a nonblocking probe could destroy a lock that cannot be restored. Use \
1354                 LOCK_NB, or run without --strict to receive ENOLCK."
1355            );
1356            return self
1357                .refuse_unserviceable_operation(guest, Sysno::flock, Errno::ENOLCK)
1358                .await;
1359        }
1360        let held = known_held.flatten();
1361
1362        // LOCK_UN cannot block, so it is forwarded exactly as the guest wrote it.
1363        let probe_operation = if releasing {
1364            operation
1365        } else {
1366            operation | LOCK_NB
1367        };
1368        let result = self
1369            .record_or_replay(guest, call.with_operation(probe_operation))
1370            .await;
1371
1372        match result {
1373            Ok(value) => {
1374                let granted = if releasing { None } else { Some(requested) };
1375                let _ = guest
1376                    .thread_state()
1377                    .with_detfd(fd, |detfd| detfd.set_flock_mode(granted));
1378                trace!(
1379                    "flock(fd={}, operation={:#x}) served, open file now holds {:?}",
1380                    fd, operation, granted
1381                );
1382                Ok(value)
1383            }
1384            Err(Errno::EWOULDBLOCK) if caller_wants_nonblocking => {
1385                // Exactly what the guest asked for, including Linux's own
1386                // non-atomic conversion behavior: if this was a conversion, the
1387                // kernel really did drop the prior lock on the way to failing,
1388                // so the cache must forget it rather than claim a lock the
1389                // guest no longer holds.
1390                if held.is_some_and(|held| held != requested) {
1391                    let _ = guest
1392                        .thread_state()
1393                        .with_detfd(fd, |detfd| detfd.set_flock_mode(None));
1394                }
1395                trace!("flock(fd={}, operation={:#x}) would block", fd, operation);
1396                Err(Errno::EWOULDBLOCK.into())
1397            }
1398            Err(Errno::EWOULDBLOCK) => {
1399                // The guest asked to wait, and Detcore substituted a probe.
1400                // Undo the probe's collateral damage before refusing.
1401                if let Some(previous) = held.filter(|previous| *previous != requested) {
1402                    let restore = call.with_operation(previous | LOCK_NB);
1403                    match self.record_or_replay(guest, restore).await {
1404                        Ok(_) => {
1405                            warn!(
1406                                "[dtid {dettid}] contended blocking flock(fd={fd}, \
1407                                 operation={operation:#x}) refused; restored this open file's \
1408                                 prior {previous:#x} lock, which Linux's non-atomic conversion \
1409                                 had dropped. The guest holds exactly what it held before the \
1410                                 call."
1411                            );
1412                        }
1413                        Err(err) => {
1414                            let _ = guest
1415                                .thread_state()
1416                                .with_detfd(fd, |detfd| detfd.set_flock_mode(None));
1417                            error!(
1418                                "[dtid {dettid}] contended blocking flock(fd={fd}, \
1419                                 operation={operation:#x}) refused, AND this open file's prior \
1420                                 {previous:#x} lock could not be restored ({err}). Linux's \
1421                                 non-atomic conversion dropped it and something outside this \
1422                                 container took it in the interval. The guest has lost a lock \
1423                                 it held; treat any mutual exclusion it was protecting as \
1424                                 broken."
1425                            );
1426                        }
1427                    }
1428                }
1429                // Waiting faithfully needs a wait queue owned by the
1430                // deterministic scheduler, the way futexes are handled; until
1431                // that exists, refuse loudly. Returning success would recreate
1432                // the mutual-exclusion bug this handler was written to fix, and
1433                // blocking in the kernel would deadlock the container.
1434                error!(
1435                    "[dtid {dettid}] blocking flock(fd={fd}, operation={operation:#x}) is \
1436                     contended, and Detcore cannot yet park a thread on a file lock \
1437                     deterministically. Refusing rather than granting a lock another guest \
1438                     holds. Use LOCK_NB, or run without --strict to receive ENOLCK."
1439                );
1440                self.refuse_unserviceable_operation(guest, Sysno::flock, Errno::ENOLCK)
1441                    .await
1442            }
1443            Err(err) => {
1444                trace!(
1445                    "flock(fd={}, operation={:#x}) refused: {}",
1446                    fd, operation, err
1447                );
1448                Err(err.into())
1449            }
1450        }
1451    }
1452
1453    async fn snapshot_procfs<G: Guest<Self>>(
1454        &self,
1455        guest: &mut G,
1456        call: syscalls::Read,
1457    ) -> Result<Vec<u8>, Error> {
1458        const MAX_SNAPSHOT_BYTES: usize = 16 * 1024 * 1024;
1459
1460        // A backend-owned read may have advanced the kernel cursor without
1461        // passing through Detcore's logical procfs cursor (KVM does this for
1462        // worker-shared descriptors). Rewind before taking the initial snapshot
1463        // so a later intercepted pread cannot snapshot from EOF.
1464        //
1465        // AUTONOMOUS-BOT-IMPLEMENTED
1466        // TODO-HUMAN-REVIEW(#1903): ESPIPE is not a failure here. Several procfs
1467        // files are legitimately non-seekable -- `/proc/net/*` single-release
1468        // seq_files return ESPIPE from `llseek` on the host, verified natively:
1469        // `lseek(fd, 0, SEEK_SET)` on `/proc/net/sockstat` gives ESPIPE while the
1470        // subsequent `read(2)` returns data. Propagating that ESPIPE made the
1471        // GUEST's `read` fail on a file Linux reads fine (`cat
1472        // /proc/net/sockstat` -> "Illegal seek"), which is a deviation from Linux
1473        // semantics, not a determinism requirement: the rewind is an internal
1474        // correction Detcore performs for its own benefit and the guest never
1475        // asked for it. A non-seekable fd also cannot have been advanced behind
1476        // our back by a seek, and a freshly opened one is already at offset 0, so
1477        // skipping the rewind loses nothing the rewind was protecting.
1478        match guest
1479            .inject_with_retry(Syscall::Lseek(
1480                syscalls::Lseek::new()
1481                    .with_fd(call.fd())
1482                    .with_offset(0)
1483                    .with_whence(Whence::SEEK_SET),
1484            ))
1485            .await
1486        {
1487            Ok(_) => {}
1488            Err(Errno::ESPIPE) => {}
1489            Err(err) => return Err(err.into()),
1490        }
1491
1492        let remote_buf = call.buf().ok_or(Errno::EFAULT)?;
1493        let mut contents = Vec::new();
1494        loop {
1495            let bytes_read = self.record_or_replay(guest, call).await? as usize;
1496            if bytes_read == 0 {
1497                return Ok(contents);
1498            }
1499            if contents.len() + bytes_read > MAX_SNAPSHOT_BYTES {
1500                return Err(Errno::EFBIG.into());
1501            }
1502
1503            let mut chunk = vec![0; bytes_read];
1504            guest.memory().read_exact(remote_buf, &mut chunk)?;
1505            contents.extend_from_slice(&chunk);
1506        }
1507    }
1508
1509    async fn initialize_procfs_snapshot<G: Guest<Self>>(
1510        &self,
1511        guest: &mut G,
1512        call: syscalls::Read,
1513    ) -> Result<(), Error> {
1514        let contents = self.snapshot_procfs(guest, call).await?;
1515        let virtual_uptime_seconds = self.calculate_procfs_uptime(guest).await?;
1516        // Only `/proc/stat` renders btime. Computing it for every snapshot let
1517        // an unrepresentable boot instant refuse unrelated files such as
1518        // `/proc/uptime`.
1519        let needs_boot_time = guest
1520            .thread_state()
1521            .with_detfd(call.fd(), |detfd| detfd.procfs_needs_boot_time())?;
1522        let virtual_boot_time_seconds = needs_boot_time
1523            .then(|| self.calculate_procfs_boot_time())
1524            .transpose()?;
1525        let virtual_realtime_seconds = i64::try_from(thread_observe_time(guest).await.as_secs())
1526            .map_err(|_| Errno::EOVERFLOW)?;
1527        // TODO-HUMAN-REVIEW(PR-863): Use configured guest memory for meminfo.
1528        let virtual_memory_kb = guest.config().memory / 1024;
1529        // TODO-HUMAN-REVIEW(PR-723): Review injected identity snapshot reads.
1530        let virtual_pid = guest.inject(syscalls::Getpid::new()).await? as i32;
1531        let virtual_ppid = guest.inject(syscalls::Getppid::new()).await? as i32;
1532        let virtual_pty_count = guest.thread_state().count_open_files_at_paths(&[
1533            std::path::Path::new("/dev/ptmx"),
1534            std::path::Path::new("/dev/pts/ptmx"),
1535        ]);
1536        let target_fd = guest
1537            .thread_state()
1538            .with_detfd(call.fd(), |detfd| detfd.procfs_target_fd())?;
1539        let fdinfo_identity = if let Some(target_fd) = target_fd {
1540            let (cached_stat, logical_flags, open_file_id, fd_type, inode_override) =
1541                guest.thread_state().with_detfd(target_fd, |detfd| {
1542                    (
1543                        detfd.stat(),
1544                        detfd.status_flags(),
1545                        detfd.open_file_id(),
1546                        detfd.ty(),
1547                        deterministic_stdio_inode_for_resource(target_fd, detfd.resource()),
1548                    )
1549                })?;
1550            let raw_inode = match cached_stat {
1551                Some(stat) => stat.inode,
1552                None => {
1553                    let stat = self.inject_fstat(guest, target_fd).await?;
1554                    stat.st_ino
1555                }
1556            };
1557            let virtual_inode = match inode_override {
1558                Some(inode) => inode,
1559                None => determinize_inode(guest, raw_inode).await.0,
1560            };
1561            let raw_mount_id =
1562                detcore_model::procfs::parse_fdinfo_mount_id(&contents).ok_or_else(|| {
1563                    Error::Tool(anyhow::anyhow!(
1564                        "kernel returned malformed /proc/*/fdinfo without one numeric mnt_id"
1565                    ))
1566                })?;
1567            // CLI container and recording paths provide the exact namespace's
1568            // row order. Replay intentionally retains the recording-time raw
1569            // IDs because ReadV2 supplies recording-time fdinfo bytes.
1570            let has_configured_mount_ids = guest.config().mountinfo_mount_ids_captured;
1571            let virtual_mount_id = if raw_mount_id == 0 {
1572                // Linux uses zero for anonymous objects such as memfd. Key the
1573                // equivalence on the observed value, not our descriptor type.
1574                0
1575            } else if !has_configured_mount_ids {
1576                // Public non-container callers have no pre-captured provenance.
1577                // Mount/unshare/setns are refused once Detcore starts, so a
1578                // tracer-side snapshot of this task's namespace is immutable.
1579                let mountinfo_path = format!("/proc/{}/mountinfo", guest.pid().as_raw());
1580                let mountinfo_contents = std::fs::read(&mountinfo_path).map_err(|error| {
1581                    Error::Tool(anyhow::anyhow!(
1582                        "failed to read {mountinfo_path} while validating fdinfo mnt_id: {error}"
1583                    ))
1584                })?;
1585                let mountinfo_rows =
1586                    crate::procfs::parse_mountinfo(&mountinfo_contents).ok_or_else(|| {
1587                        Error::Tool(anyhow::anyhow!(
1588                            "kernel returned malformed {mountinfo_path} while validating fdinfo mnt_id"
1589                        ))
1590                    })?;
1591                let snapshot = MountInfoSnapshot::new(
1592                    mountinfo_rows,
1593                    &[],
1594                    false,
1595                    BTreeMap::new(),
1596                    BTreeMap::new(),
1597                )
1598                .ok_or_else(|| {
1599                    Error::Tool(anyhow::anyhow!(
1600                        "{mountinfo_path} failed strict identity validation for fdinfo"
1601                    ))
1602                })?;
1603                determinize_mount_id(guest, raw_mount_id, Some(snapshot.raw_mount_id_order()))
1604                    .await
1605                    .ok_or_else(|| {
1606                        Error::Tool(anyhow::anyhow!(
1607                            "mountinfo mount-ID order changed after the identity snapshot while resolving fdinfo mnt_id {raw_mount_id} for {fd_type:?} from {mountinfo_path}"
1608                        ))
1609                    })?
1610            } else {
1611                determinize_mount_id(guest, raw_mount_id, None)
1612                    .await
1613                    .ok_or_else(|| {
1614                        Error::Tool(anyhow::anyhow!(
1615                            "recorded mount identity provenance was invalid while resolving fdinfo mnt_id {raw_mount_id} for {fd_type:?}"
1616                        ))
1617                    })?
1618            };
1619            Some((
1620                // Determinized immediately above (stdio-special or
1621                // `determinize_inode`); lowered to an integer only here, at the
1622                // point it is rendered into guest-visible fdinfo text.
1623                virtual_inode.as_raw(),
1624                logical_flags,
1625                open_file_id.deterministic_socket_cookie(),
1626                virtual_mount_id,
1627            ))
1628        } else {
1629            None
1630        };
1631        let needs_random_uuid = guest
1632            .thread_state()
1633            .with_detfd(call.fd(), |detfd| detfd.procfs_needs_random_uuid())?;
1634        // AUTONOMOUS-BOT-IMPLEMENTED
1635        // TODO-HUMAN-REVIEW(PR-955): Review deterministic kernel UUID generation.
1636        let random_uuid =
1637            needs_random_uuid.then(|| guest.thread_state_mut().thread_prng().random::<[u8; 16]>());
1638        // ⚠️ DETERMINIZE HERE, IN THE CALLER, THROUGH THE SAME POOLS `stat` USES.
1639        //
1640        // The sanitizers in `crate::procfs` are pure functions of content and
1641        // hold no guest handle, so they cannot reach `InodePool`/`DevicePool`.
1642        // Minting an identity down there would make the maps column stable and
1643        // STILL DISAGREE with `stat` -- deterministic, reproducible and wrong.
1644        // This mirrors how `fdinfo_identity` is built a few lines above:
1645        // determinize with `determinize_inode`/`determinize_device`, then hand
1646        // the finished values down purely to be rendered.
1647        //
1648        // BOTH COLUMNS, not just the inode. `determinize_stat` sanitizes
1649        // `st_dev` as well, so rewriting only the inode would leave the device
1650        // disagreeing -- the same defect with the reported symptom removed.
1651        let needs_mapping_identities = guest
1652            .thread_state()
1653            .with_detfd(call.fd(), |detfd| detfd.procfs_needs_mapping_identities())?;
1654        let mut mapping_identities: BTreeMap<(u64, u64), (u64, u64)> = BTreeMap::new();
1655        if needs_mapping_identities {
1656            // A mapping backed by stdio must report the SAME inode fdinfo
1657            // reports for that fd, which is the fixed `deterministic_stdio_inode`
1658            // value rather than a pooled one. Matching is by raw inode, read
1659            // from the cached stat only: injecting an fstat here would add
1660            // syscalls to every maps read and perturb the very traces this
1661            // change is meant to keep consistent.
1662            let mut stdio_by_raw_inode: BTreeMap<u64, DetInode> = BTreeMap::new();
1663            for fd in libc::STDIN_FILENO..=libc::STDERR_FILENO {
1664                let cached = guest
1665                    .thread_state()
1666                    .with_detfd(fd, |detfd| {
1667                        let inode = deterministic_stdio_inode_for_resource(fd, detfd.resource())?;
1668                        detfd.stat().map(|stat| (stat.inode, inode))
1669                    })
1670                    .ok()
1671                    .flatten();
1672                if let Some((raw, det)) = cached {
1673                    stdio_by_raw_inode.insert(raw, det);
1674                }
1675            }
1676            let raw_pairs: BTreeSet<(u64, u64)> = String::from_utf8_lossy(&contents)
1677                .lines()
1678                .filter_map(crate::procfs::mapping_header_identity)
1679                .collect();
1680            for (raw_dev, raw_inode) in raw_pairs {
1681                let det_inode = match stdio_by_raw_inode.get(&raw_inode) {
1682                    Some(inode) => *inode,
1683                    None => determinize_inode(guest, raw_inode).await.0,
1684                };
1685                let det_dev = determinize_device(guest, raw_dev).await;
1686                mapping_identities.insert((raw_dev, raw_inode), (det_dev, det_inode.as_raw()));
1687            }
1688        }
1689        let mountinfo = if guest
1690            .thread_state()
1691            .with_detfd(call.fd(), |detfd| detfd.procfs_needs_mountinfo_identities())?
1692        {
1693            let mut rows = crate::procfs::parse_mountinfo(&contents).ok_or_else(|| {
1694                Error::Tool(anyhow::anyhow!(
1695                    "kernel returned malformed /proc/*/mountinfo"
1696                ))
1697            })?;
1698            let mut device_rewrites = BTreeMap::new();
1699            for &(mountinfo_device, metadata_device) in &guest.config().mountinfo_device_rewrites {
1700                if device_rewrites
1701                    .insert(mountinfo_device, metadata_device)
1702                    .is_some()
1703                {
1704                    return Err(Error::Tool(anyhow::anyhow!(
1705                        "duplicate proven mountinfo device rewrite for raw device {mountinfo_device}"
1706                    )));
1707                }
1708            }
1709            for row in &mut rows {
1710                if let Some(metadata_device) = device_rewrites.get(&row.raw_device) {
1711                    row.raw_device = *metadata_device;
1712                }
1713            }
1714            let mut raw_devices = Vec::new();
1715            let mut seen_devices = BTreeSet::new();
1716            for row in &rows {
1717                if seen_devices.insert(row.raw_device) {
1718                    raw_devices.push(row.raw_device);
1719                }
1720            }
1721            let mut devices = BTreeMap::new();
1722            let virtualize_metadata = guest.config().virtualize_metadata;
1723            if virtualize_metadata {
1724                // Intentionally use snapshot row order to pre-populate the
1725                // same run-global DevicePool used by stat/statx. This makes
1726                // every later observation of a device agree within the run.
1727                // It does not promise that unlike host filesystem layouts
1728                // expose the same device equivalence classes or order.
1729                for raw in raw_devices {
1730                    devices.insert(raw, determinize_device(guest, raw).await);
1731                }
1732            }
1733            let mut root_rewrites = BTreeMap::new();
1734            for rewrite in &guest.config().mountinfo_root_rewrites {
1735                if root_rewrites
1736                    .insert(rewrite.raw_mount_id, rewrite.clone())
1737                    .is_some()
1738                {
1739                    return Err(Error::Tool(anyhow::anyhow!(
1740                        "duplicate proven mountinfo root rewrite for mount ID {}",
1741                        rewrite.raw_mount_id
1742                    )));
1743                }
1744            }
1745            let snapshot = MountInfoSnapshot::new(
1746                rows,
1747                if guest.config().mountinfo_mount_ids_captured {
1748                    &guest.config().mountinfo_mount_ids
1749                } else {
1750                    &[]
1751                },
1752                virtualize_metadata,
1753                devices,
1754                root_rewrites,
1755            )
1756            .ok_or_else(|| {
1757                Error::Tool(anyhow::anyhow!(
1758                    "mountinfo snapshot failed strict identity validation"
1759                ))
1760            })?;
1761            if !validate_mountinfo_identity_order(guest, snapshot.raw_mount_id_order()).await {
1762                return Err(Error::Tool(anyhow::anyhow!(
1763                    "mountinfo mount-ID order changed after the run-global identity snapshot"
1764                )));
1765            }
1766            Some(snapshot)
1767        } else {
1768            None
1769        };
1770        guest.thread_state().with_detfd(call.fd(), |detfd| {
1771            detfd.initialize_procfs(
1772                contents.clone(),
1773                ProcfsSnapshotContext {
1774                    mapping_identities: mapping_identities.clone(),
1775                    mountinfo: mountinfo.clone(),
1776                    virtual_uptime_seconds,
1777                    virtual_boot_time_seconds,
1778                    virtual_realtime_seconds,
1779                    virtual_memory_kb,
1780                    virtual_pid,
1781                    virtual_ppid,
1782                    virtual_pty_count,
1783                    fdinfo_identity,
1784                    random_uuid,
1785                },
1786            );
1787        })?;
1788        Ok(())
1789    }
1790
1791    /// SYS_read system call (MAYHANG).
1792    pub async fn handle_read<G: Guest<Self>>(
1793        &self,
1794        guest: &mut G,
1795        call: syscalls::Read,
1796    ) -> Result<i64, Error> {
1797        if self.timer_slack_binding(guest, call.fd())?.is_some() {
1798            return self
1799                .read_timer_slack(guest, call.fd(), call.buf(), call.len())
1800                .await;
1801        }
1802
1803        if call.len() == 0 {
1804            if let Ok(Some(status_flags)) = guest.thread_state().with_detfd(call.fd(), |detfd| {
1805                (detfd.ty() == FdType::Rng).then(|| detfd.status_flags())
1806            }) {
1807                // Replay reserves random-device fd slots with eventfds. A
1808                // physical zero-length read of that placeholder returns EINVAL
1809                // even though the logical random-device read must return zero.
1810                require_random_device_read_access(status_flags)?;
1811                let policy = if guest.config().backend_is_kvm {
1812                    crate::iovecs::UserAddressPolicy::Kvm
1813                } else {
1814                    crate::iovecs::UserAddressPolicy::Native
1815                };
1816                // vfs_read still checks access_ok for a zero-length buffer:
1817                // NULL is valid, but an address beyond TASK_SIZE is EFAULT.
1818                policy.validate(&[crate::iovecs::ImportedIovec {
1819                    base: call.buf().map_or(0, |address| address.as_raw()),
1820                    len: 0,
1821                }])?;
1822                return Ok(0);
1823            }
1824            // A zero-count read transfers nothing on a file or stream, but on a
1825            // datagram or SEQPACKET socket it consumes a pending message. Its
1826            // result goes through record/replay: a replayed descriptor may be
1827            // a placeholder whose own zero-count read fails differently.
1828            // AUTONOMOUS-BOT-IMPLEMENTED
1829            // TODO-HUMAN-REVIEW(PR-3601)
1830            return self
1831                .record_or_replay_preserving_tool_errors(guest, call)
1832                .await;
1833        }
1834
1835        let needs_procfs_snapshot = guest
1836            .thread_state()
1837            .with_detfd(call.fd(), |detfd| detfd.procfs_needs_snapshot())?;
1838        if needs_procfs_snapshot {
1839            self.initialize_procfs_snapshot(guest, call).await?;
1840        }
1841
1842        let procfs_bytes = guest
1843            .thread_state()
1844            .with_detfd(call.fd(), |detfd| detfd.take_procfs(call.len()))?;
1845        if let Some(bytes) = procfs_bytes {
1846            let remote_buf = call.buf().ok_or(Errno::EFAULT)?;
1847            guest.memory().write_exact(remote_buf, &bytes)?;
1848            return Ok(bytes.len() as i64);
1849        }
1850
1851        let (fd_type, physically_nonblocking, logically_nonblocking, resource, random_device) =
1852            guest.thread_state_mut().with_detfd(call.fd(), |detfd| {
1853                (
1854                    detfd.ty(),
1855                    detfd.physically_nonblocking(),
1856                    detfd.is_nonblocking(),
1857                    detfd.resource(),
1858                    detfd.clone(),
1859                )
1860            })?;
1861
1862        if let Some(resource) = resource {
1863            let mut request = guest.thread_state().mk_request(resource, Permission::R);
1864            if should_tag_sabre_internal_pipe_io(
1865                guest.config().discover_live_file_metadata,
1866                fd_type,
1867                physically_nonblocking,
1868                logically_nonblocking,
1869            ) {
1870                request.fyi(SABRE_INTERNAL_PIPE_IO_FYI);
1871            }
1872            resource_request(guest, request).await;
1873        }
1874
1875        let res = match fd_type {
1876            FdType::Rng => {
1877                trace!("Read call RNG fd {}, simulating...", call.fd());
1878                let status_flags = random_device.status_flags();
1879                random_device
1880                    .with_random_device_stream(|offset| {
1881                        require_random_device_read_access(status_flags)?;
1882                        let remote_buf = call.buf().ok_or(Errno::EFAULT)?;
1883                        self.fill_random_device_bytes(guest, remote_buf, call.len(), offset)
1884                    })
1885                    .map(|n| n as i64)
1886            }
1887            FdType::Regular => {
1888                if guest.config().deterministic_io {
1889                    self.deterministic_read(guest, call).await
1890                } else {
1891                    Ok(self.record_or_replay(guest, call).await?)
1892                }
1893            }
1894            FdType::Signalfd | FdType::Eventfd | FdType::Timerfd | FdType::Inotify => {
1895                trace!(
1896                    "Possibly blocking read call on notification fd {}, type {:?}",
1897                    call.fd(),
1898                    fd_type
1899                );
1900                self.execute_nonblockable_fd_syscall(guest, call).await
1901            }
1902            FdType::Memfd | FdType::Pidfd | FdType::Userfaultfd | FdType::Epoll => {
1903                trace!("Read call on unusual fd {}, type {:?}", call.fd(), fd_type);
1904                Ok(self.record_or_replay(guest, call).await?)
1905            }
1906
1907            FdType::Socket | FdType::Pipe => {
1908                trace!(
1909                    "Possibly blocking read call on {:?} fd {}",
1910                    fd_type,
1911                    call.fd()
1912                );
1913                self.execute_nonblockable_fd_syscall(guest, call).await
1914            }
1915        };
1916        resource_release_all(guest).await;
1917        res
1918    }
1919
1920    /// SYS_pread64 system call.
1921    pub async fn handle_pread64<G: Guest<Self>>(
1922        &self,
1923        guest: &mut G,
1924        call: syscalls::Pread64,
1925    ) -> Result<i64, Error> {
1926        if self.timer_slack_binding(guest, call.fd())?.is_some() {
1927            return self
1928                .pread_timer_slack(guest, call.fd(), call.buf(), call.len(), call.offset())
1929                .await;
1930        }
1931
1932        if call.len() == 0 {
1933            // As for read, a zero-count pread's result goes through
1934            // record/replay.
1935            // AUTONOMOUS-BOT-IMPLEMENTED
1936            // TODO-HUMAN-REVIEW(PR-3601)
1937            return self
1938                .record_or_replay_preserving_tool_errors(guest, call)
1939                .await;
1940        }
1941
1942        let offset = usize::try_from(call.offset()).map_err(|_| Errno::EINVAL)?;
1943        let needs_procfs_snapshot = guest
1944            .thread_state()
1945            .with_detfd(call.fd(), |detfd| detfd.procfs_needs_snapshot())?;
1946        if needs_procfs_snapshot {
1947            let read = syscalls::Read::new()
1948                .with_fd(call.fd())
1949                .with_buf(call.buf())
1950                .with_len(call.len());
1951            self.initialize_procfs_snapshot(guest, read).await?;
1952        }
1953
1954        let procfs_bytes = guest
1955            .thread_state()
1956            .with_detfd(call.fd(), |detfd| detfd.take_procfs_at(offset, call.len()))?;
1957        if let Some(bytes) = procfs_bytes {
1958            let remote_buf = call.buf().ok_or(Errno::EFAULT)?;
1959            guest.memory().write_exact(remote_buf, &bytes)?;
1960            return Ok(bytes.len() as i64);
1961        }
1962
1963        let (fd_type, resource) = guest
1964            .thread_state_mut()
1965            .with_detfd(call.fd(), |detfd| (detfd.ty(), detfd.resource()))?;
1966
1967        if let Some(resource) = resource {
1968            let request = guest.thread_state().mk_request(resource, Permission::R);
1969            resource_request(guest, request).await;
1970        }
1971
1972        let res = match fd_type {
1973            FdType::Rng => (|| -> Result<i64, Error> {
1974                trace!("Pread64 call RNG fd {}, simulating...", call.fd());
1975                let remote_buf = call.buf().ok_or(Errno::EFAULT)?;
1976                let n =
1977                    self.fill_random_device_bytes(guest, remote_buf, call.len(), offset as u64)?;
1978                Ok(n as i64)
1979            })(),
1980            FdType::Regular if guest.config().deterministic_io => {
1981                self.deterministic_pread64(guest, call).await
1982            }
1983            _ => match self.record_or_replay(guest, call).await {
1984                Ok(value) => Ok(value),
1985                Err(error) => Err(error.into()),
1986            },
1987        };
1988
1989        resource_release_all(guest).await;
1990        res
1991    }
1992
1993    /// SYS_lseek system call.
1994    pub async fn handle_lseek<G: Guest<Self>>(
1995        &self,
1996        guest: &mut G,
1997        call: syscalls::Lseek,
1998    ) -> Result<i64, Error> {
1999        let timer_slack_binding = self.timer_slack_binding(guest, call.fd())?;
2000        let (fd_type, status_flags, procfs_position, resource) =
2001            guest.thread_state().with_detfd(call.fd(), |detfd| {
2002                (
2003                    detfd.ty(),
2004                    detfd.status_flags(),
2005                    detfd.procfs_position(),
2006                    detfd.resource(),
2007                )
2008            })?;
2009        if fd_type == FdType::Rng {
2010            return random_device_lseek_result(status_flags, call.whence()).map_err(Into::into);
2011        }
2012        if is_inherited_container_output(resource) {
2013            return Err(Errno::ESPIPE.into());
2014        }
2015        if timer_slack_binding.is_some() && status_flags & libc::O_PATH != 0 {
2016            return Err(Errno::EBADF.into());
2017        }
2018        let Some((current, snapshot_len)) = procfs_position else {
2019            // AUTONOMOUS-BOT-IMPLEMENTED
2020            // TODO-HUMAN-REVIEW(#1044): Regular-file lseek must
2021            // flow through record_or_replay, exactly as handle_read does for
2022            // FdType::Regular. Injecting the seek live here is correct for run
2023            // and --verify (the inner NoopTool just re-injects), but under
2024            // record/replay the replay descriptor is a virtual placeholder
2025            // whose kernel position never advances (reads are served from the
2026            // log, not the file). A live SEEK_CUR then returned Ok(0) on replay
2027            // versus the recorded offset (e.g. glibc's __tzfile_read rewinds
2028            // /etc/localtime with lseek(fd, -N, SEEK_CUR)), diverging the
2029            // guest's control flow and desynchronizing the event stream. Routing
2030            // through record_or_replay records the offset once and substitutes
2031            // the recorded value on replay, keeping the two runs identical.
2032            return Ok(self.record_or_replay(guest, call).await?);
2033        };
2034
2035        if let Some(binding) = timer_slack_binding {
2036            // Linux exposes this file through seq_lseek, which accepts only
2037            // SEEK_SET and SEEK_CUR. Keep that position entirely in the
2038            // virtual open-file description even before the first read.
2039            let requested = i128::from(call.offset());
2040            let new_offset = match call.whence() {
2041                Whence::SEEK_SET => requested,
2042                Whence::SEEK_CUR => current as i128 + requested,
2043                _ => return Err(Errno::EINVAL.into()),
2044            };
2045            let new_offset = usize::try_from(new_offset).map_err(|_| Errno::EINVAL)?;
2046            let result = i64::try_from(new_offset).map_err(|_| Errno::EOVERFLOW)?;
2047            // seq_lseek does not call the show callback for a no-op or a reset
2048            // to zero. Only a traversal to another positive position observes
2049            // the target task and therefore performs lifetime/access checks.
2050            if new_offset != 0 && new_offset != current {
2051                self.require_current_timer_slack_target(guest, binding)
2052                    .await?;
2053            }
2054            guest
2055                .thread_state()
2056                .with_detfd(call.fd(), |detfd| detfd.set_procfs_offset(new_offset))?;
2057            return Ok(result);
2058        }
2059
2060        let Some(snapshot_len) = snapshot_len else {
2061            let offset = guest.inject(Syscall::from(call)).await?;
2062            let offset = usize::try_from(offset).map_err(|_| Errno::EINVAL)?;
2063            guest
2064                .thread_state()
2065                .with_detfd(call.fd(), |detfd| detfd.set_procfs_offset(offset))?;
2066            return Ok(offset as i64);
2067        };
2068
2069        let requested = i128::from(call.offset());
2070        let new_offset = match call.whence() {
2071            Whence::SEEK_SET => requested,
2072            Whence::SEEK_CUR => current as i128 + requested,
2073            Whence::SEEK_END => snapshot_len as i128 + requested,
2074            Whence::SEEK_DATA => {
2075                if requested < 0 || requested >= snapshot_len as i128 {
2076                    return Err(Errno::ENXIO.into());
2077                }
2078                requested
2079            }
2080            Whence::SEEK_HOLE => {
2081                if requested < 0 || requested >= snapshot_len as i128 {
2082                    return Err(Errno::ENXIO.into());
2083                }
2084                snapshot_len as i128
2085            }
2086            _ => return Err(Errno::EINVAL.into()),
2087        };
2088        let new_offset = usize::try_from(new_offset).map_err(|_| Errno::EINVAL)?;
2089        let result = i64::try_from(new_offset).map_err(|_| Errno::EOVERFLOW)?;
2090        guest
2091            .thread_state()
2092            .with_detfd(call.fd(), |detfd| detfd.set_procfs_offset(new_offset))?;
2093        Ok(result)
2094    }
2095
2096    /// Helper for performing a deterministic read that retries until it gets all its
2097    /// bytes.
2098    // AUTONOMOUS-BOT-IMPLEMENTED
2099    // TODO-HUMAN-REVIEW(#689): Confirm partial reads take precedence over later errors.
2100    async fn deterministic_read<G: Guest<Self>>(
2101        &self,
2102        guest: &mut G,
2103        mut call: syscalls::Read,
2104    ) -> Result<i64, Error> {
2105        let mut total_read_bytes = 0;
2106        let mut remaining_buf = call.len();
2107
2108        trace!(
2109            "[detcore/det_io]: Requested read buffer size: {:?}",
2110            remaining_buf
2111        );
2112
2113        loop {
2114            match guest.inject_with_retry(call).await {
2115                Ok(res) => {
2116                    remaining_buf -= res as usize;
2117                    total_read_bytes += res;
2118
2119                    trace!(
2120                        "[detcore/det_io]: Remaining read buffer size: {:?}",
2121                        remaining_buf
2122                    );
2123
2124                    if res == 0 || remaining_buf == 0 {
2125                        break Ok(total_read_bytes);
2126                    }
2127
2128                    // Buf is guaranteed to exist as we already issued a syscall.
2129                    let old_ptr = call.buf().unwrap().as_raw();
2130                    call = call
2131                        .with_len(remaining_buf)
2132                        .with_buf(AddrMut::<u8>::from_raw(old_ptr + res as usize));
2133                }
2134                Err(error) if total_read_bytes > 0 => {
2135                    trace!("[detcore/det_io]: returning {total_read_bytes} bytes before {error}");
2136                    break Ok(total_read_bytes);
2137                }
2138                Err(error) => break Err(error.into()),
2139            }
2140        }
2141    }
2142
2143    /// Perform a positional read until the requested buffer is full or EOF is reached.
2144    async fn deterministic_pread64<G: Guest<Self>>(
2145        &self,
2146        guest: &mut G,
2147        mut call: syscalls::Pread64,
2148    ) -> Result<i64, Error> {
2149        let mut total_read_bytes = 0;
2150        let mut remaining_buf = call.len();
2151
2152        trace!(
2153            "[detcore/det_io]: Requested pread64 buffer size: {:?}",
2154            remaining_buf
2155        );
2156
2157        loop {
2158            match guest.inject_with_retry(call).await {
2159                Ok(res) => {
2160                    remaining_buf -= res as usize;
2161                    total_read_bytes += res;
2162
2163                    trace!(
2164                        "[detcore/det_io]: Remaining pread64 buffer size: {:?}",
2165                        remaining_buf
2166                    );
2167
2168                    if res == 0 || remaining_buf == 0 {
2169                        break Ok(total_read_bytes);
2170                    }
2171
2172                    let old_ptr = call
2173                        .buf()
2174                        .expect("successful pread64 requires a valid guest buffer")
2175                        .as_raw();
2176                    let offset = call.offset().checked_add(res).ok_or(Errno::EOVERFLOW)?;
2177                    call = call
2178                        .with_len(remaining_buf)
2179                        .with_buf(AddrMut::<u8>::from_raw(old_ptr + res as usize))
2180                        .with_offset(offset);
2181                }
2182                Err(error) => break Err(error.into()),
2183            }
2184        }
2185    }
2186
2187    // AUTONOMOUS-BOT-IMPLEMENTED
2188    // TODO-HUMAN-REVIEW(PR-838): Review regular-file sendfile mediation.
2189    /// Copy data between tracked regular files or memfds.
2190    ///
2191    /// The kernel advances the input offset (or the explicit offset pointer) and
2192    /// destination offset atomically with the copy. Detcore serializes destination
2193    /// writes while the strict scheduler orders the stable input read, and routes
2194    /// the syscall through record/replay so that the result and offset update stay
2195    /// ordered with other file operations. Socket and pipe destinations can block
2196    /// and need the nonblocking scheduler path; return ENOSYS for those endpoint
2197    /// types so libc/application fallbacks use Detcore's existing read/write
2198    /// handlers instead.
2199    pub async fn handle_sendfile<G: Guest<Self>>(
2200        &self,
2201        guest: &mut G,
2202        call: syscalls::Sendfile,
2203    ) -> Result<i64, Error> {
2204        let in_type = guest
2205            .thread_state()
2206            .with_detfd(call.in_fd(), |detfd| detfd.ty())?;
2207        let (out_type, out_resource, out_inode) =
2208            guest.thread_state().with_detfd(call.out_fd(), |detfd| {
2209                (
2210                    detfd.ty(),
2211                    detfd.resource(),
2212                    detfd.stat().map(|stat| stat.inode),
2213                )
2214            })?;
2215
2216        // AUTONOMOUS-BOT-IMPLEMENTED
2217        // TODO-HUMAN-REVIEW(PR-973): Refuse sendfile from a procfs input so it
2218        // cannot bypass the deterministic ProcfsFile snapshot. A procfs fd is
2219        // classified `FdType::Regular`, so it would otherwise pass the type
2220        // guard below and copy live kernel bytes straight to the output fd,
2221        // reintroducing the nondeterminism the mediated read()/write() path
2222        // sanitizes. Failing closed with ENOSYS makes callers fall back to
2223        // that mediated path (glibc's sendfile does exactly this).
2224        let in_is_procfs = guest
2225            .thread_state()
2226            .with_detfd(call.in_fd(), |detfd| detfd.procfs_position().is_some())?;
2227        if in_is_procfs {
2228            return Err(Errno::ENOSYS.into());
2229        }
2230
2231        if !matches!(in_type, FdType::Regular | FdType::Memfd)
2232            || !matches!(out_type, FdType::Regular | FdType::Memfd)
2233        {
2234            return Err(Errno::ENOSYS.into());
2235        }
2236
2237        let dettid = guest.thread_state().dettid;
2238        let mut resources = Resources::new(dettid);
2239        // `out_inode` is the fd's cached HOST inode, so it must be
2240        // determinized before naming a resource. It is deliberately left raw
2241        // for the `touch_file` call below, which takes a `RawInode`.
2242        let out_resource = match out_resource {
2243            Some(resource) => Some(resource),
2244            None => match out_inode {
2245                Some(raw_ino) => Some(ResourceID::FileContents(
2246                    determinize_inode(guest, raw_ino).await.0,
2247                )),
2248                None => None,
2249            },
2250        };
2251        if let Some(resource) = out_resource {
2252            resources.insert(resource, Permission::W);
2253        }
2254        resources.fyi("sendfile");
2255        resource_request(guest, resources).await;
2256
2257        let result = self
2258            .record_or_replay(guest, call)
2259            .await
2260            .map_err(Error::from);
2261        if guest.config().virtualize_metadata && matches!(&result, Ok(copied) if *copied > 0) {
2262            let inode = out_inode.expect("virtualized metadata requires stat data for sendfile");
2263            touch_file(guest, inode).await;
2264        }
2265        resource_release_all(guest).await;
2266        result
2267    }
2268
2269    /// SYS_write system call.
2270    pub async fn handle_write<G: Guest<Self>>(
2271        &self,
2272        guest: &mut G,
2273        mut call: syscalls::Write,
2274    ) -> Result<i64, Error> {
2275        if self.timer_slack_binding(guest, call.fd())?.is_some() {
2276            return self
2277                .write_timer_slack(guest, call.fd(), call.buf(), call.len())
2278                .await;
2279        }
2280
2281        let (
2282            fd_type,
2283            physically_nonblocking,
2284            logically_nonblocking,
2285            open_file_id,
2286            resource,
2287            raw_ino,
2288        ) = guest.thread_state().with_detfd(call.fd(), |detfd| {
2289            (
2290                detfd.ty(),
2291                detfd.physically_nonblocking(),
2292                detfd.is_nonblocking(),
2293                detfd.open_file_id(),
2294                detfd.resource(),
2295                detfd.stat().map(|x| x.inode),
2296            )
2297        })?;
2298        // It doesn't matter much where the linearization point for this mtime bump falls:
2299        if guest.config().virtualize_metadata {
2300            let r =
2301                raw_ino.expect("Expect that when virtualize_metadata, DetFd's stat is populated!");
2302            touch_file(guest, r).await;
2303        }
2304
2305        if let Some(resource) = resource {
2306            let mut request = guest.thread_state().mk_request(resource, Permission::W);
2307            if should_tag_sabre_internal_pipe_io(
2308                guest.config().discover_live_file_metadata,
2309                fd_type,
2310                physically_nonblocking,
2311                logically_nonblocking,
2312            ) {
2313                request.fyi(SABRE_INTERNAL_PIPE_IO_FYI);
2314            }
2315            resource_request(guest, request).await;
2316        }
2317
2318        // Only route writes through the nonblockable-fd path when the fd is actually
2319        // physically nonblocking. Detcore-created pipes are physically nonblocking in every
2320        // sequential mode, including record/replay, so their logically blocking writes use
2321        // the completion helper below. A physically blocking fd instead uses the original
2322        // synchronous path: treating an internal pipe as BlockingExternalIO would assume the
2323        // writer and reader were independent and could deadlock the scheduler.
2324        let res = if physically_nonblocking && fd_type == FdType::Pipe && !logically_nonblocking {
2325            self.execute_blocking_pipe_write(guest, call, open_file_id)
2326                .await
2327        } else if physically_nonblocking
2328            && matches!(fd_type, FdType::Socket | FdType::Pipe | FdType::Eventfd)
2329        {
2330            self.execute_nonblockable_fd_syscall(guest, call).await
2331        } else if guest.config().deterministic_io {
2332            let mut total_written_bytes = 0;
2333            let mut remaining_buf = call.len();
2334
2335            trace!(
2336                "[detcore/det_io]: Requested write buffer size: {:?}",
2337                remaining_buf
2338            );
2339
2340            loop {
2341                match self
2342                    .record_or_replay_preserving_tool_errors(guest, call)
2343                    .await
2344                {
2345                    Ok(res) => {
2346                        remaining_buf -= res as usize;
2347                        total_written_bytes += res;
2348
2349                        trace!(
2350                            "[detcore/det_io]: Remaining write buffer size: {:?}",
2351                            remaining_buf
2352                        );
2353
2354                        if res == 0 || remaining_buf == 0 {
2355                            break Ok(total_written_bytes);
2356                        }
2357
2358                        // Buf is guaranteed to exist as we already issued a syscall.
2359                        let old_ptr = call.buf().unwrap().as_raw();
2360                        call = call
2361                            .with_len(remaining_buf)
2362                            .with_buf(Addr::<u8>::from_raw(old_ptr + res as usize));
2363                    }
2364                    Err(error) => {
2365                        break finish_partial_record_or_replay_write(total_written_bytes, error);
2366                    }
2367                }
2368            }
2369        } else {
2370            self.record_or_replay_preserving_tool_errors(guest, call)
2371                .await
2372        };
2373
2374        resource_release_all(guest).await;
2375        res
2376    }
2377
2378    // AUTONOMOUS-BOT-IMPLEMENTED
2379    // TODO-HUMAN-REVIEW(#683): Confirm positional-write ordering and replay semantics.
2380    /// SYS_pwrite64 system call.
2381    pub async fn handle_pwrite64<G: Guest<Self>>(
2382        &self,
2383        guest: &mut G,
2384        mut call: syscalls::Pwrite64,
2385    ) -> Result<i64, Error> {
2386        if self.timer_slack_binding(guest, call.fd())?.is_some() {
2387            return Err(if call.offset() < 0 {
2388                Errno::EINVAL.into()
2389            } else {
2390                Errno::ESPIPE.into()
2391            });
2392        }
2393
2394        let (resource, raw_ino) = guest.thread_state().with_detfd(call.fd(), |detfd| {
2395            (detfd.resource(), detfd.stat().map(|stat| stat.inode))
2396        })?;
2397        // The fd's cached `DetStat` carries the HOST inode (`DetStat` is built
2398        // straight from `fstat`/`statx`), so it must be determinized before it
2399        // can name a guest-visible resource. Passing it through directly used
2400        // to type-check only because `DetInode` was an alias for `RawInode`.
2401        let resource = match resource {
2402            Some(resource) => Some(resource),
2403            None => match raw_ino {
2404                Some(raw_ino) => Some(ResourceID::FileContents(
2405                    determinize_inode(guest, raw_ino).await.0,
2406                )),
2407                None => None,
2408            },
2409        };
2410
2411        if let Some(resource) = resource {
2412            let request = guest.thread_state().mk_request(resource, Permission::W);
2413            resource_request(guest, request).await;
2414        }
2415
2416        let result = if guest.config().deterministic_io {
2417            let mut total_written = 0_i64;
2418            let mut remaining = call.len();
2419
2420            loop {
2421                match self
2422                    .record_or_replay_preserving_tool_errors(guest, call)
2423                    .await
2424                {
2425                    Ok(written) => {
2426                        let Ok(written) = usize::try_from(written) else {
2427                            break Err(Errno::EIO.into());
2428                        };
2429                        let Ok(written_i64) = i64::try_from(written) else {
2430                            break Err(Errno::EIO.into());
2431                        };
2432                        if written > remaining {
2433                            break Err(Errno::EIO.into());
2434                        }
2435                        remaining -= written;
2436                        let Some(next_total) = total_written.checked_add(written_i64) else {
2437                            break Err(Errno::EIO.into());
2438                        };
2439                        total_written = next_total;
2440
2441                        if written == 0 || remaining == 0 {
2442                            break Ok(total_written);
2443                        }
2444
2445                        let Some(old_buf) = call.buf() else {
2446                            break Err(Errno::EFAULT.into());
2447                        };
2448                        let Some(next_buf) = old_buf.as_raw().checked_add(written) else {
2449                            break Err(Errno::EFAULT.into());
2450                        };
2451                        let Some(next_offset) = call.offset().checked_add(written_i64) else {
2452                            break Err(Errno::EFBIG.into());
2453                        };
2454                        let Some(next_buf) = Addr::<u8>::from_raw(next_buf) else {
2455                            break Err(Errno::EFAULT.into());
2456                        };
2457                        call = call
2458                            .with_buf(Some(next_buf))
2459                            .with_len(remaining)
2460                            .with_offset(next_offset);
2461                    }
2462                    Err(error) => {
2463                        break finish_partial_record_or_replay_write(total_written, error);
2464                    }
2465                }
2466            }
2467        } else {
2468            self.record_or_replay_preserving_tool_errors(guest, call)
2469                .await
2470        };
2471
2472        if guest.config().virtualize_metadata && matches!(&result, Ok(written) if *written > 0) {
2473            let inode = raw_ino.expect("virtualized metadata requires stat data for tracked fds");
2474            touch_file(guest, inode).await;
2475        }
2476
2477        resource_release_all(guest).await;
2478        result
2479    }
2480
2481    // AUTONOMOUS-BOT-IMPLEMENTED
2482    // TODO-HUMAN-REVIEW(#547)
2483    /// SYS_writev system call.
2484    ///
2485    /// Preserve the initial writev as one kernel operation so its iovec order remains intact.
2486    /// Detcore adds open-file resource ordering and nonblocking scheduler integration; a
2487    /// blocking pipe short write is completed by the helper because Hermit injected O_NONBLOCK.
2488    pub async fn handle_writev<G: Guest<Self>>(
2489        &self,
2490        guest: &mut G,
2491        call: syscalls::Writev,
2492    ) -> Result<i64, Error> {
2493        if self.timer_slack_binding(guest, call.fd())?.is_some() {
2494            self.require_timer_slack_access(guest, call.fd(), true)?;
2495            let iovecs = read_iovecs(&guest.memory(), call.iov(), call.len())?;
2496            return self.writev_timer_slack(guest, call.fd(), iovecs, 0).await;
2497        }
2498
2499        let (
2500            fd_type,
2501            physically_nonblocking,
2502            logically_nonblocking,
2503            open_file_id,
2504            resource,
2505            raw_ino,
2506        ) = guest.thread_state().with_detfd(call.fd(), |detfd| {
2507            (
2508                detfd.ty(),
2509                detfd.physically_nonblocking(),
2510                detfd.is_nonblocking(),
2511                detfd.open_file_id(),
2512                detfd.resource(),
2513                detfd.stat().map(|x| x.inode),
2514            )
2515        })?;
2516
2517        if let Some(resource) = resource {
2518            let mut request = guest.thread_state().mk_request(resource, Permission::W);
2519            if should_tag_sabre_internal_pipe_io(
2520                guest.config().discover_live_file_metadata,
2521                fd_type,
2522                physically_nonblocking,
2523                logically_nonblocking,
2524            ) {
2525                request.fyi(SABRE_INTERNAL_PIPE_IO_FYI);
2526            }
2527            resource_request(guest, request).await;
2528        }
2529
2530        let result = if physically_nonblocking && fd_type == FdType::Pipe && !logically_nonblocking
2531        {
2532            self.execute_blocking_pipe_writev(guest, call, open_file_id)
2533                .await
2534        } else if physically_nonblocking
2535            && matches!(fd_type, FdType::Socket | FdType::Pipe | FdType::Eventfd)
2536        {
2537            self.execute_nonblockable_fd_syscall(guest, call).await
2538        } else {
2539            self.record_or_replay_preserving_tool_errors(guest, call)
2540                .await
2541        };
2542
2543        if guest.config().virtualize_metadata && matches!(&result, Ok(written) if *written > 0) {
2544            let inode =
2545                raw_ino.expect("virtualized metadata requires stat data for every tracked fd");
2546            touch_file(guest, inode).await;
2547        }
2548
2549        resource_release_all(guest).await;
2550        result
2551    }
2552
2553    // AUTONOMOUS-BOT-IMPLEMENTED
2554    // TODO-HUMAN-REVIEW(#794)
2555    /// SYS_readv system call: the vectored form of `read`.
2556    ///
2557    /// Mirrors [`Self::handle_writev`] for the read direction. Detcore adds
2558    /// open-file resource ordering and, for physically nonblocking pipe/socket
2559    /// fds, the nonblocking scheduler integration. Random devices use the shared
2560    /// canonical cursor; other descriptors retain their recorded kernel operation.
2561    pub async fn handle_readv<G: Guest<Self>>(
2562        &self,
2563        guest: &mut G,
2564        call: syscalls::Readv,
2565    ) -> Result<i64, Error> {
2566        self.handle_readv_with_output(guest, call, &mut None).await
2567    }
2568
2569    /// Import after the resource wait and retain that geometry through copyout.
2570    /// The shared-cursor closure is synchronous and takes no metadata locks.
2571    fn read_random_vectors<G: Guest<Self>>(
2572        &self,
2573        guest: &mut G,
2574        detfd: &DetFd,
2575        request: RandomVectoredRead,
2576        rng_output: &mut Option<Vec<crate::io_buffers::BufferExtent>>,
2577    ) -> Result<i64, Error> {
2578        require_random_device_read_access(detfd.status_flags())?;
2579        let policy = if guest.config().backend_is_kvm {
2580            crate::iovecs::UserAddressPolicy::Kvm
2581        } else {
2582            crate::iovecs::UserAddressPolicy::Native
2583        };
2584        let iovecs = crate::iovecs::import_read_iovecs(
2585            &guest.memory(),
2586            request.address,
2587            request.count,
2588            policy,
2589        )?;
2590        let total = iovecs.iter().map(|iov| iov.len).sum();
2591        validate_random_vector_read(request.offset, total, request.flags)?;
2592        let written = if let Some(offset) = request.offset {
2593            self.fill_random_device_iovecs(guest, &iovecs, offset)?
2594        } else {
2595            detfd.with_random_device_stream(|offset| {
2596                self.fill_random_device_iovecs(guest, &iovecs, offset)
2597            })?
2598        };
2599        if written > 0 && self.cfg.detlog_io_buffers && crate::detlog_observed!() {
2600            *rng_output = Some(crate::io_buffers::rng_readv_extents(&iovecs, written)?);
2601        }
2602        Ok(written as i64)
2603    }
2604
2605    pub(crate) async fn handle_readv_with_output<G: Guest<Self>>(
2606        &self,
2607        guest: &mut G,
2608        call: syscalls::Readv,
2609        rng_output: &mut Option<Vec<crate::io_buffers::BufferExtent>>,
2610    ) -> Result<i64, Error> {
2611        if self.timer_slack_binding(guest, call.fd())?.is_some() {
2612            self.require_timer_slack_access(guest, call.fd(), false)?;
2613            let iovecs = read_iovecs(&guest.memory(), call.iov(), call.len())?;
2614            return self
2615                .readv_timer_slack(guest, call.fd(), iovecs, None, 0)
2616                .await;
2617        }
2618
2619        let is_procfs = guest
2620            .thread_state()
2621            .with_detfd(call.fd(), |detfd| detfd.procfs_position().is_some())?;
2622        if is_procfs {
2623            return Err(Errno::ENOSYS.into());
2624        }
2625
2626        let (fd_type, physically_nonblocking, logically_nonblocking, resource, detfd) =
2627            guest.thread_state().with_detfd(call.fd(), |detfd| {
2628                (
2629                    detfd.ty(),
2630                    detfd.physically_nonblocking(),
2631                    detfd.is_nonblocking(),
2632                    detfd.resource(),
2633                    detfd.clone(),
2634                )
2635            })?;
2636
2637        if let Some(resource) = resource {
2638            let mut request = guest.thread_state().mk_request(resource, Permission::R);
2639            if should_tag_sabre_internal_pipe_io(
2640                guest.config().discover_live_file_metadata,
2641                fd_type,
2642                physically_nonblocking,
2643                logically_nonblocking,
2644            ) {
2645                request.fyi(SABRE_INTERNAL_PIPE_IO_FYI);
2646            }
2647            resource_request(guest, request).await;
2648        }
2649
2650        let res = if fd_type == FdType::Rng {
2651            self.read_random_vectors(
2652                guest,
2653                &detfd,
2654                RandomVectoredRead {
2655                    address: call.iov().map_or(0, |addr| addr.as_raw()),
2656                    count: call.len(),
2657                    offset: None,
2658                    flags: 0,
2659                },
2660                rng_output,
2661            )
2662        } else if physically_nonblocking
2663            && matches!(fd_type, FdType::Socket | FdType::Pipe | FdType::Eventfd)
2664        {
2665            self.execute_nonblockable_fd_syscall(guest, call).await
2666        } else {
2667            self.record_or_replay_preserving_tool_errors(guest, call)
2668                .await
2669        };
2670
2671        resource_release_all(guest).await;
2672        res
2673    }
2674
2675    // AUTONOMOUS-BOT-IMPLEMENTED
2676    // TODO-HUMAN-REVIEW(#794)
2677    /// SYS_preadv system call: the vectored form of `pread64`.
2678    ///
2679    /// RNG reads use the canonical stream at the explicit offset without
2680    /// advancing its shared cursor. Other files retain their kernel operation.
2681    pub async fn handle_preadv<G: Guest<Self>>(
2682        &self,
2683        guest: &mut G,
2684        call: syscalls::Preadv,
2685    ) -> Result<i64, Error> {
2686        self.handle_preadv_with_output(guest, call, &mut None).await
2687    }
2688
2689    pub(crate) async fn handle_preadv_with_output<G: Guest<Self>>(
2690        &self,
2691        guest: &mut G,
2692        call: syscalls::Preadv,
2693        rng_output: &mut Option<Vec<crate::io_buffers::BufferExtent>>,
2694    ) -> Result<i64, Error> {
2695        // Linux rejects negative offsets before descriptor lookup or import.
2696        let offset = vectored_offset(call.pos_l(), call.pos_h());
2697        if offset < 0 {
2698            return Err(Errno::EINVAL.into());
2699        }
2700        if self.timer_slack_binding(guest, call.fd())?.is_some() {
2701            self.require_timer_slack_access(guest, call.fd(), false)?;
2702            let iovecs = read_iovecs(&guest.memory(), call.iov(), call.iov_len())?;
2703            return self
2704                .readv_timer_slack(guest, call.fd(), iovecs, Some(offset), 0)
2705                .await;
2706        }
2707
2708        let is_procfs = guest
2709            .thread_state()
2710            .with_detfd(call.fd(), |detfd| detfd.procfs_position().is_some())?;
2711        if is_procfs {
2712            return Err(Errno::ENOSYS.into());
2713        }
2714
2715        let detfd = guest
2716            .thread_state()
2717            .with_detfd(call.fd(), |detfd| detfd.clone())?;
2718
2719        if let Some(resource) = detfd.resource() {
2720            let request = guest.thread_state().mk_request(resource, Permission::R);
2721            resource_request(guest, request).await;
2722        }
2723
2724        let res = if detfd.ty() == FdType::Rng {
2725            self.read_random_vectors(
2726                guest,
2727                &detfd,
2728                RandomVectoredRead {
2729                    address: call.iov().map_or(0, |addr| addr.as_raw()),
2730                    count: call.iov_len(),
2731                    offset: Some(offset as u64),
2732                    flags: 0,
2733                },
2734                rng_output,
2735            )
2736        } else {
2737            self.record_or_replay_preserving_tool_errors(guest, call)
2738                .await
2739        };
2740        resource_release_all(guest).await;
2741        res
2742    }
2743
2744    // AUTONOMOUS-BOT-IMPLEMENTED
2745    // TODO-HUMAN-REVIEW(#794)
2746    /// SYS_preadv2 system call: positioned vectors, or the shared stream when
2747    /// offset is -1. RNG flag validation precedes output copying.
2748    pub async fn handle_preadv2<G: Guest<Self>>(
2749        &self,
2750        guest: &mut G,
2751        call: syscalls::Preadv2,
2752    ) -> Result<i64, Error> {
2753        self.handle_preadv2_with_output(guest, call, &mut None)
2754            .await
2755    }
2756
2757    pub(crate) async fn handle_preadv2_with_output<G: Guest<Self>>(
2758        &self,
2759        guest: &mut G,
2760        call: syscalls::Preadv2,
2761        rng_output: &mut Option<Vec<crate::io_buffers::BufferExtent>>,
2762    ) -> Result<i64, Error> {
2763        let offset = vectored_offset(call.pos_l(), call.pos_h());
2764        if offset < -1 {
2765            return Err(Errno::EINVAL.into());
2766        }
2767        if self.timer_slack_binding(guest, call.fd())?.is_some() {
2768            self.require_timer_slack_access(guest, call.fd(), false)?;
2769            let count = usize::try_from(call.iov_len()).map_err(|_| Errno::EINVAL)?;
2770            let iovecs = read_iovecs(&guest.memory(), call.iov(), count)?;
2771            return if offset == -1 {
2772                self.readv_timer_slack(guest, call.fd(), iovecs, None, call.flags())
2773                    .await
2774            } else {
2775                self.readv_timer_slack(guest, call.fd(), iovecs, Some(offset), call.flags())
2776                    .await
2777            };
2778        }
2779
2780        let is_procfs = guest
2781            .thread_state()
2782            .with_detfd(call.fd(), |detfd| detfd.procfs_position().is_some())?;
2783        if is_procfs {
2784            return Err(Errno::ENOSYS.into());
2785        }
2786
2787        let detfd = guest
2788            .thread_state()
2789            .with_detfd(call.fd(), |detfd| detfd.clone())?;
2790
2791        if let Some(resource) = detfd.resource() {
2792            let request = guest.thread_state().mk_request(resource, Permission::R);
2793            resource_request(guest, request).await;
2794        }
2795
2796        let res = if detfd.ty() == FdType::Rng {
2797            self.read_random_vectors(
2798                guest,
2799                &detfd,
2800                RandomVectoredRead {
2801                    address: call.iov().map_or(0, |addr| addr.as_raw()),
2802                    count: call.iov_len() as usize,
2803                    offset: (offset != -1).then_some(offset as u64),
2804                    flags: call.flags(),
2805                },
2806                rng_output,
2807            )
2808        } else {
2809            self.record_or_replay_preserving_tool_errors(guest, call)
2810                .await
2811        };
2812        resource_release_all(guest).await;
2813        res
2814    }
2815
2816    // AUTONOMOUS-BOT-IMPLEMENTED
2817    // TODO-HUMAN-REVIEW(#794)
2818    /// SYS_pwritev system call: the vectored form of `pwrite64`.
2819    ///
2820    /// Positioned writes target seekable files and do not block, so this mirrors
2821    /// [`Self::handle_pwrite64`]'s ordering, records/replays the single kernel
2822    /// operation, and bumps the virtual mtime on a successful write.
2823    pub async fn handle_pwritev<G: Guest<Self>>(
2824        &self,
2825        guest: &mut G,
2826        call: syscalls::Pwritev,
2827    ) -> Result<i64, Error> {
2828        if self.timer_slack_binding(guest, call.fd())?.is_some() {
2829            let offset = vectored_offset(call.pos_l(), call.pos_h());
2830            return if offset < 0 {
2831                Err(Errno::EINVAL.into())
2832            } else {
2833                Err(Errno::ESPIPE.into())
2834            };
2835        }
2836
2837        let (resource, raw_ino) = guest.thread_state().with_detfd(call.fd(), |detfd| {
2838            (detfd.resource(), detfd.stat().map(|stat| stat.inode))
2839        })?;
2840        // The fd's cached `DetStat` carries the HOST inode (`DetStat` is built
2841        // straight from `fstat`/`statx`), so it must be determinized before it
2842        // can name a guest-visible resource. Passing it through directly used
2843        // to type-check only because `DetInode` was an alias for `RawInode`.
2844        let resource = match resource {
2845            Some(resource) => Some(resource),
2846            None => match raw_ino {
2847                Some(raw_ino) => Some(ResourceID::FileContents(
2848                    determinize_inode(guest, raw_ino).await.0,
2849                )),
2850                None => None,
2851            },
2852        };
2853
2854        if let Some(resource) = resource {
2855            let request = guest.thread_state().mk_request(resource, Permission::W);
2856            resource_request(guest, request).await;
2857        }
2858
2859        let result = self
2860            .record_or_replay_preserving_tool_errors(guest, call)
2861            .await;
2862
2863        if guest.config().virtualize_metadata && matches!(&result, Ok(written) if *written > 0) {
2864            let inode = raw_ino.expect("virtualized metadata requires stat data for tracked fds");
2865            touch_file(guest, inode).await;
2866        }
2867
2868        resource_release_all(guest).await;
2869        result
2870    }
2871
2872    // AUTONOMOUS-BOT-IMPLEMENTED
2873    // TODO-HUMAN-REVIEW(#794)
2874    /// SYS_pwritev2 system call: `pwritev` with a trailing per-call flags
2875    /// argument, which record/replay forwards unchanged.
2876    pub async fn handle_pwritev2<G: Guest<Self>>(
2877        &self,
2878        guest: &mut G,
2879        call: syscalls::Pwritev2,
2880    ) -> Result<i64, Error> {
2881        if self.timer_slack_binding(guest, call.fd())?.is_some() {
2882            let offset = vectored_offset(call.pos_l(), call.pos_h());
2883            if offset < -1 {
2884                return Err(Errno::EINVAL.into());
2885            }
2886            if offset >= 0 {
2887                return Err(Errno::ESPIPE.into());
2888            }
2889            self.require_timer_slack_access(guest, call.fd(), true)?;
2890            let count = usize::try_from(call.iov_len()).map_err(|_| Errno::EINVAL)?;
2891            let iovecs = read_iovecs(&guest.memory(), call.iov(), count)?;
2892            return self
2893                .writev_timer_slack(guest, call.fd(), iovecs, call.flags())
2894                .await;
2895        }
2896
2897        let (resource, raw_ino) = guest.thread_state().with_detfd(call.fd(), |detfd| {
2898            (detfd.resource(), detfd.stat().map(|stat| stat.inode))
2899        })?;
2900        // The fd's cached `DetStat` carries the HOST inode (`DetStat` is built
2901        // straight from `fstat`/`statx`), so it must be determinized before it
2902        // can name a guest-visible resource. Passing it through directly used
2903        // to type-check only because `DetInode` was an alias for `RawInode`.
2904        let resource = match resource {
2905            Some(resource) => Some(resource),
2906            None => match raw_ino {
2907                Some(raw_ino) => Some(ResourceID::FileContents(
2908                    determinize_inode(guest, raw_ino).await.0,
2909                )),
2910                None => None,
2911            },
2912        };
2913
2914        if let Some(resource) = resource {
2915            let request = guest.thread_state().mk_request(resource, Permission::W);
2916            resource_request(guest, request).await;
2917        }
2918
2919        let result = self
2920            .record_or_replay_preserving_tool_errors(guest, call)
2921            .await;
2922
2923        if guest.config().virtualize_metadata && matches!(&result, Ok(written) if *written > 0) {
2924            let inode = raw_ino.expect("virtualized metadata requires stat data for tracked fds");
2925            touch_file(guest, inode).await;
2926        }
2927
2928        resource_release_all(guest).await;
2929        result
2930    }
2931
2932    /// SYS_mmap system call.
2933    pub async fn handle_mmap<G: Guest<Self>>(
2934        &self,
2935        guest: &mut G,
2936        call: syscalls::Mmap,
2937    ) -> Result<i64, Error> {
2938        enum SharedBacking {
2939            Anonymous,
2940            File {
2941                object: SharedMemoryObjectId,
2942                offset: u64,
2943            },
2944        }
2945
2946        let backing = if call.flags().contains(MapFlags::MAP_SHARED) {
2947            if call.fd() == -1 {
2948                Some(SharedBacking::Anonymous)
2949            } else {
2950                let offset = u64::try_from(call.offset()).map_err(|_| Errno::EINVAL)?;
2951                guest
2952                    .thread_state()
2953                    .with_detfd(call.fd(), |fd| {
2954                        let object = fd.stat().map_or_else(
2955                            || SharedMemoryObjectId::OpenFile {
2956                                id: fd.open_file_id(),
2957                            },
2958                            |stat| SharedMemoryObjectId::File {
2959                                device: stat.dev,
2960                                inode: stat.inode,
2961                            },
2962                        );
2963                        SharedBacking::File { object, offset }
2964                    })
2965                    .ok()
2966            }
2967        } else {
2968            None
2969        };
2970        let len = call.len();
2971        let result = self.record_or_replay(guest, call).await?;
2972        let start = usize::try_from(result).expect("a successful mmap must return an address");
2973
2974        guest.thread_state().unmap_memory(start, len);
2975        match backing {
2976            Some(SharedBacking::Anonymous) => {
2977                guest.thread_state().map_shared_anonymous(start, len);
2978            }
2979            Some(SharedBacking::File { object, offset }) => {
2980                guest
2981                    .thread_state()
2982                    .map_shared_object(start, len, object, offset);
2983            }
2984            None => {}
2985        }
2986        Ok(result)
2987    }
2988
2989    /// SYS_munmap system call.
2990    pub async fn handle_munmap<G: Guest<Self>>(
2991        &self,
2992        guest: &mut G,
2993        call: syscalls::Munmap,
2994    ) -> Result<i64, Error> {
2995        let start = call.addr().map(Addr::as_raw).unwrap_or(0);
2996        let len = call.len();
2997        let result = self.record_or_replay(guest, call).await?;
2998        guest.thread_state().unmap_memory(start, len);
2999        Ok(result)
3000    }
3001
3002    /// SYS_mremap system call.
3003    pub async fn handle_mremap<G: Guest<Self>>(
3004        &self,
3005        guest: &mut G,
3006        call: syscalls::Mremap,
3007    ) -> Result<i64, Error> {
3008        let old_start = call.addr().map(AddrMut::as_raw).unwrap_or(0);
3009        let old_len = call.old_len();
3010        let new_len = call.new_len();
3011        let result = self.record_or_replay(guest, call).await?;
3012        let new_start =
3013            usize::try_from(result).expect("a successful mremap must return an address");
3014        guest
3015            .thread_state()
3016            .remap_memory(old_start, old_len, new_start, new_len);
3017        Ok(result)
3018    }
3019
3020    // Determinize stat by doing:
3021    //   - using virtual inode instead of real inodes. The virtual inodes
3022    //     increase monolitically and won't be re-used (like ext4)
3023    //   - use logical modtime which could be used by program like GNU make
3024    //     to determine file changes. A file first seen with a canonical host
3025    //     mtime (`CANONICAL_FILE_MTIME_SECONDS`, e.g. the Nix store's 1) keeps
3026    //     it; any other first-seen file reports the epoch.
3027    //   - atime, ctime and btime always report the epoch: the kernel sets ctime
3028    //     and btime itself, so even for a Nix store file they are real
3029    //     timestamps, and Hermit keeps no per-file atime.
3030    async fn determinize_stat<G, S>(
3031        &self,
3032        guest: &mut G,
3033        stat: S,
3034        inode_override: Option<DetInode>,
3035    ) -> Result<DetStat, Error>
3036    where
3037        G: Guest<Self>,
3038        S: Into<DetStat>,
3039    {
3040        let cfg = guest.config().clone();
3041
3042        let mut stat: DetStat = stat.into();
3043        let (d_ino, global_mtime) = match inode_override {
3044            // The container's stdio streams have fixed inodes and always
3045            // report the epoch: whatever backs them on the host (a pipe, a
3046            // terminal, a redirected file) is not part of the guest's view.
3047            Some(inode) => {
3048                let nanos = cfg
3049                    .epoch
3050                    .timestamp_nanos_opt()
3051                    .expect("epoch cannot be represented in nanoseconds")
3052                    as u64;
3053                (inode, LogicalTime::from_nanos(nanos))
3054            }
3055            None => {
3056                // statx fills stx_mtime only when it reports STATX_MTIME.
3057                let observed = if stat.mask.contains(StatxMask::STATX_MTIME) {
3058                    ObservedMtime::from_host_mtime(stat.mtime.tv_sec, stat.mtime.tv_nsec)
3059                } else {
3060                    ObservedMtime::Unobserved
3061                };
3062                determinize_inode_observing_mtime(guest, stat.inode, observed).await
3063            }
3064        };
3065        stat.inode = d_ino.as_raw(); // Reveal only the deterministic inode.
3066
3067        // AUTONOMOUS-BOT-IMPLEMENTED
3068        // TODO-HUMAN-REVIEW(PR-1056): Deterministic st_dev remapping.
3069        // The raw st_dev leaks the kernel's host-wide anonymous block-device
3070        // number for procfs/sysfs/tmpfs mounts, which drifts between runs (and
3071        // between the two runs of `--verify`). Reveal only a deterministic
3072        // device id.
3073        stat.dev = determinize_device(guest, stat.dev).await;
3074
3075        let epoch_tp = Timespec {
3076            tv_sec: cfg.epoch.timestamp(),
3077            tv_nsec: cfg.epoch.timestamp_subsec_nanos() as i64,
3078        };
3079
3080        let mtime: Timespec = global_mtime.into();
3081        stat.atime = epoch_tp;
3082        stat.ctime = epoch_tp;
3083        stat.btime = epoch_tp;
3084
3085        stat.mtime = mtime;
3086
3087        Ok(stat)
3088    }
3089
3090    /// Handles all stat syscalls.
3091    pub async fn handle_stat_family<G: Guest<Self>>(
3092        &self,
3093        guest: &mut G,
3094        call: StatFamily,
3095    ) -> Result<i64, Error> {
3096        if guest.config().virtualize_metadata {
3097            // NB: let kernel handle error codes, it's not easy to do so without
3098            // kernel because there're many corner cases. i.e.: even access
3099            // filepath from tracer may cause tracer to hang under certain fuse
3100            // filesystem (squashfs_ll).
3101            guest.inject(Syscall::from(call)).await?;
3102            let statptr = call.stat().ok_or(Errno::EFAULT)?;
3103            let inode_override = match call {
3104                StatFamily::Fstat(call) => guest
3105                    .thread_state()
3106                    .with_detfd(call.fd(), |detfd| {
3107                        deterministic_stdio_inode_for_resource(call.fd(), detfd.resource())
3108                    })
3109                    .ok()
3110                    .flatten(),
3111                _ => None,
3112            };
3113            let mut memory = guest.memory();
3114            let stat = memory.read_value(statptr.0)?;
3115            let stat = self.determinize_stat(guest, stat, inode_override).await?;
3116            memory.write_value(statptr.0, &stat.into())?;
3117            Ok(0)
3118        } else {
3119            Ok(self.record_or_replay(guest, call).await?)
3120        }
3121    }
3122
3123    /// statx system call
3124    pub async fn handle_statx<G: Guest<Self>>(
3125        &self,
3126        guest: &mut G,
3127        call: syscalls::Statx,
3128    ) -> Result<i64, Error> {
3129        if guest.config().virtualize_metadata {
3130            // NB: let kernel handle error codes, it's not easy to do so without kernel
3131            // because there're many corner cases. i.e.: even access filepath from tracer
3132            // may cause tracer to hang under certain fuse filesystem (squashfs_ll).
3133            guest.inject(call).await?;
3134            let statptr = call.statx().ok_or(Errno::EFAULT)?;
3135            let mut memory = guest.memory();
3136            let stat = memory.read_value(statptr.0)?;
3137            let stat = self.determinize_stat(guest, stat, None).await?;
3138            memory.write_value(statptr.0, &stat.into())?;
3139            Ok(0)
3140        } else {
3141            Ok(self.record_or_replay(guest, call).await?)
3142        }
3143    }
3144
3145    /// fcntl system call
3146    pub async fn handle_fcntl<G: Guest<Self>>(
3147        &self,
3148        guest: &mut G,
3149        call: syscalls::Fcntl,
3150    ) -> Result<i64, Error> {
3151        let fd = call.fd();
3152        let o_cloexec = match call.cmd() {
3153            F_DUPFD_CLOEXEC(_) => OFlag::O_CLOEXEC,
3154            _ => OFlag::empty(),
3155        };
3156        match call.cmd() {
3157            F_GETFL => {
3158                let physical_flags = self.record_or_replay(guest, call).await?;
3159                let logical_nonblocking = guest
3160                    .thread_state()
3161                    .with_detfd(fd, |detfd| detfd.is_nonblocking())?;
3162                let nonblocking = i64::from(OFlag::O_NONBLOCK.bits());
3163                if logical_nonblocking {
3164                    Ok(physical_flags | nonblocking)
3165                } else {
3166                    Ok(physical_flags & !nonblocking)
3167                }
3168            }
3169            F_SETFL(flags) => {
3170                let fd_type = guest.thread_state().with_detfd(fd, |detfd| detfd.ty())?;
3171                let force_nonblocking = self.cfg.use_nonblocking_sockets()
3172                    && !self.cfg.recordreplay_modes
3173                    && matches!(fd_type, FdType::Socket | FdType::Pipe | FdType::Eventfd);
3174                let physical_flags = if force_nonblocking {
3175                    flags | OFlag::O_NONBLOCK.bits()
3176                } else {
3177                    flags
3178                };
3179                let result = self
3180                    .record_or_replay(guest, call.with_cmd(F_SETFL(physical_flags)))
3181                    .await?;
3182                guest.thread_state().with_detfd(fd, |detfd| {
3183                    // Record the guest's *logical* status flags (derives logical
3184                    // nonblocking); when we forced O_NONBLOCK physically without the
3185                    // guest asking, mark the description physically nonblocking too.
3186                    detfd.set_status_flags(flags);
3187                    if force_nonblocking {
3188                        detfd.set_physically_nonblocking();
3189                    }
3190                })?;
3191                Ok(result)
3192            }
3193            F_DUPFD(_) | F_DUPFD_CLOEXEC(_) => {
3194                let newfd = self.record_or_replay(guest, call).await? as RawFd;
3195                let replaced = guest.thread_state_mut().dup_fd(fd, newfd, o_cloexec)?;
3196                if let Some(open_file_id) = replaced {
3197                    self.release_port_for_open_file(guest, open_file_id).await;
3198                }
3199                Ok(newfd as i64)
3200            }
3201            F_SETFD(flags) => {
3202                let result = self.record_or_replay(guest, call).await?;
3203                guest.thread_state().with_detfd(fd, |detfd| {
3204                    detfd.set_cloexec(flags & libc::FD_CLOEXEC != 0);
3205                })?;
3206                Ok(result)
3207            }
3208            // AUTONOMOUS-BOT-IMPLEMENTED
3209            // TODO-HUMAN-REVIEW(PR-2568): Review refusing guest pipe growth.
3210            // Raising a pipe above the pinned capacity is the ONE syscall that
3211            // defeats the pin, and whether it succeeds is decided by the host's
3212            // `/proc/sys/fs/pipe-max-size`. On a default host (1 MiB ceiling)
3213            // the guest gets a 1048576-byte pipe; on a hardened host (64 KiB)
3214            // the identical guest gets EPERM and keeps 8192. Same binary, same
3215            // `--strict`, guest-visible return value and pipe capacity decided
3216            // by a host sysctl -- a determinism leak by this project's own
3217            // definition, and it survived because the capacity pin was applied
3218            // at creation and never defended afterwards.
3219            //
3220            // Refuse deterministically instead of asking Linux. EPERM is the
3221            // errno Linux itself returns when that ceiling binds, so the guest
3222            // sees a shape it must already handle rather than a novel one, and
3223            // it is the answer the hardened host would have given.
3224            //
3225            // SHRINKING IS DELIBERATELY LEFT ALONE. It is always permitted for
3226            // an unprivileged process, it is process-local with no host-derived
3227            // input, and `tests/c/pipe_capacity.c` locks
3228            // it as a guest-visible contract: that fixture shrinks to one page
3229            // and requires the value to round-trip. Clamping every
3230            // `F_SETPIPE_SZ` to the pinned capacity would break that contract
3231            // while fixing nothing that is actually nondeterministic.
3232            F_SETPIPE_SZ(requested) if pipe_capacity_request_exceeds_ceiling(requested) => {
3233                trace!(
3234                    "[detcore] refusing F_SETPIPE_SZ({}) above the deterministic pipe ceiling {}",
3235                    requested, DETERMINISTIC_PIPE_CAPACITY_BYTES
3236                );
3237                Err(Errno::EPERM.into())
3238            }
3239            _ => {
3240                trace!(
3241                    "[detcore-finishme]: fcntl unhandled cases: {:?}",
3242                    call.cmd()
3243                );
3244                Ok(self.record_or_replay(guest, call).await?)
3245            }
3246        }
3247    }
3248
3249    /// ioctl system call
3250    pub async fn handle_ioctl<G: Guest<Self>>(
3251        &self,
3252        guest: &mut G,
3253        call: syscalls::Ioctl,
3254    ) -> Result<i64, Error> {
3255        let fd = call.fd();
3256        let (cloexec, nonblocking) = match call.request() {
3257            // AUTONOMOUS-BOT-IMPLEMENTED
3258            // TODO-HUMAN-REVIEW(PR-1142): Review deterministic SIOCETHTOOL rejection.
3259            // Ethernet link state belongs to the host network namespace and can change between
3260            // runs. Match record/replay's established policy instead of exposing that state.
3261            syscalls::ioctl::Request::SIOCETHTOOL(_) => return Err(Errno::ENODEV.into()),
3262            syscalls::ioctl::Request::FIOCLEX => (Some(true), None),
3263            syscalls::ioctl::Request::FIONCLEX => (Some(false), None),
3264            syscalls::ioctl::Request::FIONBIO(value) => {
3265                let enabled = guest.memory().read_value(value.ok_or(Errno::EFAULT)?)? != 0;
3266                (None, Some(enabled))
3267            }
3268            _ => (None, None),
3269        };
3270
3271        // AUTONOMOUS-BOT-IMPLEMENTED
3272        // TODO-HUMAN-REVIEW(PR-1013): Review logical FIONBIO handling for forced fds.
3273        // Detcore already keeps scheduler-managed fds physically nonblocking. Satisfy
3274        // FIONBIO logically instead of forwarding it: some backends cannot apply the
3275        // ioctl to their proxied pipe fd, and clearing it would violate the scheduler's
3276        // nonblockize-and-retry invariant. This mirrors F_SETFL's forced state split.
3277        if let Some(enabled) = nonblocking {
3278            let (fd_type, physically_nonblocking) = guest
3279                .thread_state()
3280                .with_detfd(fd, |detfd| (detfd.ty(), detfd.physically_nonblocking()))?;
3281            let force_nonblocking = self.cfg.use_nonblocking_sockets()
3282                && !self.cfg.recordreplay_modes
3283                && matches!(fd_type, FdType::Socket | FdType::Pipe | FdType::Eventfd);
3284            if force_nonblocking && physically_nonblocking {
3285                guest.thread_state().with_detfd(fd, |detfd| {
3286                    detfd.set_logical_nonblocking(enabled);
3287                })?;
3288                return Ok(0);
3289            }
3290        }
3291
3292        let result = self.record_or_replay(guest, call).await?;
3293        if cloexec.is_some() || nonblocking.is_some() {
3294            guest.thread_state().with_detfd(fd, |detfd| {
3295                if let Some(enabled) = cloexec {
3296                    detfd.set_cloexec(enabled);
3297                }
3298                if let Some(enabled) = nonblocking {
3299                    detfd.set_nonblocking(enabled);
3300                }
3301            })?;
3302        }
3303        Ok(result)
3304    }
3305
3306    /// statfs: report deterministic filesystem statistics.
3307    ///
3308    /// The kernel's `statfs` reflects live host state: the free-block counts
3309    /// (`f_bfree`, `f_bavail`), the free-inode count (`f_ffree`) and the device
3310    /// id (`f_fsid`) all vary between runs as the underlying host filesystem
3311    /// fills and drains, which makes a bare passthrough diverge under `--verify`
3312    /// (e.g. `tar` calls statfs on its target filesystem). The static geometry
3313    /// of the mount (`f_type`, `f_bsize`, `f_blocks`, `f_namelen`, ...) is
3314    /// reproducible, so we run the real syscall and then canonicalize only the
3315    /// volatile fields.
3316    pub async fn handle_statfs<G: Guest<Self>>(
3317        &self,
3318        guest: &mut G,
3319        call: syscalls::Statfs,
3320    ) -> Result<i64, Error> {
3321        let ret = self.record_or_replay(guest, call).await?;
3322        self.canonicalize_statfs_buf(guest, call.buf())?;
3323        Ok(ret)
3324    }
3325
3326    /// fstatfs: same determinization as [`Self::handle_statfs`], keyed on an fd.
3327    pub async fn handle_fstatfs<G: Guest<Self>>(
3328        &self,
3329        guest: &mut G,
3330        call: syscalls::Fstatfs,
3331    ) -> Result<i64, Error> {
3332        let ret = self.record_or_replay(guest, call).await?;
3333        self.canonicalize_statfs_buf(guest, call.buf())?;
3334        Ok(ret)
3335    }
3336
3337    // AUTONOMOUS-BOT-IMPLEMENTED
3338    // TODO-HUMAN-REVIEW(#1851): Determinized ownership mutation. Emulate the
3339    // IDENTITY half of the call and let Linux answer the ARGUMENT half.
3340    /// The `chown` family (`chown`, `fchown`, `fchownat`, `lchown`).
3341    ///
3342    /// Detcore presents a fixed virtual-root identity, so the *permission*
3343    /// answer must be the one a real root gets: success, for any uid. But root
3344    /// privilege affects only the ownership permission check — it does not
3345    /// waive pathname, descriptor, or flag errors. A real root's
3346    /// `chown("/does/not/exist", 0, 0)` still fails with `ENOENT`.
3347    ///
3348    /// So this does not fabricate a bare `Ok(0)`. It translates the mutation
3349    /// into a side-effect-free metadata lookup with the same target-selection
3350    /// arguments, and reports success only if that lookup succeeds:
3351    ///
3352    /// * `F_GETFL` validates `fchown`'s descriptor and distinguishes an
3353    ///   `O_PATH` descriptor (valid for `fstat`, invalid for `fchown`);
3354    ///   `newfstatat` performs the corresponding path walk for the three
3355    ///   pathname variants, preserving `ENOENT`, `ENOTDIR`, `ELOOP`,
3356    ///   `ENAMETOOLONG`, `EFAULT`, and `EBADF`;
3357    /// * `fchownat` flags are checked explicitly before the lookup, so an
3358    ///   unsupported flag still returns `EINVAL` rather than being accepted by
3359    ///   a metadata syscall with a wider flag vocabulary;
3360    /// * the ownership assignment itself is not performed, but the *other*
3361    ///   consequences of a successful chown are, because Linux applies them
3362    ///   even when ownership does not change. Measured on this host with
3363    ///   `chown(path, -1, -1)`: mode `06755`, `04755` and `02755` all become
3364    ///   `0755`; `02644` keeps `S_ISGID` because the file is not
3365    ///   group-executable; a directory at `06755` keeps both bits; and ctime
3366    ///   moves in every one of those cases, including the plain `0644` file
3367    ///   with nothing to clear. Skipping that is a privilege-containment
3368    ///   regression, not a bookkeeping one: a guest that builds a setuid
3369    ///   binary and chowns it would see the setuid bit survive under hermit
3370    ///   and be cleared on the kernel.
3371    ///
3372    /// Rather than reimplement that rule, the consequence is delegated to the
3373    /// kernel by reissuing the *same* call from the same family with the
3374    /// `(-1, -1)` sentinel, which is precisely the operation whose only effects
3375    /// are `ATTR_CTIME | ATTR_KILL_SUID | ATTR_KILL_SGID`. Delegation gets the
3376    /// directory exemption, the group-executable condition on `S_ISGID`, and
3377    /// symlink handling right for free, and cannot drift from the kernel the
3378    /// way a transcribed rule would.
3379    ///
3380    /// Routed through `record_or_replay` rather than `inject`, so a replay does
3381    /// not need the guest's filesystem to still exist.
3382    ///
3383    /// **Semantic boundary, stated explicitly.** Detcore does not model
3384    /// per-file ownership, so the success is not observable through a later
3385    /// `stat`. A guest that chowns to a foreign uid and reads the owner back
3386    /// sees the unchanged owner — a divergence a single-uid container cannot
3387    /// avoid, and strictly smaller than the status quo in which the guest
3388    /// believes it is root and cannot chown at all.
3389    ///
3390    /// **Residual, also stated.** This emulates ownership permission and target
3391    /// validation, not every write-time filesystem policy. In particular a
3392    /// target on a read-only mount can pass the metadata lookup where a real
3393    /// chown would return `EROFS`. Path resolution also still requires search
3394    /// permission on the parent directories, so `EACCES` remains a function of
3395    /// the host identity under `--no-namespace`. That exposure is shared with
3396    /// every pass-through filesystem syscall (`open`, `stat`, `chmod`) and is
3397    /// not introduced here; it is recorded so the boundary is not overstated.
3398    ///
3399    /// **Residual introduced by the delegation, stated too.** The sentinel call
3400    /// needs the same `inode_owner_or_capable` permission the mode change does,
3401    /// so on a target the guest does not own it returns `EPERM` and that errno
3402    /// is propagated. Reporting a successful chown while silently failing to
3403    /// apply the consequence the kernel guarantees would be the same defect in
3404    /// a narrower form, so this fails closed instead. It is not a regression:
3405    /// that is exactly the case in which the pass-through implementation also
3406    /// returned `EPERM`. The case this change exists to fix — a guest chowning
3407    /// a file it created — is the owning case, and it succeeds.
3408    ///
3409    /// The behavioural contract is bracketed end to end by
3410    /// `hermit-cli/tests/chown_virtual_root_identity.rs`; the unit tests in
3411    /// `syscall_classification` pin membership only and cannot see this
3412    /// function's result.
3413    pub async fn handle_ownership_change_noop<G: Guest<Self>>(
3414        &self,
3415        guest: &mut G,
3416        call: Syscall,
3417    ) -> Result<i64, Error> {
3418        // Unreachable while `is_ownership_change_noop_syscall` and the matches
3419        // below name the same four syscalls. Fail closed if they drift: "not
3420        // attempted" must never be observationally identical to a validated
3421        // emulated success.
3422        if !matches!(
3423            call,
3424            Syscall::Chown(_) | Syscall::Fchown(_) | Syscall::Fchownat(_) | Syscall::Lchown(_)
3425        ) {
3426            warn!(
3427                "ownership-change no-op reached with an unexpected syscall {:?}; \
3428                 refusing unvalidated success",
3429                call.number()
3430            );
3431            return Err(Error::Errno(Errno::ENOSYS));
3432        }
3433
3434        // Flag errors precede path resolution in the kernel, so check first.
3435        if let Syscall::Fchownat(c) = &call {
3436            let allowed = AtFlags::AT_EMPTY_PATH | AtFlags::AT_SYMLINK_NOFOLLOW;
3437            if c.flags().bits() & !allowed.bits() != 0 {
3438                return Err(Error::Errno(Errno::EINVAL));
3439            }
3440        }
3441
3442        // Argument half. `F_GETFL` answers it for `fchown`: it validates the
3443        // descriptor and distinguishes an `O_PATH` descriptor, which `fstat`
3444        // accepts and `fchown` rejects. For the three pathname variants a
3445        // `newfstatat` with the same target-selection arguments performs the
3446        // corresponding path walk.
3447        if let Syscall::Fchown(c) = &call {
3448            let flags = self
3449                .record_or_replay(
3450                    guest,
3451                    syscalls::Fcntl::new().with_fd(c.fd()).with_cmd(F_GETFL),
3452                )
3453                .await?;
3454            if flags & i64::from(OFlag::O_PATH.bits()) != 0 {
3455                return Err(Error::Errno(Errno::EBADF));
3456            }
3457        } else {
3458            let mut stack = guest.stack().await;
3459            let statptr: StatPtr = StatPtr(stack.reserve());
3460            stack.commit()?;
3461
3462            let validate = match &call {
3463                Syscall::Chown(c) => Syscall::Newfstatat(
3464                    syscalls::Newfstatat::new()
3465                        .with_dirfd(libc::AT_FDCWD)
3466                        .with_path(c.path())
3467                        .with_stat(Some(statptr))
3468                        .with_flags(AtFlags::empty()),
3469                ),
3470                Syscall::Lchown(c) => Syscall::Newfstatat(
3471                    syscalls::Newfstatat::new()
3472                        .with_dirfd(libc::AT_FDCWD)
3473                        .with_path(c.path())
3474                        .with_stat(Some(statptr))
3475                        .with_flags(AtFlags::AT_SYMLINK_NOFOLLOW),
3476                ),
3477                Syscall::Fchownat(c) => Syscall::Newfstatat(
3478                    syscalls::Newfstatat::new()
3479                        .with_dirfd(c.dirfd())
3480                        .with_path(c.path())
3481                        .with_stat(Some(statptr))
3482                        .with_flags(c.flags()),
3483                ),
3484                _ => return Err(Error::Errno(Errno::ENOSYS)),
3485            };
3486
3487            // The errno of the side-effect-free validating call is the guest's
3488            // answer; only an actually executed successful validation becomes the
3489            // emulated success. Clear the scratch output on both paths.
3490            let result = self.record_or_replay(guest, validate).await;
3491            guest
3492                .memory()
3493                .write_exact(statptr.0.cast(), &[0; std::mem::size_of::<libc::stat>()])?;
3494            result?;
3495        }
3496
3497        // Metadata half, delegated to the kernel. The identity assignment is
3498        // deliberately not performed, but everything else a successful chown
3499        // does is, by reissuing the same call with the `(-1, -1)` sentinel:
3500        // clear `S_ISUID`, clear `S_ISGID` on a group-executable file, exempt
3501        // directories, and move ctime unconditionally. Letting Linux apply its
3502        // own rule keeps it from drifting here.
3503        const KEEP_ID: libc::uid_t = libc::uid_t::MAX;
3504        let consequence = match &call {
3505            Syscall::Fchown(c) => Syscall::Fchown(
3506                syscalls::Fchown::new()
3507                    .with_fd(c.fd())
3508                    .with_owner(KEEP_ID)
3509                    .with_group(KEEP_ID),
3510            ),
3511            Syscall::Chown(c) => Syscall::Chown(
3512                syscalls::Chown::new()
3513                    .with_path(c.path())
3514                    .with_owner(KEEP_ID)
3515                    .with_group(KEEP_ID),
3516            ),
3517            Syscall::Lchown(c) => Syscall::Lchown(
3518                syscalls::Lchown::new()
3519                    .with_path(c.path())
3520                    .with_owner(KEEP_ID)
3521                    .with_group(KEEP_ID),
3522            ),
3523            Syscall::Fchownat(c) => Syscall::Fchownat(
3524                syscalls::Fchownat::new()
3525                    .with_dirfd(c.dirfd())
3526                    .with_path(c.path())
3527                    .with_owner(KEEP_ID)
3528                    .with_group(KEEP_ID)
3529                    .with_flags(c.flags()),
3530            ),
3531            _ => return Err(Error::Errno(Errno::ENOSYS)),
3532        };
3533        self.record_or_replay(guest, consequence).await?;
3534        Ok(0)
3535    }
3536
3537    /// Overwrite the host-varying fields of a `statfs` result buffer with fixed
3538    /// values, leaving the static per-mount geometry intact. Shared by statfs
3539    /// and fstatfs. A null buffer (only possible on an error return, which the
3540    /// caller has already propagated) is a no-op.
3541    fn canonicalize_statfs_buf<G: Guest<Self>>(
3542        &self,
3543        guest: &mut G,
3544        buf: Option<AddrMut<libc::statfs>>,
3545    ) -> Result<(), Error> {
3546        // Fixed *caps* for the volatile counters. The exact values are
3547        // arbitrary; they only need to be constant so repeated runs agree. We
3548        // clamp each free count to the mount's (static) total so we never report
3549        // the impossible "free > total": a filesystem may be smaller than the
3550        // cap, and some (e.g. overlayfs) report no inode accounting at all
3551        // (`f_files == 0`).
3552        const FREE_BLOCKS_CAP: libc::fsblkcnt_t = 1_000_000;
3553        const FREE_INODES_CAP: libc::fsfilcnt_t = 500_000;
3554
3555        if let Some(buf) = buf {
3556            let mut sf = guest.memory().read_value(buf)?;
3557            let free_blocks = FREE_BLOCKS_CAP.min(sf.f_blocks);
3558            sf.f_bfree = free_blocks;
3559            sf.f_bavail = free_blocks;
3560            // `f_files == 0` means the filesystem does not track inodes; keep the
3561            // free count at 0 rather than inventing free inodes on a mount that
3562            // reports none.
3563            sf.f_ffree = if sf.f_files == 0 {
3564                0
3565            } else {
3566                FREE_INODES_CAP.min(sf.f_files)
3567            };
3568            // f_fsid is a device-dependent filesystem identifier; zero it. An
3569            // all-zero bit pattern is a valid `fsid_t` (a POD id pair).
3570            sf.f_fsid = unsafe { std::mem::zeroed() };
3571            guest.memory().write_value(buf, &sf)?;
3572        }
3573        Ok(())
3574    }
3575
3576    /// dup system call.
3577    pub async fn handle_dup<G: Guest<Self>>(
3578        &self,
3579        guest: &mut G,
3580        call: syscalls::Dup,
3581    ) -> Result<i64, Errno> {
3582        let old_fd = call.oldfd();
3583        let new_fd = self.record_or_replay(guest, call).await? as RawFd;
3584        let replaced = guest
3585            .thread_state_mut()
3586            .dup_fd(old_fd, new_fd, OFlag::empty())?;
3587        if let Some(open_file_id) = replaced {
3588            self.release_port_for_open_file(guest, open_file_id).await;
3589        }
3590        Ok(new_fd as i64)
3591    }
3592
3593    /// dup2 system call.
3594    pub async fn handle_dup2<G: Guest<Self>>(
3595        &self,
3596        guest: &mut G,
3597        call: syscalls::Dup2,
3598    ) -> Result<i64, Errno> {
3599        let old_fd = call.oldfd();
3600        let new_fd = call.newfd();
3601        let res = self.record_or_replay(guest, call).await?;
3602        let replaced = guest
3603            .thread_state_mut()
3604            .dup_fd(old_fd, new_fd, OFlag::empty())?;
3605        if let Some(open_file_id) = replaced {
3606            self.release_port_for_open_file(guest, open_file_id).await;
3607        }
3608        Ok(res)
3609    }
3610
3611    /// dup3 system call.
3612    pub async fn handle_dup3<G: Guest<Self>>(
3613        &self,
3614        guest: &mut G,
3615        call: syscalls::Dup3,
3616    ) -> Result<i64, Errno> {
3617        let old_fd = call.oldfd();
3618        let new_fd = call.newfd();
3619        let flags = call.flags();
3620        let res = self.record_or_replay(guest, call).await?;
3621        let replaced = guest.thread_state_mut().dup_fd(old_fd, new_fd, flags)?;
3622        if let Some(open_file_id) = replaced {
3623            self.release_port_for_open_file(guest, open_file_id).await;
3624        }
3625        Ok(res)
3626    }
3627
3628    /// pipe2 system call.
3629    pub async fn handle_pipe2<G: Guest<Self>>(
3630        &self,
3631        guest: &mut G,
3632        call: syscalls::Pipe2,
3633    ) -> Result<i64, Error> {
3634        // Pipes are unambiguously container-internal: both endpoints are owned by
3635        // guest processes. Make them physically nonblocking whenever we sequentialize
3636        // threads -- INCLUDING record/replay modes. This lets a potentially-blocking
3637        // pipe read follow the deterministic nonblockize-and-retry (InternalIOPolling)
3638        // path instead of being descheduled as BlockingExternalIO. A pipe reader and
3639        // its paired writer are NOT independent, so treating an internal pipe as
3640        // "external blocking IO" (safe to run in the background and rejoin whenever)
3641        // deadlocks the sequentialized scheduler in R/R (the documented pipe hang). The
3642        // physical O_NONBLOCK is Detcore-internal and invisible to the guest (F_GETFL is
3643        // virtualized), and mirrors what `hermit run --strict` already does for pipes.
3644        let internally_nonblocking = self.cfg.use_nonblocking_sockets();
3645        let injected = if internally_nonblocking {
3646            call.with_flags(call.flags() | OFlag::O_NONBLOCK)
3647        } else {
3648            call
3649        };
3650        // NO PRE-CALL READ OF `pipefd`. This is load-bearing on three backends, not a style
3651        // choice. `record_or_replay` below returns early on failure, so every guest memory
3652        // access AFTER it runs only when pipe2 SUCCEEDED -- and a successful pipe2 means the
3653        // kernel itself wrote two ints there, which proves the address valid. A pre-call
3654        // snapshot would be the only access to a kernel-UNVALIDATED address, and `LocalMemory`
3655        // (reverie-dbt, reverie-e9patch, reverie-liteinst) implements reads as an unsafe
3656        // `copy_nonoverlapping` that always returns `Ok`: a bad pointer is a hardware fault
3657        // that `Result::ok` cannot catch, so the guest would die with SIGSEGV before Linux
3658        // could report EFAULT. Letting the kernel touch `pipefd` first is also what keeps its
3659        // argument-validation precedence intact -- flags are checked before the pointer, so a
3660        // bad pointer with bad flags is EINVAL and with good flags EFAULT.
3661        // A C guest asserting that precedence directly is being added separately.
3662        let res = self.record_or_replay(guest, injected).await?;
3663        let memory = guest.memory();
3664
3665        if let Some(pipefd) = call.pipefd() {
3666            let fds: [i32; 2] = memory.read_value(pipefd)?;
3667            if internally_nonblocking {
3668                let capacity_result = guest
3669                    .inject(
3670                        syscalls::Fcntl::new()
3671                            .with_fd(fds[0])
3672                            .with_cmd(F_SETPIPE_SZ(DETERMINISTIC_PIPE_CAPACITY_BYTES)),
3673                    )
3674                    .await;
3675                if let Some(failure) = pipe_capacity_failure(fds, capacity_result) {
3676                    // Release the descriptors Linux already created, THEN stop the run.
3677                    //
3678                    // Returning `Err` here -- what this code did before -- unwinds without
3679                    // closing them. `pipe2` has already succeeded, so both descriptors are
3680                    // live in the guest and are not yet registered with `add_fd`, which means
3681                    // Detcore's own bookkeeping never learns they exist. Closing them is the
3682                    // only way the guest's descriptor table matches Detcore's model.
3683                    //
3684                    // We do NOT fabricate a `pipe2` errno. Linux leaves `pipefd` untouched
3685                    // when `pipe2` fails, so inventing a failure would oblige us to restore
3686                    // the caller's buffer, which needs the pre-call snapshot the comment above
3687                    // explains we must never take. And returning success with an unpinned pipe
3688                    // silently restores exactly the host-dependent capacity this path exists
3689                    // to remove. An unpinnable pipe means determinism is unavailable for this
3690                    // run, so fail closed and loudly rather than quietly.
3691                    //
3692                    // Defensive, not expected: pinning on a freshly created EMPTY pipe is a
3693                    // shrink or a no-op. EBUSY needs buffered data and EPERM needs to exceed
3694                    // `pipe-max-size`; neither can hold here.
3695                    for close in failure.close_syscalls() {
3696                        let _ = guest.inject(close).await;
3697                    }
3698                    error!(
3699                        "[detcore] cannot pin scheduler-managed pipe to {} bytes (fds {:?}): {}. \
3700                         Determinism is unavailable for this run.",
3701                        DETERMINISTIC_PIPE_CAPACITY_BYTES, failure.created_fds, failure.error,
3702                    );
3703                    // Fail-closed policy: determinism is unavailable for this run.
3704                    unrecoverable_shutdown(guest, detcore_model::HERMIT_POLICY_REFUSAL_EXIT).await;
3705                }
3706            }
3707            self.add_fd(guest, fds[0], call.flags(), FdType::Pipe)
3708                .await?;
3709            self.add_fd(guest, fds[1], call.flags(), FdType::Pipe)
3710                .await?;
3711            if internally_nonblocking {
3712                self.maybe_set_nonblocking_fd(guest, fds[0]);
3713                self.maybe_set_nonblocking_fd(guest, fds[1]);
3714            }
3715        }
3716
3717        Ok(res)
3718    }
3719
3720    /// utime syscall: update access/modification time on a file
3721    pub async fn handle_utime<G: Guest<Self>>(
3722        &self,
3723        guest: &mut G,
3724        call: syscalls::Utime,
3725    ) -> Result<i64, Errno> {
3726        let times = match call.times() {
3727            None => {
3728                let now: Timespec = thread_observe_time(guest).await.into();
3729                [now, now]
3730            }
3731            Some(times) => {
3732                let utimbuf = guest.memory().read_value(times)?;
3733                [
3734                    Timespec {
3735                        tv_sec: utimbuf.actime,
3736                        tv_nsec: 0,
3737                    },
3738                    Timespec {
3739                        tv_sec: utimbuf.modtime,
3740                        tv_nsec: 0,
3741                    },
3742                ]
3743            }
3744        };
3745
3746        let utimensat = syscalls::Utimensat::new()
3747            .with_dirfd(libc::AT_FDCWD)
3748            .with_path(call.path());
3749
3750        self.set_file_times(guest, utimensat, Some(times)).await
3751    }
3752
3753    /// utimes syscall
3754    pub async fn handle_utimes<G: Guest<Self>>(
3755        &self,
3756        guest: &mut G,
3757        call: syscalls::Utimes,
3758    ) -> Result<i64, Errno> {
3759        let utimensat = syscalls::Utimensat::new()
3760            .with_dirfd(libc::AT_FDCWD)
3761            .with_path(call.filename());
3762
3763        match call.times() {
3764            None => {
3765                let now: Timespec = thread_observe_time(guest).await.into();
3766                self.set_file_times(guest, utimensat, Some([now, now]))
3767                    .await
3768            }
3769            Some(times) => {
3770                // Convert the timeval array to a timespec array.
3771                let mut memory = guest.memory();
3772                let tvs = memory.read_value(times)?;
3773                let tp: Addr<[Timespec; 2]> = times.cast();
3774
3775                // Safety: The address could point to read-only memory and the
3776                // write below could fail.
3777                let tp = unsafe { tp.into_mut() };
3778
3779                memory.write_value(tp, &[tvs[0].into(), tvs[1].into()])?;
3780                self.set_file_times(guest, utimensat.with_times(Some(tp.into())), None)
3781                    .await
3782            }
3783        }
3784    }
3785
3786    /// ustimensat syscall
3787    pub async fn handle_utimensat<G: Guest<Self>>(
3788        &self,
3789        guest: &mut G,
3790        call: syscalls::Utimensat,
3791    ) -> Result<i64, Errno> {
3792        self.set_file_times(guest, call, None).await
3793    }
3794
3795    /// Performs a utimensat call and copies the mtime it sets into the virtual
3796    /// mtime. With `staged` the times are first pushed onto the guest's scratch
3797    /// stack and replace the call's `times` pointer; utime and utimes(NULL)
3798    /// build their times that way.
3799    async fn set_file_times<G: Guest<Self>>(
3800        &self,
3801        guest: &mut G,
3802        call: syscalls::Utimensat,
3803        staged: Option<[Timespec; 2]>,
3804    ) -> Result<i64, Errno> {
3805        if !guest.config().virtualize_metadata {
3806            return self.utimensat_without_lookup(guest, call, staged).await;
3807        }
3808        // Without thread sequentialization another guest thread can swap the
3809        // target away and back around the call, so that both lookups name a
3810        // file the kernel did not update. No lookup can tell the two apart,
3811        // so the virtual mtime is left alone.
3812        if !guest.config().sequentialize_threads {
3813            return self.utimensat_without_lookup(guest, call, staged).await;
3814        }
3815        // The scratch stack starts 128 bytes below the stack pointer and its
3816        // addresses are computed by subtraction, so a stack pointer too close
3817        // to zero to hold the staged times and the lookup buffer would stop
3818        // Hermit rather than reach Linux.
3819        let scratch = 128
3820            + staged.map_or(0, |_| std::mem::size_of::<[Timespec; 2]>())
3821            + std::mem::size_of::<libc::stat>();
3822        if usize::try_from(guest.regs().await.rsp).map_or(true, |rsp| rsp <= scratch) {
3823            info!(
3824                "Guest stack pointer cannot hold the utimensat target lookup; \
3825                 leaving the virtual mtime unchanged."
3826            );
3827            return self.utimensat_without_lookup(guest, call, staged).await;
3828        }
3829
3830        // The staged times and the buffer for the target lookups share one
3831        // scratch stack, and its guard is held until the last injected syscall
3832        // has run, so that neither the kernel's read of the times nor a lookup
3833        // finds the other's bytes, or a restored stack, at its address.
3834        let mut stack = guest.stack().await;
3835        let staged_call = match staged {
3836            Some(times) => call.with_times(Some(stack.push(times))),
3837            None => call,
3838        };
3839        let statptr: StatPtr = StatPtr(stack.reserve());
3840        // The scratch stack lies below the guest's red zone, where a raw
3841        // syscall may keep its own path or times. Writing the lookup buffer
3842        // over either would change the call before Linux reads it.
3843        if utimensat_input_overlaps(&guest.memory(), &call, staged.is_none(), statptr) {
3844            info!(
3845                "utimensat inputs overlap the target lookup buffer; \
3846                 leaving the virtual mtime unchanged."
3847            );
3848            drop(stack);
3849            return self.utimensat_without_lookup(guest, call, staged).await;
3850        }
3851        // The buffer's address range can still share its pages with guest data
3852        // through a second shared mapping, which no address comparison sees.
3853        // Its bytes are saved here and put back after the commit and after
3854        // each lookup, so the lookups and the call read their inputs unchanged
3855        // and the guest keeps its memory.
3856        let mut saved = [0u8; std::mem::size_of::<libc::stat>()];
3857        if guest
3858            .memory()
3859            .read_exact(statptr.0.cast(), &mut saved)
3860            .is_err()
3861        {
3862            info!(
3863                "Guest stack scratch cannot hold the utimensat target lookup; \
3864                 leaving the virtual mtime unchanged."
3865            );
3866            drop(stack);
3867            return self.utimensat_without_lookup(guest, call, staged).await;
3868        }
3869        let _guard = match stack.commit() {
3870            Ok(guard) => guard,
3871            // The lookup buffer only serves the virtual update, so a scratch
3872            // stack that cannot hold it, for example one next to the guard
3873            // page, must not keep the guest's own call from reaching Linux.
3874            // The commit writes page by page, so it can fail after writing a
3875            // lower writable page; those bytes are put back first.
3876            Err(_) => {
3877                info!(
3878                    "Guest stack scratch cannot hold the utimensat target lookup; \
3879                     leaving the virtual mtime unchanged."
3880                );
3881                restore_lookup_buffer(&mut guest.memory(), statptr, &saved);
3882                return self.utimensat_without_lookup(guest, call, staged).await;
3883            }
3884        };
3885        restore_lookup_buffer(&mut guest.memory(), statptr, &saved);
3886        let call = staged_call;
3887
3888        // The kernel applies the new times to the real file, but the guest
3889        // observes the virtual mtime, which otherwise only moves on writes. Copy
3890        // the requested mtime into it so that `tar` extraction, `cp -p` and
3891        // `touch -r` restore a file's mtime and `make` compares the times the
3892        // build asked for rather than the order in which files were unpacked.
3893        //
3894        // Nothing below may change the syscall's result: an unreadable `times`
3895        // or a failed lookup only skips the virtual update, and the kernel
3896        // reports its own error for the call itself.
3897        let mtime = match (staged, call.times()) {
3898            (Some([_, mtime]), _) => Some(mtime),
3899            (None, None) => Some(Timespec {
3900                tv_sec: 0,
3901                tv_nsec: libc::UTIME_NOW,
3902            }),
3903            (None, Some(times)) => guest
3904                .memory()
3905                .read_value(times)
3906                .ok()
3907                .map(|[_, mtime]| mtime),
3908        }
3909        .filter(|mtime| mtime.tv_nsec != libc::UTIME_OMIT);
3910        let before = match mtime {
3911            Some(_) => self.utimensat_target(guest, &call, statptr).await,
3912            None => None,
3913        };
3914        restore_lookup_buffer(&mut guest.memory(), statptr, &saved);
3915
3916        let res = self.record_or_replay(guest, call).await?;
3917
3918        // Update only the inode the kernel modified. No other guest thread
3919        // runs during the call, but a process outside the container can still
3920        // rename, unlink or replace the target; the target is resolved before
3921        // and after the call, and the update is skipped unless both name the
3922        // same file. Equal lookups alone do not show that the kernel updated
3923        // that file: the name may have been swapped away and back during the
3924        // call, or an inode number reused. So an explicit mtime is copied only
3925        // when the file holds it, truncated to the filesystem's granularity,
3926        // and the virtual mtime takes the value the kernel stored, which is
3927        // what stat reports on Linux. `UTIME_NOW` has no such witness.
3928        let (Some(mtime), Some(before)) = (mtime, before) else {
3929            return Ok(res);
3930        };
3931        let after = self.utimensat_target(guest, &call, statptr).await;
3932        restore_lookup_buffer(&mut guest.memory(), statptr, &saved);
3933        let Some(after) = after else {
3934            return Ok(res);
3935        };
3936        if (after.st_dev, after.st_ino) != (before.st_dev, before.st_ino) {
3937            return Ok(res);
3938        }
3939        if mtime.tv_nsec == libc::UTIME_NOW {
3940            touch_file(guest, after.st_ino).await;
3941            return Ok(res);
3942        }
3943        const NANOS_PER_SEC: i128 = 1_000_000_000;
3944        let requested = i128::from(mtime.tv_sec) * NANOS_PER_SEC + i128::from(mtime.tv_nsec);
3945        let stored = i128::from(after.st_mtime) * NANOS_PER_SEC + i128::from(after.st_mtime_nsec);
3946        // Linux truncates the requested time down to the filesystem's
3947        // granularity, at most a second on the filesystems builds use.
3948        if (0..NANOS_PER_SEC).contains(&(requested - stored)) {
3949            let nanos = u64::try_from(stored.max(0)).unwrap_or(u64::MAX);
3950            set_file_mtime(guest, after.st_ino, LogicalTime::from_nanos(nanos)).await;
3951        }
3952        Ok(res)
3953    }
3954
3955    /// Performs a utimensat call without the virtual mtime update. Staged
3956    /// times are then the only scratch-stack allocation, as they were before
3957    /// the update existed.
3958    async fn utimensat_without_lookup<G: Guest<Self>>(
3959        &self,
3960        guest: &mut G,
3961        call: syscalls::Utimensat,
3962        staged: Option<[Timespec; 2]>,
3963    ) -> Result<i64, Errno> {
3964        let Some(times) = staged else {
3965            return self.record_or_replay(guest, call).await;
3966        };
3967        let mut stack = guest.stack().await;
3968        let call = call.with_times(Some(stack.push(times)));
3969        let _guard = stack.commit()?;
3970        self.record_or_replay(guest, call).await
3971    }
3972
3973    /// The metadata of the file a utimensat call targets, with the same target
3974    /// selection: the descriptor itself for `futimens` (a NULL path), else a
3975    /// path walk honoring the flags utimensat accepts. `None` if the lookup
3976    /// fails.
3977    async fn utimensat_target<G: Guest<Self>>(
3978        &self,
3979        guest: &mut G,
3980        call: &syscalls::Utimensat,
3981        statptr: StatPtr<'_>,
3982    ) -> Option<libc::stat> {
3983        let lookup = match call.path() {
3984            None => Syscall::Fstat(
3985                syscalls::Fstat::new()
3986                    .with_fd(call.dirfd())
3987                    .with_stat(Some(statptr)),
3988            ),
3989            Some(path) => {
3990                let allowed = libc::AT_SYMLINK_NOFOLLOW | libc::AT_EMPTY_PATH;
3991                Syscall::Newfstatat(
3992                    syscalls::Newfstatat::new()
3993                        .with_dirfd(call.dirfd())
3994                        .with_path(Some(path))
3995                        .with_stat(Some(statptr))
3996                        .with_flags(AtFlags::from_bits_truncate(call.flags() & allowed)),
3997                )
3998            }
3999        };
4000        self.record_or_replay(guest, lookup).await.ok()?;
4001        statptr.read(&guest.memory()).ok()
4002    }
4003
4004    /// socket system call.
4005    pub async fn handle_socket<G: Guest<Self>>(
4006        &self,
4007        guest: &mut G,
4008        call: syscalls::Socket,
4009    ) -> Result<i64, Error> {
4010        // The socket syscall itself is not blocking, but we must decide whether to make the socket
4011        // returned physically nonblocking.
4012        if !self.cfg.sequentialize_threads || self.cfg.recordreplay_modes {
4013            // Allow possibly blocking syscall in record mode
4014            let fd = self.record_or_replay(guest, call).await? as RawFd;
4015            self.add_fd(
4016                guest,
4017                fd,
4018                OFlag::from_bits_truncate(call.r#type()),
4019                FdType::Socket,
4020            )
4021            .await?;
4022            self.mark_sock_diag_fd(guest, fd, &call);
4023            Ok(fd as i64)
4024        } else {
4025            // Under run mode, force all sockets to be registered to be nonblocking in the OS:
4026            let call2 = if self.cfg.use_nonblocking_sockets() {
4027                call.with_type(call.r#type() | libc::SOCK_NONBLOCK)
4028            } else {
4029                call
4030            };
4031            let fd = self.record_or_replay(guest, call2).await? as RawFd; // Cannot hang.
4032            self.add_fd(
4033                guest,
4034                fd,
4035                OFlag::from_bits_truncate(
4036                    call.r#type() & (libc::SOCK_NONBLOCK | libc::SOCK_CLOEXEC),
4037                ),
4038                FdType::Socket,
4039            )
4040            .await?;
4041            self.maybe_set_nonblocking_fd(guest, fd);
4042            self.mark_sock_diag_fd(guest, fd, &call);
4043
4044            Ok(fd as i64)
4045        }
4046    }
4047
4048    // AUTONOMOUS-BOT-IMPLEMENTED
4049    // TODO-HUMAN-REVIEW(PR-1064)
4050    /// Flag `AF_NETLINK`/`NETLINK_SOCK_DIAG` sockets so `handle_recvmsg`
4051    /// determinizes the socket inode numbers carried by their binary dump
4052    /// replies (see `crate::sock_diag`). Best-effort: if the descriptor lookup
4053    /// fails the reply is simply left unsanitized.
4054    fn mark_sock_diag_fd<G: Guest<Self>>(&self, guest: &mut G, fd: RawFd, call: &syscalls::Socket) {
4055        if call.family() != libc::AF_NETLINK {
4056            return;
4057        }
4058        if call.protocol() == libc::NETLINK_SOCK_DIAG {
4059            let _ = guest
4060                .thread_state()
4061                .with_detfd(fd, |detfd| detfd.set_sock_diag());
4062        }
4063        // TODO-HUMAN-REVIEW(PR-2478)
4064        // NETLINK_ROUTE link dumps carry live interface counters. They were
4065        // invisible until IO-buffer hashing went on by default, because the
4066        // reply's LENGTH and return value are identical between runs and only
4067        // the payload bytes move.
4068        if call.protocol() == libc::NETLINK_ROUTE {
4069            let _ = guest
4070                .thread_state()
4071                .with_detfd(fd, |detfd| detfd.set_netlink_route());
4072        }
4073    }
4074
4075    /// socketpair system call.
4076    pub async fn handle_socketpair<G: Guest<Self>>(
4077        &self,
4078        guest: &mut G,
4079        call: syscalls::Socketpair,
4080    ) -> Result<i64, Error> {
4081        let call2 = if self.cfg.sequentialize_threads && !self.cfg.debug_externalize_sockets {
4082            call.with_type(call.r#type() | libc::SOCK_NONBLOCK)
4083        } else {
4084            call
4085        };
4086        let res = self.record_or_replay(guest, call2).await?;
4087        if let Some(usockvec) = call.usockvec() {
4088            let memory = guest.memory();
4089            let fds: [i32; 2] = memory.read_value(usockvec)?;
4090
4091            // Logical flags are as requested:
4092            self.add_fd(
4093                guest,
4094                fds[0],
4095                OFlag::from_bits_truncate(
4096                    call.r#type() & (libc::SOCK_NONBLOCK | libc::SOCK_CLOEXEC),
4097                ),
4098                FdType::Socket,
4099            )
4100            .await?;
4101            self.add_fd(
4102                guest,
4103                fds[1],
4104                OFlag::from_bits_truncate(
4105                    call.r#type() & (libc::SOCK_NONBLOCK | libc::SOCK_CLOEXEC),
4106                ),
4107                FdType::Socket,
4108            )
4109            .await?;
4110
4111            self.maybe_set_nonblocking_fd(guest, fds[0]);
4112            self.maybe_set_nonblocking_fd(guest, fds[1]);
4113        }
4114        Ok(res)
4115    }
4116
4117    // AUTONOMOUS-BOT-IMPLEMENTED
4118    // TODO-HUMAN-REVIEW(#663)
4119    /// Apply a socket option to an already tracked socket. Record mode captures
4120    /// the result; replay re-applies a successful option before later socket I/O,
4121    /// which remains mediated by Detcore's nonblocking scheduler paths.
4122    pub async fn handle_setsockopt<G: Guest<Self>>(
4123        &self,
4124        guest: &mut G,
4125        call: syscalls::Setsockopt,
4126    ) -> Result<i64, Error> {
4127        Ok(self.record_or_replay(guest, call).await?)
4128    }
4129
4130    // AUTONOMOUS-BOT-IMPLEMENTED
4131    // TODO-HUMAN-REVIEW(#663)
4132    /// Transition an already tracked socket into listening state.
4133    pub async fn handle_listen<G: Guest<Self>>(
4134        &self,
4135        guest: &mut G,
4136        call: syscalls::Listen,
4137    ) -> Result<i64, Error> {
4138        Ok(self.record_or_replay(guest, call).await?)
4139    }
4140
4141    // AUTONOMOUS-BOT-IMPLEMENTED
4142    // TODO-HUMAN-REVIEW(#663)
4143    /// Return the local address of a tracked socket.
4144    pub async fn handle_getsockname<G: Guest<Self>>(
4145        &self,
4146        guest: &mut G,
4147        call: syscalls::Getsockname,
4148    ) -> Result<i64, Error> {
4149        Ok(self.record_or_replay(guest, call).await?)
4150    }
4151
4152    // AUTONOMOUS-BOT-IMPLEMENTED
4153    // TODO-HUMAN-REVIEW(#663)
4154    /// Return the peer address of a tracked socket.
4155    pub async fn handle_getpeername<G: Guest<Self>>(
4156        &self,
4157        guest: &mut G,
4158        call: syscalls::Getpeername,
4159    ) -> Result<i64, Error> {
4160        Ok(self.record_or_replay(guest, call).await?)
4161    }
4162
4163    // AUTONOMOUS-BOT-IMPLEMENTED
4164    // TODO-HUMAN-REVIEW(#663)
4165    /// Return an option value from a tracked socket. Hermit only promises normal
4166    /// run determinism for isolated guest networking; record/replay captures the
4167    /// result when external socket state is part of the recording boundary.
4168    pub async fn handle_getsockopt<G: Guest<Self>>(
4169        &self,
4170        guest: &mut G,
4171        call: syscalls::Getsockopt,
4172    ) -> Result<i64, Error> {
4173        // TODO-HUMAN-REVIEW(PR-894): Review deterministic network-namespace identity.
4174        let requested_length =
4175            if call.level() == libc::SOL_SOCKET && call.optname() == libc::SO_NETNS_COOKIE {
4176                let fd_type = guest
4177                    .thread_state()
4178                    .with_detfd(call.fd(), |detfd| detfd.ty())?;
4179                if fd_type == FdType::Socket {
4180                    call.optlen()
4181                        .map(|length| guest.memory().read_value(length))
4182                        .transpose()?
4183                } else {
4184                    None
4185                }
4186            } else {
4187                None
4188            };
4189
4190        // TODO-HUMAN-REVIEW(PR-886): Review deterministic SO_COOKIE identities.
4191        let deterministic_cookie =
4192            if call.level() == libc::SOL_SOCKET && call.optname() == libc::SO_COOKIE {
4193                let requested_length = call
4194                    .optlen()
4195                    .map(|length| guest.memory().read_value(length))
4196                    .transpose()?;
4197                let open_file_id = guest
4198                    .thread_state()
4199                    .with_detfd(call.fd(), |detfd| detfd.open_file_id())?;
4200                Some((open_file_id.deterministic_socket_cookie(), requested_length))
4201            } else {
4202                None
4203            };
4204
4205        let result = self.record_or_replay(guest, call).await?;
4206
4207        // TODO-HUMAN-REVIEW(PR-898): Hermit exposes one virtual CPU, so do not
4208        // leak the host CPU that processed a socket's most recent packet.
4209        if result == 0
4210            && call.level() == libc::SOL_SOCKET
4211            && call.optname() == libc::SO_INCOMING_CPU
4212            && let (Some(optval), Some(optlen)) = (call.optval(), call.optlen())
4213        {
4214            let returned_len: libc::socklen_t = guest.memory().read_value(optlen)?;
4215            let zero_cpu = 0_i32.to_ne_bytes();
4216            let returned_len = (returned_len as usize).min(zero_cpu.len());
4217            guest
4218                .memory()
4219                .write_exact(optval.cast::<u8>(), &zero_cpu[..returned_len])?;
4220        }
4221        if result == 0
4222            && call.level() == libc::IPPROTO_TCP
4223            && call.optname() == libc::TCP_INFO
4224            && let (Some(optval), Some(optlen)) = (call.optval(), call.optlen())
4225        {
4226            let returned_len: libc::socklen_t = guest.memory().read_value(optlen)?;
4227            let mut info = vec![0; returned_len as usize];
4228            let optval = optval.cast::<u8>();
4229            guest.memory().read_exact(optval, info.as_mut_slice())?;
4230            canonicalize_tcp_info(&mut info);
4231            guest.memory().write_exact(optval, info.as_slice())?;
4232        }
4233        if let Some(requested_length) = requested_length
4234            && let Some(value) = call.optval()
4235        {
4236            let bytes = DETERMINISTIC_NETNS_COOKIE.to_ne_bytes();
4237            let write_length = (requested_length as usize).min(bytes.len());
4238            guest
4239                .memory()
4240                .write_exact(value.cast(), &bytes[..write_length])?;
4241        }
4242        if let Some((cookie, Some(requested_length))) = deterministic_cookie
4243            && let Some(value) = call.optval()
4244        {
4245            let bytes = cookie.to_ne_bytes();
4246            let write_length = (requested_length as usize).min(bytes.len());
4247            guest
4248                .memory()
4249                .write_exact(value.cast(), &bytes[..write_length])?;
4250        }
4251        Ok(result)
4252    }
4253    // AUTONOMOUS-BOT-IMPLEMENTED
4254    // TODO-HUMAN-REVIEW(#818)
4255    /// Half-close the read and/or write direction of an already tracked socket.
4256    /// shutdown never blocks and returns no data; its effect is deterministic
4257    /// given the container's socket state, so it forwards via record_or_replay
4258    /// exactly like the rest of the socket family (KVM ratchet round 12).
4259    pub async fn handle_shutdown<G: Guest<Self>>(
4260        &self,
4261        guest: &mut G,
4262        call: syscalls::Shutdown,
4263    ) -> Result<i64, Error> {
4264        Ok(self.record_or_replay(guest, call).await?)
4265    }
4266
4267    /// bind system call.
4268    pub async fn handle_bind<G: Guest<Self>>(
4269        &self,
4270        guest: &mut G,
4271        call: syscalls::Bind,
4272    ) -> Result<i64, Error> {
4273        // WIP!
4274        if guest.config().sched_heuristic == SchedHeuristic::ConnectBind {
4275            trace!("Scheduling heuristic: reprioritizing bind");
4276            let resource = ResourceID::PriorityChangePoint(
4277                LAST_PRIORITY,
4278                guest.thread_state().thread_logical_time.as_nanos(),
4279                guest.thread_state().committed_clock_value,
4280                Vec::new(),
4281            );
4282            let req = guest.thread_state().mk_request(resource, Permission::W);
4283            resource_request(guest, req).await;
4284        }
4285        let addr = call.umyaddr().ok_or(Errno::EFAULT)?;
4286        let sock_fd = call.fd();
4287        let open_file_id = guest
4288            .thread_state()
4289            .with_detfd(sock_fd, |detfd| detfd.open_file_id())?;
4290
4291        let sockaddr_family = guest.memory().read_value(addr.cast::<u16>())?;
4292        // AUTONOMOUS-BOT-IMPLEMENTED
4293        // TODO-HUMAN-REVIEW(PR-872): Review deterministic AF_UNIX autobind identity.
4294        if sockaddr_family == libc::AF_UNIX as u16
4295            && call.addrlen() == std::mem::offset_of!(libc::sockaddr_un, sun_path) as i32
4296        {
4297            let resp = send_and_update_time(guest, GlobalRequest::RequestPort(open_file_id)).await;
4298            let port = match resp.1 {
4299                GlobalResponse::RequestPort(port) => port,
4300                GlobalResponse::PortFull => {
4301                    return Err(reverie::Error::from(nix::errno::Errno::EADDRINUSE));
4302                }
4303                _ => unreachable!(),
4304            };
4305
4306            let mut stack = guest.stack().await;
4307            let autobind_addr: AddrMut<libc::sockaddr_un> = stack.reserve();
4308            let _stack_guard = stack.commit()?;
4309            guest
4310                .memory()
4311                .write_value(autobind_addr, &unix_autobind_address(port))?;
4312            let deterministic_bind = call
4313                .with_umyaddr(Some(autobind_addr.cast()))
4314                .with_addrlen(unix_autobind_addrlen());
4315            return Ok(self.record_or_replay(guest, deterministic_bind).await?);
4316        // AUTONOMOUS-BOT-IMPLEMENTED
4317        // TODO-HUMAN-REVIEW(PR-880): Review deterministic Netlink autobind identities.
4318        } else if sockaddr_family == libc::AF_NETLINK as u16
4319            && call.addrlen() >= std::mem::size_of::<libc::sockaddr_nl>() as i32
4320        {
4321            let mut sockaddr_nl: libc::sockaddr_nl = guest
4322                .memory()
4323                .read_value(addr.cast::<libc::sockaddr_nl>())?;
4324            if sockaddr_nl.nl_pid == 0 {
4325                let resp =
4326                    send_and_update_time(guest, GlobalRequest::RequestPort(open_file_id)).await;
4327                match resp.1 {
4328                    GlobalResponse::RequestPort(port) => {
4329                        sockaddr_nl.nl_pid = DETERMINISTIC_NETLINK_PORT_ID_BASE | u32::from(port);
4330                        let mut stack = guest.stack().await;
4331                        let deterministic_addr: AddrMut<libc::sockaddr_nl> = stack.reserve();
4332                        let _stack_guard = stack.commit()?;
4333                        guest
4334                            .memory()
4335                            .write_value(deterministic_addr, &sockaddr_nl)?;
4336                        let deterministic_bind = call.with_umyaddr(Some(deterministic_addr.cast()));
4337                        return Ok(self.record_or_replay(guest, deterministic_bind).await?);
4338                    }
4339                    GlobalResponse::PortFull => {
4340                        return Err(reverie::Error::from(nix::errno::Errno::EADDRINUSE));
4341                    }
4342                    _ => unreachable!(),
4343                }
4344            }
4345        } else if sockaddr_family == libc::AF_INET as u16 {
4346            // For IPv4
4347            let mut sockaddr_in: libc::sockaddr_in = guest
4348                .memory()
4349                .read_value(addr.cast::<libc::sockaddr_in>())?;
4350
4351            let port = sockaddr_in.sin_port.to_be();
4352            let ipaddr = Ipv4Addr::from(sockaddr_in.sin_addr.s_addr);
4353            if port != 0 {
4354                if guest.config().warn_non_zero_binds {
4355                    warn!(
4356                        "Analyze Networking: Non-zero port detected: {:?}:{:?}",
4357                        ipaddr, port
4358                    );
4359                }
4360                // Send RPC to make sure already used ports are not used.
4361                let resp =
4362                    send_and_update_time(guest, GlobalRequest::AddUsedPort(port, open_file_id))
4363                        .await;
4364                match resp.1 {
4365                    GlobalResponse::AddUsedPort => {
4366                        trace!("Added to used port {}", port);
4367                    }
4368                    _ => unreachable!(),
4369                }
4370            } else {
4371                // Request a determinzed port
4372                let resp =
4373                    send_and_update_time(guest, GlobalRequest::RequestPort(open_file_id)).await;
4374                match resp.1 {
4375                    GlobalResponse::RequestPort(port_assigned) => {
4376                        sockaddr_in.sin_port = port_assigned.to_be();
4377                        guest
4378                            .memory()
4379                            .write_value(addr.cast::<libc::sockaddr_in>(), &sockaddr_in)?;
4380                    }
4381                    GlobalResponse::PortFull => {
4382                        return Err(reverie::Error::from(nix::errno::Errno::EADDRINUSE));
4383                    }
4384                    _ => unreachable!(),
4385                }
4386            }
4387        } else if sockaddr_family == libc::AF_INET6 as u16 {
4388            // For IPv6
4389            let mut sockfaddr_in: libc::sockaddr_in6 = guest
4390                .memory()
4391                .read_value(addr.cast::<libc::sockaddr_in6>())?;
4392            let port = sockfaddr_in.sin6_port.to_be();
4393            let ipaddr = Ipv6Addr::from(sockfaddr_in.sin6_addr.s6_addr);
4394            if port != 0 {
4395                if guest.config().warn_non_zero_binds {
4396                    warn!(
4397                        "Analyze Networking: Non-zero port detected: {:?}:{:?}",
4398                        ipaddr, port
4399                    );
4400                }
4401                let resp =
4402                    send_and_update_time(guest, GlobalRequest::AddUsedPort(port, open_file_id))
4403                        .await;
4404                match resp.1 {
4405                    GlobalResponse::AddUsedPort => {
4406                        trace!("Added to used port {}", port);
4407                    }
4408                    _ => unreachable!(),
4409                }
4410            } else {
4411                let resp =
4412                    send_and_update_time(guest, GlobalRequest::RequestPort(open_file_id)).await;
4413                match resp.1 {
4414                    GlobalResponse::RequestPort(port_assigned) => {
4415                        sockfaddr_in.sin6_port = port_assigned.to_be();
4416                        guest
4417                            .memory()
4418                            .write_value(addr.cast::<libc::sockaddr_in6>(), &sockfaddr_in)?;
4419                        trace!("Port assigned {}", port_assigned)
4420                    }
4421                    GlobalResponse::PortFull => {
4422                        return Err(reverie::Error::from(nix::errno::Errno::EADDRINUSE));
4423                    }
4424                    _ => unreachable!(),
4425                }
4426            }
4427        }
4428        let res = self.record_or_replay(guest, call).await?;
4429
4430        Ok(res)
4431    }
4432
4433    /// Create and register an event notification counter.
4434    ///
4435    /// Determinism: strict execution serializes creation, so the initial counter, guest-visible
4436    /// flags, and descriptor number depend only on syscall arguments and the reconstructed file
4437    /// table. Any internally added nonblocking flag remains hidden from the guest.
4438    pub async fn handle_eventfd2<G: Guest<Self>>(
4439        &self,
4440        guest: &mut G,
4441        call: syscalls::Eventfd2,
4442    ) -> Result<i64, Error> {
4443        let internally_nonblocking =
4444            self.cfg.use_nonblocking_sockets() && !self.cfg.recordreplay_modes;
4445        let injected = if internally_nonblocking {
4446            call.with_flags(call.flags() | syscalls::EfdFlags::EFD_NONBLOCK)
4447        } else {
4448            call
4449        };
4450        let fd = self.record_or_replay(guest, injected).await? as RawFd;
4451        self.add_fd(
4452            guest,
4453            fd,
4454            OFlag::from_bits_truncate(
4455                call.flags().bits() & (libc::EFD_CLOEXEC | libc::EFD_NONBLOCK),
4456            ),
4457            FdType::Eventfd,
4458        )
4459        .await?;
4460        if internally_nonblocking {
4461            self.maybe_set_nonblocking_fd(guest, fd);
4462        }
4463        Ok(fd as i64)
4464    }
4465
4466    /// signalfd4 system call.
4467    pub async fn handle_signalfd4<G: Guest<Self>>(
4468        &self,
4469        guest: &mut G,
4470        call: syscalls::Signalfd4,
4471    ) -> Result<i64, Error> {
4472        let signalfd = self.record_or_replay(guest, call).await? as RawFd;
4473        self.add_fd(
4474            guest,
4475            signalfd,
4476            OFlag::from_bits_truncate(
4477                call.flags().bits() & (libc::SFD_CLOEXEC | libc::SFD_NONBLOCK),
4478            ),
4479            FdType::Signalfd,
4480        )
4481        .await?;
4482        Ok(signalfd as i64)
4483    }
4484
4485    /// Create and register a timer notification descriptor.
4486    ///
4487    /// Determinism: strict execution serializes creation, which exposes only kernel validation,
4488    /// guest-visible flags, and a descriptor number; this operation does not read the clock.
4489    pub async fn handle_timerfd_create<G: Guest<Self>>(
4490        &self,
4491        guest: &mut G,
4492        call: syscalls::TimerfdCreate,
4493    ) -> Result<i64, Error> {
4494        let fd = self.record_or_replay(guest, call).await? as RawFd;
4495        self.add_fd(
4496            guest,
4497            fd,
4498            OFlag::from_bits_truncate(
4499                call.flags().bits() & (libc::TFD_CLOEXEC | libc::TFD_NONBLOCK),
4500            ),
4501            FdType::Timerfd,
4502        )
4503        .await?;
4504        Ok(fd as i64)
4505    }
4506
4507    /// Serialize a notification descriptor control operation.
4508    async fn notification_fd_control<G: Guest<Self>>(
4509        &self,
4510        guest: &mut G,
4511        call: Syscall,
4512    ) -> Result<i64, Error> {
4513        let dettid = guest.thread_state().dettid;
4514        resource_request(guest, Resources::new(dettid)).await;
4515        Ok(self.record_or_replay(guest, call).await?)
4516    }
4517
4518    /// timerfd_settime system call.
4519    pub async fn handle_timerfd_settime<G: Guest<Self>>(
4520        &self,
4521        guest: &mut G,
4522        call: syscalls::TimerfdSettime,
4523    ) -> Result<i64, Error> {
4524        self.notification_fd_control(guest, call.into()).await
4525    }
4526
4527    /// timerfd_gettime system call.
4528    pub async fn handle_timerfd_gettime<G: Guest<Self>>(
4529        &self,
4530        guest: &mut G,
4531        call: syscalls::TimerfdGettime,
4532    ) -> Result<i64, Error> {
4533        self.notification_fd_control(guest, call.into()).await
4534    }
4535
4536    /// inotify_init1 system call.
4537    pub async fn handle_inotify_init1<G: Guest<Self>>(
4538        &self,
4539        guest: &mut G,
4540        call: syscalls::InotifyInit1,
4541    ) -> Result<i64, Error> {
4542        let fd = self.record_or_replay(guest, call).await? as RawFd;
4543        self.add_fd(
4544            guest,
4545            fd,
4546            OFlag::from_bits_truncate(call.flags().bits() & (libc::IN_CLOEXEC | libc::IN_NONBLOCK)),
4547            FdType::Inotify,
4548        )
4549        .await?;
4550        Ok(fd as i64)
4551    }
4552
4553    /// inotify_add_watch system call.
4554    pub async fn handle_inotify_add_watch<G: Guest<Self>>(
4555        &self,
4556        guest: &mut G,
4557        call: syscalls::InotifyAddWatch,
4558    ) -> Result<i64, Error> {
4559        self.notification_fd_control(guest, call.into()).await
4560    }
4561
4562    /// inotify_rm_watch system call.
4563    pub async fn handle_inotify_rm_watch<G: Guest<Self>>(
4564        &self,
4565        guest: &mut G,
4566        call: syscalls::InotifyRmWatch,
4567    ) -> Result<i64, Error> {
4568        self.notification_fd_control(guest, call.into()).await
4569    }
4570
4571    /// memfd_create system call.
4572    pub async fn handle_memfd_create<G: Guest<Self>>(
4573        &self,
4574        guest: &mut G,
4575        call: syscalls::MemfdCreate,
4576    ) -> Result<i64, Error> {
4577        let fd = self.record_or_replay(guest, call).await? as RawFd;
4578        self.add_fd(
4579            guest,
4580            fd,
4581            OFlag::from_bits_truncate((call.flags() & libc::MFD_CLOEXEC) as i32),
4582            FdType::Memfd,
4583        )
4584        .await?;
4585        Ok(fd as i64)
4586    }
4587
4588    // AUTONOMOUS-BOT-IMPLEMENTED
4589    // TODO-HUMAN-REVIEW(PR-862): pidfd creation and Detcore FD registration.
4590    /// Create a pidfd through record/replay and synchronize the descriptor with
4591    /// Detcore's metadata before fcntl, poll, close, or waitid can observe it.
4592    pub async fn handle_pidfd_open<G: Guest<Self>>(
4593        &self,
4594        guest: &mut G,
4595        call: syscalls::PidfdOpen,
4596    ) -> Result<i64, Error> {
4597        let allowed_flags = libc::O_NONBLOCK as u32;
4598        if call.flags() & !allowed_flags != 0 {
4599            return Err(Errno::EINVAL.into());
4600        }
4601
4602        let fd = self.record_or_replay(guest, call).await? as RawFd;
4603        let flags = OFlag::O_CLOEXEC | OFlag::from_bits_truncate(call.flags() as libc::c_int);
4604        self.add_fd(guest, fd, flags, FdType::Pidfd).await?;
4605        let target = DetPid::from_raw(call.pid() as i32);
4606        guest
4607            .thread_state()
4608            .with_detfd(fd, |detfd| detfd.set_pidfd_target(target))?;
4609        Ok(fd as i64)
4610    }
4611
4612    // AUTONOMOUS-BOT-IMPLEMENTED
4613    // TODO-HUMAN-REVIEW(PR-1175): pidfd_send_signal(2) determinization.
4614    /// Deliver a signal to the process referred to by a pidfd.
4615    ///
4616    /// `pidfd_send_signal(pidfd, sig, info, flags)` names its target by an open
4617    /// kernel descriptor rather than a numeric PID. Unlike `kill(2)`, there is
4618    /// therefore no host-PID/virtual-PID ambiguity for Detcore to resolve: the
4619    /// pidfd was bound to one specific process at `pidfd_open` time. Signal
4620    /// generation runs inside this thread's serialized scheduler turn, exactly
4621    /// like `tgkill`/`tkill`/`rt_tgsigqueueinfo` (which also just forward through
4622    /// record/replay), so forwarding the kernel call is deterministic by
4623    /// construction. This handler adds deterministic argument validation ahead of
4624    /// the forward: a descriptor that Detcore does not model as a pidfd fails
4625    /// closed with `EBADF`, and the flags field the current kernel reserves is
4626    /// required to be zero (`EINVAL` otherwise), so the guest-visible errno is
4627    /// fixed and host-independent.
4628    ///
4629    /// `call` is the raw `Syscall::Other`; `pidfd`/`flags` are pre-extracted from
4630    /// its arguments by the dispatcher.
4631    pub async fn handle_pidfd_send_signal<G: Guest<Self>>(
4632        &self,
4633        guest: &mut G,
4634        call: Syscall,
4635        pidfd: RawFd,
4636        flags: u32,
4637    ) -> Result<i64, Error> {
4638        // The kernel currently reserves `flags`; a nonzero value is EINVAL.
4639        if flags != 0 {
4640            return Err(Errno::EINVAL.into());
4641        }
4642        // Fail closed unless Detcore models this descriptor as a pidfd. This also
4643        // yields a deterministic EBADF for an unknown/closed descriptor.
4644        let is_pidfd = guest
4645            .thread_state()
4646            .with_detfd(pidfd, |detfd| matches!(detfd.ty(), FdType::Pidfd))?;
4647        if !is_pidfd {
4648            return Err(Errno::EBADF.into());
4649        }
4650        Ok(self.record_or_replay(guest, call).await?)
4651    }
4652
4653    // AUTONOMOUS-BOT-IMPLEMENTED
4654    // TODO-HUMAN-REVIEW(PR-1175): pidfd_getfd(2) determinization and the
4655    // FdType of the duplicated descriptor.
4656    /// Duplicate a descriptor from the process referred to by a pidfd.
4657    ///
4658    /// `pidfd_getfd(pidfd, targetfd, flags)` returns a fresh descriptor in the
4659    /// caller that aliases `targetfd` in the target process. The source pidfd
4660    /// names one specific process fixed at `pidfd_open` time, the returned
4661    /// descriptor number is chosen through record/replay (so it is stable across
4662    /// runs), and a successful modeled operation executes inside this thread's
4663    /// serialized turn, so the result is deterministic. For zero flags, Detcore
4664    /// fails closed with `EBADF` unless it models the descriptor as a pidfd.
4665    /// Linux checks the kernel-reserved `flags` first, however, so a nonzero
4666    /// value takes the raw record/replay path and preserves the kernel's exact
4667    /// `EINVAL` across valid and invalid descriptor combinations.
4668    ///
4669    /// The modeled path is narrower than "same process": the caller must be the
4670    /// thread-group leader named by the pidfd. `CLONE_THREAD` does not imply
4671    /// `CLONE_FILES`, so a nonleader can share the target's TGID while using a
4672    /// different descriptor table. Requiring `target == getpid() == gettid()`
4673    /// proves that `targetfd` is resolved in the caller's exact table. The
4674    /// returned descriptor is then modeled as a real alias of the source open
4675    /// file description. Broader support needs a cross-task OFD channel that
4676    /// Detcore does not have today, so every other target is refused with
4677    /// `EOPNOTSUPP`. Failed calls leave descriptor state unchanged.
4678    pub async fn handle_pidfd_getfd<G: Guest<Self>>(
4679        &self,
4680        guest: &mut G,
4681        call: Syscall,
4682        pidfd: RawFd,
4683        targetfd: RawFd,
4684        flags: u32,
4685    ) -> Result<i64, Error> {
4686        if flags != 0 {
4687            return match self.record_or_replay(guest, call).await {
4688                Err(error) => Err(error.into()),
4689                Ok(fd) => {
4690                    let fd = fd as RawFd;
4691                    let close_result = guest.inject(syscalls::Close::new().with_fd(fd)).await;
4692                    Err(Error::Tool(anyhow::anyhow!(
4693                        "pidfd_getfd unexpectedly accepted reserved flags and returned fd {fd}; cleanup close result: {close_result:?}"
4694                    )))
4695                }
4696            };
4697        }
4698
4699        if !guest.config().sequentialize_threads {
4700            return self
4701                .refuse_unserviceable_operation(guest, Sysno::pidfd_getfd, Errno::EOPNOTSUPP)
4702                .await;
4703        }
4704
4705        let current_tgid = DetPid::from_raw(guest.inject(syscalls::Getpid::new()).await? as i32);
4706        let current_tid = DetTid::from_raw(guest.inject(syscalls::Gettid::new()).await? as i32);
4707        let source = guest.thread_state().capture_pidfd_getfd_source(
4708            pidfd,
4709            targetfd,
4710            current_tgid,
4711            current_tid,
4712        )?;
4713        let fd = match self.record_or_replay(guest, call).await {
4714            Ok(fd) => fd as RawFd,
4715            Err(error) => {
4716                if let Some(open_file_id) = guest.thread_state().abandon_captured_fd(source) {
4717                    self.release_port_for_open_file(guest, open_file_id).await;
4718                }
4719                return Err(error.into());
4720            }
4721        };
4722        // pidfd_getfd always sets FD_CLOEXEC on the returned descriptor.
4723        let replaced = match guest.thread_state_mut().install_captured_fd(
4724            source,
4725            fd,
4726            OFlag::O_CLOEXEC,
4727        ) {
4728            Ok(replaced) => replaced,
4729            Err(error @ CapturedDetFdInstallError { .. }) => {
4730                let expected_files_id = error.expected_files_id;
4731                let actual_files_id = error.actual_files_id;
4732                let cleanup = error.into_cleanup();
4733                let close_result = guest
4734                    .inject(syscalls::Close::new().with_fd(cleanup.close_fd))
4735                    .await;
4736                if let Some(open_file_id) = cleanup.release_open_file {
4737                    self.release_port_for_open_file(guest, open_file_id).await;
4738                }
4739                if let Err(close_error) = close_result {
4740                    return Err(Error::Tool(anyhow::anyhow!(
4741                        "pidfd_getfd returned fd {fd}, but its captured source table changed from {expected_files_id:?} to {actual_files_id:?}; cleanup close failed with {close_error}"
4742                    )));
4743                }
4744                warn!(
4745                    "pidfd_getfd returned fd {fd}, but its captured source table changed from {expected_files_id:?} to {actual_files_id:?}; closed the result and refusing with EOPNOTSUPP"
4746                );
4747                return Err(Errno::EOPNOTSUPP.into());
4748            }
4749        };
4750        if let Some(open_file_id) = replaced {
4751            self.release_port_for_open_file(guest, open_file_id).await;
4752        }
4753        Ok(fd as i64)
4754    }
4755
4756    /// userfaultfd system call.
4757    pub async fn handle_userfaultfd<G: Guest<Self>>(
4758        &self,
4759        guest: &mut G,
4760        call: syscalls::Userfaultfd,
4761    ) -> Result<i64, Error> {
4762        let fd = self.record_or_replay(guest, call).await? as RawFd;
4763        self.add_fd(
4764            guest,
4765            fd,
4766            OFlag::from_bits_truncate(call.flags()),
4767            FdType::Userfaultfd,
4768        )
4769        .await?;
4770        Ok(fd as i64)
4771    }
4772
4773    /// accept4 system call (MAYHANG).
4774    ///
4775    /// Category: External OR Internal IO
4776    /// ---------------------------------
4777    /// When do we know?  We only know if an accept4 did an extra-container IO AFTER it returns.
4778    /// I.e. we could accept a connection from another endpoint in the container, or from the outside,
4779    /// and we don't know which at the point where `accept4` is called.
4780    pub async fn handle_accept4<G: Guest<Self>>(
4781        &self,
4782        guest: &mut G,
4783        call: syscalls::Accept4,
4784    ) -> Result<i64, Error> {
4785        // This option applies both to the socket we're doing the accept call on, and the connection
4786        // that we return. We don't have any smart detection yet to separate internal/external, so
4787        // applies to everything.
4788        let call2 = if self.cfg.use_nonblocking_sockets() {
4789            // Let the socket returned from accept4 be physically nonblocking:
4790            call.with_flags(call.flags() | SockFlag::SOCK_NONBLOCK)
4791        } else {
4792            call
4793        };
4794        // This will do blocking/polling as appropriate based on the fd status:
4795        let fd = self.execute_nonblockable_fd_syscall(guest, call2).await? as RawFd;
4796
4797        self.add_fd(
4798            guest,
4799            fd,
4800            // This will specify whether the socket returned is logically non-blocking:
4801            oflag_from_sock_bits(call.flags().bits()),
4802            FdType::Socket,
4803        )
4804        .await?;
4805
4806        self.maybe_set_nonblocking_fd(guest, fd);
4807
4808        Ok(fd as i64)
4809    }
4810
4811    /// getdents system call.
4812    pub async fn handle_getdents<G: Guest<Self>>(
4813        &self,
4814        guest: &mut G,
4815        call: syscalls::Getdents,
4816    ) -> Result<i64, Error> {
4817        if !guest.config().virtualize_metadata {
4818            return Ok(self.record_or_replay(guest, call).await?);
4819        }
4820
4821        let dirent = call.dirent().ok_or(Errno::EFAULT)?;
4822
4823        let nb = self.record_or_replay(guest, call).await?;
4824        if nb == 0 {
4825            return Ok(0);
4826        }
4827
4828        let mut dents_bytes = vec![0; nb as usize];
4829        dents_bytes.reserve_exact(128);
4830
4831        guest
4832            .memory()
4833            .read_exact(dirent.cast(), dents_bytes.as_mut_slice())?;
4834
4835        let mut dents = unsafe { deserialize_dirents(&dents_bytes) };
4836        dents.sort();
4837        for dent in &mut dents {
4838            let (d_ino, _) = determinize_inode(guest, dent.ino).await;
4839            dent.ino = d_ino.as_raw();
4840        }
4841
4842        let mut dents_bytes = vec![0; dents_bytes.len()];
4843        let _ = unsafe { serialize_dirents(&dents, &mut dents_bytes) };
4844
4845        guest
4846            .memory()
4847            .write_exact(dirent.cast(), dents_bytes.as_slice())?;
4848        Ok(nb)
4849    }
4850
4851    /// getdents64 system call.
4852    pub async fn handle_getdents64<G: Guest<Self>>(
4853        &self,
4854        guest: &mut G,
4855        call: syscalls::Getdents64,
4856    ) -> Result<i64, Error> {
4857        if !guest.config().virtualize_metadata {
4858            return Ok(self.record_or_replay(guest, call).await?);
4859        }
4860
4861        let dirent = call.dirent().ok_or(Errno::EFAULT)?;
4862
4863        let nb = self.record_or_replay(guest, call).await?;
4864        if nb == 0 {
4865            return Ok(0);
4866        }
4867
4868        let mut dents_bytes = vec![0; nb as usize];
4869        dents_bytes.reserve_exact(128);
4870
4871        guest
4872            .memory()
4873            .read_exact(dirent.cast(), dents_bytes.as_mut_slice())?;
4874
4875        let mut dents = unsafe { deserialize_dirents64(&dents_bytes) };
4876        dents.sort();
4877        for dent in &mut dents {
4878            let (d_ino, _) = determinize_inode(guest, dent.ino).await;
4879            dent.ino = d_ino.as_raw();
4880        }
4881
4882        let mut dents_bytes = vec![0; dents_bytes.len()];
4883        let _ = unsafe { serialize_dirents64(&dents, &mut dents_bytes) };
4884
4885        guest
4886            .memory()
4887            .write_exact(dirent.cast(), dents_bytes.as_slice())?;
4888        Ok(nb)
4889    }
4890}
4891
4892#[cfg(test)]
4893mod procfs_wiring_guard {
4894    //! The procfs snapshot WIRING, guarded where the wiring lives.
4895    //!
4896    //! WHY THIS IS A SOURCE-LEVEL GUARD AND NOT A BEHAVIOURAL TEST. The thing
4897    //! at risk is not `ProcfsFile`'s logic -- that is exercised elsewhere. It is
4898    //! the CALL from each read handler into the snapshot initialiser. Those
4899    //! handlers are `async fn`s on the `Tool` trait taking a live `Guest`, so a
4900    //! unit test cannot invoke one without standing up a traced guest process;
4901    //! that is exactly why the only thing guarding this today is one heavyweight
4902    //! integration test that compiles a C probe and runs hermit.
4903    //!
4904    //! MEASURED GAP (2026-08-07, hermit 75506005d): deleting the pread64
4905    //! snapshot-initialisation block leaves ALL 386 detcore lib tests green.
4906    //! A mechanism whose only proof of life is one fixture is one deletion away
4907    //! from vanishing unnoticed -- the positioned-read determinism bug this
4908    //! wiring fixes would silently return.
4909    //!
4910    //! These assertions are deliberately narrow: they bind to the CALL, name the
4911    //! mechanism when they fail, and cost nothing to run. They do not claim to
4912    //! verify that the snapshot is correct.
4913
4914    /// The production source only. `include_str!` pulls in THIS module too, so a
4915    /// naive scan counts the guard's own string literals and reports phantom
4916    /// duplicates -- it did exactly that on first run. Truncating at the guard's
4917    /// own header makes every assertion below immune to self-reference.
4918    fn production_source() -> &'static str {
4919        const WHOLE: &str = include_str!("files.rs");
4920        const GUARD: &str = "#[cfg(test)]\nmod procfs_wiring_guard {";
4921        match WHOLE.find(GUARD) {
4922            Some(cut) => &WHOLE[..cut],
4923            None => WHOLE,
4924        }
4925    }
4926
4927    /// The body of `fn <name>` up to the next top-level `    }` at fn indent.
4928    fn handler_body(name: &str) -> &'static str {
4929        let start = production_source()
4930            .find(&format!("fn {}<G: Guest<Self>>", name))
4931            .unwrap_or_else(|| {
4932                panic!(
4933                    "procfs wiring guard: handler `{name}` not found in files.rs.\n\
4934                     TWO VERY DIFFERENT CAUSES, and the guard cannot tell them apart:\n\
4935                       (a) the handler was RENAMED or its signature changed -- the \
4936                     mechanism is fine, update the name in this guard; or\n\
4937                       (b) the handler was DELETED -- the procfs snapshot wiring is gone.\n\
4938                     Check which before editing. This guard binds to source text on \
4939                     purpose: the handlers are async `Tool` methods taking a live Guest, \
4940                     so nothing cheaper can observe the call. It is deliberately loud \
4941                     when it cannot see the code, because silently passing is the \
4942                     failure it exists to prevent."
4943                )
4944            });
4945        let rest = &production_source()[start..];
4946        let end = rest
4947            .find(
4948                "
4949    }
4950",
4951            )
4952            .map(|e| e + 6)
4953            .unwrap_or(rest.len());
4954        &rest[..end]
4955    }
4956
4957    #[test]
4958    fn pread64_initializes_the_procfs_snapshot() {
4959        let body = handler_body("handle_pread64");
4960        assert!(
4961            body.contains("procfs_needs_snapshot") && body.contains("initialize_procfs_snapshot"),
4962            "MISSING MECHANISM: the pread64 handler no longer initialises the procfs \
4963             snapshot. Positioned reads will fall through to LIVE KERNEL BYTES instead of \
4964             the sanitized ProcfsFile snapshot, reintroducing the positioned-read \
4965             nondeterminism that hermit-cli/tests/procfs_positioned_determinism.rs exists \
4966             to catch. Restore the `procfs_needs_snapshot` -> `initialize_procfs_snapshot` \
4967             call in handle_pread64."
4968        );
4969    }
4970
4971    #[test]
4972    fn read_initializes_the_procfs_snapshot() {
4973        let body = handler_body("handle_read");
4974        assert!(
4975            body.contains("procfs_needs_snapshot") && body.contains("initialize_procfs_snapshot"),
4976            "MISSING MECHANISM: the sequential read handler no longer initialises the \
4977             procfs snapshot. Reads of /proc will observe live kernel bytes. Restore the \
4978             `procfs_needs_snapshot` -> `initialize_procfs_snapshot` call in handle_read."
4979        );
4980    }
4981
4982    #[test]
4983    fn both_read_paths_share_one_snapshot_initializer() {
4984        // The original defect was exactly this asymmetry: `read` consumed the
4985        // sanitized snapshot while `pread64` did not. Forking the logic into two
4986        // initialisers is how that asymmetry comes back.
4987        let n = production_source()
4988            .matches("async fn initialize_procfs_snapshot")
4989            .count();
4990        assert_eq!(
4991            n, 1,
4992            "MISSING MECHANISM: expected exactly ONE `initialize_procfs_snapshot` \
4993             definition so every read path shares it; found {n}. Two initialisers is how \
4994             read/pread64 drifted apart in the first place."
4995        );
4996        for handler in ["handle_read", "handle_pread64"] {
4997            assert!(
4998                handler_body(handler).contains("self.initialize_procfs_snapshot("),
4999                "MISSING MECHANISM: `{handler}` does not call the shared \
5000                 initialize_procfs_snapshot."
5001            );
5002        }
5003    }
5004
5005    #[test]
5006    fn the_guard_can_actually_see_the_handlers() {
5007        // Positive control: if the extractor silently returned empty bodies the
5008        // three assertions above would be vacuous rather than protective.
5009        for handler in ["handle_read", "handle_pread64"] {
5010            let body = handler_body(handler);
5011            assert!(
5012                body.len() > 200 && body.contains("call.fd()"),
5013                "guard extractor did not find a real body for `{handler}` \
5014                 (len {}), so the wiring assertions would be vacuous",
5015                body.len()
5016            );
5017        }
5018    }
5019}
5020
5021#[cfg(test)]
5022mod test {
5023    use nix::fcntl::OFlag;
5024    use reverie::syscalls::FromToRaw;
5025    use reverie::syscalls::Whence;
5026
5027    use super::DETERMINISTIC_PIPE_CAPACITY_BYTES;
5028    use super::pipe_capacity_request_exceeds_ceiling;
5029    /// The ceiling is inclusive. A guest that reads the advertised
5030    /// `pipe-max-size` and asks for exactly that must be allowed to have it;
5031    /// refusing at the boundary would advertise a size that cannot be set.
5032    #[test]
5033    fn the_pipe_ceiling_admits_exactly_the_pinned_capacity() {
5034        assert!(!pipe_capacity_request_exceeds_ceiling(
5035            DETERMINISTIC_PIPE_CAPACITY_BYTES
5036        ));
5037        assert!(!pipe_capacity_request_exceeds_ceiling(
5038            DETERMINISTIC_PIPE_CAPACITY_BYTES - 1
5039        ));
5040        assert!(pipe_capacity_request_exceeds_ceiling(
5041            DETERMINISTIC_PIPE_CAPACITY_BYTES + 1
5042        ));
5043    }
5044
5045    /// Shrinking stays legal. `tests/c/pipe_capacity.c`
5046    /// shrinks to one page and requires the value to round-trip, so a blanket
5047    /// clamp to the pinned capacity would break a guest-visible contract this
5048    /// repository already locked.
5049    #[test]
5050    fn shrinking_is_never_refused_by_the_ceiling() {
5051        for requested in [1, 4096, DETERMINISTIC_PIPE_CAPACITY_BYTES / 2] {
5052            assert!(
5053                !pipe_capacity_request_exceeds_ceiling(requested),
5054                "shrink to {requested} must remain permitted"
5055            );
5056        }
5057    }
5058
5059    /// The host's own ceiling is the value this change exists to stop
5060    /// consulting: 1048576 on a default host, 65536 on a hardened one. Both are
5061    /// refused now, so the guest-visible answer no longer depends on which host
5062    /// it is.
5063    #[test]
5064    fn host_ceilings_are_refused_identically_on_any_host() {
5065        for host_ceiling in [65536, 1048576] {
5066            assert!(
5067                pipe_capacity_request_exceeds_ceiling(host_ceiling),
5068                "{host_ceiling} must be refused regardless of the host sysctl"
5069            );
5070        }
5071    }
5072
5073    use super::Errno;
5074    use super::TimerSlackBinding;
5075    use super::UNIX_AUTOBIND_NAME_LEN;
5076    use super::canonicalize_tcp_info;
5077    use super::classify_timer_slack_binding;
5078    use super::is_inherited_container_output;
5079    use super::parse_timer_slack_write;
5080    use super::pipe_capacity_failure;
5081    use super::random_device_lseek_result;
5082    use super::should_tag_sabre_internal_pipe_io;
5083    use super::unix_autobind_address;
5084    use super::unix_autobind_addrlen;
5085    use super::vectored_offset;
5086    use crate::fd::FdType;
5087    use crate::resources::Device;
5088    use crate::resources::ResourceID;
5089
5090    /// This is an assumption we're making about flags.  Probably these flags can never be
5091    /// changed, but let's check just in case.
5092    #[test]
5093    fn linux_flags_assumptions() {
5094        assert_eq!(libc::SOCK_NONBLOCK, OFlag::O_NONBLOCK.bits());
5095        assert_eq!(libc::SOCK_CLOEXEC, OFlag::O_CLOEXEC.bits());
5096    }
5097
5098    #[test]
5099    fn pipe_capacity_failure_classifies_the_errno() {
5100        let created = [17, 18];
5101
5102        // The only success shape: Linux applied EXACTLY the capacity we asked for.
5103        assert_eq!(
5104            pipe_capacity_failure(created, Ok(i64::from(DETERMINISTIC_PIPE_CAPACITY_BYTES))),
5105            None
5106        );
5107
5108        // Successful-but-wrong capacity is a pipe whose size we did not choose. That is not a
5109        // kernel errno, so it is reported as EIO rather than dressed up as one.
5110        let mismatch = pipe_capacity_failure(
5111            created,
5112            Ok(i64::from(DETERMINISTIC_PIPE_CAPACITY_BYTES) * 2),
5113        )
5114        .expect("a capacity Linux rounded away from the pin must not read as success");
5115        assert_eq!(mismatch.created_fds, created);
5116        assert_eq!(mismatch.error, Errno::EIO);
5117
5118        // A real kernel errno is preserved rather than rewritten.
5119        let denied = pipe_capacity_failure(created, Err(Errno::EPERM))
5120            .expect("a kernel refusal must not read as success");
5121        assert_eq!(denied.created_fds, created);
5122        assert_eq!(denied.error, Errno::EPERM);
5123
5124        // The descriptors are carried through every failure shape, because they are what the
5125        // caller has to close; losing them here is how they would leak.
5126        assert_eq!(denied.created_fds, created);
5127    }
5128
5129    #[test]
5130    fn pipe_capacity_failure_closes_both_created_descriptors() {
5131        let mut created = [-1; 2];
5132        assert_eq!(
5133            unsafe { libc::pipe2(created.as_mut_ptr(), libc::O_CLOEXEC) },
5134            0
5135        );
5136
5137        let pin_result = unsafe { libc::fcntl(created[0], libc::F_SETPIPE_SZ, -1) };
5138        assert_eq!(pin_result, -1);
5139        let pin_error = Errno::last();
5140        // Pin the errno, not just the failure. Without this the test asserts only that
5141        // `fcntl` returned -1, so it would still pass if the call failed for a reason we
5142        // did not engineer -- an invalid `created[0]` fails EBADF, every assertion below
5143        // still holds, and the capacity-pin path is never exercised at all.
5144        //
5145        // EINVAL is structural here, not a property of this host. `fcntl`'s argument is an
5146        // `unsigned long`, so -1 arrives as ULONG_MAX; `round_pipe_size` returns 0 for any
5147        // size above 2^31, and `pipe_set_size` maps that 0 to -EINVAL BEFORE it consults
5148        // `CAP_SYS_RESOURCE`. So the result does not depend on privilege or on
5149        // `/proc/sys/fs/pipe-max-size`.
5150        assert_eq!(pin_error, Errno::EINVAL);
5151
5152        let failure = pipe_capacity_failure(created, Err(pin_error))
5153            .expect("the forced capacity-pin failure must enter the cleanup path");
5154        for close in failure.close_syscalls() {
5155            assert_eq!(unsafe { libc::close(close.fd()) }, 0);
5156        }
5157
5158        for fd in created {
5159            assert_eq!(unsafe { libc::fcntl(fd, libc::F_GETFD) }, -1);
5160            assert_eq!(Errno::last(), Errno::EBADF);
5161        }
5162    }
5163
5164    #[test]
5165    fn sabre_pipe_marker_requires_nonblockize_retry_semantics() {
5166        assert!(should_tag_sabre_internal_pipe_io(
5167            true,
5168            FdType::Pipe,
5169            true,
5170            false
5171        ));
5172        assert!(!should_tag_sabre_internal_pipe_io(
5173            true,
5174            FdType::Pipe,
5175            true,
5176            true
5177        ));
5178        assert!(!should_tag_sabre_internal_pipe_io(
5179            true,
5180            FdType::Pipe,
5181            false,
5182            false
5183        ));
5184        assert!(!should_tag_sabre_internal_pipe_io(
5185            false,
5186            FdType::Pipe,
5187            true,
5188            false
5189        ));
5190        assert!(!should_tag_sabre_internal_pipe_io(
5191            true,
5192            FdType::Regular,
5193            true,
5194            false
5195        ));
5196    }
5197
5198    #[test]
5199    fn random_device_lseek_matches_linux_noop_llseek() {
5200        for whence in [
5201            Whence::SEEK_SET,
5202            Whence::SEEK_CUR,
5203            Whence::SEEK_END,
5204            Whence::SEEK_DATA,
5205            Whence::SEEK_HOLE,
5206        ] {
5207            for status_flags in [
5208                OFlag::empty().bits(),
5209                OFlag::O_WRONLY.bits(),
5210                OFlag::O_RDWR.bits(),
5211            ] {
5212                assert_eq!(random_device_lseek_result(status_flags, whence), Ok(0));
5213            }
5214            assert_eq!(
5215                random_device_lseek_result(OFlag::O_PATH.bits(), whence),
5216                Err(Errno::EBADF)
5217            );
5218        }
5219        assert_eq!(
5220            random_device_lseek_result(OFlag::empty().bits(), Whence::from_raw(99)),
5221            Err(Errno::EINVAL)
5222        );
5223        assert_eq!(
5224            random_device_lseek_result(OFlag::O_PATH.bits(), Whence::from_raw(99)),
5225            Err(Errno::EBADF)
5226        );
5227    }
5228
5229    #[test]
5230    fn timer_slack_write_parser_matches_decimal_procfs_contract() {
5231        assert_eq!(parse_timer_slack_write(b"0"), Ok(0));
5232        assert_eq!(parse_timer_slack_write(b"+123\n"), Ok(123));
5233        assert_eq!(parse_timer_slack_write(b"456\0ignored"), Ok(456));
5234        assert_eq!(
5235            parse_timer_slack_write(u64::MAX.to_string().as_bytes()),
5236            Ok(u64::MAX)
5237        );
5238
5239        for invalid in [b"".as_slice(), b"+", b"-1", b" 1", b"1 ", b"1\n2", b"0x10"] {
5240            assert_eq!(parse_timer_slack_write(invalid), Err(Errno::EINVAL));
5241        }
5242        assert_eq!(
5243            parse_timer_slack_write(b"18446744073709551616"),
5244            Err(Errno::ERANGE)
5245        );
5246    }
5247
5248    #[test]
5249    fn timer_slack_vectored_offset_preserves_minus_one_sentinel() {
5250        assert_eq!(vectored_offset(u64::MAX, u64::MAX), -1);
5251        assert_eq!(vectored_offset(0, 0), 0);
5252        assert_eq!(vectored_offset(7, 0), 7);
5253    }
5254
5255    #[test]
5256    fn timer_slack_binding_rejects_exit_reuse_and_other_tasks() {
5257        let binding = TimerSlackBinding {
5258            target: 202,
5259            device: 11,
5260            inode: 22,
5261        };
5262        assert_eq!(
5263            classify_timer_slack_binding(binding, Some((11, 22)), 202),
5264            Ok(())
5265        );
5266        assert_eq!(
5267            classify_timer_slack_binding(binding, Some((11, 22)), 303),
5268            Err(Errno::EPERM)
5269        );
5270        assert_eq!(
5271            classify_timer_slack_binding(binding, None, 202),
5272            Err(Errno::ESRCH)
5273        );
5274        assert_eq!(
5275            classify_timer_slack_binding(binding, Some((11, 23)), 202),
5276            Err(Errno::ESRCH),
5277            "a recycled numeric TID must not revive an old proc inode"
5278        );
5279    }
5280
5281    #[test]
5282    fn only_inherited_container_output_is_nonseekable() {
5283        assert!(is_inherited_container_output(Some(ResourceID::Device(
5284            Device::ContainerStdout
5285        ))));
5286        assert!(is_inherited_container_output(Some(ResourceID::Device(
5287            Device::ContainerStderr
5288        ))));
5289        assert!(!is_inherited_container_output(Some(ResourceID::Device(
5290            Device::ContainerStdin
5291        ))));
5292        assert!(!is_inherited_container_output(None));
5293    }
5294
5295    #[test]
5296    fn unix_autobind_address_matches_linux_shape() {
5297        let address = unix_autobind_address(0x2af);
5298        assert_eq!(address.sun_family, libc::AF_UNIX as libc::sa_family_t);
5299        assert_eq!(address.sun_path[0], 0);
5300        let name = address.sun_path[1..UNIX_AUTOBIND_NAME_LEN]
5301            .iter()
5302            .map(|byte| *byte as u8)
5303            .collect::<Vec<_>>();
5304        assert_eq!(name, b"002af");
5305        assert_eq!(
5306            unix_autobind_addrlen() as usize,
5307            std::mem::offset_of!(libc::sockaddr_un, sun_path) + UNIX_AUTOBIND_NAME_LEN
5308        );
5309    }
5310
5311    #[test]
5312    fn tcp_info_retains_only_logical_connection_header() {
5313        let mut info = [0xff; 16];
5314        canonicalize_tcp_info(&mut info);
5315
5316        for (offset, byte) in info.into_iter().enumerate() {
5317            let expected = if matches!(offset, 0 | 1 | 5 | 6) {
5318                0xff
5319            } else {
5320                0
5321            };
5322            assert_eq!(byte, expected, "unexpected byte at offset {offset}");
5323        }
5324
5325        for len in 0..8 {
5326            canonicalize_tcp_info(&mut [0xff; 8][..len]);
5327        }
5328    }
5329}
5330
5331/// `inject_fstat` and `add_fd` against a scripted guest whose "address space"
5332/// is this test process, so every injected syscall runs for real on host
5333/// memory and a real descriptor. Regression coverage for
5334/// <https://github.com/rrnewton/hermit/issues/3328>; the traced end-to-end case
5335/// is `tests_misc::tight_stack_openat`.
5336#[cfg(test)]
5337mod inject_fstat_scratch {
5338    use std::os::fd::IntoRawFd;
5339    use std::os::fd::RawFd;
5340    use std::os::unix::fs::MetadataExt;
5341    use std::sync::Arc;
5342    use std::sync::atomic::AtomicBool;
5343    use std::sync::atomic::Ordering;
5344
5345    use reverie::GlobalRPC;
5346    use reverie::GlobalTool;
5347    use reverie::Pid;
5348    use reverie::Tool;
5349    use reverie::syscalls::LocalMemory;
5350    use reverie::syscalls::ProtFlags;
5351
5352    use super::*;
5353    use crate::Config;
5354    use crate::GlobalState;
5355    use crate::ThreadState;
5356    use crate::types::DetPid;
5357
5358    /// Room for one `libc::stat`, 8-byte aligned like a real stack slot.
5359    const ARENA_WORDS: usize = 32;
5360
5361    /// A guest stack scratch that is either writable or faults on commit,
5362    /// like the ptrace scratch below an `rsp` with no writable memory under
5363    /// it. The writable arena belongs to the guest and outlives every guard,
5364    /// so an early guard drop is reported by `guard_live` rather than by a
5365    /// write into freed memory.
5366    struct ScriptedStack {
5367        writable: bool,
5368        arena: usize,
5369        guard_live: Arc<AtomicBool>,
5370    }
5371
5372    struct ScriptedStackGuard {
5373        guard_live: Arc<AtomicBool>,
5374    }
5375
5376    impl Drop for ScriptedStackGuard {
5377        fn drop(&mut self) {
5378            self.guard_live.store(false, Ordering::SeqCst);
5379        }
5380    }
5381
5382    impl reverie::Stack for ScriptedStack {
5383        type StackGuard = ScriptedStackGuard;
5384
5385        fn size(&self) -> usize {
5386            panic!("inject_fstat must not query the scratch size")
5387        }
5388        fn capacity(&self) -> usize {
5389            panic!("inject_fstat must not query the scratch capacity")
5390        }
5391        fn push<'stack, T>(&mut self, _: T) -> Addr<'stack, T> {
5392            panic!("inject_fstat reserves its buffer rather than pushing one")
5393        }
5394        fn reserve<'stack, T>(&mut self) -> AddrMut<'stack, T> {
5395            assert!(std::mem::size_of::<T>() <= ARENA_WORDS * std::mem::size_of::<u64>());
5396            AddrMut::from_raw(self.arena).unwrap()
5397        }
5398        fn commit(self) -> Result<Self::StackGuard, Errno> {
5399            if !self.writable {
5400                return Err(Errno::EFAULT);
5401            }
5402            self.guard_live.store(true, Ordering::SeqCst);
5403            Ok(ScriptedStackGuard {
5404                guard_live: self.guard_live,
5405            })
5406        }
5407    }
5408
5409    struct ScriptedGuest {
5410        config: Config,
5411        thread: ThreadState<()>,
5412        stack_writable: bool,
5413        mmap_fails: bool,
5414        arena: Box<[u64; ARENA_WORDS]>,
5415        guard_live: Arc<AtomicBool>,
5416        injected: Vec<Sysno>,
5417        /// Whether a stack guard was live when each fstat was injected.
5418        fstat_guard_live: Vec<bool>,
5419        /// Buffer address of each injected fstat.
5420        fstat_buffers: Vec<usize>,
5421        /// (address, length) of each page the guest mapped.
5422        mapped: Vec<(usize, usize)>,
5423        /// (address, length) of each successful munmap.
5424        unmapped: Vec<(usize, usize)>,
5425        /// Descriptors closed through injection.
5426        closed: Vec<RawFd>,
5427    }
5428
5429    impl ScriptedGuest {
5430        fn new(stack_writable: bool, mmap_fails: bool) -> (Detcore, Self) {
5431            let config = Config {
5432                virtualize_metadata: true,
5433                ..Config::default()
5434            };
5435            let pid = DetPid::from_raw(1);
5436            let mut thread = ThreadState::new(pid, &config, ());
5437            thread.detpid = Some(pid);
5438            let tool = <Detcore as Tool>::new(Pid::from_raw(1), &config);
5439            let guest = Self {
5440                config,
5441                thread,
5442                stack_writable,
5443                mmap_fails,
5444                arena: Box::new([u64::MAX; ARENA_WORDS]),
5445                guard_live: Arc::new(AtomicBool::new(false)),
5446                injected: Vec::new(),
5447                fstat_guard_live: Vec::new(),
5448                fstat_buffers: Vec::new(),
5449                mapped: Vec::new(),
5450                unmapped: Vec::new(),
5451                closed: Vec::new(),
5452            };
5453            (tool, guest)
5454        }
5455    }
5456
5457    #[reverie::tool]
5458    impl GlobalRPC<GlobalState> for ScriptedGuest {
5459        async fn send_rpc(
5460            &self,
5461            message: <GlobalState as GlobalTool>::Request,
5462        ) -> <GlobalState as GlobalTool>::Response {
5463            panic!("fd registration must not send an RPC: {:?}", message.2)
5464        }
5465        fn config(&self) -> &Config {
5466            &self.config
5467        }
5468    }
5469
5470    #[reverie::tool]
5471    impl Guest<Detcore> for ScriptedGuest {
5472        type Memory = LocalMemory;
5473        type Stack = ScriptedStack;
5474
5475        fn tid(&self) -> Pid {
5476            Pid::from_raw(1)
5477        }
5478        fn pid(&self) -> Pid {
5479            Pid::from_raw(1)
5480        }
5481        fn ppid(&self) -> Option<Pid> {
5482            None
5483        }
5484        fn memory(&self) -> Self::Memory {
5485            LocalMemory::new()
5486        }
5487        fn thread_state_mut(&mut self) -> &mut ThreadState<()> {
5488            &mut self.thread
5489        }
5490        fn thread_state(&self) -> &ThreadState<()> {
5491            &self.thread
5492        }
5493        async fn regs(&mut self) -> libc::user_regs_struct {
5494            panic!("fd registration must not read registers")
5495        }
5496        async fn stack(&mut self) -> Self::Stack {
5497            ScriptedStack {
5498                writable: self.stack_writable,
5499                arena: self.arena.as_mut_ptr() as usize,
5500                guard_live: self.guard_live.clone(),
5501            }
5502        }
5503        async fn daemonize(&mut self) {
5504            panic!("fd registration must not daemonize")
5505        }
5506        async fn inject<S: SyscallInfo>(&mut self, syscall: S) -> Result<i64, Errno> {
5507            let (number, args) = syscall.into_parts();
5508            self.injected.push(number);
5509            // SAFETY: each arm runs the syscall Detcore asked for against this
5510            // process, on addresses Detcore obtained from this guest.
5511            let raw = match Syscall::from_raw(number, args) {
5512                Syscall::Mmap(call) => {
5513                    assert!(call.addr().is_none(), "the kernel must choose the address");
5514                    assert_eq!(call.prot(), ProtFlags::PROT_READ | ProtFlags::PROT_WRITE);
5515                    assert_eq!(
5516                        call.flags(),
5517                        MapFlags::MAP_PRIVATE | MapFlags::MAP_ANONYMOUS
5518                    );
5519                    if self.mmap_fails {
5520                        return Err(Errno::ENOMEM);
5521                    }
5522                    let address = unsafe {
5523                        libc::mmap(
5524                            std::ptr::null_mut(),
5525                            call.len(),
5526                            libc::PROT_READ | libc::PROT_WRITE,
5527                            libc::MAP_PRIVATE | libc::MAP_ANONYMOUS,
5528                            -1,
5529                            0,
5530                        )
5531                    };
5532                    if address == libc::MAP_FAILED {
5533                        -1
5534                    } else {
5535                        self.mapped.push((address as usize, call.len()));
5536                        address as i64
5537                    }
5538                }
5539                Syscall::Fstat(call) => {
5540                    let buffer = call.stat().expect("fstat without a buffer").0.as_raw();
5541                    self.fstat_buffers.push(buffer);
5542                    self.fstat_guard_live
5543                        .push(self.guard_live.load(Ordering::SeqCst));
5544                    i64::from(unsafe { libc::fstat(call.fd(), buffer as *mut libc::stat) })
5545                }
5546                Syscall::Munmap(call) => {
5547                    let address = call.addr().expect("munmap without an address").as_raw();
5548                    let raw = i64::from(unsafe { libc::munmap(address as *mut _, call.len()) });
5549                    if raw == 0 {
5550                        self.unmapped.push((address, call.len()));
5551                    }
5552                    raw
5553                }
5554                Syscall::Close(call) => {
5555                    self.closed.push(call.fd());
5556                    i64::from(unsafe { libc::close(call.fd()) })
5557                }
5558                other => panic!("unexpected injected syscall {other:?}"),
5559            };
5560            Errno::result(raw)
5561        }
5562        async fn tail_inject<S: SyscallInfo>(&mut self, _: S) -> reverie::Never {
5563            panic!("fd registration must not retire the guest")
5564        }
5565        fn set_timer(&mut self, _: reverie::TimerSchedule) -> Result<(), Error> {
5566            panic!("fd registration must not set a timer")
5567        }
5568        fn set_timer_precise(&mut self, _: reverie::TimerSchedule) -> Result<(), Error> {
5569            panic!("fd registration must not set a timer")
5570        }
5571        fn read_clock(&mut self) -> Result<u64, Error> {
5572            panic!("fd registration must not read a clock")
5573        }
5574    }
5575
5576    /// A real descriptor the test owns by number, and its inode. Ownership is
5577    /// raw so a descriptor that Detcore closes is never closed a second time.
5578    fn open_file() -> (RawFd, u64) {
5579        let file = tempfile::tempfile().unwrap();
5580        let inode = file.metadata().unwrap().ino();
5581        (file.into_raw_fd(), inode)
5582    }
5583
5584    fn close_unless_detcore_did(guest: &ScriptedGuest, fd: RawFd) {
5585        if !guest.closed.contains(&fd) {
5586            assert_eq!(unsafe { libc::close(fd) }, 0);
5587        }
5588    }
5589
5590    fn recorded_inode(guest: &ScriptedGuest, fd: RawFd) -> Option<u64> {
5591        guest
5592            .thread
5593            .with_detfd(fd, |detfd| detfd.stat().map(|stat| stat.inode))
5594            .unwrap()
5595    }
5596
5597    #[tokio::test]
5598    async fn writable_stack_scratch_is_used_while_its_guard_is_live() {
5599        let (fd, inode) = open_file();
5600        let (tool, mut guest) = ScriptedGuest::new(true, false);
5601
5602        let result = tool
5603            .add_fd(&mut guest, fd, OFlag::O_RDONLY, FdType::Regular)
5604            .await;
5605        close_unless_detcore_did(&guest, fd);
5606
5607        assert_eq!(result, Ok(()));
5608        assert_eq!(guest.injected, [Sysno::fstat]);
5609        assert_eq!(
5610            guest.fstat_guard_live,
5611            [true],
5612            "the stack guard must outlive the injected fstat: backends whose \
5613             scratch is an arena free it when the guard drops"
5614        );
5615        assert_eq!(
5616            guest.fstat_buffers,
5617            [guest.arena.as_ptr() as usize],
5618            "fstat must write into the stack scratch"
5619        );
5620        let used_words = std::mem::size_of::<libc::stat>().div_ceil(8);
5621        assert!(
5622            guest.arena[..used_words].iter().all(|word| *word == 0),
5623            "the stat must not be left in the guest's stack scratch"
5624        );
5625        assert_eq!(recorded_inode(&guest, fd), Some(inode));
5626    }
5627
5628    #[tokio::test]
5629    async fn faulting_stack_scratch_falls_back_to_a_transient_page() {
5630        let (fd, inode) = open_file();
5631        let (tool, mut guest) = ScriptedGuest::new(false, false);
5632
5633        let result = tool
5634            .add_fd(&mut guest, fd, OFlag::O_RDONLY, FdType::Regular)
5635            .await;
5636        close_unless_detcore_did(&guest, fd);
5637
5638        assert_eq!(
5639            result,
5640            Ok(()),
5641            "a stack that cannot hold the fstat buffer must not fail the open"
5642        );
5643        assert_eq!(guest.injected, [Sysno::mmap, Sysno::fstat, Sysno::munmap]);
5644        let [(page, len)] = guest.mapped[..] else {
5645            panic!(
5646                "expected exactly one transient page, got {:?}",
5647                guest.mapped
5648            );
5649        };
5650        assert_eq!(
5651            guest.fstat_buffers,
5652            [page],
5653            "fstat must write into the transient page"
5654        );
5655        assert_eq!(
5656            guest.unmapped,
5657            [(page, len)],
5658            "the transient page must be unmapped, whole"
5659        );
5660        assert_eq!(recorded_inode(&guest, fd), Some(inode));
5661    }
5662
5663    #[tokio::test]
5664    async fn descriptor_is_closed_when_no_scratch_can_be_found() {
5665        let (fd, _) = open_file();
5666        let (tool, mut guest) = ScriptedGuest::new(false, true);
5667
5668        let result = tool
5669            .add_fd(&mut guest, fd, OFlag::O_RDONLY, FdType::Regular)
5670            .await;
5671        close_unless_detcore_did(&guest, fd);
5672
5673        assert_eq!(result, Err(Errno::ENOMEM));
5674        assert_eq!(guest.injected, [Sysno::mmap, Sysno::close]);
5675        assert_eq!(
5676            guest.closed,
5677            [fd],
5678            "the descriptor must not stay open behind the error"
5679        );
5680        assert_eq!(
5681            guest.thread.with_detfd(fd, |_| ()),
5682            Err(Errno::EBADF),
5683            "a descriptor that failed registration must not be modeled"
5684        );
5685    }
5686}