Skip to main content

detcore/syscalls/
threads.rs

1/*
2 * Copyright (c) Meta Platforms, Inc. and affiliates.
3 * All rights reserved.
4 *
5 * This source code is licensed under the BSD-style license found in the
6 * LICENSE file in the root directory of this source tree.
7 */
8
9//! System calls for dealing with threads and concurrency.
10
11use std::marker::PhantomData;
12use std::sync::Arc;
13use std::sync::Mutex;
14use std::time::Duration;
15
16use procfs::process::Process;
17use rand::Rng;
18use reverie::Error;
19use reverie::Guest;
20use reverie::Pid;
21use reverie::Stack;
22use reverie::syscalls;
23use reverie::syscalls::Addr;
24use reverie::syscalls::AddrMut;
25use reverie::syscalls::CloneFlags;
26use reverie::syscalls::Errno;
27use reverie::syscalls::MemoryAccess;
28use reverie::syscalls::Syscall;
29use reverie::syscalls::SyscallInfo;
30use reverie::syscalls::Timespec;
31use reverie::syscalls::WaitPidFlag;
32use tracing::debug;
33use tracing::info;
34use tracing::trace;
35
36use crate::config::BlockingMode;
37use crate::memory::MemoryMetadata;
38use crate::record_or_replay::RecordOrReplay;
39use crate::resources::ExternalOpId;
40use crate::resources::Permission;
41use crate::resources::ResourceID;
42use crate::resources::Resources;
43use crate::scheduler::SchedValue;
44use crate::syscalls::helpers::record_retry_event;
45use crate::syscalls::helpers::retry_nonblocking_syscall;
46use crate::syscalls::helpers::retry_nonblocking_syscall_with_timeout;
47use crate::syscalls::robust_list;
48use crate::tool_global::FutexAction;
49use crate::tool_global::ResumeStatus;
50use crate::tool_global::await_exact_child_physical_exit;
51use crate::tool_global::cancel_exec;
52use crate::tool_global::child_tid_clear_address;
53use crate::tool_global::consume_child_wait;
54use crate::tool_global::create_child_thread;
55use crate::tool_global::futex_action;
56use crate::tool_global::prepare_exec;
57use crate::tool_global::process_group;
58use crate::tool_global::ready_child_wait;
59use crate::tool_global::resource_request;
60use crate::tool_global::set_child_tid_address;
61use crate::tool_global::thread_is_live;
62use crate::tool_global::thread_observe_time;
63use crate::tool_global::wait_for_child_lifecycle;
64use crate::tool_global::yield_once;
65use crate::tool_local::Detcore;
66use crate::tool_local::PendingVfork;
67use crate::tool_local::RobustListExit;
68use crate::tool_local::RobustListWake;
69use crate::types::ChildWaitExitClass;
70use crate::types::ChildWaitSelector;
71use crate::types::ChildWaitSpec;
72use crate::types::DetPid;
73use crate::types::DetTid;
74use crate::types::ExactChildWaitState;
75use crate::types::LogicalTime;
76use crate::types::SigWrapper;
77
78// Preserve the historical Detcore ABI while hiding the host's configured CPU
79// count. This represents one virtual CPU in a fixed 128-bit kernel mask.
80const VIRTUAL_CPUSET_BYTES: usize = 16;
81
82const IOPRIO_WHO_PROCESS: libc::c_int = 1;
83const IOPRIO_WHO_PGRP: libc::c_int = 2;
84const IOPRIO_WHO_USER: libc::c_int = 3;
85const IOPRIO_CLASS_SHIFT: libc::c_int = 13;
86const IOPRIO_CLASS_BE: libc::c_int = 2;
87const IOPRIO_BE_NORM: libc::c_int = 4;
88const IOPRIO_DEFAULT_EFFECTIVE: libc::c_int =
89    (IOPRIO_CLASS_BE << IOPRIO_CLASS_SHIFT) | IOPRIO_BE_NORM;
90
91// sched_attr wire contract, from include/uapi/linux/sched/types.h. These are
92// spelled out rather than derived from `size_of::<libc::sched_attr>()`: the
93// value the kernel reports back on the E2BIG paths and the offset past which
94// trailing bytes must be zero are properties of the KERNEL's struct, which is
95// larger than the libc crate's 48-byte mirror. Deriving either from a Rust type
96// would silently re-point the contract if that type ever grows a field.
97const SCHED_ATTR_SIZE_VER0: u32 = 48; // first published struct
98const SCHED_ATTR_SIZE_VER1: u32 = 56; // adds sched_util_{min,max}
99/// The kernel's own `sizeof(struct sched_attr)`. Two things key off it: bytes at
100/// or beyond this offset must be zero, and it is the value written back into
101/// `uattr->size` when the request is refused with E2BIG.
102const SCHED_ATTR_KERNEL_SIZE: u32 = SCHED_ATTR_SIZE_VER1;
103/// `sched_copy_attr` refuses a buffer larger than one page. Hermit is x86-64
104/// only, where PAGE_SIZE is 4096.
105const SCHED_ATTR_MAX_SIZE: u32 = 4096;
106
107/// The scheduling policy Detcore presents as every thread's current one.
108///
109/// This is not a guess about the host. It is the value `handle_sched_getattr`
110/// writes back for every thread, so it is the policy a guest observes and the
111/// only policy `SCHED_FLAG_KEEP_POLICY` can be keeping.
112const VIRTUAL_CURRENT_POLICY: u32 = libc::SCHED_OTHER as u32;
113
114/// Outcome of scanning the bytes past the kernel's `struct sched_attr`.
115#[derive(Debug, Eq, PartialEq)]
116enum TailVerdict {
117    AllZero,
118    /// A byte the kernel does not understand was set: `copy_struct_from_user`
119    /// reports this as E2BIG, after storing its own size back.
120    NotZeroed,
121    /// Guest memory could not be read before any non-zero byte was seen.
122    Faulted,
123}
124
125/// Reads of eight bytes or fewer take safeptrace's `PTRACE_PEEKDATA` path,
126/// which reads a whole aligned word and BYPASSES GUEST PAGE PROTECTIONS. Only a
127/// read strictly larger than that reaches `process_vm_readv`, which honours
128/// them. Note this is not symmetric with the write side: `write` special-cases
129/// a length of exactly eight, so splitting a write in half escapes it, while
130/// `read` special-cases eight *or fewer* and splitting only makes it worse.
131const MIN_PROTECTION_RESPECTING_READ: usize = std::mem::size_of::<u64>() + 1;
132
133/// Scan `tail_len` bytes at `base + tail_off` the way `check_zeroed_user` does.
134///
135/// Two things here are load-bearing and neither is visible from the outcome
136/// alone, only from the ORDER the outcome is reached in.
137///
138/// First, Linux scans FORWARD and stops at the first thing it finds. A non-zero
139/// byte followed later by an unreadable page is E2BIG, because the scan never
140/// reaches the page; an unreadable page reached before any non-zero byte is
141/// EFAULT. Reading the whole tail up front and judging afterwards collapses
142/// both into EFAULT and silently changes the errno a guest sees.
143///
144/// Second, every read here is kept strictly larger than eight bytes, for the
145/// reason on [`MIN_PROTECTION_RESPECTING_READ`]. Where the remainder is
146/// smaller, the window is extended BACKWARDS over bytes already scanned rather
147/// than shortened; re-reading a byte is free and only the new bytes are judged.
148fn scan_tail_is_zeroed<M: MemoryAccess>(
149    memory: &M,
150    base: AddrMut<u8>,
151    tail_off: usize,
152    tail_len: usize,
153) -> TailVerdict {
154    const CHUNK: usize = 256;
155    let mut done = 0usize;
156    let mut buf = [0u8; CHUNK];
157    while done < tail_len {
158        let remaining = tail_len - done;
159        let want = remaining.min(CHUNK);
160        // Extend backwards when the remainder alone would be a small read.
161        // The window may reach back past the start of the tail into the
162        // `sched_attr` prefix: the kernel has already read those bytes, so they
163        // are readable, and only the new bytes are judged below. Clamping this
164        // to `done` alone left a 1-byte tail with nowhere to grow into and
165        // issued exactly the small read this constant exists to avoid.
166        let back = MIN_PROTECTION_RESPECTING_READ
167            .saturating_sub(want)
168            .min(done + tail_off);
169        let read_len = want + back;
170        let start = tail_off + done - back;
171        let Some(addr) = base
172            .as_raw()
173            .checked_add(start)
174            .and_then(AddrMut::<u8>::from_raw)
175        else {
176            return TailVerdict::Faulted;
177        };
178        // A failed read means zero NEW bytes were obtained; the `got < read_len`
179        // check below is what turns that into `Faulted`. Written `unwrap_or(0)`
180        // rather than `unwrap_or_default()` so the 0 stays visible.
181        let got = memory.read(addr, &mut buf[..read_len]).unwrap_or(0);
182        // Judge only the bytes that are genuinely new AND genuinely read.
183        let new_from = back.min(got);
184        if buf[new_from..got].iter().any(|byte| *byte != 0) {
185            return TailVerdict::NotZeroed;
186        }
187        if got < read_len {
188            // The scan stopped at an unreadable byte with nothing non-zero
189            // before it, which is `check_zeroed_user`'s -EFAULT.
190            return TailVerdict::Faulted;
191        }
192        done += want;
193    }
194    TailVerdict::AllZero
195}
196
197// Scheduling policies, from include/uapi/linux/sched.h.
198const SCHED_FIFO: u32 = 1;
199const SCHED_RR: u32 = 2;
200const SCHED_DEADLINE: u32 = 6;
201const SCHED_EXT: u32 = 7;
202/// The kernel's `MAX_RT_PRIO`; valid real-time priorities are 1..=99.
203const MAX_RT_PRIO: u32 = 100;
204
205// Byte offsets of the fields this handler inspects. Taken from the UAPI struct
206// rather than from a Rust mirror, for the reason given above.
207const SCHED_ATTR_OFF_POLICY: usize = 4;
208const SCHED_ATTR_OFF_FLAGS: usize = 8;
209const SCHED_ATTR_OFF_PRIORITY: usize = 20;
210const SCHED_ATTR_OFF_RUNTIME: usize = 24;
211const SCHED_ATTR_OFF_DEADLINE: usize = 32;
212const SCHED_ATTR_OFF_PERIOD: usize = 40;
213
214// sched_attr.sched_flags bits, from include/uapi/linux/sched.h.
215const SCHED_FLAG_RESET_ON_FORK: u64 = 0x01;
216const SCHED_FLAG_RECLAIM: u64 = 0x02;
217const SCHED_FLAG_DL_OVERRUN: u64 = 0x04;
218const SCHED_FLAG_KEEP_POLICY: u64 = 0x08;
219const SCHED_FLAG_KEEP_PARAMS: u64 = 0x10;
220const SCHED_FLAG_UTIL_CLAMP_MIN: u64 = 0x20;
221const SCHED_FLAG_UTIL_CLAMP_MAX: u64 = 0x40;
222const SCHED_FLAG_UTIL_CLAMP: u64 = SCHED_FLAG_UTIL_CLAMP_MIN | SCHED_FLAG_UTIL_CLAMP_MAX;
223const SCHED_FLAG_ALL: u64 = SCHED_FLAG_RESET_ON_FORK
224    | SCHED_FLAG_RECLAIM
225    | SCHED_FLAG_DL_OVERRUN
226    | SCHED_FLAG_KEEP_POLICY
227    | SCHED_FLAG_KEEP_PARAMS
228    | SCHED_FLAG_UTIL_CLAMP;
229
230/// Whether Linux accepts `policy` as a scheduling policy for `sched_setattr`.
231/// This is the kernel's `valid_policy()`: idle, fair, rt, deadline or ext. Note
232/// the gap at 4 -- SCHED_ISO is reserved and never valid -- and that 7 is
233/// SCHED_EXT, which `valid_policy()` accepts on a kernel built with
234/// CONFIG_SCHED_CLASS_EXT.
235fn is_valid_sched_policy(policy: u32) -> bool {
236    matches!(
237        policy,
238        // SCHED_OTHER/NORMAL, FIFO, RR, BATCH
239        0 | 1 | 2 | 3
240        // SCHED_IDLE
241        | 5
242        | SCHED_DEADLINE
243        | SCHED_EXT
244    )
245}
246
247/// The kernel's `rt_policy()`: the two fixed-priority real-time policies.
248fn is_rt_policy(policy: u32) -> bool {
249    policy == SCHED_FIFO || policy == SCHED_RR
250}
251
252/// The kernel's `__checkparam_dl()`, restricted to the parts that are pure ABI.
253///
254/// The kernel's final test compares the period against
255/// `sysctl_sched_dl_period_{min,max}`, which are runtime-tunable. That bound is
256/// a property of the host's configuration rather than of the interface, so it
257/// is deliberately not reproduced here -- see the handler's doc comment for the
258/// other cases in that class.
259fn deadline_params_are_valid(runtime: u64, deadline: u64, period: u64) -> bool {
260    // deadline != 0
261    if deadline == 0 {
262        return false;
263    }
264    // The kernel truncates DL_SCALE (10) bits, so runtime must be at least that
265    // big to survive the truncation.
266    if runtime < (1u64 << 10) {
267        return false;
268    }
269    // The MSB is reserved for wrap-around and sign handling.
270    if deadline & (1u64 << 63) != 0 || period & (1u64 << 63) != 0 {
271        return false;
272    }
273    // A zero period means "same as the deadline".
274    let period = if period == 0 { deadline } else { period };
275    // runtime <= deadline <= period
276    runtime <= deadline && deadline <= period
277}
278
279/// Apply `sched_copy_attr`'s size rules to the `size` the guest declared.
280///
281/// `Ok(n)` is the effective size to copy with; `Err(())` means the request is
282/// refused with E2BIG *and* the kernel first writes its own struct size back
283/// into `uattr->size`, so the caller owes that store.
284///
285/// The zero case is not a mistake: the kernel carries an explicit ABI
286/// compatibility quirk, `if (!size) size = SCHED_ATTR_SIZE_VER0;`, so a
287/// zero-sized request is a well-formed VER0 request and succeeds.
288fn sched_attr_effective_size(declared: u32) -> Result<u32, ()> {
289    let size = if declared == 0 {
290        SCHED_ATTR_SIZE_VER0
291    } else {
292        declared
293    };
294    if !(SCHED_ATTR_SIZE_VER0..=SCHED_ATTR_MAX_SIZE).contains(&size) {
295        return Err(());
296    }
297    Ok(size)
298}
299
300/// The descriptor fields this handler inspects, decoded from the guest's bytes.
301#[derive(Clone, Copy, Debug)]
302struct SchedAttrFields {
303    policy: u32,
304    sched_flags: u64,
305    priority: u32,
306    runtime: u64,
307    deadline: u64,
308    period: u64,
309}
310
311/// The checks that run in `sched_copy_attr` and in the `sched_setattr` wrapper,
312/// i.e. everything the kernel decides *before* it looks the target pid up.
313///
314/// `size` is the effective size from [`sched_attr_effective_size`], so the
315/// util-clamp rule can be stated in the same terms the kernel uses.
316///
317/// Measured: with a pid that does not exist, both of these still report their
318/// own errno rather than ESRCH, which is what places them on this side of the
319/// lookup.
320fn validate_sched_attr_before_lookup(size: u32, attr: &SchedAttrFields) -> Result<(), Errno> {
321    // `sched_copy_attr`: util-clamp lives past the VER0 tail, so asking for it
322    // with a VER0 buffer is incoherent.
323    if attr.sched_flags & SCHED_FLAG_UTIL_CLAMP != 0 && size < SCHED_ATTR_SIZE_VER1 {
324        return Err(Errno::EINVAL);
325    }
326    // `sched_setattr`: the policy is compared as a *signed* int, and this test
327    // sits above the KEEP_POLICY substitution below, so the sign bit is refused
328    // even when the policy field is otherwise ignored.
329    if (attr.policy as i32) < 0 {
330        return Err(Errno::EINVAL);
331    }
332    Ok(())
333}
334
335/// The checks inside `__sched_setscheduler`, i.e. everything the kernel decides
336/// *after* it has resolved the target pid.
337///
338/// Measured: with a pid that does not exist, every rule here reports ESRCH
339/// instead, which is what places them on this side of the lookup.
340fn validate_sched_attr_after_lookup(attr: &SchedAttrFields) -> Result<(), Errno> {
341    // `sched_setattr` rewrites the policy to SETPARAM_POLICY (-1) when
342    // KEEP_POLICY is set, and `__sched_setscheduler` then takes its
343    // `policy < 0` branch. That branch does not skip the policy-dependent
344    // rules -- it REUSES THE TASK'S CURRENT POLICY and applies them against
345    // that. `valid_policy()` is the only thing it skips, which is why an
346    // undefined value in the ignored field still passes.
347    //
348    // Detcore's current policy is not unknown: `handle_sched_getattr` reports
349    // SCHED_OTHER, nice 0, priority 0 for every thread, unconditionally, and
350    // that is the whole of the guest-visible scheduling state this sandbox
351    // exposes. So KEEP_POLICY substitutes SCHED_OTHER here. Treating it as
352    // "no policy" and skipping the rules accepted, for instance,
353    // KEEP_POLICY with sched_priority=1, which Linux refuses because a
354    // non-real-time current policy requires priority 0.
355    //
356    // These two sites must stay in step: if `handle_sched_getattr` ever
357    // reports a different virtual policy, this substitution follows it.
358    let policy = if attr.sched_flags & SCHED_FLAG_KEEP_POLICY != 0 {
359        Some(VIRTUAL_CURRENT_POLICY)
360    } else {
361        if !is_valid_sched_policy(attr.policy) {
362            return Err(Errno::EINVAL);
363        }
364        Some(attr.policy)
365    };
366    // No undefined sched_flags bits.
367    if attr.sched_flags & !SCHED_FLAG_ALL != 0 {
368        return Err(Errno::EINVAL);
369    }
370    // Valid priorities are 1..=MAX_RT_PRIO-1 for the real-time policies and
371    // exactly 0 for every other policy, so the priority and the policy have to
372    // agree; the kernel states that as `rt_policy(policy) != (prio != 0)`.
373    if attr.priority > MAX_RT_PRIO - 1 {
374        return Err(Errno::EINVAL);
375    }
376    if let Some(policy) = policy {
377        if is_rt_policy(policy) != (attr.priority != 0) {
378            return Err(Errno::EINVAL);
379        }
380        if policy == SCHED_DEADLINE
381            && !deadline_params_are_valid(attr.runtime, attr.deadline, attr.period)
382        {
383            return Err(Errno::EINVAL);
384        }
385    }
386    Ok(())
387}
388
389// AUTONOMOUS-BOT-IMPLEMENTED
390// TODO-HUMAN-REVIEW(PR-881)
391fn virtual_ioprio(which: libc::c_int) -> Result<i64, Errno> {
392    match which {
393        IOPRIO_WHO_PROCESS => Ok(0),
394        IOPRIO_WHO_PGRP | IOPRIO_WHO_USER => Ok(i64::from(IOPRIO_DEFAULT_EFFECTIVE)),
395        _ => Err(Errno::EINVAL),
396    }
397}
398
399#[repr(C)]
400#[derive(Clone, Copy)]
401struct WaitidSigchldFields {
402    pid: libc::pid_t,
403    uid: libc::uid_t,
404    status: libc::c_int,
405    utime: libc::c_long,
406    stime: libc::c_long,
407}
408
409#[repr(C)]
410union WaitidSiginfoFields {
411    _alignment: *mut libc::c_void,
412    sigchld: WaitidSigchldFields,
413}
414
415#[repr(C)]
416struct WaitidSiginfoHead {
417    _base: [libc::c_int; 3],
418    fields: WaitidSiginfoFields,
419}
420
421fn wait_status_is_termination(status: libc::c_int) -> bool {
422    libc::WIFEXITED(status) || libc::WIFSIGNALED(status)
423}
424
425fn waitid_code_is_termination(code: libc::c_int) -> bool {
426    matches!(code, libc::CLD_EXITED | libc::CLD_KILLED | libc::CLD_DUMPED)
427}
428
429fn canonicalize_waitid_siginfo(info: &mut libc::siginfo_t) {
430    debug_assert!(
431        std::mem::size_of::<WaitidSiginfoHead>() <= std::mem::size_of::<libc::siginfo_t>()
432    );
433    // SAFETY: Linux siginfo_t starts with three c_int fields followed by a
434    // pointer-aligned union. Its SIGCHLD member is pid, uid, status, utime,
435    // and stime in that order. The local repr(C) mirror changes only the two
436    // host CPU-accounting fields and preserves the kernel-populated event.
437    let sigchld = unsafe {
438        &mut (*(info as *mut libc::siginfo_t).cast::<WaitidSiginfoHead>())
439            .fields
440            .sigchld
441    };
442    sigchld.utime = 0;
443    sigchld.stime = 0;
444}
445
446fn finish_waitid_result<T, G>(
447    guest: &mut G,
448    call: syscalls::Waitid,
449    value: i64,
450    mut info_value: libc::siginfo_t,
451) -> Result<i64, Error>
452where
453    T: RecordOrReplay,
454    G: Guest<Detcore<T>>,
455{
456    // SAFETY: waitid writes either zeroed output or the SIGCHLD siginfo_t
457    // variant, for which libc exposes si_pid.
458    let child_pid = unsafe { info_value.si_pid() };
459    if child_pid != 0 {
460        canonicalize_waitid_siginfo(&mut info_value);
461        guest.memory().write_value(
462            call.info().expect("waitid infop checked before execution"),
463            &info_value,
464        )?;
465        if call.options() & libc::WNOWAIT == 0 && waitid_code_is_termination(info_value.si_code) {
466            guest
467                .thread_state_mut()
468                .reap_child_process_cpu_time(DetPid::from_raw(child_pid));
469        }
470        if let Some(rusage) = call.rusage() {
471            // Host CPU and scheduling counters are not deterministic.
472            let usage: libc::rusage = unsafe { std::mem::zeroed() };
473            guest.memory().write_value(rusage, &usage)?;
474        }
475    }
476    Ok(value)
477}
478
479#[derive(Debug, Eq, PartialEq)]
480enum ExactWaitPollDecision {
481    ChildReady,
482    AwaitPhysicalExit,
483    ReapAfterLogicalExit,
484    Interrupted,
485    Retry,
486}
487
488fn exact_wait_poll_decision(
489    child_ready: bool,
490    signaled: bool,
491    lifecycle: Option<ExactChildWaitState>,
492) -> ExactWaitPollDecision {
493    if child_ready {
494        ExactWaitPollDecision::ChildReady
495    } else if lifecycle == Some(ExactChildWaitState::PhysicalExitPending) {
496        ExactWaitPollDecision::AwaitPhysicalExit
497    } else if matches!(
498        lifecycle,
499        Some(ExactChildWaitState::LogicallyExited | ExactChildWaitState::PhysicallyExited)
500    ) {
501        ExactWaitPollDecision::ReapAfterLogicalExit
502    } else if signaled {
503        ExactWaitPollDecision::Interrupted
504    } else {
505        ExactWaitPollDecision::Retry
506    }
507}
508
509fn stale_any_wait_must_interrupt(signaled: bool, next_ready: Option<DetPid>) -> bool {
510    signaled && next_ready.is_none()
511}
512
513fn terminal_child_wait_spec(
514    selector: ChildWaitSelector,
515    caller: DetTid,
516    options: libc::c_int,
517) -> ChildWaitSpec {
518    let exit_class = if options & libc::__WALL != 0 {
519        ChildWaitExitClass::Any
520    } else if options & libc::__WCLONE != 0 {
521        ChildWaitExitClass::Clone
522    } else {
523        ChildWaitExitClass::Sigchld
524    };
525    ChildWaitSpec {
526        selector,
527        owner: (options & libc::__WNOTHREAD != 0).then_some(caller),
528        exit_class,
529    }
530}
531
532fn validate_wait4_arguments(pid: libc::pid_t, options: WaitPidFlag) -> Result<(), Errno> {
533    let allowed_options = WaitPidFlag::WNOHANG
534        | WaitPidFlag::WUNTRACED
535        | WaitPidFlag::WCONTINUED
536        | WaitPidFlag::__WNOTHREAD
537        | WaitPidFlag::__WCLONE
538        | WaitPidFlag::__WALL;
539
540    // Linux kernel/exit.c:kernel_wait4 rejects unknown option bits before it
541    // interprets the pid selector.
542    if options.bits() & !allowed_options.bits() != 0 {
543        return Err(Errno::EINVAL);
544    }
545    // The same function returns ESRCH for INT_MIN before looking for children,
546    // because negating that selector is not representable as a pid_t.
547    if pid == libc::pid_t::MIN {
548        return Err(Errno::ESRCH);
549    }
550    Ok(())
551}
552
553fn child_wait_can_retry_after_stale(spec: ChildWaitSpec) -> bool {
554    !matches!(spec.selector, ChildWaitSelector::Exact(_))
555}
556
557pub(super) type KernelSigset = u64;
558pub(super) const KERNEL_SIGSET_SIZE: usize = std::mem::size_of::<KernelSigset>();
559
560fn signal_is_blocked(mask: &KernelSigset, signal: SigWrapper) -> bool {
561    let raw_signal = signal.raw();
562    (1..=KernelSigset::BITS as i32).contains(&raw_signal)
563        && mask & (1_u64 << (raw_signal as u32 - 1)) != 0
564}
565
566#[repr(C)]
567#[derive(Clone, Copy)]
568pub(super) struct KernelSigaction {
569    pub(super) handler: u64,
570    pub(super) flags: u64,
571    pub(super) restorer: u64,
572    pub(super) mask: KernelSigset,
573}
574
575#[derive(Clone, Copy, Debug, Eq, PartialEq)]
576pub(super) enum WaitSignalDisposition {
577    Interrupt,
578    Restart,
579}
580
581fn signal_default_disposition_does_not_interrupt_child_wait(signal: SigWrapper) -> bool {
582    matches!(
583        signal.raw(),
584        libc::SIGCHLD | libc::SIGCONT | libc::SIGURG | libc::SIGWINCH
585    )
586}
587
588fn signal_has_uncatchable_default_disposition(signal: SigWrapper) -> bool {
589    matches!(signal.raw(), libc::SIGKILL | libc::SIGSTOP)
590}
591
592pub(super) async fn wait_signal_disposition<G, T>(
593    guest: &mut G,
594    status: ResumeStatus,
595    guest_signal_mask: &KernelSigset,
596    action_addr: AddrMut<'_, KernelSigaction>,
597    inspect_action: bool,
598) -> Result<Option<WaitSignalDisposition>, Error>
599where
600    G: Guest<Detcore<T>>,
601    T: RecordOrReplay,
602{
603    let ResumeStatus::Signaled(signals) = status else {
604        return Ok(None);
605    };
606    let Some(mut signals) = signals else {
607        return Ok(Some(WaitSignalDisposition::Interrupt));
608    };
609    signals.sort_by_key(|signal| signal.raw());
610    for signal in signals {
611        if signal_is_blocked(guest_signal_mask, signal) {
612            continue;
613        }
614        if signal_has_uncatchable_default_disposition(signal) {
615            return Ok(Some(WaitSignalDisposition::Interrupt));
616        }
617        if !inspect_action {
618            return Ok(Some(WaitSignalDisposition::Interrupt));
619        }
620        let call = syscalls::RtSigaction::new()
621            .with_signum(signal.raw())
622            .with_action(None)
623            .with_old_action(Some(action_addr.cast()))
624            .with_sigsetsize(std::mem::size_of::<u64>());
625        guest.inject_with_retry(call).await?;
626        let action: KernelSigaction = guest.memory().read_value(action_addr)?;
627        if action.handler == libc::SIG_IGN as u64
628            || action.handler == libc::SIG_DFL as u64
629                && signal_default_disposition_does_not_interrupt_child_wait(signal)
630        {
631            continue;
632        }
633        return Ok(Some(
634            if action.handler != libc::SIG_DFL as u64 && action.flags & libc::SA_RESTART as u64 != 0
635            {
636                WaitSignalDisposition::Restart
637            } else {
638                WaitSignalDisposition::Interrupt
639            },
640        ));
641    }
642    Ok(None)
643}
644
645pub(super) fn blocked_signal_mask() -> KernelSigset {
646    // Preserve libc's definition of the blockable set (notably its reserved
647    // NPTL signals) while converting the result to the kernel's one-word ABI.
648    let mut libc_mask: libc::sigset_t = unsafe { std::mem::zeroed() };
649    unsafe {
650        libc::sigfillset(&mut libc_mask);
651        libc::sigdelset(&mut libc_mask, reverie::PERF_EVENT_SIGNAL as i32);
652    }
653    (1..=KernelSigset::BITS as i32).fold(0, |mask, raw_signal| {
654        if unsafe { libc::sigismember(&libc_mask, raw_signal) } == 1 {
655            mask | (1_u64 << (raw_signal as u32 - 1))
656        } else {
657            mask
658        }
659    })
660}
661
662pub(super) async fn block_signals_for_disposition<G, T>(
663    guest: &mut G,
664    blocked_mask_addr: Addr<'_, KernelSigset>,
665    old_mask_addr: AddrMut<'_, KernelSigset>,
666) -> Result<KernelSigset, Error>
667where
668    G: Guest<Detcore<T>>,
669    T: RecordOrReplay,
670{
671    let block_signals = syscalls::RtSigprocmask::new()
672        .with_how(libc::SIG_SETMASK)
673        .with_set(
674            (!guest
675                .config()
676                .backend_requires_thread_directed_process_signals)
677                .then_some(blocked_mask_addr.cast()),
678        )
679        .with_oldset(Some(old_mask_addr.cast()))
680        .with_sigsetsize(KERNEL_SIGSET_SIZE);
681    guest.inject_with_retry(block_signals).await?;
682    Ok(guest.memory().read_value(old_mask_addr)?)
683}
684
685pub(super) async fn restore_signals_after_disposition<G, T>(
686    guest: &mut G,
687    old_mask_addr: AddrMut<'_, KernelSigset>,
688) -> Result<(), Error>
689where
690    G: Guest<Detcore<T>>,
691    T: RecordOrReplay,
692{
693    if !guest
694        .config()
695        .backend_requires_thread_directed_process_signals
696    {
697        let old_mask: Addr<'_, KernelSigset> = old_mask_addr.into();
698        let restore_signals = syscalls::RtSigprocmask::new()
699            .with_how(libc::SIG_SETMASK)
700            .with_set(Some(old_mask.cast()))
701            .with_oldset(None)
702            .with_sigsetsize(KERNEL_SIGSET_SIZE);
703        guest.inject_with_retry(restore_signals).await?;
704    }
705    Ok(())
706}
707
708async fn interrupted_child_wait_result<G, T, S>(
709    guest: &mut G,
710    call: S,
711    disposition: WaitSignalDisposition,
712) -> Result<i64, Error>
713where
714    G: Guest<Detcore<T>>,
715    T: RecordOrReplay,
716    S: SyscallInfo,
717{
718    if !guest
719        .config()
720        .backend_requires_thread_directed_process_signals
721    {
722        return Err(Errno::ERESTARTSYS.into());
723    }
724    if disposition == WaitSignalDisposition::Interrupt {
725        return Err(Errno::EINTR.into());
726    }
727
728    guest.tail_inject(call).await
729}
730
731fn snapshot_process_group(pid: Pid) -> Result<libc::pid_t, Errno> {
732    let pgrp = Process::new(pid.as_raw())
733        .and_then(|process| process.stat())
734        .map(|stat| stat.pgrp)
735        .map_err(|_| Errno::ESRCH)?;
736    if pgrp == 0 {
737        Err(Errno::EOPNOTSUPP)
738    } else {
739        Ok(pgrp)
740    }
741}
742
743fn guest_fd_status_flags(pid: Pid, fd: libc::c_int) -> Result<libc::c_int, Errno> {
744    let path = format!("/proc/{}/fdinfo/{}", pid.as_raw(), fd);
745    let contents = std::fs::read_to_string(path).map_err(|_| Errno::EBADF)?;
746    let flags = contents
747        .lines()
748        .find_map(|line| line.strip_prefix("flags:"))
749        .map(str::trim)
750        .ok_or(Errno::EINVAL)?;
751    libc::c_int::from_str_radix(flags, 8).map_err(|_| Errno::EINVAL)
752}
753
754#[derive(Debug, Clone, Copy, PartialEq, Eq)]
755enum FutexTimeout {
756    Relative(u64),
757    Absolute(LogicalTime),
758}
759
760fn parse_futex_timeout(futex_op: i32, timeout: Timespec) -> Result<FutexTimeout, Errno> {
761    let seconds = u64::try_from(timeout.tv_sec).map_err(|_| Errno::EINVAL)?;
762    let nanoseconds = u64::try_from(timeout.tv_nsec).map_err(|_| Errno::EINVAL)?;
763    if nanoseconds >= 1_000_000_000 {
764        return Err(Errno::EINVAL);
765    }
766
767    let timeout_nanos = seconds
768        .checked_mul(1_000_000_000)
769        .and_then(|nanos| nanos.checked_add(nanoseconds))
770        .ok_or(Errno::EINVAL)?;
771    // Mask off FUTEX_PRIVATE_FLAG / FUTEX_CLOCK_REALTIME before matching the
772    // command: FUTEX_WAIT_BITSET measures its timeout as an *absolute* deadline,
773    // whereas plain FUTEX_WAIT uses a *relative* one. A private-flagged
774    // FUTEX_WAIT_BITSET (e.g. 0x89) must still be recognized as the BITSET
775    // command; comparing the raw op would misclassify it as relative and add
776    // the absolute deadline to the current time (leaking the epoch).
777    if futex_op & libc::FUTEX_CMD_MASK == libc::FUTEX_WAIT_BITSET {
778        Ok(FutexTimeout::Absolute(LogicalTime::from_nanos(
779            timeout_nanos,
780        )))
781    } else {
782        Ok(FutexTimeout::Relative(timeout_nanos))
783    }
784}
785
786fn rebase_absolute_timeout(
787    deadline: LogicalTime,
788    clock_now: LogicalTime,
789    logical_now: LogicalTime,
790) -> LogicalTime {
791    logical_now + Duration::from_nanos(deadline.as_nanos().saturating_sub(clock_now.as_nanos()))
792}
793
794fn absolute_timeout_uses_host_clock(
795    deadline: LogicalTime,
796    host_clock_now: LogicalTime,
797    logical_now: LogicalTime,
798) -> bool {
799    deadline.as_nanos().abs_diff(host_clock_now.as_nanos())
800        < deadline.as_nanos().abs_diff(logical_now.as_nanos())
801}
802
803impl<T: RecordOrReplay> Detcore<T> {
804    async fn futex_timeout_deadline<G: Guest<Self>>(
805        &self,
806        guest: &mut G,
807        futex_flags: i32,
808        timeout: Option<Addr<'_, Timespec>>,
809    ) -> Result<Option<LogicalTime>, Error> {
810        let Some(timeout) = timeout else {
811            return Ok(None);
812        };
813        let timeout = parse_futex_timeout(futex_flags, guest.memory().read_value(timeout)?)?;
814        match timeout {
815            FutexTimeout::Relative(nanos) => {
816                let now = thread_observe_time(guest).await;
817                Ok(Some(now + Duration::from_nanos(nanos)))
818            }
819            FutexTimeout::Absolute(deadline)
820                if self.cfg.virtualize_time && !self.cfg.detect_host_clock_futex_timeouts =>
821            {
822                Ok(Some(deadline))
823            }
824            FutexTimeout::Absolute(deadline) => {
825                let clockid = if futex_flags & libc::FUTEX_CLOCK_REALTIME != 0 {
826                    syscalls::ClockId::CLOCK_REALTIME
827                } else {
828                    syscalls::ClockId::CLOCK_MONOTONIC
829                };
830
831                let mut stack = guest.stack().await;
832                let clock_output = syscalls::TimespecMutPtr(stack.reserve());
833                let _stack_guard = stack.commit()?;
834                let clock_call = syscalls::ClockGettime::new()
835                    .with_clockid(clockid)
836                    .with_tp(Some(clock_output));
837                if self.cfg.virtualize_time && self.cfg.detect_host_clock_futex_timeouts {
838                    // Read the same live host clock as a direct guest vDSO call. Replaying a
839                    // recorded value here would compare this run's host-domain deadline with the
840                    // previous run's clock and turn a short timeout into an arbitrary long one.
841                    guest.inject(Syscall::from(clock_call)).await?;
842                } else {
843                    self.record_or_replay(guest, clock_call).await?;
844                }
845                let clock_now = match parse_futex_timeout(
846                    libc::FUTEX_WAIT_BITSET,
847                    guest.memory().read_value(clock_output.0)?,
848                )? {
849                    FutexTimeout::Absolute(time) => time,
850                    FutexTimeout::Relative(_) => unreachable!(),
851                };
852                let logical_now = thread_observe_time(guest).await;
853                if self.cfg.virtualize_time
854                    && !absolute_timeout_uses_host_clock(deadline, clock_now, logical_now)
855                {
856                    return Ok(Some(deadline));
857                }
858                Ok(Some(rebase_absolute_timeout(
859                    deadline,
860                    clock_now,
861                    logical_now,
862                )))
863            }
864        }
865    }
866
867    /// Clone, clone3, fork, vfork system calls
868    pub async fn handle_clone_family<G: Guest<Self>>(
869        &self,
870        guest: &mut G,
871        clone_family: syscalls::family::CloneFamily,
872    ) -> Result<i64, Error> {
873        let flags = clone_family.flags(&guest.memory());
874        let exit_signal = match clone_family {
875            #[cfg(not(target_arch = "aarch64"))]
876            syscalls::family::CloneFamily::Fork(_) | syscalls::family::CloneFamily::Vfork(_) => {
877                libc::SIGCHLD
878            }
879            syscalls::family::CloneFamily::Clone(clone) => {
880                (clone.flags().bits() & 0xff) as libc::c_int
881            }
882            syscalls::family::CloneFamily::Clone3(clone) => clone
883                .args()
884                .and_then(|address| guest.memory().read_value(address).ok())
885                .map_or(0, |args: syscalls::CloneArgs| {
886                    args.exit_signal as libc::c_int
887                }),
888        };
889        let ctid = child_tid_clear_address(flags, clone_family.child_tid(&guest.memory()));
890        let is_vfork = flags.contains(CloneFlags::CLONE_VFORK);
891        let parent_blocks_for_child = is_vfork
892            || (self.cfg.backend_serializes_fork_children
893                && !flags.contains(CloneFlags::CLONE_THREAD));
894        let backend_uninstrumented_thread =
895            flags.contains(CloneFlags::CLONE_THREAD) && !self.cfg.backend_dispatches_thread_tools;
896
897        let ts = guest.thread_state_mut();
898        assert_eq!(ts.clone_flags, None);
899        assert!(ts.pending_vfork.is_none());
900        ts.clone_flags = Some(flags);
901
902        let parent_dettid = ts.dettid;
903        let child_priority_entropy = if parent_blocks_for_child
904            && self.cfg.chaos
905            && self.cfg.replay_preemptions_from.is_none()
906            && self.cfg.replay_schedule_from.is_none()
907        {
908            let mut parent_chaos_prng = ts.chaos_prng.clone();
909            Some(parent_chaos_prng.next_u64())
910        } else {
911            None
912        };
913        if parent_blocks_for_child {
914            ts.pending_vfork = Some(PendingVfork {
915                parent_dettid,
916                parent_detpid: ts.detpid.expect("detpid unset"),
917                child_tid_addr: ctid,
918                flags,
919                exit_signal,
920                child_priority_entropy,
921            });
922        }
923
924        trace!("[detcore, dtid {}] parent invoking clone.", parent_dettid);
925        let blocking_child_op_id =
926            ExternalOpId::new(parent_dettid, guest.thread_state().stats.syscall_count);
927
928        // A CLONE_VFORK parent and a backend-serialized fork parent cannot
929        // resume until the child exits. Relinquish the parent's scheduler turn
930        // before entering either blocking operation.
931        if parent_blocks_for_child && self.cfg.sequentialize_threads {
932            let mut resources = Resources::new(parent_dettid);
933            resources.insert(
934                ResourceID::BlockingVfork(blocking_child_op_id),
935                Permission::RW,
936            );
937            resources.fyi(if is_vfork {
938                "clone_vfork"
939            } else {
940                "clone_serialized_child"
941            });
942            resource_request(guest, resources).await;
943        }
944
945        let maybe_res = guest.inject(Syscall::from(clone_family)).await;
946
947        if parent_blocks_for_child && self.cfg.sequentialize_threads {
948            let mut resources = Resources::new(parent_dettid);
949            if maybe_res.is_err() {
950                // TODO-HUMAN-REVIEW(PR-1152): Review failed deferred-vfork cancellation.
951                // A deferred-spawn backend cannot infer failure from the absence of a registered
952                // child: a successful child also registers after this continuation. Report the
953                // known injected-syscall outcome explicitly so the scheduler can cancel the
954                // barrier and re-admit this parent before we propagate the original error below.
955                resources.insert(
956                    ResourceID::VforkFailed(blocking_child_op_id),
957                    Permission::RW,
958                );
959                resources.fyi(if is_vfork {
960                    "clone_vfork_failed"
961                } else {
962                    "clone_serialized_child_failed"
963                });
964            } else {
965                resources.insert(
966                    ResourceID::BlockedExternalContinue(blocking_child_op_id),
967                    Permission::RW,
968                );
969                resources.fyi(if is_vfork {
970                    "clone_vfork"
971                } else {
972                    "clone_serialized_child"
973                });
974            }
975            resource_request(guest, resources).await;
976        }
977
978        let ts = guest.thread_state_mut();
979        ts.clone_flags = None; // Unset, now that it has been read by the child.
980        ts.pending_vfork = None;
981
982        let res = maybe_res?;
983
984        if !flags.contains(CloneFlags::CLONE_THREAD) {
985            // Only a successful process clone can let another process mutate
986            // inherited open file descriptions. Failed clone-family calls leave
987            // the previously observed flock state authoritative.
988            guest.thread_state().forget_flock_modes();
989        }
990
991        // Match ordinary clone: the parent consumes the priority entropy after
992        // the child has inherited the parent state.
993        if parent_blocks_for_child
994            && self.cfg.chaos
995            && self.cfg.replay_preemptions_from.is_none()
996            && self.cfg.replay_schedule_from.is_none()
997        {
998            let _ = guest
999                .thread_state_mut()
1000                .chaos_prng_next_u64("child_priority");
1001        }
1002
1003        let child_tid = Pid::from_raw(res as i32);
1004        let child_dettid = DetTid::from_raw(child_tid.into()); // TODO(T78538674), virtualized tid/pid
1005        trace!(
1006            "[detcore] dtid {} cloned, continuing parent + register new thread.",
1007            child_dettid
1008        );
1009
1010        if !parent_blocks_for_child && !backend_uninstrumented_thread {
1011            create_child_thread(guest, child_dettid, ctid, Some(flags), exit_signal, None).await;
1012        }
1013
1014        {
1015            // The child will have updated their pedigree, we update ours before continuing.
1016            let parent_pedigree = &mut guest.thread_state_mut().pedigree;
1017            let child_pedigree = parent_pedigree.fork_mut();
1018            debug!(
1019                "[dtid {}] after creating child thread (tid {}, pedigree {}) parents pedigree becomes {}",
1020                parent_dettid, child_dettid, child_pedigree, parent_pedigree,
1021            );
1022        }
1023
1024        Ok(child_dettid.as_raw() as i64)
1025    }
1026
1027    // TODO-HUMAN-REVIEW(PR-2985): Review scheduler tracking of set_tid_address.
1028    /// `set_tid_address` system call.
1029    ///
1030    /// Linux owns the guest-visible registration and return value. Detcore
1031    /// mirrors the accepted address into the scheduler so its modeled
1032    /// CHILD_CLEARTID wake targets the same word as the backend.
1033    pub async fn handle_set_tid_address<G: Guest<Self>>(
1034        &self,
1035        guest: &mut G,
1036        call: syscalls::SetTidAddress,
1037    ) -> Result<i64, Error> {
1038        let address = call.tidptr().map_or(0, |pointer| pointer.as_raw());
1039        let result = self
1040            .record_or_replay(guest, Syscall::SetTidAddress(call))
1041            .await?;
1042        if guest.config().sequentialize_threads {
1043            set_child_tid_address(guest, address).await;
1044        }
1045        trace!(
1046            "[detcore, dtid {}] child TID clear address registered: {address:#x}",
1047            guest.thread_state().dettid,
1048        );
1049        Ok(result)
1050    }
1051
1052    /// `set_robust_list` system call.
1053    ///
1054    /// Still a pass-through: Linux owns the registration and supplies the
1055    /// result, and Detcore only records the head address so it can replay
1056    /// `exit_robust_list()` when the thread dies (see
1057    /// `Self::run_robust_list_owner_death`). Recording happens only after the
1058    /// kernel accepts the call, so a rejected length or address never becomes
1059    /// Detcore state.
1060    ///
1061    /// The call goes through `record_or_replay`, like every other pass-through.
1062    /// Using `Guest::inject` directly would keep the classification but drop the
1063    /// behavior it implies: the syscall would vanish from a `hermit record`
1064    /// trace, and a log recorded by a build that did record it would no longer
1065    /// replay.
1066    // AUTONOMOUS-BOT-IMPLEMENTED
1067    pub async fn handle_set_robust_list<G: Guest<Self>>(
1068        &self,
1069        guest: &mut G,
1070        call: syscalls::SetRobustList,
1071    ) -> Result<i64, Error> {
1072        let head = call.head().map(AddrMut::as_raw);
1073        let len = call.len();
1074        let res = self
1075            .record_or_replay(guest, Syscall::SetRobustList(call))
1076            .await?;
1077        // TODO-HUMAN-REVIEW(PR-2223): Review robust-list
1078        // head tracking used to drive owner-death wakeups.
1079        let recorded = match head {
1080            // `set_robust_list(NULL, ...)` unregisters the list.
1081            None => None,
1082            // Fail closed, and not dead code: `len` is the guest's own argument,
1083            // so a 32-bit guest (or a bug) can present the 12-byte
1084            // `compat_robust_list_head`, whose fields sit at different offsets.
1085            // The 64-bit walk would misread it, so refuse to record it at all
1086            // rather than walk a layout we cannot parse.
1087            Some(_) if len != robust_list::ROBUST_LIST_HEAD_LEN => None,
1088            Some(addr) => Some(addr),
1089        };
1090        guest.thread_state_mut().record_robust_list_head(recorded);
1091        trace!(
1092            "[detcore, dtid {}] robust-list head registered: {:?}",
1093            guest.thread_state().dettid,
1094            recorded,
1095        );
1096        Ok(res)
1097    }
1098
1099    /// Replay Linux's `exit_robust_list()` for the calling thread.
1100    ///
1101    /// Linux runs this from `mm_release()` while a task dies. Detcore performs
1102    /// the same walk and issues the wake against its own modeled waiter pool,
1103    /// which is where precise-mode futex waiters actually live. Ptrace leaves
1104    /// the atomic word transition to Linux; backends without that ordering do
1105    /// not enter this path. The algorithm is in
1106    /// [`crate::syscalls::robust_list`]; this function only supplies guest
1107    /// memory and the scheduler wake.
1108    ///
1109    /// Only the precise futex model needs this. Polling and external modes park
1110    /// waiters in the host kernel, which performs its own robust-list cleanup;
1111    /// and without thread sequentialization Detcore does not model futexes at
1112    /// all.
1113    // AUTONOMOUS-BOT-IMPLEMENTED
1114    // TODO-HUMAN-REVIEW(PR-2223): Review owner-death wakeup
1115    // emulation, which changes how a dying thread's peers are scheduled.
1116    async fn run_robust_list_owner_death<G: Guest<Self>>(&self, guest: &mut G) {
1117        if !self.cfg.backend_runs_exit_robust_list
1118            || !self.cfg.sequentialize_threads
1119            || self.cfg.debug_futex_mode != BlockingMode::Precise
1120        {
1121            return;
1122        }
1123        let Some(head) = guest.thread_state().robust_list_head else {
1124            return;
1125        };
1126        let dettid = guest.thread_state().dettid;
1127        // Which TID goes in the comparison? Not one Detcore hands out: `gettid`
1128        // is a reviewed pass-through, and glibc puts the value the kernel
1129        // supplied through `CLONE_PARENT_SETTID` into the lock word, not a
1130        // `gettid` result. The comparison is valid for a narrower reason —
1131        // `DetTid` currently *is* the raw namespaced `Tid`. `init_thread_state`
1132        // builds every one with `DetPid::from_raw(tid.into())` (detcore/src/lib.rs,
1133        // `lib.rs:1257` at this commit), four lines under
1134        // `// TODO(T78538674): virtualize tid, extend tid<=>dettid mapping here.`
1135        // Completing that TODO breaks this comparison, which would then have to
1136        // map back to the guest-visible TID.
1137        // `hermit-cli/tests/robust_futex_owner_death.rs` is the tripwire: it
1138        // fails the moment the two diverge.
1139        let owner_tid = dettid.as_raw() as u32;
1140
1141        let mut effects = GuestRobustEffects::<'_, G, T> {
1142            guest,
1143            dettid,
1144            defer_owner_death_to_backend: true,
1145            staged_wakes: None,
1146            tool: PhantomData,
1147        };
1148        let outcome = robust_list::exit_robust_list(&mut effects, head, owner_tid).await;
1149        if outcome.head_unreadable {
1150            trace!(
1151                "[detcore, dtid {}] unreadable robust-list head at {:#x}; no owner-death wakeups",
1152                dettid, head,
1153            );
1154        } else if outcome.aborted || outcome.next_faulted || outcome.truncated {
1155            trace!(
1156                "[detcore, dtid {}] robust-list walk stopped early after {} entr(ies): {:?}",
1157                dettid, outcome.entries_visited, outcome,
1158            );
1159        }
1160    }
1161
1162    /// Read every registered robust list in the current thread group while
1163    /// its address space still exists, but hold its modeled wakes until the
1164    /// corresponding ptrace exit callback confirms Linux has completed the
1165    /// atomic owner-word update.
1166    pub(crate) async fn stage_thread_group_robust_list_wakes<G: Guest<Self>>(
1167        &self,
1168        guest: &mut G,
1169        reason: RobustListExit,
1170    ) {
1171        if !self.cfg.backend_runs_exit_robust_list
1172            || !self.cfg.sequentialize_threads
1173            || self.cfg.debug_futex_mode != BlockingMode::Precise
1174        {
1175            return;
1176        }
1177
1178        let heads = guest.thread_state().robust_list_heads();
1179        let mut staged = Vec::with_capacity(heads.len());
1180        for (owner, head) in heads {
1181            let mut wakes = Vec::new();
1182            let mut effects = GuestRobustEffects::<'_, G, T> {
1183                guest,
1184                dettid: owner,
1185                defer_owner_death_to_backend: true,
1186                staged_wakes: Some(&mut wakes),
1187                tool: PhantomData,
1188            };
1189            let outcome =
1190                robust_list::exit_robust_list(&mut effects, head, owner.as_raw() as u32).await;
1191            if outcome.head_unreadable || outcome.aborted {
1192                trace!(
1193                    "[detcore, dtid {}] could not stage complete robust-list owner-death effects: {:?}",
1194                    owner, outcome,
1195                );
1196            }
1197            staged.push((owner, wakes));
1198        }
1199        guest.thread_state().stage_robust_list_wakes(reason, staged);
1200    }
1201
1202    /// Exit system call
1203    pub async fn handle_exit<G: Guest<Self>>(
1204        &self,
1205        guest: &mut G,
1206        call: syscalls::Exit,
1207    ) -> Result<i64, Error> {
1208        let request = guest.thread_state().mk_request(
1209            ResourceID::Exit {
1210                group: false,
1211                process: guest.thread_state().detpid.expect("detpid unset"),
1212                mm: guest.thread_state().mm_id,
1213            },
1214            Permission::RW,
1215        );
1216        resource_request(guest, request).await;
1217        self.run_robust_list_owner_death(guest).await;
1218        // It's ok here that we skip running the posthook:
1219        guest.tail_inject(call).await
1220    }
1221
1222    /// Exit_group system call
1223    pub async fn handle_exit_group<G: Guest<Self>>(
1224        &self,
1225        guest: &mut G,
1226        call: syscalls::ExitGroup,
1227    ) -> Result<i64, Error> {
1228        let request = guest.thread_state().mk_request(
1229            ResourceID::Exit {
1230                group: true,
1231                process: guest.thread_state().detpid.expect("detpid unset"),
1232                mm: guest.thread_state().mm_id,
1233            },
1234            Permission::RW,
1235        );
1236        resource_request(guest, request).await;
1237        self.stage_thread_group_robust_list_wakes(guest, RobustListExit::ExitGroup)
1238            .await;
1239        // It's ok here that we skip running the posthook:
1240        guest.tail_inject(call).await
1241    }
1242
1243    /// Futex system call, which can block.
1244    pub async fn handle_futex<G: Guest<Self>>(
1245        &self,
1246        guest: &mut G,
1247        call: syscalls::Futex,
1248    ) -> Result<i64, Error> {
1249        let dettid = guest.thread_state().dettid;
1250        let ptr = match call.uaddr() {
1251            None => {
1252                // null pointer error:
1253                return Ok(guest.inject(call).await?);
1254            }
1255            Some(x) => x,
1256        };
1257        let init_val = guest.memory().read_value(ptr)?;
1258        trace!(
1259            "[detcore, dtid {}] futex op with memory address containing value {}",
1260            &dettid, init_val
1261        );
1262
1263        if !self.cfg.sequentialize_threads {
1264            Ok(guest.inject(call).await?)
1265        } else {
1266            match self.cfg.debug_futex_mode {
1267                BlockingMode::Precise => self.handle_futex_blocking(guest, call, init_val).await,
1268                BlockingMode::Polling => self.handle_futex_polling(guest, call, init_val).await,
1269                BlockingMode::External => self.record_or_replay_blocking(guest, call.into()).await,
1270            }
1271        }
1272    }
1273
1274    /// Blocking (precise) Futex implementation.
1275    /// Here we use a two-phase request to the scheduler: before and after the futex wait/wake
1276    /// side effects. We EMULATE futex calls and NEVER run them inside the kernel.
1277    pub async fn handle_futex_blocking<G: Guest<Self>>(
1278        &self,
1279        guest: &mut G,
1280        call: syscalls::Futex,
1281        init_val: i32,
1282    ) -> Result<i64, Error> {
1283        let ptr = call.uaddr().unwrap();
1284        let futexid = guest.thread_state().futex_id(
1285            AddrMut::as_raw(ptr),
1286            call.futex_op() & libc::FUTEX_PRIVATE_FLAG != 0,
1287        );
1288        let futex_op = call.futex_op() & libc::FUTEX_CMD_MASK;
1289        let bitset = match futex_op {
1290            libc::FUTEX_WAKE_BITSET | libc::FUTEX_WAIT_BITSET => call.val3() as u32,
1291            _ => u32::MAX,
1292        };
1293        if bitset == 0 {
1294            return Err(Error::Errno(Errno::EINVAL));
1295        }
1296        let dettid = guest.thread_state().dettid;
1297        match futex_op {
1298            libc::FUTEX_WAKE | libc::FUTEX_WAKE_BITSET => {
1299                let num = match futex_action(
1300                    guest,
1301                    FutexAction::WakeRequest(call.val()),
1302                    &futexid,
1303                    init_val,
1304                    bitset,
1305                )
1306                .await
1307                .expect("futex wake must return value")
1308                {
1309                    SchedValue::Value(num) => num,
1310                    SchedValue::TimeOut => panic!("impossible, futex wake doesn't have a timeout"),
1311                };
1312                // AUTONOMOUS-BOT-IMPLEMENTED
1313                // TODO-HUMAN-REVIEW(#845): Review exited-thread futex diagnostics.
1314                match guest.memory().read_value(ptr) {
1315                    Ok(observed) => trace!(
1316                        "[detcore, dtid {}] emulated futex wake committed, memory value is {}, expected {}",
1317                        &dettid,
1318                        observed,
1319                        call.val(),
1320                    ),
1321                    Err(error) => trace!(
1322                        "[detcore, dtid {}] skipped post-wake futex memory diagnostic: {}",
1323                        &dettid, error,
1324                    ),
1325                }
1326                let _ = futex_action(
1327                    guest,
1328                    FutexAction::WakeFinished(0),
1329                    &futexid,
1330                    init_val,
1331                    bitset,
1332                )
1333                .await;
1334                Ok(num as i64)
1335            }
1336            libc::FUTEX_WAIT | libc::FUTEX_WAIT_BITSET => {
1337                if init_val != call.val() {
1338                    info!(
1339                        "[detcore, dtid {}] Futex wait running immediately because it will fizzle ({} != {}).",
1340                        &dettid,
1341                        init_val,
1342                        call.val()
1343                    );
1344                    Err(Error::Errno(Errno::EAGAIN))
1345                } else {
1346                    let maybe_timeout_lt = self
1347                        .futex_timeout_deadline(guest, call.futex_op(), call.timeout())
1348                        .await?;
1349                    let ans = futex_action(
1350                        guest,
1351                        FutexAction::WaitRequest(maybe_timeout_lt),
1352                        &futexid,
1353                        init_val,
1354                        bitset,
1355                    )
1356                    .await;
1357                    let res = if ans != Some(SchedValue::TimeOut) {
1358                        let expected = call.val();
1359                        // AUTONOMOUS-BOT-IMPLEMENTED
1360                        // TODO-HUMAN-REVIEW(#845): Review exited-thread futex diagnostics.
1361                        match guest.memory().read_value(ptr) {
1362                            Ok(observed) => {
1363                                trace!(
1364                                    "[detcore, dtid {}] after (emulated) futex wait, memory value is {}, expected {}",
1365                                    &dettid, observed, expected,
1366                                );
1367                                if expected == observed {
1368                                    debug!(
1369                                        "WARNING: fishy that the futex value did not change before wakeup. Weird application-level protocol.\n"
1370                                    );
1371                                }
1372                            }
1373                            Err(error) => trace!(
1374                                "[detcore, dtid {}] skipped post-wait futex memory diagnostic: {}",
1375                                &dettid, error,
1376                            ),
1377                        }
1378                        Ok(0)
1379                    } else {
1380                        trace!("[detcore, dtid {}] futex wait timed out", &dettid);
1381                        Err(Error::Errno(Errno::ETIMEDOUT))
1382                    };
1383                    futex_action(guest, FutexAction::WaitFinished, &futexid, init_val, bitset)
1384                        .await;
1385                    res
1386                }
1387            }
1388            libc::FUTEX_FD => {
1389                panic!("[detcore] refusing to execute FUTEX_FD, which was removed in Linux 2.6.26.")
1390            }
1391            other => {
1392                panic!("[detcore] futex op not handled yet: {}", other);
1393            }
1394        }
1395    }
1396
1397    /// Futex system call, alternative implemenattion where we treat futexes as InternalIOPolling
1398    /// operations.
1399    pub async fn handle_futex_polling<G: Guest<Self>>(
1400        &self,
1401        guest: &mut G,
1402        call: syscalls::Futex,
1403        init_val: i32,
1404    ) -> Result<i64, Error> {
1405        fn make_futex_wake_request(dettid: DetTid) -> Resources {
1406            let mut rsrc = Resources::new(dettid);
1407            rsrc.fyi("futex_wake");
1408            rsrc
1409        }
1410
1411        fn make_futex_wait_request(dettid: DetTid) -> Resources {
1412            let mut rsrc = Resources::new(dettid);
1413            rsrc.insert(ResourceID::InternalIOPolling, Permission::W);
1414            rsrc.fyi("futex_wait");
1415            rsrc
1416        }
1417
1418        let dettid = guest.thread_state().dettid;
1419        let futex_op = call.futex_op() & libc::FUTEX_CMD_MASK;
1420        match futex_op {
1421            libc::FUTEX_WAKE | libc::FUTEX_WAKE_BITSET => {
1422                let rsrc = make_futex_wake_request(dettid);
1423                resource_request(guest, rsrc.clone()).await; // Linearize this operation as a separate COMMIT.
1424                let res = guest.inject(call).await;
1425                // FIXME: With the non-blocking version of futex_wait, `res` will always be 0.  It
1426                // is quite difficult to tell how many polling waiters we unblocked with a given
1427                // wake, without going back to modeling futexes like `handle_futex_blocking` does.
1428                Ok(res?)
1429            }
1430            libc::FUTEX_WAIT | libc::FUTEX_WAIT_BITSET => {
1431                if init_val != call.val() {
1432                    info!(
1433                        "[detcore, dtid {}] Futex wait running immediately because it will fizzle ({} != {}).",
1434                        dettid,
1435                        init_val,
1436                        call.val()
1437                    );
1438                    let res = guest.inject(call).await;
1439                    Ok(res?)
1440                } else {
1441                    let rsrc = make_futex_wait_request(dettid);
1442                    let deadline = self
1443                        .futex_timeout_deadline(guest, call.futex_op(), call.timeout())
1444                        .await?;
1445                    let res =
1446                        retry_nonblocking_syscall_with_timeout(guest, call, rsrc, deadline).await?;
1447                    trace!(
1448                        "[detcore, dtid {}] after futex wait, memory value is {}",
1449                        &dettid,
1450                        guest.memory().read_value(call.uaddr().unwrap()).unwrap()
1451                    );
1452                    Ok(res)
1453                }
1454            }
1455            libc::FUTEX_FD => {
1456                panic!("[detcore] refusing to execute FUTEX_FD, which was removed in Linux 2.6.26.")
1457            }
1458            other => {
1459                panic!("[detcore] futex op not handled yet: {}", other);
1460            }
1461        }
1462    }
1463
1464    /// Execveat system call.  Doesn't return if successful.
1465    pub async fn handle_execveat<G: Guest<Self>>(
1466        &self,
1467        guest: &mut G,
1468        call: syscalls::Execveat,
1469    ) -> Result<i64, Error> {
1470        let (old_metadata, old_memory_metadata, table_is_shared, dettid, detpid, old_mm_id) = {
1471            let thread_state = guest.thread_state();
1472            (
1473                Arc::clone(&thread_state.file_metadata),
1474                Arc::clone(&thread_state.memory_metadata),
1475                Arc::strong_count(&thread_state.file_metadata) > 1,
1476                thread_state.dettid,
1477                thread_state.detpid.expect("detpid unset"),
1478                thread_state.mm_id,
1479            )
1480        };
1481        let (new_metadata, closed_open_files, exec_fd_blocking) = {
1482            let metadata = old_metadata.lock().unwrap();
1483            let new_metadata = metadata.for_exec(dettid);
1484            (
1485                new_metadata.clone(),
1486                metadata.open_files_closed_on_exec(table_is_shared),
1487                new_metadata.exec_blocking_overrides(),
1488            )
1489        };
1490        let preserve_exec_fd_status = guest.thread_state().discover_live_file_metadata;
1491
1492        prepare_exec(
1493            guest,
1494            old_mm_id,
1495            if preserve_exec_fd_status {
1496                exec_fd_blocking
1497            } else {
1498                Default::default()
1499            },
1500        )
1501        .await;
1502
1503        let mut released_ports = Vec::new();
1504        for open_file_id in closed_open_files {
1505            if let Some(port) = self.release_port_for_open_file(guest, open_file_id).await {
1506                released_ports.push((open_file_id, port));
1507            }
1508        }
1509
1510        // A successful execve replaces the address space, and Linux clears
1511        // `task->robust_list` with it; the new image re-registers its own.
1512        let old_robust_list_head;
1513
1514        {
1515            let thread_state = guest.thread_state_mut();
1516            thread_state.file_metadata = Arc::new(Mutex::new(new_metadata));
1517            thread_state.memory_metadata = Arc::new(Mutex::new(MemoryMetadata::new()));
1518            thread_state.mm_id = old_mm_id.for_exec(detpid);
1519            old_robust_list_head = thread_state.take_robust_list_for_exec();
1520        }
1521
1522        // execve(2) doesn't return upon success.
1523        let errno = self.record_or_replay(guest, call).await.unwrap_err();
1524
1525        {
1526            let thread_state = guest.thread_state_mut();
1527            thread_state.file_metadata = old_metadata;
1528            thread_state.memory_metadata = old_memory_metadata;
1529            thread_state.mm_id = old_mm_id;
1530            thread_state.restore_robust_list_after_failed_exec(old_robust_list_head);
1531        }
1532
1533        cancel_exec(guest).await;
1534        for (open_file_id, port) in released_ports {
1535            self.restore_port_for_open_file(guest, open_file_id, port)
1536                .await;
1537        }
1538
1539        // KVM validates the replacement image before refusing a worker's
1540        // promotion to group leader. Diagnose that refusal only after normal
1541        // failed-exec rollback, preserving errors such as ENOENT and ENOEXEC.
1542        if self.cfg.backend_is_kvm && dettid != detpid && errno == Errno::ENOSYS {
1543            tracing::error!(
1544                "[detcore, dtid {dettid}] KVM nonleader exec is unsupported; \
1545                 the replacement image did not run"
1546            );
1547            if !self.cfg.panic_on_unsupported_syscalls {
1548                crate::tool_global::report_unsupported_syscall(guest, call.number()).await;
1549            }
1550            return self
1551                .refuse_unserviceable_operation(guest, call.number(), errno)
1552                .await;
1553        }
1554
1555        Err(errno.into())
1556    }
1557
1558    // AUTONOMOUS-BOT-IMPLEMENTED
1559    // TODO-HUMAN-REVIEW(#258): Confirm one-turn exclusion semantics across scheduler modes.
1560    /// End the current logical timeslice for a sequentialized sched_yield.
1561    pub async fn handle_sched_yield<G: Guest<Self>>(
1562        &self,
1563        guest: &mut G,
1564        call: syscalls::SchedYield,
1565    ) -> Result<i64, Error> {
1566        if self.cfg.sequentialize_threads {
1567            // In chaos mode, thread-interleaving diversity (and thus fairness)
1568            // comes entirely from re-randomizing thread priorities at
1569            // preemption-timer expirations. When timer preemption is disabled
1570            // (`--max-timeslice disabled`), priorities are fixed at thread
1571            // creation and never change. A plain yield only re-enqueues the
1572            // caller at the back of its own (fixed) priority level, so a thread
1573            // that spins on sched_yield while holding the numerically-lowest
1574            // priority is always reselected first and starves every thread it is
1575            // waiting on (GH #81). Treat sched_yield as an explicit chaos
1576            // reprioritization point: draw a fresh random priority for the
1577            // caller so it cedes the CPU and other runnable threads can make
1578            // progress. This mirrors what `end_timeslice` does at a timer-driven
1579            // preemption point, and is recorded for chaos replay.
1580            if self.cfg.chaos && self.cfg.max_timeslice.is_none() {
1581                let change_time = guest.thread_state().thread_logical_time.as_nanos();
1582                let request = Self::random_priority_changepoint_request(guest, change_time);
1583                resource_request(guest, request).await;
1584            } else if !self.cfg.chaos && self.cfg.replay_preemptions_from.is_some() {
1585                if self.cfg.max_timeslice.is_some() {
1586                    guest
1587                        .thread_state_mut()
1588                        .reset_timeslice_for_explicit_yield();
1589                }
1590                let request = Self::sched_yield_request(guest);
1591                resource_request(guest, request).await;
1592            } else if self.cfg.chaos || self.cfg.replay_schedule_from.is_some() {
1593                let request = Self::yield_request(guest);
1594                resource_request(guest, request).await;
1595            } else {
1596                self.end_timeslice_for_sched_yield(guest).await;
1597            }
1598            trace!("sched_yield yielded to the scheduler; NOT performing actual syscall");
1599            Ok(0)
1600        } else {
1601            Ok(self.record_or_replay(guest, call).await?)
1602        }
1603    }
1604
1605    /// wait4 system call
1606    /// This is handled by the scheduler and not passed to the record/replay layer.
1607    // TODO-HUMAN-REVIEW(PR-587): Confirm wait4 rusage canonicalization boundaries.
1608    pub async fn handle_wait4<G: Guest<Self>>(
1609        &self,
1610        guest: &mut G,
1611        call: syscalls::Wait4,
1612    ) -> Result<i64, Error> {
1613        let dettid = guest.thread_state().dettid;
1614        let mut rsrc = Resources::new(dettid);
1615        rsrc.insert(ResourceID::InternalIOPolling, Permission::W);
1616        rsrc.fyi("wait4");
1617
1618        validate_wait4_arguments(call.pid(), call.options())?;
1619
1620        let parent = guest.thread_state().detpid.expect("detpid unset");
1621        // Stop/continue events need a backend waitability callback. Non-SIGCHLD
1622        // process clones also remain legacy until backends distinguish
1623        // PTRACE_EVENT_CLONE from CLONE_THREAD. The common matcher already
1624        // carries those filters so activation does not require another model.
1625        let selector = if call.options().intersects(
1626            WaitPidFlag::WUNTRACED
1627                | WaitPidFlag::WCONTINUED
1628                | WaitPidFlag::__WCLONE
1629                | WaitPidFlag::__WALL,
1630        ) {
1631            None
1632        } else {
1633            match call.pid() {
1634                pid if pid > 0 => Some(ChildWaitSelector::Exact(DetPid::from_raw(pid))),
1635                -1 => Some(ChildWaitSelector::Any),
1636                0 => process_group(guest, parent)
1637                    .await
1638                    .map(ChildWaitSelector::ProcessGroup),
1639                pid if pid < -1 => Some(ChildWaitSelector::ProcessGroup(DetPid::from_raw(-pid))),
1640                _ => unreachable!(),
1641            }
1642        };
1643        let spec = selector
1644            .map(|selector| terminal_child_wait_spec(selector, dettid, call.options().bits()));
1645        let complete_lineage = guest.config().backend_tracks_process_children;
1646        let managed_spec = if let Some(spec) = spec {
1647            let (_, has_child) = ready_child_wait(guest, spec).await;
1648            (has_child || complete_lineage).then_some(spec)
1649        } else {
1650            None
1651        };
1652
1653        let value = if call.options().contains(WaitPidFlag::WNOHANG) {
1654            resource_request(guest, rsrc.clone()).await;
1655            info!(
1656                "[dtid {}] Executing non-blocking wait4 in one shot.",
1657                dettid
1658            );
1659            if let Some(spec) = spec {
1660                'select_child: loop {
1661                    let (ready, has_child) = ready_child_wait(guest, spec).await;
1662                    let Some(child) = ready else {
1663                        break if has_child {
1664                            0
1665                        } else if complete_lineage {
1666                            return Err(Errno::ECHILD.into());
1667                        } else {
1668                            guest.inject_with_retry(call).await?
1669                        };
1670                    };
1671                    let _ = await_exact_child_physical_exit(guest, child).await;
1672                    let exact_call = call.with_pid(child.as_raw());
1673                    loop {
1674                        match guest.inject_with_retry(exact_call).await {
1675                            Ok(value) if value != 0 => break 'select_child value,
1676                            Ok(_) => yield_once().await,
1677                            Err(Errno::ECHILD) => {
1678                                let _ = consume_child_wait(guest, child).await;
1679                                if child_wait_can_retry_after_stale(spec) {
1680                                    continue 'select_child;
1681                                }
1682                                return Err(Errno::ECHILD.into());
1683                            }
1684                            Err(errno) => return Err(errno.into()),
1685                        }
1686                    }
1687                }
1688            } else {
1689                guest.inject_with_retry(call).await?
1690            }
1691        } else if let Some(spec) = managed_spec {
1692            {
1693                // The ptrace backend must block ordinary signals until child
1694                // readiness is resolved. DBT already delays application signal
1695                // delivery while this callback is active, and replacing its
1696                // application mask here also hides those signals from
1697                // rt_sigpending. Read that mask without changing it instead.
1698                let blocked_mask = blocked_signal_mask();
1699                let mut stack = guest.stack().await;
1700                let blocked_mask_addr = stack.push(blocked_mask);
1701                let old_mask_addr = stack.reserve::<KernelSigset>();
1702                let action_addr = stack.reserve::<KernelSigaction>();
1703                let _mask_guard = stack.commit()?;
1704                let guest_signal_mask =
1705                    block_signals_for_disposition(guest, blocked_mask_addr, old_mask_addr).await?;
1706                let inspect_signal_action = guest
1707                    .config()
1708                    .backend_requires_thread_directed_process_signals;
1709
1710                let poll_call = call.with_options(call.options() | WaitPidFlag::WNOHANG);
1711                let mut pending_signal = None;
1712                let result: Result<i64, Error> = loop {
1713                    let status = wait_for_child_lifecycle(guest, spec).await;
1714                    if pending_signal.is_none() {
1715                        pending_signal = wait_signal_disposition(
1716                            guest,
1717                            status,
1718                            &guest_signal_mask,
1719                            action_addr,
1720                            inspect_signal_action,
1721                        )
1722                        .await?;
1723                    }
1724                    let (ready, has_child) = ready_child_wait(guest, spec).await;
1725                    if let Some(child) = ready {
1726                        let _ = await_exact_child_physical_exit(guest, child).await;
1727                        match guest.inject_with_retry(call.with_pid(child.as_raw())).await {
1728                            Ok(value) => break Ok(value),
1729                            Err(Errno::ECHILD) => {
1730                                let _ = consume_child_wait(guest, child).await;
1731                                if child_wait_can_retry_after_stale(spec) {
1732                                    let (next_ready, _) = ready_child_wait(guest, spec).await;
1733                                    if stale_any_wait_must_interrupt(
1734                                        pending_signal.is_some(),
1735                                        next_ready,
1736                                    ) {
1737                                        break interrupted_child_wait_result(
1738                                            guest,
1739                                            call,
1740                                            pending_signal.expect("signal checked above"),
1741                                        )
1742                                        .await;
1743                                    }
1744                                    continue;
1745                                }
1746                                break Err(Errno::ECHILD.into());
1747                            }
1748                            Err(errno) => break Err(errno.into()),
1749                        }
1750                    }
1751                    if !has_child {
1752                        break Err(Errno::ECHILD.into());
1753                    }
1754                    match guest.inject(poll_call).await {
1755                        Ok(value) => {
1756                            if value > 0 {
1757                                break Ok(value);
1758                            }
1759                            if let Some(disposition) = pending_signal {
1760                                break interrupted_child_wait_result(guest, call, disposition)
1761                                    .await;
1762                            }
1763                        }
1764                        Err(errno) => break Err(errno.into()),
1765                    }
1766                };
1767
1768                restore_signals_after_disposition(guest, old_mask_addr).await?;
1769                result?
1770            }
1771        } else {
1772            // wait4 is a scheduler poll, not a record/replay data read (see doc above),
1773            // so it is not routed through the record/replay subtool.
1774            retry_nonblocking_syscall(guest, call, rsrc, None).await?
1775        };
1776        let consumed_termination = if value <= 0 {
1777            false
1778        } else if let Some(status) = call.wstatus() {
1779            wait_status_is_termination(guest.memory().read_value(status)?)
1780        } else {
1781            guest
1782                .thread_state()
1783                .has_exited_child_process_cpu_time(DetPid::from_raw(value as i32))
1784        };
1785        if consumed_termination {
1786            guest
1787                .thread_state_mut()
1788                .reap_child_process_cpu_time(DetPid::from_raw(value as i32));
1789            let _ = consume_child_wait(guest, DetPid::from_raw(value as i32)).await;
1790        }
1791        if value > 0
1792            && let Some(rusage) = call.rusage()
1793        {
1794            // Host CPU and scheduling counters are not deterministic.
1795            let usage: libc::rusage = unsafe { std::mem::zeroed() };
1796            guest.memory().write_value(rusage, &usage)?;
1797        }
1798        Ok(value)
1799    }
1800
1801    // AUTONOMOUS-BOT-IMPLEMENTED
1802    // TODO-HUMAN-REVIEW(#274): Review waitid polling and compatibility boundaries.
1803    // TODO-HUMAN-REVIEW(#2246): Review waitid progress and child-ready signal precedence.
1804    /// waitid system call
1805    /// This is handled by the scheduler and not passed to the record/replay layer.
1806    pub async fn handle_waitid<G: Guest<Self>>(
1807        &self,
1808        guest: &mut G,
1809        mut call: syscalls::Waitid,
1810    ) -> Result<i64, Error> {
1811        let dettid = guest.thread_state().dettid;
1812        let mut rsrc = Resources::new(dettid);
1813        rsrc.insert(ResourceID::InternalIOPolling, Permission::W);
1814        rsrc.fyi("waitid");
1815
1816        let event_options = libc::WEXITED | libc::WSTOPPED | libc::WCONTINUED;
1817        let allowed_options = event_options
1818            | libc::WNOHANG
1819            | libc::WNOWAIT
1820            | libc::__WNOTHREAD
1821            | libc::__WALL
1822            | libc::__WCLONE;
1823        if call.options() & event_options == 0 || call.options() & !allowed_options != 0 {
1824            return Err(Errno::EINVAL.into());
1825        }
1826
1827        // POSIX requires non-null infop. Linux accepts null, but that form can
1828        // expose host rusage and requires backend-neutral scratch memory for
1829        // deterministic polling. Reject it uniformly instead of diverging or
1830        // panicking on DBT's unsupported scratch stack.
1831        if call.info().is_none() {
1832            return Err(Errno::EFAULT.into());
1833        }
1834
1835        // Keep clone-class waits on the legacy path until ptrace and KVM both
1836        // register non-SIGCHLD clone children as processes rather than threads.
1837        let terminal_events_only = call.options() & libc::WEXITED != 0
1838            && call.options() & (libc::WSTOPPED | libc::WCONTINUED | libc::__WCLONE | libc::__WALL)
1839                == 0;
1840
1841        // The parked nonterminal path still uses kernel polling. Preserve its
1842        // P_PGID(0) entry snapshot until backend lifecycle events replace it.
1843        if !terminal_events_only && call.which() == libc::P_PGID as i32 && call.pid() == 0 {
1844            call = call.with_pid(snapshot_process_group(guest.pid())?);
1845        }
1846
1847        // A blocking waitid on an O_NONBLOCK pidfd must return EAGAIN rather
1848        // than being converted to WNOHANG. Acquire the scheduler resource first,
1849        // then snapshot fdinfo and issue the one-shot wait without another yield.
1850        let pidfd_nonblocking =
1851            if call.which() == libc::P_PIDFD as i32 && call.options() & libc::WNOHANG == 0 {
1852                resource_request(guest, rsrc.clone()).await;
1853                guest_fd_status_flags(guest.pid(), call.pid())? & libc::O_NONBLOCK != 0
1854            } else {
1855                false
1856            };
1857        if call.which() == libc::P_PIDFD as i32
1858            && call.options() & libc::WNOHANG == 0
1859            && !pidfd_nonblocking
1860        {
1861            // Polling a numeric pidfd cannot preserve Linux's held file
1862            // reference if another thread closes and reuses the descriptor.
1863            // Reject the blocking form until Detcore can retain that identity.
1864            return Err(Errno::EOPNOTSUPP.into());
1865        }
1866        let info = call.info().expect("waitid infop checked above");
1867
1868        // Unlike wait4, waitid returns zero both when it reports a child event and
1869        // when WNOHANG finds nothing. Polling must inspect si_pid to distinguish
1870        // those cases.
1871        // Known limitation: without backend-neutral scratch memory, an invalid
1872        // non-null infop faults on the first physical poll rather than after a
1873        // child becomes waitable.
1874        // siginfo_t has no portable initializer. An all-zero value is the
1875        // waitid WNOHANG sentinel defined by POSIX and Linux.
1876        let empty_info: libc::siginfo_t = unsafe { std::mem::zeroed() };
1877
1878        // The lifecycle scheduler currently models terminal child events for
1879        // exact and any-child selectors. Group membership and stop/continue
1880        // state remain on the legacy kernel-polling path.
1881        let terminal_spec = if terminal_events_only {
1882            let selector = match call.which() {
1883                which if which == libc::P_PID as i32 => {
1884                    Some(ChildWaitSelector::Exact(DetPid::from_raw(call.pid())))
1885                }
1886                which if which == libc::P_ALL as i32 => Some(ChildWaitSelector::Any),
1887                which if which == libc::P_PGID as i32 => {
1888                    let group = if call.pid() == 0 {
1889                        process_group(guest, guest.thread_state().detpid.expect("detpid unset"))
1890                            .await
1891                    } else {
1892                        Some(DetPid::from_raw(call.pid()))
1893                    };
1894                    group.map(ChildWaitSelector::ProcessGroup)
1895                }
1896                _ => None,
1897            };
1898            selector.map(|selector| terminal_child_wait_spec(selector, dettid, call.options()))
1899        } else {
1900            None
1901        };
1902        let complete_lineage = guest.config().backend_tracks_process_children;
1903        let managed_terminal_spec = if let Some(spec) = terminal_spec {
1904            let (_, has_child) = ready_child_wait(guest, spec).await;
1905            (has_child || complete_lineage).then_some(spec)
1906        } else {
1907            None
1908        };
1909
1910        if call.options() & libc::WNOHANG != 0 || pidfd_nonblocking {
1911            if !pidfd_nonblocking {
1912                resource_request(guest, rsrc).await;
1913            }
1914            info!(
1915                "[dtid {}] Executing non-blocking waitid in one shot.",
1916                dettid
1917            );
1918            'select_child: loop {
1919                let selected = if let Some(spec) = terminal_spec {
1920                    let (ready, has_child) = ready_child_wait(guest, spec).await;
1921                    if ready.is_none() && has_child {
1922                        guest.memory().write_value(info, &empty_info)?;
1923                        return Ok(0);
1924                    }
1925                    if ready.is_none() && complete_lineage {
1926                        return Err(Errno::ECHILD.into());
1927                    }
1928                    ready
1929                } else {
1930                    None
1931                };
1932                if let Some(child) = selected {
1933                    let _ = await_exact_child_physical_exit(guest, child).await;
1934                }
1935                let effective_call = selected.map_or(call, |child| {
1936                    call.with_which(libc::P_PID as i32).with_pid(child.as_raw())
1937                });
1938                loop {
1939                    guest.memory().write_value(info, &empty_info)?;
1940                    let value = match guest.inject_with_retry(effective_call).await {
1941                        Ok(value) => value,
1942                        Err(Errno::ECHILD) if selected.is_some() => {
1943                            let child = selected.expect("selected child checked above");
1944                            let _ = consume_child_wait(guest, child).await;
1945                            if terminal_spec.is_some_and(child_wait_can_retry_after_stale) {
1946                                continue 'select_child;
1947                            }
1948                            return Err(Errno::ECHILD.into());
1949                        }
1950                        Err(errno) => return Err(errno.into()),
1951                    };
1952                    let info_value: libc::siginfo_t = guest.memory().read_value(info)?;
1953                    let child_pid = unsafe { info_value.si_pid() };
1954                    if child_pid == 0 && selected.is_some() {
1955                        yield_once().await;
1956                        continue;
1957                    }
1958                    let consumed = child_pid != 0
1959                        && call.options() & libc::WNOWAIT == 0
1960                        && waitid_code_is_termination(info_value.si_code);
1961                    let result = finish_waitid_result(guest, call, value, info_value)?;
1962                    if consumed {
1963                        let _ = consume_child_wait(guest, DetPid::from_raw(child_pid)).await;
1964                    }
1965                    return Ok(result);
1966                }
1967            }
1968        }
1969
1970        {
1971            // A signal can arrive after the scheduler wakes this logical wait but
1972            // before the zero-timeout kernel probe that resolves Linux's
1973            // child-ready-versus-interrupt precedence. The ptrace backend blocks
1974            // ordinary signals across that probe, then restores the guest's exact
1975            // mask before returning. DBT reads the mask without replacing it because
1976            // DynamoRIO already delays application delivery while this callback runs.
1977            // The tracer's private preemption signal must remain unblocked.
1978            let blocked_mask = blocked_signal_mask();
1979            let mut stack = guest.stack().await;
1980            let blocked_mask_addr = stack.push(blocked_mask);
1981            let old_mask_addr = stack.reserve::<KernelSigset>();
1982            let action_addr = stack.reserve::<KernelSigaction>();
1983            let _mask_guard = stack.commit()?;
1984            let guest_signal_mask =
1985                block_signals_for_disposition(guest, blocked_mask_addr, old_mask_addr).await?;
1986            let inspect_signal_action = guest
1987                .config()
1988                .backend_requires_thread_directed_process_signals;
1989
1990            let poll_call = call.with_options(call.options() | libc::WNOHANG);
1991            let mut pending_signal = None;
1992            let result: Result<i64, Error> = loop {
1993                // Match the polling protocol used by wait4: the first request with
1994                // poll_attempt zero establishes an ordinary runnable turn, while later
1995                // nonzero attempts receive the scheduler's poller backoff. Omitting the
1996                // first request starts directly as a poller and can keep the run queue
1997                // nonempty forever, preventing logical time from reaching a pending
1998                // signal's deadline.
1999                //
2000                // Do not return on Signaled yet. Linux lets an already-waitable child
2001                // status win over an interrupt, so the zero-timeout kernel probe below
2002                // remains authoritative when readiness and a signal coincide.
2003                let managed_spec = managed_terminal_spec;
2004                // Both ways of parking inside waitid -- the scheduler-managed child
2005                // wait and the legacy kernel-polling loop -- can now be resumed with
2006                // the signals that woke the thread, so both consult the guest mask.
2007                let status = if let Some(spec) = managed_spec {
2008                    wait_for_child_lifecycle(guest, spec).await
2009                } else {
2010                    resource_request(guest, rsrc.clone()).await
2011                };
2012                if pending_signal.is_none() {
2013                    pending_signal = wait_signal_disposition(
2014                        guest,
2015                        status,
2016                        &guest_signal_mask,
2017                        action_addr,
2018                        inspect_signal_action,
2019                    )
2020                    .await?;
2021                }
2022                let (ready, has_child) = if let Some(spec) = managed_spec {
2023                    ready_child_wait(guest, spec).await
2024                } else {
2025                    (None, true)
2026                };
2027                if let Some(child) = ready {
2028                    let _ = await_exact_child_physical_exit(guest, child).await;
2029                    if let Err(error) = guest.memory().write_value(info, &empty_info) {
2030                        break Err(error.into());
2031                    }
2032                    let exact_call = call.with_which(libc::P_PID as i32).with_pid(child.as_raw());
2033                    match guest.inject_with_retry(exact_call).await {
2034                        Ok(value) => {
2035                            let info_value = match guest.memory().read_value(info) {
2036                                Ok(value) => value,
2037                                Err(error) => break Err(error.into()),
2038                            };
2039                            break finish_waitid_result(guest, call, value, info_value);
2040                        }
2041                        Err(Errno::ECHILD) => {
2042                            let _ = consume_child_wait(guest, child).await;
2043                            if managed_spec.is_some_and(child_wait_can_retry_after_stale) {
2044                                let (next_ready, _) =
2045                                    ready_child_wait(guest, managed_spec.expect("managed spec"))
2046                                        .await;
2047                                if stale_any_wait_must_interrupt(
2048                                    pending_signal.is_some(),
2049                                    next_ready,
2050                                ) {
2051                                    break interrupted_child_wait_result(
2052                                        guest,
2053                                        call,
2054                                        pending_signal.expect("signal checked above"),
2055                                    )
2056                                    .await;
2057                                }
2058                                continue;
2059                            }
2060                            break Err(Errno::ECHILD.into());
2061                        }
2062                        Err(errno) => break Err(errno.into()),
2063                    }
2064                }
2065                if managed_spec.is_some() && !has_child {
2066                    break Err(Errno::ECHILD.into());
2067                }
2068
2069                if let Err(error) = guest.memory().write_value(info, &empty_info) {
2070                    break Err(error.into());
2071                }
2072                let result = guest.inject(poll_call).await;
2073                match result {
2074                    Ok(value) => {
2075                        let info_value: libc::siginfo_t = match guest.memory().read_value(info) {
2076                            Ok(value) => value,
2077                            Err(error) => break Err(error.into()),
2078                        };
2079                        // waitid writes the SIGCHLD variant of siginfo_t. A zeroed
2080                        // structure is used only for the no-event WNOHANG result.
2081                        let child_pid = unsafe { info_value.si_pid() };
2082                        match exact_wait_poll_decision(
2083                            child_pid != 0,
2084                            pending_signal.is_some(),
2085                            None,
2086                        ) {
2087                            ExactWaitPollDecision::ChildReady => {
2088                                break finish_waitid_result(guest, call, value, info_value);
2089                            }
2090                            ExactWaitPollDecision::Interrupted => {
2091                                break interrupted_child_wait_result(
2092                                    guest,
2093                                    call,
2094                                    pending_signal.expect("signal checked above"),
2095                                )
2096                                .await;
2097                            }
2098                            ExactWaitPollDecision::Retry => {}
2099                            ExactWaitPollDecision::AwaitPhysicalExit
2100                            | ExactWaitPollDecision::ReapAfterLogicalExit => unreachable!(),
2101                        }
2102                        if managed_spec.is_some() {
2103                            if !has_child {
2104                                break Ok(value);
2105                            }
2106                            continue;
2107                        }
2108                        rsrc.poll_attempt += 1;
2109                        trace!(
2110                            "Retry #{} for waitid because no child state is ready",
2111                            rsrc.poll_attempt
2112                        );
2113                        record_retry_event(guest, poll_call).await;
2114                    }
2115                    Err(Errno::ERESTARTSYS) if pending_signal.is_some() => {
2116                        break Err(Errno::EINTR.into());
2117                    }
2118                    Err(errno) => break Err(errno.into()),
2119                }
2120            };
2121
2122            restore_signals_after_disposition(guest, old_mask_addr).await?;
2123            if result.is_ok() && call.options() & libc::WNOWAIT == 0 {
2124                let info_value: libc::siginfo_t = guest.memory().read_value(info)?;
2125                let child_pid = unsafe { info_value.si_pid() };
2126                if child_pid != 0 && waitid_code_is_termination(info_value.si_code) {
2127                    let _ = consume_child_wait(guest, DetPid::from_raw(child_pid)).await;
2128                }
2129            }
2130            result
2131        }
2132    }
2133
2134    // AUTONOMOUS-BOT-IMPLEMENTED
2135    // TODO-HUMAN-REVIEW(#546)
2136    /// Accept valid affinity masks without changing the host scheduler.
2137    pub async fn handle_sched_setaffinity<G: Guest<Self>>(
2138        &self,
2139        guest: &mut G,
2140        call: syscalls::SchedSetaffinity,
2141    ) -> Result<i64, Error> {
2142        let size_bytes = call.len() as usize;
2143        if size_bytes == 0 {
2144            return Err(Errno::EINVAL.into());
2145        }
2146
2147        let mask = call.mask().ok_or(Errno::EFAULT)?;
2148        let mask: Addr<u8> = mask.cast();
2149        let mut requested = [0u8; VIRTUAL_CPUSET_BYTES];
2150        let bytes_to_read = size_bytes.min(VIRTUAL_CPUSET_BYTES);
2151        guest
2152            .memory()
2153            .read_exact(mask, &mut requested[..bytes_to_read])?;
2154        info!(
2155            "Suppressing sched_setaffinity mask {:?}; affinity remains virtual CPU 0",
2156            requested
2157        );
2158        Ok(0)
2159    }
2160
2161    // AUTONOMOUS-BOT-IMPLEMENTED
2162    // TODO-HUMAN-REVIEW(#546)
2163    /// Report that we are on cpu 0, irrespective of what physical CPU we are on.
2164    pub async fn handle_sched_getaffinity<G: Guest<Self>>(
2165        &self,
2166        guest: &mut G,
2167        call: syscalls::SchedGetaffinity,
2168    ) -> Result<i64, Error> {
2169        let size_bytes: usize = call.len() as usize;
2170        if size_bytes < VIRTUAL_CPUSET_BYTES
2171            || !size_bytes.is_multiple_of(std::mem::size_of::<libc::c_ulong>())
2172        {
2173            return Err(Errno::EINVAL.into());
2174        }
2175
2176        // N.B. we can't use an opaque, type-safe representation such as
2177        // nix::sched::CpuSet currently.  The problem is that the
2178        // SchedGetAffinity syscall treats this field as a u64.
2179        let mut cpu_set = [0u8; VIRTUAL_CPUSET_BYTES];
2180        cpu_set[0] = 1;
2181
2182        info!(
2183            "Suppressing sched_getaffinity and returning {}-byte virtualized result, {:?}",
2184            VIRTUAL_CPUSET_BYTES, cpu_set
2185        );
2186        if let Some(mask) = call.mask() {
2187            let mask: AddrMut<u8> = mask.cast();
2188            guest.memory().write_exact(mask, &cpu_set)?;
2189            // From the man page:
2190            // > On success, the raw sched_getaffinity() system call returns the size (in bytes) of
2191            // > the cpumask_t data type that is used internally by the kernel to represent the CPU
2192            // > set bit mask.
2193            Ok(VIRTUAL_CPUSET_BYTES as i64)
2194        } else {
2195            Err(Error::Errno(Errno::EFAULT))
2196        }
2197    }
2198
2199    /// sched_getparam under Hermit. Detcore replaces the Linux scheduler with its
2200    /// own deterministic one, so a thread's Linux scheduling parameters are
2201    /// inert. Report a fixed SCHED_OTHER priority of 0. The value is emulated
2202    /// (never injected), so it is identical across --verify runs and
2203    /// record/replay.
2204    // AUTONOMOUS-BOT-IMPLEMENTED
2205    // TODO-HUMAN-REVIEW(#720)
2206    pub async fn handle_sched_getparam<G: Guest<Self>>(
2207        &self,
2208        guest: &mut G,
2209        call: syscalls::SchedGetparam,
2210    ) -> Result<i64, Error> {
2211        if let Some(param) = call.param() {
2212            let p = libc::sched_param { sched_priority: 0 };
2213            guest.memory().write_value(param, &p)?;
2214        }
2215        Ok(0)
2216    }
2217
2218    /// sched_rr_get_interval under Hermit. The round-robin quantum is a property
2219    /// of the Linux scheduler, which Detcore does not use, so report a fixed zero
2220    /// interval. Being a constant, it is deterministic across --verify and
2221    /// record/replay.
2222    // AUTONOMOUS-BOT-IMPLEMENTED
2223    // TODO-HUMAN-REVIEW(#720)
2224    pub async fn handle_sched_rr_get_interval<G: Guest<Self>>(
2225        &self,
2226        guest: &mut G,
2227        call: syscalls::SchedRrGetInterval,
2228    ) -> Result<i64, Error> {
2229        if let Some(tp) = call.tp() {
2230            let t = Timespec {
2231                tv_sec: 0,
2232                tv_nsec: 0,
2233            };
2234            guest.memory().write_value(tp, &t)?;
2235        }
2236        Ok(0)
2237    }
2238
2239    /// sched_getattr under Hermit. Detcore replaces the Linux scheduler with its
2240    /// own deterministic one, so a thread's Linux scheduling attributes are inert.
2241    /// Report a fixed SCHED_OTHER policy with zeroed nice/priority/flags. The
2242    /// value is emulated (never injected), so it is identical across --verify runs
2243    /// and record/replay. Re-enables `chrt` under --strict.
2244    // AUTONOMOUS-BOT-IMPLEMENTED
2245    // TODO-HUMAN-REVIEW(#791)
2246    pub async fn handle_sched_getattr<G: Guest<Self>>(
2247        &self,
2248        guest: &mut G,
2249        call: syscalls::SchedGetattr,
2250    ) -> Result<i64, Error> {
2251        // `flags` is reserved and must be zero; the kernel rejects anything else.
2252        if call.flags() != 0 {
2253            return Err(Errno::EINVAL.into());
2254        }
2255        // The caller must provide room for at least the base sched_attr (VER0).
2256        let attr_size = std::mem::size_of::<libc::sched_attr>();
2257        if (call.size() as usize) < attr_size {
2258            return Err(Errno::EINVAL.into());
2259        }
2260        let attr = call.attr().ok_or(Errno::EINVAL)?;
2261        // SAFETY: sched_attr is a plain-old-data struct; an all-zero bit pattern is
2262        // a valid SCHED_OTHER descriptor (nice/priority/flags/runtime/... all 0).
2263        let mut sa: libc::sched_attr = unsafe { std::mem::zeroed() };
2264        sa.size = attr_size as u32;
2265        sa.sched_policy = libc::SCHED_OTHER as u32;
2266        // SAFETY: reinterpret the POD struct as its raw bytes to copy it into the
2267        // guest's buffer.
2268        let bytes = unsafe {
2269            std::slice::from_raw_parts(&sa as *const libc::sched_attr as *const u8, attr_size)
2270        };
2271        let dst: AddrMut<u8> = attr.cast();
2272        guest.memory().write_exact(dst, bytes)?;
2273        info!(
2274            "Emulating sched_getattr(pid={}): fixed SCHED_OTHER, nice 0, priority 0",
2275            call.pid()
2276        );
2277        Ok(0)
2278    }
2279
2280    // AUTONOMOUS-BOT-IMPLEMENTED
2281    // TODO-HUMAN-REVIEW(PR-841): Review virtual sched_setattr no-op policy.
2282    /// Linux scheduler attributes cannot affect Detcore's replacement
2283    /// scheduler, so a *well-formed* request is accepted as a deterministic
2284    /// no-op, matching the existing sched_setscheduler and sched_setparam
2285    /// policy.
2286    ///
2287    /// Suppressing the effect is not the same as accepting arguments Linux
2288    /// refuses, nor as refusing arguments Linux accepts. Both directions are
2289    /// guest-visible: a probe that expects EINVAL and sees success takes the
2290    /// wrong branch, and so does one that expects success and sees E2BIG.
2291    ///
2292    /// The order below is the kernel's, and it is guest-visible when two
2293    /// arguments are wrong at once, because the kernel returns the *first*
2294    /// applicable error. `sched_setattr()` screens `uattr`, `pid` and `flags`
2295    /// together; `sched_copy_attr()` then handles the size, the trailing bytes
2296    /// and the util-clamp size rule; the signed-policy test follows; the target
2297    /// pid is resolved *there*, in the middle; and only then does
2298    /// `__sched_setscheduler()` judge the policy, the flags, the priority and
2299    /// the deadline parameters. Putting any of that last group before the pid
2300    /// lookup makes a request against a nonexistent pid report EINVAL where
2301    /// Linux reports ESRCH.
2302    ///
2303    /// # Determinism
2304    ///
2305    /// Every check is a pure function of the guest's own arguments. The one
2306    /// piece of state consulted is the pid lookup, which asks the scheduler's
2307    /// own task table via `thread_is_live` -- Detcore state, replayed
2308    /// identically -- rather than the host's process table, which would leak
2309    /// unrelated host processes into a guest-visible answer.
2310    ///
2311    /// Specifically NOT `tool_global::resolve_kill_targets`, which
2312    /// models `kill(2)` and so recognises only thread-group leaders; asking it
2313    /// this question reports ESRCH for a live non-leader thread.
2314    ///
2315    /// # Deliberately not emulated
2316    ///
2317    /// Three behaviours are excluded, under one rule: **an answer that depends
2318    /// on the host's kernel configuration or the caller's privileges is not
2319    /// reproduced, because a deterministic sandbox must not vary with the
2320    /// machine underneath it.** Each was measured natively and would differ on
2321    /// a differently-built or differently-privileged host:
2322    ///
2323    /// * EPERM for a real-time priority, a SCHED_DEADLINE admission, or a
2324    ///   negative nice, which depends on `CAP_SYS_NICE` and `RLIMIT_RTPRIO`.
2325    /// * EOPNOTSUPP for util-clamp on a VER1 buffer, which depends on
2326    ///   `CONFIG_UCLAMP_TASK`.
2327    /// * The `sysctl_sched_dl_period_{min,max}` bound on SCHED_DEADLINE
2328    ///   periods, which is runtime-tunable.
2329    ///
2330    /// In all three Hermit accepts the request as the same no-op as any other
2331    /// well-formed one. The same rule is why SCHED_EXT (policy 7) is accepted
2332    /// unconditionally even though `valid_policy()` admits it only with
2333    /// `CONFIG_SCHED_CLASS_EXT`: picking one answer keeps the sandbox stable
2334    /// across hosts, and accepting is the choice consistent with suppressing
2335    /// the effect rather than refusing the request.
2336    ///
2337    /// The bracketed regression test in `hermit-cli/tests/sched_setattr_abi.rs`
2338    /// compares Hermit against the running kernel case by case, and therefore
2339    /// deliberately omits exactly these cases -- including them would make its
2340    /// verdict depend on the host it runs on.
2341    pub async fn handle_sched_setattr<G: Guest<Self>>(
2342        &self,
2343        guest: &mut G,
2344        call: syscalls::SchedSetattr,
2345    ) -> Result<i64, Error> {
2346        // `if (!uattr || pid < 0 || flags)` -- one clause in the kernel, so all
2347        // three are EINVAL and none of them is ordered against the others. A
2348        // null pointer is EINVAL rather than EFAULT because it is refused
2349        // before any access is attempted.
2350        if call.flags() != 0 || call.pid() < 0 {
2351            return Err(Errno::EINVAL.into());
2352        }
2353        let attr = call.attr().ok_or(Errno::EINVAL)?;
2354
2355        // `get_user(size, &uattr->size)` -- the declared size is read before
2356        // anything else, and a fault here is EFAULT with no store back.
2357        // `AddrMut` is not `Copy` and `cast` consumes it, so each access below
2358        // re-derives the address from the syscall argument.
2359        // A fault reading guest memory is EFAULT for this syscall. The
2360        // backend's memory layer reports a failed peek as EIO, which is not an
2361        // errno `sched_setattr` can return, so it is mapped here; measured
2362        // native, an unmapped `uattr` gives EFAULT.
2363        let declared: u32 = guest
2364            .memory()
2365            .read_value(attr.cast())
2366            .map_err(|_| Errno::EFAULT)?;
2367
2368        // Both E2BIG exits below run the kernel's `err_size` path, which stores
2369        // the kernel's own struct size into `uattr->size` before returning. A
2370        // guest that reads the field back learns the size it should have sent,
2371        // which is the entire point of the store; the kernel ignores whether it
2372        // succeeds, and so do we.
2373        fn refuse_too_big<S: MemoryAccess>(mut memory: S, attr: AddrMut<libc::c_void>) -> Error {
2374            let _ = memory.write_value(attr.cast::<u32>(), &SCHED_ATTR_KERNEL_SIZE);
2375            Errno::E2BIG.into()
2376        }
2377
2378        let size = match sched_attr_effective_size(declared) {
2379            Ok(size) => size,
2380            Err(()) => {
2381                let back = call.attr().ok_or(Errno::EINVAL)?;
2382                return Err(refuse_too_big(guest.memory(), back));
2383            }
2384        };
2385
2386        // `copy_struct_from_user` with a user size past the kernel's struct:
2387        // every trailing byte must be zero, or the guest is sending a field
2388        // this kernel does not know and must not silently drop. A fault while
2389        // scanning is EFAULT and skips the store back -- only the not-all-zero
2390        // verdict is E2BIG.
2391        if size > SCHED_ATTR_KERNEL_SIZE {
2392            let base: AddrMut<u8> = call.attr().ok_or(Errno::EINVAL)?.cast();
2393            match scan_tail_is_zeroed(
2394                &guest.memory(),
2395                base,
2396                SCHED_ATTR_KERNEL_SIZE as usize,
2397                (size - SCHED_ATTR_KERNEL_SIZE) as usize,
2398            ) {
2399                TailVerdict::AllZero => {}
2400                TailVerdict::NotZeroed => {
2401                    let back = call.attr().ok_or(Errno::EINVAL)?;
2402                    return Err(refuse_too_big(guest.memory(), back));
2403                }
2404                TailVerdict::Faulted => return Err(Errno::EFAULT.into()),
2405            }
2406        }
2407
2408        // Copy the interoperable prefix, zero-filling anything the guest's
2409        // buffer is too short to carry, exactly as `copy_struct_from_user`
2410        // does for a short read.
2411        let copied = std::cmp::min(size, SCHED_ATTR_KERNEL_SIZE) as usize;
2412        let mut raw = [0u8; SCHED_ATTR_KERNEL_SIZE as usize];
2413        let base: AddrMut<u8> = call.attr().ok_or(Errno::EINVAL)?.cast();
2414        guest
2415            .memory()
2416            .read_exact(base, &mut raw[..copied])
2417            .map_err(|_| Errno::EFAULT)?;
2418        let field32 = |offset: usize| -> u32 {
2419            u32::from_ne_bytes(raw[offset..offset + 4].try_into().expect("4 bytes"))
2420        };
2421        let field64 = |offset: usize| -> u64 {
2422            u64::from_ne_bytes(raw[offset..offset + 8].try_into().expect("8 bytes"))
2423        };
2424        let fields = SchedAttrFields {
2425            policy: field32(SCHED_ATTR_OFF_POLICY),
2426            sched_flags: field64(SCHED_ATTR_OFF_FLAGS),
2427            priority: field32(SCHED_ATTR_OFF_PRIORITY),
2428            runtime: field64(SCHED_ATTR_OFF_RUNTIME),
2429            deadline: field64(SCHED_ATTR_OFF_DEADLINE),
2430            period: field64(SCHED_ATTR_OFF_PERIOD),
2431        };
2432
2433        // Everything the kernel decides before it resolves the pid.
2434        validate_sched_attr_before_lookup(size, &fields)?;
2435
2436        // `find_process_by_pid()` sits here, between the two groups of checks,
2437        // and the position is guest-visible: a request that is malformed only
2438        // in a way judged below reports ESRCH rather than EINVAL when the pid
2439        // does not exist. pid 0 means the calling thread, which always exists;
2440        // otherwise ask the scheduler's own task table, so the answer comes
2441        // from Detcore's state rather than from the host's process table.
2442        // `find_task_by_vpid` resolves ANY LIVE TASK, not just a thread-group
2443        // leader. The kill-target resolver is the wrong question here: it
2444        // models `kill(2)`, whose first act is to refuse anything that is not a
2445        // leader, so a perfectly live non-leader thread's tid reported ESRCH.
2446        if call.pid() != 0 && !thread_is_live(guest, DetTid::from_raw(call.pid())).await {
2447            return Err(Errno::ESRCH.into());
2448        }
2449
2450        // And everything `__sched_setscheduler` decides after it.
2451        validate_sched_attr_after_lookup(&fields)?;
2452
2453        info!(
2454            "Suppressing sched_setattr(pid={}, flags={}); Linux scheduler attributes are virtual",
2455            call.pid(),
2456            call.flags()
2457        );
2458        Ok(0)
2459    }
2460
2461    /// ioprio_set under Hermit. Detcore serializes guest threads onto one virtual
2462    /// CPU, so the block-layer I/O scheduling class and priority cannot change
2463    /// guest-visible computation. Accept and suppress the request as a
2464    /// deterministic no-op success, mirroring how sched_setaffinity is handled.
2465    /// Re-enables `ionice` under --strict.
2466    // AUTONOMOUS-BOT-IMPLEMENTED
2467    // TODO-HUMAN-REVIEW(#791)
2468    pub async fn handle_ioprio_set<G: Guest<Self>>(
2469        &self,
2470        _guest: &mut G,
2471        call: syscalls::IoprioSet,
2472    ) -> Result<i64, Error> {
2473        info!(
2474            "Suppressing ioprio_set(which={}, who={}, priority={}); I/O priority is virtual",
2475            call.which(),
2476            call.who(),
2477            call.priority()
2478        );
2479        Ok(0)
2480    }
2481
2482    /// ioprio_get under Hermit. I/O priority is inert under Detcore's serialized
2483    /// scheduler, so process queries observe the fixed raw IOPRIO_CLASS_NONE
2484    /// value while group/user queries observe the effective SCHED_OTHER default
2485    /// of IOPRIO_CLASS_BE/4, without consulting host block-scheduler state.
2486    // AUTONOMOUS-BOT-IMPLEMENTED
2487    // TODO-HUMAN-REVIEW(PR-881)
2488    pub async fn handle_ioprio_get<G: Guest<Self>>(
2489        &self,
2490        _guest: &mut G,
2491        call: syscalls::IoprioGet,
2492    ) -> Result<i64, Error> {
2493        let priority = virtual_ioprio(call.which())?;
2494
2495        info!(
2496            "Emulating ioprio_get(which={}, who={}): fixed priority {}",
2497            call.which(),
2498            call.who(),
2499            priority
2500        );
2501        Ok(priority)
2502    }
2503}
2504
2505/// The one real implementation of [`robust_list::RobustDeathEffects`]: guest
2506/// memory plus a wake against Detcore's modeled futex waiter pool.
2507///
2508/// It exists so the walk itself never mentions `Guest`, and can therefore be
2509/// unit-tested against fake memory.
2510struct GuestRobustEffects<'a, G, T> {
2511    guest: &'a mut G,
2512    dettid: DetTid,
2513    defer_owner_death_to_backend: bool,
2514    staged_wakes: Option<&'a mut Vec<RobustListWake>>,
2515    tool: PhantomData<T>,
2516}
2517
2518impl<G, T> robust_list::RobustDeathEffects for GuestRobustEffects<'_, G, T>
2519where
2520    G: Guest<Detcore<T>>,
2521    T: RecordOrReplay,
2522{
2523    fn read_u64(&mut self, address: usize) -> Option<u64> {
2524        let at = Addr::<u64>::from_raw(address)?;
2525        self.guest.memory().read_value::<_, u64>(at).ok()
2526    }
2527
2528    fn read_u32(&mut self, address: usize) -> Option<u32> {
2529        let at = Addr::<u32>::from_raw(address)?;
2530        self.guest.memory().read_value::<_, u32>(at).ok()
2531    }
2532
2533    fn compare_and_swap(
2534        &mut self,
2535        address: usize,
2536        expected: u32,
2537        desired: u32,
2538    ) -> robust_list::FutexCasOutcome {
2539        use robust_list::FutexCasOutcome;
2540
2541        let (Some(read_at), Some(write_at)) = (
2542            Addr::<u32>::from_raw(address),
2543            AddrMut::<u32>::from_raw(address),
2544        ) else {
2545            return FutexCasOutcome::Faulted;
2546        };
2547        // Reverie's guest-memory interface offers reads and writes, not a
2548        // cross-address-space cmpxchg, so this re-reads the word immediately
2549        // before storing and refuses to store when the value has moved.
2550        //
2551        // The dying task keeps the scheduler turn for the whole walk, so no
2552        // modeled thread can run between this read and the write. That does not
2553        // make the operations atomic against a process outside the model. Only
2554        // ptrace can safely take the deferred branch below; DBT and SaBRe run
2555        // their tool-exit callbacks before native cleanup, and KVM has none.
2556        let observed = match self.guest.memory().read_value::<_, u32>(read_at) {
2557            Ok(value) => value,
2558            Err(_) => return FutexCasOutcome::Faulted,
2559        };
2560        if observed != expected {
2561            return FutexCasOutcome::Changed(observed);
2562        }
2563        if self.defer_owner_death_to_backend {
2564            // Ptrace keeps this syscall handler pending through the native
2565            // exit and does not release another scheduler turn until Linux has
2566            // repeated the owner check and changed the word atomically. Do not
2567            // perform a separate write here: a process outside Hermit's
2568            // scheduler can share the mapping and acquire the mutex between
2569            // this read and that write.
2570            debug!(
2571                "[detcore, dtid {}] robust-list owner death: leaving futex word {:#x} for backend exit cleanup",
2572                self.dettid, address,
2573            );
2574            return FutexCasOutcome::Deferred;
2575        }
2576        // `MemoryAccess::write_value` writes through to the guest immediately
2577        // (`write_exact`); it is not the scratch-stack path, which buffers
2578        // until `commit()`.
2579        if self.guest.memory().write_value(write_at, &desired).is_err() {
2580            return FutexCasOutcome::Faulted;
2581        }
2582        // Deliberately DEBUG, not INFO: this line carries a raw guest address,
2583        // and INFO is the surface `--verify-strict` compares. The
2584        // KVM diagnostics read this from `--log=debug`; keep it below INFO
2585        // because the address is not a deterministic observation.
2586        debug!(
2587            "[detcore, dtid {}] robust-list owner death: futex word {:#x} {:#x} -> {:#x}",
2588            self.dettid, address, expected, desired,
2589        );
2590        FutexCasOutcome::Stored
2591    }
2592
2593    async fn wake_one(&mut self, address: usize, observed: u32) {
2594        // glibc always issues robust-mutex futex operations with the shared
2595        // flag, and so does the kernel's owner-death wake; resolve the same key.
2596        let futexid = self.guest.thread_state().futex_id(address, false);
2597        if let Some(wakes) = self.staged_wakes.as_mut() {
2598            wakes.push(RobustListWake { futex: futexid });
2599            debug!(
2600                "[detcore, dtid {}] staged robust-list owner-death wake until physical exit",
2601                self.dettid,
2602            );
2603            return;
2604        }
2605        let woken = match futex_action(
2606            self.guest,
2607            FutexAction::WakeRequest(1),
2608            &futexid,
2609            observed as i32,
2610            u32::MAX,
2611        )
2612        .await
2613        {
2614            Some(SchedValue::Value(count)) => count,
2615            // A wake never carries a timeout, and a cancelled RPC wakes nobody.
2616            Some(SchedValue::TimeOut) | None => 0,
2617        };
2618        // Guest-level identities only: dettid, the modeled futex key, and a
2619        // count. No host pointers and no iteration order leak into this line,
2620        // which is compared exactly under `--verify-strict`.
2621        info!(
2622            "[detcore, dtid {}] robust-list owner death woke {} waiter(s) on futex {:?}",
2623            self.dettid, woken, futexid,
2624        );
2625        let _ = futex_action(
2626            self.guest,
2627            FutexAction::WakeFinished(0),
2628            &futexid,
2629            observed as i32,
2630            u32::MAX,
2631        )
2632        .await;
2633    }
2634}
2635
2636#[cfg(test)]
2637mod tests {
2638    use std::io::Read;
2639    use std::io::Seek;
2640    use std::os::fd::AsRawFd;
2641
2642    use reverie::GlobalRPC;
2643    use reverie::GlobalTool;
2644    use reverie::Tid;
2645    use reverie::Tool;
2646
2647    use super::*;
2648    use crate::config::Config;
2649    use crate::tool_global::GlobalRequest;
2650    use crate::tool_global::GlobalState;
2651    use crate::types::MmId;
2652
2653    struct FailedExecStack;
2654    struct FailedExecStackGuard;
2655
2656    impl Drop for FailedExecStackGuard {
2657        fn drop(&mut self) {}
2658    }
2659
2660    impl reverie::Stack for FailedExecStack {
2661        type StackGuard = FailedExecStackGuard;
2662
2663        fn size(&self) -> usize {
2664            panic!("failed exec must not use the guest stack")
2665        }
2666        fn capacity(&self) -> usize {
2667            panic!("failed exec must not use the guest stack")
2668        }
2669        fn push<'stack, T>(&mut self, _: T) -> Addr<'stack, T> {
2670            panic!("failed exec must not use the guest stack")
2671        }
2672        fn reserve<'stack, T>(&mut self) -> AddrMut<'stack, T> {
2673            panic!("failed exec must not use the guest stack")
2674        }
2675        fn commit(self) -> Result<Self::StackGuard, Errno> {
2676            panic!("failed exec must not use the guest stack")
2677        }
2678    }
2679
2680    // Inject only the backend's errno. Preparation, rollback, cancellation,
2681    // unsupported reporting and the configured refusal policy are real code.
2682    struct FailedExecGuest<'a> {
2683        config: &'a Config,
2684        global: &'a GlobalState,
2685        thread: crate::ThreadState<()>,
2686        sender: Tid,
2687        process: Tid,
2688        old_mm: MmId,
2689        errno: Errno,
2690        injections: usize,
2691        requests: Mutex<Vec<GlobalRequest>>,
2692    }
2693
2694    #[reverie::tool]
2695    impl GlobalRPC<GlobalState> for FailedExecGuest<'_> {
2696        async fn send_rpc(
2697            &self,
2698            message: <GlobalState as GlobalTool>::Request,
2699        ) -> <GlobalState as GlobalTool>::Response {
2700            assert_eq!(message.1, self.old_mm, "RPC follows restored exec identity");
2701            self.requests.lock().unwrap().push(message.2.clone());
2702            self.global.receive_rpc(self.sender, message).await
2703        }
2704        fn config(&self) -> &Config {
2705            self.config
2706        }
2707    }
2708
2709    #[reverie::tool]
2710    impl Guest<Detcore> for FailedExecGuest<'_> {
2711        type Memory = reverie::syscalls::LocalMemory;
2712        type Stack = FailedExecStack;
2713
2714        fn tid(&self) -> Tid {
2715            self.sender
2716        }
2717        fn pid(&self) -> Tid {
2718            self.process
2719        }
2720        fn ppid(&self) -> Option<Tid> {
2721            None
2722        }
2723        fn memory(&self) -> Self::Memory {
2724            panic!("failed exec rollback must not read guest memory")
2725        }
2726        fn thread_state(&self) -> &crate::ThreadState<()> {
2727            &self.thread
2728        }
2729        fn thread_state_mut(&mut self) -> &mut crate::ThreadState<()> {
2730            &mut self.thread
2731        }
2732        async fn regs(&mut self) -> libc::user_regs_struct {
2733            panic!("failed exec rollback must not read registers")
2734        }
2735        async fn stack(&mut self) -> Self::Stack {
2736            panic!("failed exec rollback must not use a guest stack")
2737        }
2738        async fn daemonize(&mut self) {
2739            panic!("failed exec rollback must not daemonize")
2740        }
2741        async fn inject<S: SyscallInfo>(&mut self, call: S) -> Result<i64, Errno> {
2742            assert_eq!(call.number(), syscalls::Sysno::execveat);
2743            assert_eq!(
2744                self.thread.mm_id,
2745                self.old_mm.for_exec(self.thread.detpid.unwrap())
2746            );
2747            assert_eq!(self.thread.robust_list_head, None);
2748            self.injections += 1;
2749            Err(self.errno)
2750        }
2751        async fn tail_inject<S: SyscallInfo>(&mut self, _: S) -> reverie::Never {
2752            panic!("failed exec must not retire a live guest")
2753        }
2754        fn set_timer(&mut self, _: reverie::TimerSchedule) -> Result<(), Error> {
2755            panic!("failed exec rollback must not replace the timer")
2756        }
2757        fn set_timer_precise(&mut self, _: reverie::TimerSchedule) -> Result<(), Error> {
2758            panic!("failed exec rollback must not replace the timer")
2759        }
2760        fn read_clock(&mut self) -> Result<u64, Error> {
2761            panic!("failed exec rollback must not sample the clock")
2762        }
2763    }
2764
2765    #[tokio::test]
2766    async fn kvm_nonleader_exec_refusal_preserves_failed_exec_rollback_and_policy() {
2767        let process = DetPid::from_raw(17);
2768        let worker = DetTid::from_raw(18);
2769        for (backend_is_kvm, caller, errno, fail_closed, refused, reported) in [
2770            (true, worker, Errno::ENOSYS, true, true, false),
2771            (true, worker, Errno::ENOSYS, false, false, true),
2772            (true, worker, Errno::ENOENT, true, false, false),
2773            (true, worker, Errno::ENOEXEC, true, false, false),
2774            (true, worker, Errno::EFAULT, true, false, false),
2775            (true, worker, Errno::EACCES, true, false, false),
2776            (true, worker, Errno::EOPNOTSUPP, true, false, false),
2777            (true, process, Errno::ENOSYS, true, false, false),
2778            (false, worker, Errno::ENOSYS, true, false, false),
2779        ] {
2780            let mut report = tempfile::tempfile().unwrap();
2781            let config = Config {
2782                backend_is_kvm,
2783                sequentialize_threads: false,
2784                panic_on_unsupported_syscalls: fail_closed,
2785                exit_on_unsupported_syscall: true,
2786                shutdown_on_unsupported_syscall: false,
2787                unsupported_syscall_report_fd: Some(report.as_raw_fd()),
2788                ..Config::default()
2789            };
2790            let global = GlobalState::init_global_state(&config).await;
2791            let tool = Detcore::new(Tid::from_raw(process.as_raw()), &config);
2792            let mut thread = crate::ThreadState::new(caller, &config, ());
2793            thread.detpid = Some(process);
2794            thread.mm_id = MmId::initial(process);
2795            thread.record_robust_list_head(Some(0x12340));
2796            thread.thread_logical_time.add_syscall_with_cost(123);
2797            let old_time = thread.thread_logical_time.as_nanos();
2798            let old_files = Arc::clone(&thread.file_metadata);
2799            let old_memory = Arc::clone(&thread.memory_metadata);
2800            let old_mm = thread.mm_id;
2801            let mut guest = FailedExecGuest {
2802                config: &config,
2803                global: &global,
2804                thread,
2805                sender: Tid::from_raw(caller.as_raw()),
2806                process: Tid::from_raw(process.as_raw()),
2807                old_mm,
2808                errno,
2809                injections: 0,
2810                requests: Mutex::new(Vec::new()),
2811            };
2812            let result = tool
2813                .handle_execveat(&mut guest, syscalls::Execveat::new())
2814                .await;
2815            match result {
2816                Err(Error::Tool(error)) if refused => assert_eq!(
2817                    error
2818                        .downcast_ref::<crate::UnsupportedSyscallError>()
2819                        .unwrap()
2820                        .0,
2821                    syscalls::Sysno::execveat
2822                ),
2823                Err(Error::Errno(actual)) if !refused => assert_eq!(actual, errno),
2824                result => panic!("wrong failed-exec policy: {result:?}"),
2825            }
2826            assert_eq!(guest.injections, 1, "backend preflight must run first");
2827            assert_eq!(guest.thread.dettid, caller);
2828            assert_eq!(guest.thread.detpid, Some(process));
2829            assert_eq!(guest.thread.mm_id, old_mm);
2830            assert_eq!(guest.thread.robust_list_head, Some(0x12340));
2831            assert_eq!(guest.thread.thread_logical_time.as_nanos(), old_time);
2832            assert!(Arc::ptr_eq(&guest.thread.file_metadata, &old_files));
2833            assert!(Arc::ptr_eq(&guest.thread.memory_metadata, &old_memory));
2834            let requests = guest.requests.lock().unwrap();
2835            assert!(matches!(requests[0], GlobalRequest::PrepareExec(..)));
2836            assert!(matches!(requests[1], GlobalRequest::CancelExec(..)));
2837            assert_eq!(requests.len(), if reported { 3 } else { 2 });
2838            if reported {
2839                assert!(
2840                    matches!(&requests[2], GlobalRequest::ReportUnsupportedSyscall(name) if name == "execveat")
2841                );
2842            }
2843            report.rewind().unwrap();
2844            let mut aggregate = String::new();
2845            report.read_to_string(&mut aggregate).unwrap();
2846            assert_eq!(aggregate, if reported { "execveat\n" } else { "" });
2847        }
2848    }
2849
2850    #[test]
2851    fn kernel_blocked_mask_preserves_libc_signal_membership() {
2852        let mask = blocked_signal_mask();
2853        let mut libc_mask: libc::sigset_t = unsafe { std::mem::zeroed() };
2854        unsafe {
2855            libc::sigfillset(&mut libc_mask);
2856            libc::sigdelset(&mut libc_mask, reverie::PERF_EVENT_SIGNAL as i32);
2857        }
2858        for raw_signal in 1..=KernelSigset::BITS as i32 {
2859            assert_eq!(
2860                signal_is_blocked(&mask, SigWrapper(raw_signal)),
2861                unsafe { libc::sigismember(&libc_mask, raw_signal) == 1 },
2862                "signal {raw_signal} membership changed while converting to the kernel ABI"
2863            );
2864        }
2865    }
2866
2867    #[test]
2868    fn clone_without_child_cleartid_does_not_register_the_pointer() {
2869        assert_eq!(
2870            child_tid_clear_address(CloneFlags::CLONE_CHILD_SETTID, 0x1234),
2871            0
2872        );
2873    }
2874
2875    #[test]
2876    fn clone_with_child_cleartid_registers_the_pointer() {
2877        assert_eq!(
2878            child_tid_clear_address(CloneFlags::CLONE_CHILD_CLEARTID, 0x1234),
2879            0x1234
2880        );
2881    }
2882
2883    #[test]
2884    fn linux_default_dispositions_that_do_not_interrupt_child_waits() {
2885        for signal in [libc::SIGCHLD, libc::SIGCONT, libc::SIGURG, libc::SIGWINCH] {
2886            assert!(signal_default_disposition_does_not_interrupt_child_wait(
2887                SigWrapper(signal)
2888            ));
2889        }
2890        for signal in [libc::SIGALRM, libc::SIGSTOP, libc::SIGUSR1] {
2891            assert!(!signal_default_disposition_does_not_interrupt_child_wait(
2892                SigWrapper(signal)
2893            ));
2894        }
2895    }
2896
2897    #[test]
2898    fn uncatchable_signals_do_not_require_a_sigaction_query() {
2899        assert!(signal_has_uncatchable_default_disposition(SigWrapper(
2900            libc::SIGKILL
2901        )));
2902        assert!(signal_has_uncatchable_default_disposition(SigWrapper(
2903            libc::SIGSTOP
2904        )));
2905        assert!(!signal_has_uncatchable_default_disposition(SigWrapper(
2906            libc::SIGUSR1
2907        )));
2908    }
2909
2910    #[test]
2911    fn waitid_ready_child_wins_when_scheduler_also_reports_a_signal() {
2912        assert_eq!(
2913            exact_wait_poll_decision(true, true, Some(ExactChildWaitState::Running)),
2914            ExactWaitPollDecision::ChildReady
2915        );
2916        assert_eq!(
2917            exact_wait_poll_decision(false, true, Some(ExactChildWaitState::LogicallyExited)),
2918            ExactWaitPollDecision::ReapAfterLogicalExit
2919        );
2920        assert_eq!(
2921            exact_wait_poll_decision(false, true, Some(ExactChildWaitState::PhysicalExitPending)),
2922            ExactWaitPollDecision::AwaitPhysicalExit
2923        );
2924        assert_eq!(
2925            exact_wait_poll_decision(false, true, Some(ExactChildWaitState::Running)),
2926            ExactWaitPollDecision::Interrupted
2927        );
2928        assert_eq!(
2929            exact_wait_poll_decision(false, false, Some(ExactChildWaitState::Running)),
2930            ExactWaitPollDecision::Retry
2931        );
2932    }
2933
2934    #[test]
2935    fn wait4_argument_validation_follows_linux_precedence() {
2936        let valid_bits = [0, 1, 3, 29, 30, 31];
2937        for bit in valid_bits {
2938            let options = WaitPidFlag::from_bits_retain((1_u32 << bit) as libc::c_int);
2939            assert_eq!(validate_wait4_arguments(-1, options), Ok(()), "bit {bit}");
2940        }
2941
2942        for bit in (0..u32::BITS).filter(|bit| !valid_bits.contains(bit)) {
2943            let options = WaitPidFlag::from_bits_retain((1_u32 << bit) as libc::c_int);
2944            assert_eq!(
2945                validate_wait4_arguments(-1, options),
2946                Err(Errno::EINVAL),
2947                "bit {bit}"
2948            );
2949        }
2950
2951        // Only INT_MIN is refused: every other pid, including the nearest
2952        // process-group selector, must still reach the wait.
2953        for pid in [libc::pid_t::MIN + 1, -2, 0, 1, libc::pid_t::MAX] {
2954            assert_eq!(
2955                validate_wait4_arguments(pid, WaitPidFlag::empty()),
2956                Ok(()),
2957                "pid {pid}"
2958            );
2959        }
2960        assert_eq!(
2961            validate_wait4_arguments(libc::pid_t::MIN, WaitPidFlag::empty()),
2962            Err(Errno::ESRCH)
2963        );
2964        assert_eq!(
2965            validate_wait4_arguments(libc::pid_t::MIN, WaitPidFlag::from_bits_retain(0x10)),
2966            Err(Errno::EINVAL),
2967            "invalid options must win over the INT_MIN selector"
2968        );
2969    }
2970
2971    #[test]
2972    fn stale_any_child_preserves_interrupt_until_no_ready_child_remains() {
2973        let next_child = DetPid::from_raw(200);
2974
2975        assert!(
2976            !stale_any_wait_must_interrupt(true, Some(next_child)),
2977            "another ready child must retain child-ready precedence"
2978        );
2979        assert!(
2980            stale_any_wait_must_interrupt(true, None),
2981            "a pending signal must interrupt before the wait parks again"
2982        );
2983        assert!(!stale_any_wait_must_interrupt(false, None));
2984    }
2985
2986    #[test]
2987    fn ioprio_query_reports_fixed_raw_and_effective_defaults() {
2988        assert_eq!(virtual_ioprio(IOPRIO_WHO_PROCESS), Ok(0));
2989        assert_eq!(
2990            virtual_ioprio(IOPRIO_WHO_PGRP),
2991            Ok(i64::from(IOPRIO_DEFAULT_EFFECTIVE))
2992        );
2993        assert_eq!(
2994            virtual_ioprio(IOPRIO_WHO_USER),
2995            Ok(i64::from(IOPRIO_DEFAULT_EFFECTIVE))
2996        );
2997        assert_eq!(virtual_ioprio(0), Err(Errno::EINVAL));
2998        assert_eq!(virtual_ioprio(4), Err(Errno::EINVAL));
2999    }
3000
3001    #[test]
3002    fn waitid_siginfo_canonicalization_clears_only_cpu_accounting() {
3003        let mut info: libc::siginfo_t = unsafe { std::mem::zeroed() };
3004        info.si_signo = libc::SIGCHLD;
3005        info.si_code = libc::CLD_EXITED;
3006        // SAFETY: This uses the same Linux SIGCHLD layout mirror validated by
3007        // canonicalize_waitid_siginfo.
3008        let fields = unsafe {
3009            &mut (*(std::ptr::addr_of_mut!(info)).cast::<WaitidSiginfoHead>())
3010                .fields
3011                .sigchld
3012        };
3013        fields.pid = 123;
3014        fields.uid = 456;
3015        fields.status = 7;
3016        fields.utime = 8;
3017        fields.stime = 9;
3018
3019        canonicalize_waitid_siginfo(&mut info);
3020
3021        assert_eq!(info.si_signo, libc::SIGCHLD);
3022        assert_eq!(info.si_code, libc::CLD_EXITED);
3023        assert_eq!(unsafe { info.si_pid() }, 123);
3024        assert_eq!(unsafe { info.si_uid() }, 456);
3025        assert_eq!(unsafe { info.si_status() }, 7);
3026        assert_eq!(unsafe { info.si_utime() }, 0);
3027        assert_eq!(unsafe { info.si_stime() }, 0);
3028    }
3029
3030    #[test]
3031    fn wait_status_rollup_only_accepts_process_termination() {
3032        assert!(wait_status_is_termination(0));
3033        assert!(wait_status_is_termination(libc::SIGTERM));
3034        assert!(!wait_status_is_termination((libc::SIGSTOP << 8) | 0x7f));
3035        assert!(!wait_status_is_termination(0xffff));
3036
3037        assert!(waitid_code_is_termination(libc::CLD_EXITED));
3038        assert!(waitid_code_is_termination(libc::CLD_KILLED));
3039        assert!(waitid_code_is_termination(libc::CLD_DUMPED));
3040        assert!(!waitid_code_is_termination(libc::CLD_STOPPED));
3041        assert!(!waitid_code_is_termination(libc::CLD_CONTINUED));
3042        assert!(!waitid_code_is_termination(libc::CLD_TRAPPED));
3043    }
3044
3045    #[test]
3046    fn futex_timeout_units_and_modes_match_linux() {
3047        let timeout = Timespec {
3048            tv_sec: 2,
3049            tv_nsec: 3,
3050        };
3051        assert_eq!(
3052            parse_futex_timeout(libc::FUTEX_WAIT, timeout),
3053            Ok(FutexTimeout::Relative(2_000_000_003))
3054        );
3055        assert_eq!(
3056            parse_futex_timeout(libc::FUTEX_WAIT_BITSET, timeout),
3057            Ok(FutexTimeout::Absolute(LogicalTime::from_nanos(
3058                2_000_000_003
3059            )))
3060        );
3061        // The command bits must be matched after masking off FUTEX_PRIVATE_FLAG
3062        // (and FUTEX_CLOCK_REALTIME): a private FUTEX_WAIT_BITSET still uses an
3063        // absolute deadline, and a private FUTEX_WAIT still uses a relative one.
3064        assert_eq!(
3065            parse_futex_timeout(libc::FUTEX_WAIT_BITSET | libc::FUTEX_PRIVATE_FLAG, timeout),
3066            Ok(FutexTimeout::Absolute(LogicalTime::from_nanos(
3067                2_000_000_003
3068            )))
3069        );
3070        assert_eq!(
3071            parse_futex_timeout(libc::FUTEX_WAIT | libc::FUTEX_PRIVATE_FLAG, timeout),
3072            Ok(FutexTimeout::Relative(2_000_000_003))
3073        );
3074    }
3075
3076    #[test]
3077    fn absolute_futex_timeout_is_rebased_to_logical_time() {
3078        let logical_now = LogicalTime::from_secs(100);
3079        let clock_now = LogicalTime::from_secs(5_000);
3080        let deadline = clock_now + Duration::from_millis(100);
3081        assert_eq!(
3082            rebase_absolute_timeout(deadline, clock_now, logical_now),
3083            logical_now + Duration::from_millis(100)
3084        );
3085        assert_eq!(
3086            rebase_absolute_timeout(
3087                clock_now - LogicalTime::from_nanos(1),
3088                clock_now,
3089                logical_now
3090            ),
3091            logical_now
3092        );
3093    }
3094
3095    #[test]
3096    fn absolute_futex_timeout_detects_host_and_logical_clock_domains() {
3097        let host_monotonic_now = LogicalTime::from_secs(374_766);
3098        let logical_now = LogicalTime::from_secs(1_640_995_199);
3099        let delta = Duration::from_millis(100);
3100
3101        assert!(absolute_timeout_uses_host_clock(
3102            host_monotonic_now + delta,
3103            host_monotonic_now,
3104            logical_now
3105        ));
3106        assert!(!absolute_timeout_uses_host_clock(
3107            logical_now + delta,
3108            host_monotonic_now,
3109            logical_now
3110        ));
3111
3112        let host_realtime_now = LogicalTime::from_secs(1_785_142_800);
3113        assert!(absolute_timeout_uses_host_clock(
3114            host_realtime_now + delta,
3115            host_realtime_now,
3116            logical_now
3117        ));
3118    }
3119
3120    #[test]
3121    fn futex_timeout_rejects_invalid_timespecs() {
3122        assert_eq!(
3123            parse_futex_timeout(
3124                libc::FUTEX_WAIT,
3125                Timespec {
3126                    tv_sec: -1,
3127                    tv_nsec: 0,
3128                },
3129            ),
3130            Err(Errno::EINVAL)
3131        );
3132        assert_eq!(
3133            parse_futex_timeout(
3134                libc::FUTEX_WAIT_BITSET,
3135                Timespec {
3136                    tv_sec: 0,
3137                    tv_nsec: 1_000_000_000,
3138                },
3139            ),
3140            Err(Errno::EINVAL)
3141        );
3142    }
3143
3144    // The expectations below are not guesses at the kernel's behaviour: each is
3145    // a row measured natively on this host (Linux 6.19, x86-64) with a raw
3146    // `sched_setattr` probe. Where a row differs from what the handler used to
3147    // do, the difference is called out.
3148
3149    /// A well-formed VER0 SCHED_OTHER descriptor, as the decoded fields.
3150    fn plain_attr() -> SchedAttrFields {
3151        SchedAttrFields {
3152            policy: 0,
3153            sched_flags: 0,
3154            priority: 0,
3155            runtime: 0,
3156            deadline: 0,
3157            period: 0,
3158        }
3159    }
3160
3161    #[test]
3162    fn size_zero_is_a_well_formed_ver0_request() {
3163        // ABI compatibility quirk: `if (!size) size = SCHED_ATTR_SIZE_VER0;`.
3164        // Measured native: ret 0. PR #2288 returned E2BIG.
3165        assert_eq!(sched_attr_effective_size(0), Ok(SCHED_ATTR_SIZE_VER0));
3166    }
3167
3168    #[test]
3169    fn size_below_ver0_or_past_a_page_is_too_big() {
3170        // Measured native: size 1, 47 and 4097 all give E2BIG.
3171        assert_eq!(sched_attr_effective_size(1), Err(()));
3172        assert_eq!(sched_attr_effective_size(SCHED_ATTR_SIZE_VER0 - 1), Err(()));
3173        assert_eq!(sched_attr_effective_size(SCHED_ATTR_MAX_SIZE + 1), Err(()));
3174    }
3175
3176    #[test]
3177    fn size_from_ver0_through_one_page_is_accepted_unchanged() {
3178        // Measured native: 48, 56, 57 and 4096 all give ret 0.
3179        for size in [
3180            SCHED_ATTR_SIZE_VER0,
3181            SCHED_ATTR_SIZE_VER1,
3182            SCHED_ATTR_SIZE_VER1 + 1,
3183            SCHED_ATTR_MAX_SIZE,
3184        ] {
3185            assert_eq!(sched_attr_effective_size(size), Ok(size), "size {}", size);
3186        }
3187    }
3188
3189    #[test]
3190    fn sched_ext_is_a_valid_policy_and_sched_iso_is_not() {
3191        // Measured native: policy 7 gives ret 0; policy 4 gives EINVAL. PR
3192        // #2288 refused 7.
3193        assert!(is_valid_sched_policy(SCHED_EXT), "SCHED_EXT is accepted");
3194        assert!(!is_valid_sched_policy(4), "SCHED_ISO is reserved");
3195        for policy in [0, 1, 2, 3, 5, 6] {
3196            assert!(is_valid_sched_policy(policy), "policy {}", policy);
3197        }
3198        // Measured native: policy 8 and 99 both give EINVAL.
3199        for policy in [8, 99] {
3200            assert!(!is_valid_sched_policy(policy), "policy {}", policy);
3201        }
3202    }
3203
3204    // ---- which side of the pid lookup each rule falls on --------------------
3205    // The kernel resolves the pid between `sched_setattr()` and
3206    // `__sched_setscheduler()`, so a request naming a nonexistent pid reports
3207    // ESRCH for every rule on the far side and its own errno for every rule on
3208    // the near side. Measured natively with pid 0x3fffffff:
3209    //     bad sched_flags    -> ESRCH    (far side)
3210    //     bad priority       -> ESRCH    (far side)
3211    //     negative policy    -> EINVAL   (near side)
3212    //     util-clamp size 48 -> EINVAL   (near side)
3213    // The split between the two functions below is exactly that boundary.
3214
3215    #[test]
3216    fn util_clamp_size_rule_is_decided_before_the_pid_lookup() {
3217        // Measured native: EINVAL even with a nonexistent pid.
3218        let mut attr = plain_attr();
3219        attr.sched_flags = SCHED_FLAG_UTIL_CLAMP_MIN;
3220        assert_eq!(
3221            validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER0, &attr),
3222            Err(Errno::EINVAL)
3223        );
3224        assert_eq!(
3225            validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER1, &attr),
3226            Ok(())
3227        );
3228        // ...and it is not re-decided on the far side.
3229        assert_eq!(validate_sched_attr_after_lookup(&attr), Ok(()));
3230    }
3231
3232    #[test]
3233    fn a_negative_policy_is_decided_before_the_pid_lookup() {
3234        // Measured native: EINVAL even with a nonexistent pid, and EINVAL with
3235        // KEEP_POLICY too -- the signed test sits above the substitution.
3236        let mut attr = plain_attr();
3237        attr.policy = 0x8000_0000;
3238        assert_eq!(
3239            validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER0, &attr),
3240            Err(Errno::EINVAL)
3241        );
3242        attr.sched_flags = SCHED_FLAG_KEEP_POLICY;
3243        assert_eq!(
3244            validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER0, &attr),
3245            Err(Errno::EINVAL),
3246            "KEEP_POLICY must not hide a negative policy"
3247        );
3248    }
3249
3250    #[test]
3251    fn policy_and_flag_rules_are_decided_after_the_pid_lookup() {
3252        // Measured native: with a nonexistent pid both of these report ESRCH,
3253        // so neither may be decided on the near side.
3254        let mut bad_policy = plain_attr();
3255        bad_policy.policy = 99;
3256        assert_eq!(
3257            validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER0, &bad_policy),
3258            Ok(()),
3259            "an undefined policy must survive the near side so ESRCH can win"
3260        );
3261        assert_eq!(
3262            validate_sched_attr_after_lookup(&bad_policy),
3263            Err(Errno::EINVAL)
3264        );
3265
3266        let mut bad_flag = plain_attr();
3267        bad_flag.sched_flags = 0x80;
3268        assert_eq!(
3269            validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER0, &bad_flag),
3270            Ok(()),
3271            "an undefined sched_flags bit must survive the near side"
3272        );
3273        assert_eq!(
3274            validate_sched_attr_after_lookup(&bad_flag),
3275            Err(Errno::EINVAL)
3276        );
3277    }
3278
3279    #[test]
3280    fn keep_policy_makes_the_policy_field_irrelevant() {
3281        // Measured native: policy 99 alone is EINVAL, but policy 99 with
3282        // SCHED_FLAG_KEEP_POLICY is ret 0 -- the kernel overwrites the field
3283        // with SETPARAM_POLICY and never runs valid_policy(). PR #2288 refused
3284        // it.
3285        let mut attr = plain_attr();
3286        attr.policy = 99;
3287        assert_eq!(validate_sched_attr_after_lookup(&attr), Err(Errno::EINVAL));
3288        attr.sched_flags = SCHED_FLAG_KEEP_POLICY;
3289        assert_eq!(validate_sched_attr_after_lookup(&attr), Ok(()));
3290    }
3291
3292    #[test]
3293    fn defined_sched_flags_bits_are_accepted_and_undefined_ones_are_not() {
3294        // Measured native: sched_flags 0x80 gives EINVAL; RESET_ON_FORK,
3295        // KEEP_PARAMS, KEEP_ALL, RECLAIM and DL_OVERRUN are all ret 0 at
3296        // size 48. PR #2288 ignored sched_flags entirely.
3297        let mut attr = plain_attr();
3298        attr.sched_flags = 0x80;
3299        assert_eq!(validate_sched_attr_after_lookup(&attr), Err(Errno::EINVAL));
3300        for flag in [
3301            SCHED_FLAG_RESET_ON_FORK,
3302            SCHED_FLAG_RECLAIM,
3303            SCHED_FLAG_DL_OVERRUN,
3304            SCHED_FLAG_KEEP_PARAMS,
3305            SCHED_FLAG_KEEP_POLICY,
3306        ] {
3307            let mut ok = plain_attr();
3308            ok.sched_flags = flag;
3309            assert_eq!(
3310                validate_sched_attr_after_lookup(&ok),
3311                Ok(()),
3312                "flag {:#x}",
3313                flag
3314            );
3315        }
3316        // The whole mask at once is refused on the NEAR side, and for the
3317        // util-clamp size reason rather than the flag-validity one: measured
3318        // native EINVAL at size 48.
3319        let mut all = plain_attr();
3320        all.sched_flags = SCHED_FLAG_ALL;
3321        assert_eq!(
3322            validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER0, &all),
3323            Err(Errno::EINVAL)
3324        );
3325        assert_eq!(
3326            validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER1, &all),
3327            Ok(())
3328        );
3329        assert_eq!(validate_sched_attr_after_lookup(&all), Ok(()));
3330    }
3331
3332    #[test]
3333    fn priority_must_agree_with_the_policy() {
3334        // The kernel states this as `rt_policy(policy) != (prio != 0)`.
3335        // Measured native, all EINVAL: OTHER/BATCH/IDLE with priority 1, and
3336        // FIFO/RR with priority 0. PR #2288 accepted every one of them.
3337        for policy in [0u32, 3, 5] {
3338            let mut attr = plain_attr();
3339            attr.policy = policy;
3340            attr.priority = 1;
3341            assert_eq!(
3342                validate_sched_attr_after_lookup(&attr),
3343                Err(Errno::EINVAL),
3344                "policy {} with priority 1",
3345                policy
3346            );
3347            attr.priority = 0;
3348            assert_eq!(
3349                validate_sched_attr_after_lookup(&attr),
3350                Ok(()),
3351                "policy {} with priority 0",
3352                policy
3353            );
3354        }
3355        for policy in [SCHED_FIFO, SCHED_RR] {
3356            let mut attr = plain_attr();
3357            attr.policy = policy;
3358            attr.priority = 0;
3359            assert_eq!(
3360                validate_sched_attr_after_lookup(&attr),
3361                Err(Errno::EINVAL),
3362                "rt policy {} with priority 0",
3363                policy
3364            );
3365            // A real-time priority is accepted here; whether the *caller* may
3366            // ask for it is EPERM, which depends on privilege and is not
3367            // emulated.
3368            attr.priority = 1;
3369            assert_eq!(validate_sched_attr_after_lookup(&attr), Ok(()));
3370        }
3371    }
3372
3373    #[test]
3374    fn a_priority_past_max_rt_prio_is_refused() {
3375        // Measured native: FIFO with priority 100 is EINVAL rather than the
3376        // EPERM that priorities 1..=99 give, so the range test runs first.
3377        let mut attr = plain_attr();
3378        attr.policy = SCHED_FIFO;
3379        attr.priority = MAX_RT_PRIO;
3380        assert_eq!(validate_sched_attr_after_lookup(&attr), Err(Errno::EINVAL));
3381        attr.priority = MAX_RT_PRIO - 1;
3382        assert_eq!(validate_sched_attr_after_lookup(&attr), Ok(()));
3383    }
3384
3385    #[test]
3386    fn deadline_parameters_are_checked() {
3387        // Measured native: SCHED_DEADLINE with all-zero parameters is EINVAL,
3388        // and with runtime > deadline is EINVAL. PR #2288 accepted both.
3389        let mut zeroed = plain_attr();
3390        zeroed.policy = SCHED_DEADLINE;
3391        assert_eq!(
3392            validate_sched_attr_after_lookup(&zeroed),
3393            Err(Errno::EINVAL),
3394            "all-zero deadline parameters"
3395        );
3396
3397        let mut inverted = plain_attr();
3398        inverted.policy = SCHED_DEADLINE;
3399        inverted.runtime = 90_000_000;
3400        inverted.deadline = 30_000_000;
3401        inverted.period = 30_000_000;
3402        assert_eq!(
3403            validate_sched_attr_after_lookup(&inverted),
3404            Err(Errno::EINVAL),
3405            "runtime > deadline"
3406        );
3407
3408        // Sane parameters pass the ABI rules. Natively this host answers EPERM,
3409        // which is a privilege question rather than an ABI one.
3410        let mut sane = plain_attr();
3411        sane.policy = SCHED_DEADLINE;
3412        sane.runtime = 10_000_000;
3413        sane.deadline = 30_000_000;
3414        sane.period = 30_000_000;
3415        assert_eq!(validate_sched_attr_after_lookup(&sane), Ok(()));
3416    }
3417
3418    #[test]
3419    fn deadline_parameter_edges_follow_checkparam_dl() {
3420        // deadline == 0 is refused whatever else is set.
3421        assert!(!deadline_params_are_valid(1 << 20, 0, 0));
3422        // runtime below the DL_SCALE truncation floor (1 << 10) is refused.
3423        assert!(!deadline_params_are_valid((1 << 10) - 1, 1 << 20, 1 << 20));
3424        assert!(deadline_params_are_valid(1 << 10, 1 << 20, 1 << 20));
3425        // The MSB is reserved on both deadline and period.
3426        assert!(!deadline_params_are_valid(1 << 20, 1 << 63, 0));
3427        assert!(!deadline_params_are_valid(1 << 20, 1 << 20, 1 << 63));
3428        // A zero period means "same as the deadline", so this is runtime <=
3429        // deadline <= deadline and is accepted.
3430        assert!(deadline_params_are_valid(1 << 20, 1 << 21, 0));
3431        // deadline > period is refused.
3432        assert!(!deadline_params_are_valid(1 << 20, 1 << 22, 1 << 21));
3433    }
3434
3435    #[test]
3436    fn the_size_written_back_on_refusal_is_the_kernel_struct_not_the_libc_mirror() {
3437        // Measured native: every E2BIG row leaves uattr->size holding 56, not
3438        // 48. Deriving that from the libc crate's `sched_attr` would report 48
3439        // and tell the guest to retry with a size the kernel already accepted.
3440        assert_eq!(SCHED_ATTR_KERNEL_SIZE, 56);
3441        assert_eq!(std::mem::size_of::<libc::sched_attr>(), 48);
3442        assert_ne!(
3443            SCHED_ATTR_KERNEL_SIZE as usize,
3444            std::mem::size_of::<libc::sched_attr>()
3445        );
3446    }
3447
3448    /// A `MemoryAccess` whose readable region ends at `readable_len`, recording
3449    /// every read length it is asked for.
3450    struct BoundedMemory {
3451        bytes: Vec<u8>,
3452        readable_len: usize,
3453        reads: std::cell::RefCell<Vec<usize>>,
3454    }
3455
3456    impl reverie::syscalls::MemoryAccess for BoundedMemory {
3457        fn read_vectored(
3458            &self,
3459            read_from: &[std::io::IoSlice<'_>],
3460            write_to: &mut [std::io::IoSliceMut<'_>],
3461        ) -> Result<usize, Errno> {
3462            let start = read_from[0].as_ptr() as usize;
3463            let want = read_from.iter().map(|slice| slice.len()).sum::<usize>();
3464            self.reads.borrow_mut().push(want);
3465            if start >= self.readable_len {
3466                return Err(Errno::EFAULT);
3467            }
3468            let avail = (self.readable_len - start).min(want);
3469            let mut copied = 0;
3470            for out in write_to.iter_mut() {
3471                if copied == avail {
3472                    break;
3473                }
3474                let take = out.len().min(avail - copied);
3475                out[..take].copy_from_slice(&self.bytes[start + copied..start + copied + take]);
3476                copied += take;
3477            }
3478            Ok(copied)
3479        }
3480
3481        fn write_vectored(
3482            &mut self,
3483            _read_from: &[std::io::IoSlice<'_>],
3484            _write_to: &mut [std::io::IoSliceMut<'_>],
3485        ) -> Result<usize, Errno> {
3486            unimplemented!("this fixture never writes")
3487        }
3488    }
3489
3490    fn bounded(bytes: Vec<u8>, readable_len: usize) -> BoundedMemory {
3491        BoundedMemory {
3492            bytes,
3493            readable_len,
3494            reads: std::cell::RefCell::new(Vec::new()),
3495        }
3496    }
3497
3498    fn tail_base() -> AddrMut<'static, u8> {
3499        AddrMut::<u8>::from_raw(1).expect("nonzero base")
3500    }
3501
3502    /// ITEM 2 REGRESSION, ORDERING HALF: a non-zero byte BEFORE an unreadable
3503    /// page is E2BIG, not EFAULT.
3504    ///
3505    /// `check_zeroed_user` scans forward and stops at the first thing it finds,
3506    /// so it never reaches the fault. Reading the whole tail up front and
3507    /// judging afterwards turns this into EFAULT and changes the errno the
3508    /// guest sees.
3509    #[test]
3510    fn a_nonzero_byte_before_a_fault_is_e2big_not_efault() {
3511        let off = SCHED_ATTR_KERNEL_SIZE as usize;
3512        let mut bytes = vec![0u8; off + 4096];
3513        bytes[off + 1] = 0xAA; // non-zero, well before the boundary
3514        let readable = off + 512; // everything past this faults
3515        let memory = bounded(bytes, readable);
3516        assert_eq!(
3517            scan_tail_is_zeroed(&memory, tail_base(), off, 4096),
3518            TailVerdict::NotZeroed,
3519            "a non-zero byte the scan reaches first must win over a later fault"
3520        );
3521    }
3522
3523    /// ITEM 2 REGRESSION, the other side of the same order: a fault with
3524    /// nothing non-zero before it is EFAULT.
3525    #[test]
3526    fn a_fault_before_any_nonzero_byte_is_efault() {
3527        let off = SCHED_ATTR_KERNEL_SIZE as usize;
3528        let bytes = vec![0u8; off + 4096];
3529        let memory = bounded(bytes, off + 512);
3530        assert_eq!(
3531            scan_tail_is_zeroed(&memory, tail_base(), off, 4096),
3532            TailVerdict::Faulted,
3533            "an unreadable byte reached before anything non-zero must be EFAULT"
3534        );
3535    }
3536
3537    /// ITEM 2 REGRESSION, PROTECTION HALF: never issue a read of eight bytes or
3538    /// fewer.
3539    ///
3540    /// safeptrace serves those through `PTRACE_PEEKDATA`, which reads a whole
3541    /// aligned word and bypasses guest page protections, so a small tail would
3542    /// be readable under Hermit where Linux reports EFAULT. The assertion is on
3543    /// the READ LENGTHS ASKED FOR, because that is the mechanism; the returned
3544    /// verdict cannot show it.
3545    #[test]
3546    fn tail_reads_are_never_small_enough_to_bypass_guest_protection() {
3547        let off = SCHED_ATTR_KERNEL_SIZE as usize;
3548        for tail_len in 1..=16usize {
3549            let bytes = vec![0u8; off + tail_len + 64];
3550            let memory = bounded(bytes, off + tail_len + 64);
3551            assert_eq!(
3552                scan_tail_is_zeroed(&memory, tail_base(), off, tail_len),
3553                TailVerdict::AllZero
3554            );
3555            let reads = memory.reads.borrow().clone();
3556            assert!(!reads.is_empty(), "tail_len {tail_len} issued no read");
3557            for length in reads {
3558                assert!(
3559                    length > std::mem::size_of::<u64>(),
3560                    "tail_len {tail_len} issued a {length}-byte read, which safeptrace \
3561                     serves with PTRACE_PEEKDATA and which therefore bypasses guest \
3562                     page protections"
3563                );
3564            }
3565        }
3566    }
3567
3568    /// ITEM 3 REGRESSION: KEEP_POLICY reuses the CURRENT policy; it does not
3569    /// switch the policy-dependent rules off.
3570    ///
3571    /// Detcore's current policy is the one `handle_sched_getattr` reports for
3572    /// every thread, SCHED_OTHER, and a non-real-time policy requires priority
3573    /// 0. Skipping the rules accepted this.
3574    #[test]
3575    fn keep_policy_validates_against_the_virtual_current_policy() {
3576        let attr = SchedAttrFields {
3577            policy: 99, // ignored under KEEP_POLICY, and deliberately invalid
3578            sched_flags: SCHED_FLAG_KEEP_POLICY,
3579            priority: 1,
3580            runtime: 0,
3581            deadline: 0,
3582            period: 0,
3583        };
3584        assert_eq!(
3585            validate_sched_attr_after_lookup(&attr),
3586            Err(Errno::EINVAL),
3587            "priority 1 under the virtual SCHED_OTHER current policy must be refused"
3588        );
3589
3590        // The companion that must keep passing: the ignored policy field really
3591        // is ignored, so the same request at priority 0 is accepted.
3592        let ok = SchedAttrFields {
3593            priority: 0,
3594            ..attr
3595        };
3596        assert_eq!(
3597            validate_sched_attr_after_lookup(&ok),
3598            Ok(()),
3599            "KEEP_POLICY must still ignore the policy field itself"
3600        );
3601    }
3602
3603    /// ITEM 3 COHERENCE: the substituted policy is the one the guest can
3604    /// actually observe, so the two sites cannot drift apart silently.
3605    #[test]
3606    fn the_virtual_current_policy_is_what_sched_getattr_reports() {
3607        assert_eq!(
3608            VIRTUAL_CURRENT_POLICY,
3609            libc::SCHED_OTHER as u32,
3610            "handle_sched_getattr writes SCHED_OTHER into sched_policy for every thread; \
3611             KEEP_POLICY must substitute that same value"
3612        );
3613    }
3614}