use std::marker::PhantomData;
use std::sync::Arc;
use std::sync::Mutex;
use std::time::Duration;
use procfs::process::Process;
use rand::Rng;
use reverie::Error;
use reverie::Guest;
use reverie::Pid;
use reverie::Stack;
use reverie::syscalls;
use reverie::syscalls::Addr;
use reverie::syscalls::AddrMut;
use reverie::syscalls::CloneFlags;
use reverie::syscalls::Errno;
use reverie::syscalls::MemoryAccess;
use reverie::syscalls::Syscall;
use reverie::syscalls::SyscallInfo;
use reverie::syscalls::Timespec;
use reverie::syscalls::WaitPidFlag;
use tracing::debug;
use tracing::info;
use tracing::trace;
use crate::config::BlockingMode;
use crate::memory::MemoryMetadata;
use crate::record_or_replay::RecordOrReplay;
use crate::resources::ExternalOpId;
use crate::resources::Permission;
use crate::resources::ResourceID;
use crate::resources::Resources;
use crate::scheduler::SchedValue;
use crate::syscalls::helpers::record_retry_event;
use crate::syscalls::helpers::retry_nonblocking_syscall;
use crate::syscalls::helpers::retry_nonblocking_syscall_with_timeout;
use crate::syscalls::robust_list;
use crate::tool_global::FutexAction;
use crate::tool_global::ResumeStatus;
use crate::tool_global::await_exact_child_physical_exit;
use crate::tool_global::cancel_exec;
use crate::tool_global::child_tid_clear_address;
use crate::tool_global::consume_child_wait;
use crate::tool_global::create_child_thread;
use crate::tool_global::futex_action;
use crate::tool_global::prepare_exec;
use crate::tool_global::process_group;
use crate::tool_global::ready_child_wait;
use crate::tool_global::resource_request;
use crate::tool_global::set_child_tid_address;
use crate::tool_global::thread_is_live;
use crate::tool_global::thread_observe_time;
use crate::tool_global::wait_for_child_lifecycle;
use crate::tool_global::yield_once;
use crate::tool_local::Detcore;
use crate::tool_local::PendingVfork;
use crate::tool_local::RobustListExit;
use crate::tool_local::RobustListWake;
use crate::types::ChildWaitExitClass;
use crate::types::ChildWaitSelector;
use crate::types::ChildWaitSpec;
use crate::types::DetPid;
use crate::types::DetTid;
use crate::types::ExactChildWaitState;
use crate::types::LogicalTime;
use crate::types::SigWrapper;
const VIRTUAL_CPUSET_BYTES: usize = 16;
const IOPRIO_WHO_PROCESS: libc::c_int = 1;
const IOPRIO_WHO_PGRP: libc::c_int = 2;
const IOPRIO_WHO_USER: libc::c_int = 3;
const IOPRIO_CLASS_SHIFT: libc::c_int = 13;
const IOPRIO_CLASS_BE: libc::c_int = 2;
const IOPRIO_BE_NORM: libc::c_int = 4;
const IOPRIO_DEFAULT_EFFECTIVE: libc::c_int =
(IOPRIO_CLASS_BE << IOPRIO_CLASS_SHIFT) | IOPRIO_BE_NORM;
const SCHED_ATTR_SIZE_VER0: u32 = 48; const SCHED_ATTR_SIZE_VER1: u32 = 56; const SCHED_ATTR_KERNEL_SIZE: u32 = SCHED_ATTR_SIZE_VER1;
const SCHED_ATTR_MAX_SIZE: u32 = 4096;
const VIRTUAL_CURRENT_POLICY: u32 = libc::SCHED_OTHER as u32;
#[derive(Debug, Eq, PartialEq)]
enum TailVerdict {
AllZero,
NotZeroed,
Faulted,
}
const MIN_PROTECTION_RESPECTING_READ: usize = std::mem::size_of::<u64>() + 1;
fn scan_tail_is_zeroed<M: MemoryAccess>(
memory: &M,
base: AddrMut<u8>,
tail_off: usize,
tail_len: usize,
) -> TailVerdict {
const CHUNK: usize = 256;
let mut done = 0usize;
let mut buf = [0u8; CHUNK];
while done < tail_len {
let remaining = tail_len - done;
let want = remaining.min(CHUNK);
let back = MIN_PROTECTION_RESPECTING_READ
.saturating_sub(want)
.min(done + tail_off);
let read_len = want + back;
let start = tail_off + done - back;
let Some(addr) = base
.as_raw()
.checked_add(start)
.and_then(AddrMut::<u8>::from_raw)
else {
return TailVerdict::Faulted;
};
let got = memory.read(addr, &mut buf[..read_len]).unwrap_or(0);
let new_from = back.min(got);
if buf[new_from..got].iter().any(|byte| *byte != 0) {
return TailVerdict::NotZeroed;
}
if got < read_len {
return TailVerdict::Faulted;
}
done += want;
}
TailVerdict::AllZero
}
const SCHED_FIFO: u32 = 1;
const SCHED_RR: u32 = 2;
const SCHED_DEADLINE: u32 = 6;
const SCHED_EXT: u32 = 7;
const MAX_RT_PRIO: u32 = 100;
const SCHED_ATTR_OFF_POLICY: usize = 4;
const SCHED_ATTR_OFF_FLAGS: usize = 8;
const SCHED_ATTR_OFF_PRIORITY: usize = 20;
const SCHED_ATTR_OFF_RUNTIME: usize = 24;
const SCHED_ATTR_OFF_DEADLINE: usize = 32;
const SCHED_ATTR_OFF_PERIOD: usize = 40;
const SCHED_FLAG_RESET_ON_FORK: u64 = 0x01;
const SCHED_FLAG_RECLAIM: u64 = 0x02;
const SCHED_FLAG_DL_OVERRUN: u64 = 0x04;
const SCHED_FLAG_KEEP_POLICY: u64 = 0x08;
const SCHED_FLAG_KEEP_PARAMS: u64 = 0x10;
const SCHED_FLAG_UTIL_CLAMP_MIN: u64 = 0x20;
const SCHED_FLAG_UTIL_CLAMP_MAX: u64 = 0x40;
const SCHED_FLAG_UTIL_CLAMP: u64 = SCHED_FLAG_UTIL_CLAMP_MIN | SCHED_FLAG_UTIL_CLAMP_MAX;
const SCHED_FLAG_ALL: u64 = SCHED_FLAG_RESET_ON_FORK
| SCHED_FLAG_RECLAIM
| SCHED_FLAG_DL_OVERRUN
| SCHED_FLAG_KEEP_POLICY
| SCHED_FLAG_KEEP_PARAMS
| SCHED_FLAG_UTIL_CLAMP;
fn is_valid_sched_policy(policy: u32) -> bool {
matches!(
policy,
0 | 1 | 2 | 3
| 5
| SCHED_DEADLINE
| SCHED_EXT
)
}
fn is_rt_policy(policy: u32) -> bool {
policy == SCHED_FIFO || policy == SCHED_RR
}
fn deadline_params_are_valid(runtime: u64, deadline: u64, period: u64) -> bool {
if deadline == 0 {
return false;
}
if runtime < (1u64 << 10) {
return false;
}
if deadline & (1u64 << 63) != 0 || period & (1u64 << 63) != 0 {
return false;
}
let period = if period == 0 { deadline } else { period };
runtime <= deadline && deadline <= period
}
fn sched_attr_effective_size(declared: u32) -> Result<u32, ()> {
let size = if declared == 0 {
SCHED_ATTR_SIZE_VER0
} else {
declared
};
if !(SCHED_ATTR_SIZE_VER0..=SCHED_ATTR_MAX_SIZE).contains(&size) {
return Err(());
}
Ok(size)
}
#[derive(Clone, Copy, Debug)]
struct SchedAttrFields {
policy: u32,
sched_flags: u64,
priority: u32,
runtime: u64,
deadline: u64,
period: u64,
}
fn validate_sched_attr_before_lookup(size: u32, attr: &SchedAttrFields) -> Result<(), Errno> {
if attr.sched_flags & SCHED_FLAG_UTIL_CLAMP != 0 && size < SCHED_ATTR_SIZE_VER1 {
return Err(Errno::EINVAL);
}
if (attr.policy as i32) < 0 {
return Err(Errno::EINVAL);
}
Ok(())
}
fn validate_sched_attr_after_lookup(attr: &SchedAttrFields) -> Result<(), Errno> {
let policy = if attr.sched_flags & SCHED_FLAG_KEEP_POLICY != 0 {
Some(VIRTUAL_CURRENT_POLICY)
} else {
if !is_valid_sched_policy(attr.policy) {
return Err(Errno::EINVAL);
}
Some(attr.policy)
};
if attr.sched_flags & !SCHED_FLAG_ALL != 0 {
return Err(Errno::EINVAL);
}
if attr.priority > MAX_RT_PRIO - 1 {
return Err(Errno::EINVAL);
}
if let Some(policy) = policy {
if is_rt_policy(policy) != (attr.priority != 0) {
return Err(Errno::EINVAL);
}
if policy == SCHED_DEADLINE
&& !deadline_params_are_valid(attr.runtime, attr.deadline, attr.period)
{
return Err(Errno::EINVAL);
}
}
Ok(())
}
fn virtual_ioprio(which: libc::c_int) -> Result<i64, Errno> {
match which {
IOPRIO_WHO_PROCESS => Ok(0),
IOPRIO_WHO_PGRP | IOPRIO_WHO_USER => Ok(i64::from(IOPRIO_DEFAULT_EFFECTIVE)),
_ => Err(Errno::EINVAL),
}
}
#[repr(C)]
#[derive(Clone, Copy)]
struct WaitidSigchldFields {
pid: libc::pid_t,
uid: libc::uid_t,
status: libc::c_int,
utime: libc::c_long,
stime: libc::c_long,
}
#[repr(C)]
union WaitidSiginfoFields {
_alignment: *mut libc::c_void,
sigchld: WaitidSigchldFields,
}
#[repr(C)]
struct WaitidSiginfoHead {
_base: [libc::c_int; 3],
fields: WaitidSiginfoFields,
}
fn wait_status_is_termination(status: libc::c_int) -> bool {
libc::WIFEXITED(status) || libc::WIFSIGNALED(status)
}
fn waitid_code_is_termination(code: libc::c_int) -> bool {
matches!(code, libc::CLD_EXITED | libc::CLD_KILLED | libc::CLD_DUMPED)
}
fn canonicalize_waitid_siginfo(info: &mut libc::siginfo_t) {
debug_assert!(
std::mem::size_of::<WaitidSiginfoHead>() <= std::mem::size_of::<libc::siginfo_t>()
);
let sigchld = unsafe {
&mut (*(info as *mut libc::siginfo_t).cast::<WaitidSiginfoHead>())
.fields
.sigchld
};
sigchld.utime = 0;
sigchld.stime = 0;
}
fn finish_waitid_result<T, G>(
guest: &mut G,
call: syscalls::Waitid,
value: i64,
mut info_value: libc::siginfo_t,
) -> Result<i64, Error>
where
T: RecordOrReplay,
G: Guest<Detcore<T>>,
{
let child_pid = unsafe { info_value.si_pid() };
if child_pid != 0 {
canonicalize_waitid_siginfo(&mut info_value);
guest.memory().write_value(
call.info().expect("waitid infop checked before execution"),
&info_value,
)?;
if call.options() & libc::WNOWAIT == 0 && waitid_code_is_termination(info_value.si_code) {
guest
.thread_state_mut()
.reap_child_process_cpu_time(DetPid::from_raw(child_pid));
}
if let Some(rusage) = call.rusage() {
let usage: libc::rusage = unsafe { std::mem::zeroed() };
guest.memory().write_value(rusage, &usage)?;
}
}
Ok(value)
}
#[derive(Debug, Eq, PartialEq)]
enum ExactWaitPollDecision {
ChildReady,
AwaitPhysicalExit,
ReapAfterLogicalExit,
Interrupted,
Retry,
}
fn exact_wait_poll_decision(
child_ready: bool,
signaled: bool,
lifecycle: Option<ExactChildWaitState>,
) -> ExactWaitPollDecision {
if child_ready {
ExactWaitPollDecision::ChildReady
} else if lifecycle == Some(ExactChildWaitState::PhysicalExitPending) {
ExactWaitPollDecision::AwaitPhysicalExit
} else if matches!(
lifecycle,
Some(ExactChildWaitState::LogicallyExited | ExactChildWaitState::PhysicallyExited)
) {
ExactWaitPollDecision::ReapAfterLogicalExit
} else if signaled {
ExactWaitPollDecision::Interrupted
} else {
ExactWaitPollDecision::Retry
}
}
fn stale_any_wait_must_interrupt(signaled: bool, next_ready: Option<DetPid>) -> bool {
signaled && next_ready.is_none()
}
fn terminal_child_wait_spec(
selector: ChildWaitSelector,
caller: DetTid,
options: libc::c_int,
) -> ChildWaitSpec {
let exit_class = if options & libc::__WALL != 0 {
ChildWaitExitClass::Any
} else if options & libc::__WCLONE != 0 {
ChildWaitExitClass::Clone
} else {
ChildWaitExitClass::Sigchld
};
ChildWaitSpec {
selector,
owner: (options & libc::__WNOTHREAD != 0).then_some(caller),
exit_class,
}
}
fn validate_wait4_arguments(pid: libc::pid_t, options: WaitPidFlag) -> Result<(), Errno> {
let allowed_options = WaitPidFlag::WNOHANG
| WaitPidFlag::WUNTRACED
| WaitPidFlag::WCONTINUED
| WaitPidFlag::__WNOTHREAD
| WaitPidFlag::__WCLONE
| WaitPidFlag::__WALL;
if options.bits() & !allowed_options.bits() != 0 {
return Err(Errno::EINVAL);
}
if pid == libc::pid_t::MIN {
return Err(Errno::ESRCH);
}
Ok(())
}
fn child_wait_can_retry_after_stale(spec: ChildWaitSpec) -> bool {
!matches!(spec.selector, ChildWaitSelector::Exact(_))
}
pub(super) type KernelSigset = u64;
pub(super) const KERNEL_SIGSET_SIZE: usize = std::mem::size_of::<KernelSigset>();
fn signal_is_blocked(mask: &KernelSigset, signal: SigWrapper) -> bool {
let raw_signal = signal.raw();
(1..=KernelSigset::BITS as i32).contains(&raw_signal)
&& mask & (1_u64 << (raw_signal as u32 - 1)) != 0
}
#[repr(C)]
#[derive(Clone, Copy)]
pub(super) struct KernelSigaction {
pub(super) handler: u64,
pub(super) flags: u64,
pub(super) restorer: u64,
pub(super) mask: KernelSigset,
}
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
pub(super) enum WaitSignalDisposition {
Interrupt,
Restart,
}
fn signal_default_disposition_does_not_interrupt_child_wait(signal: SigWrapper) -> bool {
matches!(
signal.raw(),
libc::SIGCHLD | libc::SIGCONT | libc::SIGURG | libc::SIGWINCH
)
}
fn signal_has_uncatchable_default_disposition(signal: SigWrapper) -> bool {
matches!(signal.raw(), libc::SIGKILL | libc::SIGSTOP)
}
pub(super) async fn wait_signal_disposition<G, T>(
guest: &mut G,
status: ResumeStatus,
guest_signal_mask: &KernelSigset,
action_addr: AddrMut<'_, KernelSigaction>,
inspect_action: bool,
) -> Result<Option<WaitSignalDisposition>, Error>
where
G: Guest<Detcore<T>>,
T: RecordOrReplay,
{
let ResumeStatus::Signaled(signals) = status else {
return Ok(None);
};
let Some(mut signals) = signals else {
return Ok(Some(WaitSignalDisposition::Interrupt));
};
signals.sort_by_key(|signal| signal.raw());
for signal in signals {
if signal_is_blocked(guest_signal_mask, signal) {
continue;
}
if signal_has_uncatchable_default_disposition(signal) {
return Ok(Some(WaitSignalDisposition::Interrupt));
}
if !inspect_action {
return Ok(Some(WaitSignalDisposition::Interrupt));
}
let call = syscalls::RtSigaction::new()
.with_signum(signal.raw())
.with_action(None)
.with_old_action(Some(action_addr.cast()))
.with_sigsetsize(std::mem::size_of::<u64>());
guest.inject_with_retry(call).await?;
let action: KernelSigaction = guest.memory().read_value(action_addr)?;
if action.handler == libc::SIG_IGN as u64
|| action.handler == libc::SIG_DFL as u64
&& signal_default_disposition_does_not_interrupt_child_wait(signal)
{
continue;
}
return Ok(Some(
if action.handler != libc::SIG_DFL as u64 && action.flags & libc::SA_RESTART as u64 != 0
{
WaitSignalDisposition::Restart
} else {
WaitSignalDisposition::Interrupt
},
));
}
Ok(None)
}
pub(super) fn blocked_signal_mask() -> KernelSigset {
let mut libc_mask: libc::sigset_t = unsafe { std::mem::zeroed() };
unsafe {
libc::sigfillset(&mut libc_mask);
libc::sigdelset(&mut libc_mask, reverie::PERF_EVENT_SIGNAL as i32);
}
(1..=KernelSigset::BITS as i32).fold(0, |mask, raw_signal| {
if unsafe { libc::sigismember(&libc_mask, raw_signal) } == 1 {
mask | (1_u64 << (raw_signal as u32 - 1))
} else {
mask
}
})
}
pub(super) async fn block_signals_for_disposition<G, T>(
guest: &mut G,
blocked_mask_addr: Addr<'_, KernelSigset>,
old_mask_addr: AddrMut<'_, KernelSigset>,
) -> Result<KernelSigset, Error>
where
G: Guest<Detcore<T>>,
T: RecordOrReplay,
{
let block_signals = syscalls::RtSigprocmask::new()
.with_how(libc::SIG_SETMASK)
.with_set(
(!guest
.config()
.backend_requires_thread_directed_process_signals)
.then_some(blocked_mask_addr.cast()),
)
.with_oldset(Some(old_mask_addr.cast()))
.with_sigsetsize(KERNEL_SIGSET_SIZE);
guest.inject_with_retry(block_signals).await?;
Ok(guest.memory().read_value(old_mask_addr)?)
}
pub(super) async fn restore_signals_after_disposition<G, T>(
guest: &mut G,
old_mask_addr: AddrMut<'_, KernelSigset>,
) -> Result<(), Error>
where
G: Guest<Detcore<T>>,
T: RecordOrReplay,
{
if !guest
.config()
.backend_requires_thread_directed_process_signals
{
let old_mask: Addr<'_, KernelSigset> = old_mask_addr.into();
let restore_signals = syscalls::RtSigprocmask::new()
.with_how(libc::SIG_SETMASK)
.with_set(Some(old_mask.cast()))
.with_oldset(None)
.with_sigsetsize(KERNEL_SIGSET_SIZE);
guest.inject_with_retry(restore_signals).await?;
}
Ok(())
}
async fn interrupted_child_wait_result<G, T, S>(
guest: &mut G,
call: S,
disposition: WaitSignalDisposition,
) -> Result<i64, Error>
where
G: Guest<Detcore<T>>,
T: RecordOrReplay,
S: SyscallInfo,
{
if !guest
.config()
.backend_requires_thread_directed_process_signals
{
return Err(Errno::ERESTARTSYS.into());
}
if disposition == WaitSignalDisposition::Interrupt {
return Err(Errno::EINTR.into());
}
guest.tail_inject(call).await
}
fn snapshot_process_group(pid: Pid) -> Result<libc::pid_t, Errno> {
let pgrp = Process::new(pid.as_raw())
.and_then(|process| process.stat())
.map(|stat| stat.pgrp)
.map_err(|_| Errno::ESRCH)?;
if pgrp == 0 {
Err(Errno::EOPNOTSUPP)
} else {
Ok(pgrp)
}
}
fn guest_fd_status_flags(pid: Pid, fd: libc::c_int) -> Result<libc::c_int, Errno> {
let path = format!("/proc/{}/fdinfo/{}", pid.as_raw(), fd);
let contents = std::fs::read_to_string(path).map_err(|_| Errno::EBADF)?;
let flags = contents
.lines()
.find_map(|line| line.strip_prefix("flags:"))
.map(str::trim)
.ok_or(Errno::EINVAL)?;
libc::c_int::from_str_radix(flags, 8).map_err(|_| Errno::EINVAL)
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
enum FutexTimeout {
Relative(u64),
Absolute(LogicalTime),
}
fn parse_futex_timeout(futex_op: i32, timeout: Timespec) -> Result<FutexTimeout, Errno> {
let seconds = u64::try_from(timeout.tv_sec).map_err(|_| Errno::EINVAL)?;
let nanoseconds = u64::try_from(timeout.tv_nsec).map_err(|_| Errno::EINVAL)?;
if nanoseconds >= 1_000_000_000 {
return Err(Errno::EINVAL);
}
let timeout_nanos = seconds
.checked_mul(1_000_000_000)
.and_then(|nanos| nanos.checked_add(nanoseconds))
.ok_or(Errno::EINVAL)?;
if futex_op & libc::FUTEX_CMD_MASK == libc::FUTEX_WAIT_BITSET {
Ok(FutexTimeout::Absolute(LogicalTime::from_nanos(
timeout_nanos,
)))
} else {
Ok(FutexTimeout::Relative(timeout_nanos))
}
}
fn rebase_absolute_timeout(
deadline: LogicalTime,
clock_now: LogicalTime,
logical_now: LogicalTime,
) -> LogicalTime {
logical_now + Duration::from_nanos(deadline.as_nanos().saturating_sub(clock_now.as_nanos()))
}
fn absolute_timeout_uses_host_clock(
deadline: LogicalTime,
host_clock_now: LogicalTime,
logical_now: LogicalTime,
) -> bool {
deadline.as_nanos().abs_diff(host_clock_now.as_nanos())
< deadline.as_nanos().abs_diff(logical_now.as_nanos())
}
impl<T: RecordOrReplay> Detcore<T> {
async fn futex_timeout_deadline<G: Guest<Self>>(
&self,
guest: &mut G,
futex_flags: i32,
timeout: Option<Addr<'_, Timespec>>,
) -> Result<Option<LogicalTime>, Error> {
let Some(timeout) = timeout else {
return Ok(None);
};
let timeout = parse_futex_timeout(futex_flags, guest.memory().read_value(timeout)?)?;
match timeout {
FutexTimeout::Relative(nanos) => {
let now = thread_observe_time(guest).await;
Ok(Some(now + Duration::from_nanos(nanos)))
}
FutexTimeout::Absolute(deadline)
if self.cfg.virtualize_time && !self.cfg.detect_host_clock_futex_timeouts =>
{
Ok(Some(deadline))
}
FutexTimeout::Absolute(deadline) => {
let clockid = if futex_flags & libc::FUTEX_CLOCK_REALTIME != 0 {
syscalls::ClockId::CLOCK_REALTIME
} else {
syscalls::ClockId::CLOCK_MONOTONIC
};
let mut stack = guest.stack().await;
let clock_output = syscalls::TimespecMutPtr(stack.reserve());
let _stack_guard = stack.commit()?;
let clock_call = syscalls::ClockGettime::new()
.with_clockid(clockid)
.with_tp(Some(clock_output));
if self.cfg.virtualize_time && self.cfg.detect_host_clock_futex_timeouts {
guest.inject(Syscall::from(clock_call)).await?;
} else {
self.record_or_replay(guest, clock_call).await?;
}
let clock_now = match parse_futex_timeout(
libc::FUTEX_WAIT_BITSET,
guest.memory().read_value(clock_output.0)?,
)? {
FutexTimeout::Absolute(time) => time,
FutexTimeout::Relative(_) => unreachable!(),
};
let logical_now = thread_observe_time(guest).await;
if self.cfg.virtualize_time
&& !absolute_timeout_uses_host_clock(deadline, clock_now, logical_now)
{
return Ok(Some(deadline));
}
Ok(Some(rebase_absolute_timeout(
deadline,
clock_now,
logical_now,
)))
}
}
}
pub async fn handle_clone_family<G: Guest<Self>>(
&self,
guest: &mut G,
clone_family: syscalls::family::CloneFamily,
) -> Result<i64, Error> {
let flags = clone_family.flags(&guest.memory());
let exit_signal = match clone_family {
#[cfg(not(target_arch = "aarch64"))]
syscalls::family::CloneFamily::Fork(_) | syscalls::family::CloneFamily::Vfork(_) => {
libc::SIGCHLD
}
syscalls::family::CloneFamily::Clone(clone) => {
(clone.flags().bits() & 0xff) as libc::c_int
}
syscalls::family::CloneFamily::Clone3(clone) => clone
.args()
.and_then(|address| guest.memory().read_value(address).ok())
.map_or(0, |args: syscalls::CloneArgs| {
args.exit_signal as libc::c_int
}),
};
let ctid = child_tid_clear_address(flags, clone_family.child_tid(&guest.memory()));
let is_vfork = flags.contains(CloneFlags::CLONE_VFORK);
let parent_blocks_for_child = is_vfork
|| (self.cfg.backend_serializes_fork_children
&& !flags.contains(CloneFlags::CLONE_THREAD));
let backend_uninstrumented_thread =
flags.contains(CloneFlags::CLONE_THREAD) && !self.cfg.backend_dispatches_thread_tools;
let ts = guest.thread_state_mut();
assert_eq!(ts.clone_flags, None);
assert!(ts.pending_vfork.is_none());
ts.clone_flags = Some(flags);
let parent_dettid = ts.dettid;
let child_priority_entropy = if parent_blocks_for_child
&& self.cfg.chaos
&& self.cfg.replay_preemptions_from.is_none()
&& self.cfg.replay_schedule_from.is_none()
{
let mut parent_chaos_prng = ts.chaos_prng.clone();
Some(parent_chaos_prng.next_u64())
} else {
None
};
if parent_blocks_for_child {
ts.pending_vfork = Some(PendingVfork {
parent_dettid,
parent_detpid: ts.detpid.expect("detpid unset"),
child_tid_addr: ctid,
flags,
exit_signal,
child_priority_entropy,
});
}
trace!("[detcore, dtid {}] parent invoking clone.", parent_dettid);
let blocking_child_op_id =
ExternalOpId::new(parent_dettid, guest.thread_state().stats.syscall_count);
if parent_blocks_for_child && self.cfg.sequentialize_threads {
let mut resources = Resources::new(parent_dettid);
resources.insert(
ResourceID::BlockingVfork(blocking_child_op_id),
Permission::RW,
);
resources.fyi(if is_vfork {
"clone_vfork"
} else {
"clone_serialized_child"
});
resource_request(guest, resources).await;
}
let maybe_res = guest.inject(Syscall::from(clone_family)).await;
if parent_blocks_for_child && self.cfg.sequentialize_threads {
let mut resources = Resources::new(parent_dettid);
if maybe_res.is_err() {
resources.insert(
ResourceID::VforkFailed(blocking_child_op_id),
Permission::RW,
);
resources.fyi(if is_vfork {
"clone_vfork_failed"
} else {
"clone_serialized_child_failed"
});
} else {
resources.insert(
ResourceID::BlockedExternalContinue(blocking_child_op_id),
Permission::RW,
);
resources.fyi(if is_vfork {
"clone_vfork"
} else {
"clone_serialized_child"
});
}
resource_request(guest, resources).await;
}
let ts = guest.thread_state_mut();
ts.clone_flags = None; ts.pending_vfork = None;
let res = maybe_res?;
if !flags.contains(CloneFlags::CLONE_THREAD) {
guest.thread_state().forget_flock_modes();
}
if parent_blocks_for_child
&& self.cfg.chaos
&& self.cfg.replay_preemptions_from.is_none()
&& self.cfg.replay_schedule_from.is_none()
{
let _ = guest
.thread_state_mut()
.chaos_prng_next_u64("child_priority");
}
let child_tid = Pid::from_raw(res as i32);
let child_dettid = DetTid::from_raw(child_tid.into()); trace!(
"[detcore] dtid {} cloned, continuing parent + register new thread.",
child_dettid
);
if !parent_blocks_for_child && !backend_uninstrumented_thread {
create_child_thread(guest, child_dettid, ctid, Some(flags), exit_signal, None).await;
}
{
let parent_pedigree = &mut guest.thread_state_mut().pedigree;
let child_pedigree = parent_pedigree.fork_mut();
debug!(
"[dtid {}] after creating child thread (tid {}, pedigree {}) parents pedigree becomes {}",
parent_dettid, child_dettid, child_pedigree, parent_pedigree,
);
}
Ok(child_dettid.as_raw() as i64)
}
pub async fn handle_set_tid_address<G: Guest<Self>>(
&self,
guest: &mut G,
call: syscalls::SetTidAddress,
) -> Result<i64, Error> {
let address = call.tidptr().map_or(0, |pointer| pointer.as_raw());
let result = self
.record_or_replay(guest, Syscall::SetTidAddress(call))
.await?;
if guest.config().sequentialize_threads {
set_child_tid_address(guest, address).await;
}
trace!(
"[detcore, dtid {}] child TID clear address registered: {address:#x}",
guest.thread_state().dettid,
);
Ok(result)
}
pub async fn handle_set_robust_list<G: Guest<Self>>(
&self,
guest: &mut G,
call: syscalls::SetRobustList,
) -> Result<i64, Error> {
let head = call.head().map(AddrMut::as_raw);
let len = call.len();
let res = self
.record_or_replay(guest, Syscall::SetRobustList(call))
.await?;
let recorded = match head {
None => None,
Some(_) if len != robust_list::ROBUST_LIST_HEAD_LEN => None,
Some(addr) => Some(addr),
};
guest.thread_state_mut().record_robust_list_head(recorded);
trace!(
"[detcore, dtid {}] robust-list head registered: {:?}",
guest.thread_state().dettid,
recorded,
);
Ok(res)
}
async fn run_robust_list_owner_death<G: Guest<Self>>(&self, guest: &mut G) {
if !self.cfg.backend_runs_exit_robust_list
|| !self.cfg.sequentialize_threads
|| self.cfg.debug_futex_mode != BlockingMode::Precise
{
return;
}
let Some(head) = guest.thread_state().robust_list_head else {
return;
};
let dettid = guest.thread_state().dettid;
let owner_tid = dettid.as_raw() as u32;
let mut effects = GuestRobustEffects::<'_, G, T> {
guest,
dettid,
defer_owner_death_to_backend: true,
staged_wakes: None,
tool: PhantomData,
};
let outcome = robust_list::exit_robust_list(&mut effects, head, owner_tid).await;
if outcome.head_unreadable {
trace!(
"[detcore, dtid {}] unreadable robust-list head at {:#x}; no owner-death wakeups",
dettid, head,
);
} else if outcome.aborted || outcome.next_faulted || outcome.truncated {
trace!(
"[detcore, dtid {}] robust-list walk stopped early after {} entr(ies): {:?}",
dettid, outcome.entries_visited, outcome,
);
}
}
pub(crate) async fn stage_thread_group_robust_list_wakes<G: Guest<Self>>(
&self,
guest: &mut G,
reason: RobustListExit,
) {
if !self.cfg.backend_runs_exit_robust_list
|| !self.cfg.sequentialize_threads
|| self.cfg.debug_futex_mode != BlockingMode::Precise
{
return;
}
let heads = guest.thread_state().robust_list_heads();
let mut staged = Vec::with_capacity(heads.len());
for (owner, head) in heads {
let mut wakes = Vec::new();
let mut effects = GuestRobustEffects::<'_, G, T> {
guest,
dettid: owner,
defer_owner_death_to_backend: true,
staged_wakes: Some(&mut wakes),
tool: PhantomData,
};
let outcome =
robust_list::exit_robust_list(&mut effects, head, owner.as_raw() as u32).await;
if outcome.head_unreadable || outcome.aborted {
trace!(
"[detcore, dtid {}] could not stage complete robust-list owner-death effects: {:?}",
owner, outcome,
);
}
staged.push((owner, wakes));
}
guest.thread_state().stage_robust_list_wakes(reason, staged);
}
pub async fn handle_exit<G: Guest<Self>>(
&self,
guest: &mut G,
call: syscalls::Exit,
) -> Result<i64, Error> {
let request = guest.thread_state().mk_request(
ResourceID::Exit {
group: false,
process: guest.thread_state().detpid.expect("detpid unset"),
mm: guest.thread_state().mm_id,
},
Permission::RW,
);
resource_request(guest, request).await;
self.run_robust_list_owner_death(guest).await;
guest.tail_inject(call).await
}
pub async fn handle_exit_group<G: Guest<Self>>(
&self,
guest: &mut G,
call: syscalls::ExitGroup,
) -> Result<i64, Error> {
let request = guest.thread_state().mk_request(
ResourceID::Exit {
group: true,
process: guest.thread_state().detpid.expect("detpid unset"),
mm: guest.thread_state().mm_id,
},
Permission::RW,
);
resource_request(guest, request).await;
self.stage_thread_group_robust_list_wakes(guest, RobustListExit::ExitGroup)
.await;
guest.tail_inject(call).await
}
pub async fn handle_futex<G: Guest<Self>>(
&self,
guest: &mut G,
call: syscalls::Futex,
) -> Result<i64, Error> {
let dettid = guest.thread_state().dettid;
let ptr = match call.uaddr() {
None => {
return Ok(guest.inject(call).await?);
}
Some(x) => x,
};
let init_val = guest.memory().read_value(ptr)?;
trace!(
"[detcore, dtid {}] futex op with memory address containing value {}",
&dettid, init_val
);
if !self.cfg.sequentialize_threads {
Ok(guest.inject(call).await?)
} else {
match self.cfg.debug_futex_mode {
BlockingMode::Precise => self.handle_futex_blocking(guest, call, init_val).await,
BlockingMode::Polling => self.handle_futex_polling(guest, call, init_val).await,
BlockingMode::External => self.record_or_replay_blocking(guest, call.into()).await,
}
}
}
pub async fn handle_futex_blocking<G: Guest<Self>>(
&self,
guest: &mut G,
call: syscalls::Futex,
init_val: i32,
) -> Result<i64, Error> {
let ptr = call.uaddr().unwrap();
let futexid = guest.thread_state().futex_id(
AddrMut::as_raw(ptr),
call.futex_op() & libc::FUTEX_PRIVATE_FLAG != 0,
);
let futex_op = call.futex_op() & libc::FUTEX_CMD_MASK;
let bitset = match futex_op {
libc::FUTEX_WAKE_BITSET | libc::FUTEX_WAIT_BITSET => call.val3() as u32,
_ => u32::MAX,
};
if bitset == 0 {
return Err(Error::Errno(Errno::EINVAL));
}
let dettid = guest.thread_state().dettid;
match futex_op {
libc::FUTEX_WAKE | libc::FUTEX_WAKE_BITSET => {
let num = match futex_action(
guest,
FutexAction::WakeRequest(call.val()),
&futexid,
init_val,
bitset,
)
.await
.expect("futex wake must return value")
{
SchedValue::Value(num) => num,
SchedValue::TimeOut => panic!("impossible, futex wake doesn't have a timeout"),
};
match guest.memory().read_value(ptr) {
Ok(observed) => trace!(
"[detcore, dtid {}] emulated futex wake committed, memory value is {}, expected {}",
&dettid,
observed,
call.val(),
),
Err(error) => trace!(
"[detcore, dtid {}] skipped post-wake futex memory diagnostic: {}",
&dettid, error,
),
}
let _ = futex_action(
guest,
FutexAction::WakeFinished(0),
&futexid,
init_val,
bitset,
)
.await;
Ok(num as i64)
}
libc::FUTEX_WAIT | libc::FUTEX_WAIT_BITSET => {
if init_val != call.val() {
info!(
"[detcore, dtid {}] Futex wait running immediately because it will fizzle ({} != {}).",
&dettid,
init_val,
call.val()
);
Err(Error::Errno(Errno::EAGAIN))
} else {
let maybe_timeout_lt = self
.futex_timeout_deadline(guest, call.futex_op(), call.timeout())
.await?;
let ans = futex_action(
guest,
FutexAction::WaitRequest(maybe_timeout_lt),
&futexid,
init_val,
bitset,
)
.await;
let res = if ans != Some(SchedValue::TimeOut) {
let expected = call.val();
match guest.memory().read_value(ptr) {
Ok(observed) => {
trace!(
"[detcore, dtid {}] after (emulated) futex wait, memory value is {}, expected {}",
&dettid, observed, expected,
);
if expected == observed {
debug!(
"WARNING: fishy that the futex value did not change before wakeup. Weird application-level protocol.\n"
);
}
}
Err(error) => trace!(
"[detcore, dtid {}] skipped post-wait futex memory diagnostic: {}",
&dettid, error,
),
}
Ok(0)
} else {
trace!("[detcore, dtid {}] futex wait timed out", &dettid);
Err(Error::Errno(Errno::ETIMEDOUT))
};
futex_action(guest, FutexAction::WaitFinished, &futexid, init_val, bitset)
.await;
res
}
}
libc::FUTEX_FD => {
panic!("[detcore] refusing to execute FUTEX_FD, which was removed in Linux 2.6.26.")
}
other => {
panic!("[detcore] futex op not handled yet: {}", other);
}
}
}
pub async fn handle_futex_polling<G: Guest<Self>>(
&self,
guest: &mut G,
call: syscalls::Futex,
init_val: i32,
) -> Result<i64, Error> {
fn make_futex_wake_request(dettid: DetTid) -> Resources {
let mut rsrc = Resources::new(dettid);
rsrc.fyi("futex_wake");
rsrc
}
fn make_futex_wait_request(dettid: DetTid) -> Resources {
let mut rsrc = Resources::new(dettid);
rsrc.insert(ResourceID::InternalIOPolling, Permission::W);
rsrc.fyi("futex_wait");
rsrc
}
let dettid = guest.thread_state().dettid;
let futex_op = call.futex_op() & libc::FUTEX_CMD_MASK;
match futex_op {
libc::FUTEX_WAKE | libc::FUTEX_WAKE_BITSET => {
let rsrc = make_futex_wake_request(dettid);
resource_request(guest, rsrc.clone()).await; let res = guest.inject(call).await;
Ok(res?)
}
libc::FUTEX_WAIT | libc::FUTEX_WAIT_BITSET => {
if init_val != call.val() {
info!(
"[detcore, dtid {}] Futex wait running immediately because it will fizzle ({} != {}).",
dettid,
init_val,
call.val()
);
let res = guest.inject(call).await;
Ok(res?)
} else {
let rsrc = make_futex_wait_request(dettid);
let deadline = self
.futex_timeout_deadline(guest, call.futex_op(), call.timeout())
.await?;
let res =
retry_nonblocking_syscall_with_timeout(guest, call, rsrc, deadline).await?;
trace!(
"[detcore, dtid {}] after futex wait, memory value is {}",
&dettid,
guest.memory().read_value(call.uaddr().unwrap()).unwrap()
);
Ok(res)
}
}
libc::FUTEX_FD => {
panic!("[detcore] refusing to execute FUTEX_FD, which was removed in Linux 2.6.26.")
}
other => {
panic!("[detcore] futex op not handled yet: {}", other);
}
}
}
pub async fn handle_execveat<G: Guest<Self>>(
&self,
guest: &mut G,
call: syscalls::Execveat,
) -> Result<i64, Error> {
let (old_metadata, old_memory_metadata, table_is_shared, dettid, detpid, old_mm_id) = {
let thread_state = guest.thread_state();
(
Arc::clone(&thread_state.file_metadata),
Arc::clone(&thread_state.memory_metadata),
Arc::strong_count(&thread_state.file_metadata) > 1,
thread_state.dettid,
thread_state.detpid.expect("detpid unset"),
thread_state.mm_id,
)
};
let (new_metadata, closed_open_files, exec_fd_blocking) = {
let metadata = old_metadata.lock().unwrap();
let new_metadata = metadata.for_exec(dettid);
(
new_metadata.clone(),
metadata.open_files_closed_on_exec(table_is_shared),
new_metadata.exec_blocking_overrides(),
)
};
let preserve_exec_fd_status = guest.thread_state().discover_live_file_metadata;
prepare_exec(
guest,
old_mm_id,
if preserve_exec_fd_status {
exec_fd_blocking
} else {
Default::default()
},
)
.await;
let mut released_ports = Vec::new();
for open_file_id in closed_open_files {
if let Some(port) = self.release_port_for_open_file(guest, open_file_id).await {
released_ports.push((open_file_id, port));
}
}
let old_robust_list_head;
{
let thread_state = guest.thread_state_mut();
thread_state.file_metadata = Arc::new(Mutex::new(new_metadata));
thread_state.memory_metadata = Arc::new(Mutex::new(MemoryMetadata::new()));
thread_state.mm_id = old_mm_id.for_exec(detpid);
old_robust_list_head = thread_state.take_robust_list_for_exec();
}
let errno = self.record_or_replay(guest, call).await.unwrap_err();
{
let thread_state = guest.thread_state_mut();
thread_state.file_metadata = old_metadata;
thread_state.memory_metadata = old_memory_metadata;
thread_state.mm_id = old_mm_id;
thread_state.restore_robust_list_after_failed_exec(old_robust_list_head);
}
cancel_exec(guest).await;
for (open_file_id, port) in released_ports {
self.restore_port_for_open_file(guest, open_file_id, port)
.await;
}
if self.cfg.backend_is_kvm && dettid != detpid && errno == Errno::ENOSYS {
tracing::error!(
"[detcore, dtid {dettid}] KVM nonleader exec is unsupported; \
the replacement image did not run"
);
if !self.cfg.panic_on_unsupported_syscalls {
crate::tool_global::report_unsupported_syscall(guest, call.number()).await;
}
return self
.refuse_unserviceable_operation(guest, call.number(), errno)
.await;
}
Err(errno.into())
}
pub async fn handle_sched_yield<G: Guest<Self>>(
&self,
guest: &mut G,
call: syscalls::SchedYield,
) -> Result<i64, Error> {
if self.cfg.sequentialize_threads {
if self.cfg.chaos && self.cfg.max_timeslice.is_none() {
let change_time = guest.thread_state().thread_logical_time.as_nanos();
let request = Self::random_priority_changepoint_request(guest, change_time);
resource_request(guest, request).await;
} else if !self.cfg.chaos && self.cfg.replay_preemptions_from.is_some() {
if self.cfg.max_timeslice.is_some() {
guest
.thread_state_mut()
.reset_timeslice_for_explicit_yield();
}
let request = Self::sched_yield_request(guest);
resource_request(guest, request).await;
} else if self.cfg.chaos || self.cfg.replay_schedule_from.is_some() {
let request = Self::yield_request(guest);
resource_request(guest, request).await;
} else {
self.end_timeslice_for_sched_yield(guest).await;
}
trace!("sched_yield yielded to the scheduler; NOT performing actual syscall");
Ok(0)
} else {
Ok(self.record_or_replay(guest, call).await?)
}
}
pub async fn handle_wait4<G: Guest<Self>>(
&self,
guest: &mut G,
call: syscalls::Wait4,
) -> Result<i64, Error> {
let dettid = guest.thread_state().dettid;
let mut rsrc = Resources::new(dettid);
rsrc.insert(ResourceID::InternalIOPolling, Permission::W);
rsrc.fyi("wait4");
validate_wait4_arguments(call.pid(), call.options())?;
let parent = guest.thread_state().detpid.expect("detpid unset");
let selector = if call.options().intersects(
WaitPidFlag::WUNTRACED
| WaitPidFlag::WCONTINUED
| WaitPidFlag::__WCLONE
| WaitPidFlag::__WALL,
) {
None
} else {
match call.pid() {
pid if pid > 0 => Some(ChildWaitSelector::Exact(DetPid::from_raw(pid))),
-1 => Some(ChildWaitSelector::Any),
0 => process_group(guest, parent)
.await
.map(ChildWaitSelector::ProcessGroup),
pid if pid < -1 => Some(ChildWaitSelector::ProcessGroup(DetPid::from_raw(-pid))),
_ => unreachable!(),
}
};
let spec = selector
.map(|selector| terminal_child_wait_spec(selector, dettid, call.options().bits()));
let complete_lineage = guest.config().backend_tracks_process_children;
let managed_spec = if let Some(spec) = spec {
let (_, has_child) = ready_child_wait(guest, spec).await;
(has_child || complete_lineage).then_some(spec)
} else {
None
};
let value = if call.options().contains(WaitPidFlag::WNOHANG) {
resource_request(guest, rsrc.clone()).await;
info!(
"[dtid {}] Executing non-blocking wait4 in one shot.",
dettid
);
if let Some(spec) = spec {
'select_child: loop {
let (ready, has_child) = ready_child_wait(guest, spec).await;
let Some(child) = ready else {
break if has_child {
0
} else if complete_lineage {
return Err(Errno::ECHILD.into());
} else {
guest.inject_with_retry(call).await?
};
};
let _ = await_exact_child_physical_exit(guest, child).await;
let exact_call = call.with_pid(child.as_raw());
loop {
match guest.inject_with_retry(exact_call).await {
Ok(value) if value != 0 => break 'select_child value,
Ok(_) => yield_once().await,
Err(Errno::ECHILD) => {
let _ = consume_child_wait(guest, child).await;
if child_wait_can_retry_after_stale(spec) {
continue 'select_child;
}
return Err(Errno::ECHILD.into());
}
Err(errno) => return Err(errno.into()),
}
}
}
} else {
guest.inject_with_retry(call).await?
}
} else if let Some(spec) = managed_spec {
{
let blocked_mask = blocked_signal_mask();
let mut stack = guest.stack().await;
let blocked_mask_addr = stack.push(blocked_mask);
let old_mask_addr = stack.reserve::<KernelSigset>();
let action_addr = stack.reserve::<KernelSigaction>();
let _mask_guard = stack.commit()?;
let guest_signal_mask =
block_signals_for_disposition(guest, blocked_mask_addr, old_mask_addr).await?;
let inspect_signal_action = guest
.config()
.backend_requires_thread_directed_process_signals;
let poll_call = call.with_options(call.options() | WaitPidFlag::WNOHANG);
let mut pending_signal = None;
let result: Result<i64, Error> = loop {
let status = wait_for_child_lifecycle(guest, spec).await;
if pending_signal.is_none() {
pending_signal = wait_signal_disposition(
guest,
status,
&guest_signal_mask,
action_addr,
inspect_signal_action,
)
.await?;
}
let (ready, has_child) = ready_child_wait(guest, spec).await;
if let Some(child) = ready {
let _ = await_exact_child_physical_exit(guest, child).await;
match guest.inject_with_retry(call.with_pid(child.as_raw())).await {
Ok(value) => break Ok(value),
Err(Errno::ECHILD) => {
let _ = consume_child_wait(guest, child).await;
if child_wait_can_retry_after_stale(spec) {
let (next_ready, _) = ready_child_wait(guest, spec).await;
if stale_any_wait_must_interrupt(
pending_signal.is_some(),
next_ready,
) {
break interrupted_child_wait_result(
guest,
call,
pending_signal.expect("signal checked above"),
)
.await;
}
continue;
}
break Err(Errno::ECHILD.into());
}
Err(errno) => break Err(errno.into()),
}
}
if !has_child {
break Err(Errno::ECHILD.into());
}
match guest.inject(poll_call).await {
Ok(value) => {
if value > 0 {
break Ok(value);
}
if let Some(disposition) = pending_signal {
break interrupted_child_wait_result(guest, call, disposition)
.await;
}
}
Err(errno) => break Err(errno.into()),
}
};
restore_signals_after_disposition(guest, old_mask_addr).await?;
result?
}
} else {
retry_nonblocking_syscall(guest, call, rsrc, None).await?
};
let consumed_termination = if value <= 0 {
false
} else if let Some(status) = call.wstatus() {
wait_status_is_termination(guest.memory().read_value(status)?)
} else {
guest
.thread_state()
.has_exited_child_process_cpu_time(DetPid::from_raw(value as i32))
};
if consumed_termination {
guest
.thread_state_mut()
.reap_child_process_cpu_time(DetPid::from_raw(value as i32));
let _ = consume_child_wait(guest, DetPid::from_raw(value as i32)).await;
}
if value > 0
&& let Some(rusage) = call.rusage()
{
let usage: libc::rusage = unsafe { std::mem::zeroed() };
guest.memory().write_value(rusage, &usage)?;
}
Ok(value)
}
pub async fn handle_waitid<G: Guest<Self>>(
&self,
guest: &mut G,
mut call: syscalls::Waitid,
) -> Result<i64, Error> {
let dettid = guest.thread_state().dettid;
let mut rsrc = Resources::new(dettid);
rsrc.insert(ResourceID::InternalIOPolling, Permission::W);
rsrc.fyi("waitid");
let event_options = libc::WEXITED | libc::WSTOPPED | libc::WCONTINUED;
let allowed_options = event_options
| libc::WNOHANG
| libc::WNOWAIT
| libc::__WNOTHREAD
| libc::__WALL
| libc::__WCLONE;
if call.options() & event_options == 0 || call.options() & !allowed_options != 0 {
return Err(Errno::EINVAL.into());
}
if call.info().is_none() {
return Err(Errno::EFAULT.into());
}
let terminal_events_only = call.options() & libc::WEXITED != 0
&& call.options() & (libc::WSTOPPED | libc::WCONTINUED | libc::__WCLONE | libc::__WALL)
== 0;
if !terminal_events_only && call.which() == libc::P_PGID as i32 && call.pid() == 0 {
call = call.with_pid(snapshot_process_group(guest.pid())?);
}
let pidfd_nonblocking =
if call.which() == libc::P_PIDFD as i32 && call.options() & libc::WNOHANG == 0 {
resource_request(guest, rsrc.clone()).await;
guest_fd_status_flags(guest.pid(), call.pid())? & libc::O_NONBLOCK != 0
} else {
false
};
if call.which() == libc::P_PIDFD as i32
&& call.options() & libc::WNOHANG == 0
&& !pidfd_nonblocking
{
return Err(Errno::EOPNOTSUPP.into());
}
let info = call.info().expect("waitid infop checked above");
let empty_info: libc::siginfo_t = unsafe { std::mem::zeroed() };
let terminal_spec = if terminal_events_only {
let selector = match call.which() {
which if which == libc::P_PID as i32 => {
Some(ChildWaitSelector::Exact(DetPid::from_raw(call.pid())))
}
which if which == libc::P_ALL as i32 => Some(ChildWaitSelector::Any),
which if which == libc::P_PGID as i32 => {
let group = if call.pid() == 0 {
process_group(guest, guest.thread_state().detpid.expect("detpid unset"))
.await
} else {
Some(DetPid::from_raw(call.pid()))
};
group.map(ChildWaitSelector::ProcessGroup)
}
_ => None,
};
selector.map(|selector| terminal_child_wait_spec(selector, dettid, call.options()))
} else {
None
};
let complete_lineage = guest.config().backend_tracks_process_children;
let managed_terminal_spec = if let Some(spec) = terminal_spec {
let (_, has_child) = ready_child_wait(guest, spec).await;
(has_child || complete_lineage).then_some(spec)
} else {
None
};
if call.options() & libc::WNOHANG != 0 || pidfd_nonblocking {
if !pidfd_nonblocking {
resource_request(guest, rsrc).await;
}
info!(
"[dtid {}] Executing non-blocking waitid in one shot.",
dettid
);
'select_child: loop {
let selected = if let Some(spec) = terminal_spec {
let (ready, has_child) = ready_child_wait(guest, spec).await;
if ready.is_none() && has_child {
guest.memory().write_value(info, &empty_info)?;
return Ok(0);
}
if ready.is_none() && complete_lineage {
return Err(Errno::ECHILD.into());
}
ready
} else {
None
};
if let Some(child) = selected {
let _ = await_exact_child_physical_exit(guest, child).await;
}
let effective_call = selected.map_or(call, |child| {
call.with_which(libc::P_PID as i32).with_pid(child.as_raw())
});
loop {
guest.memory().write_value(info, &empty_info)?;
let value = match guest.inject_with_retry(effective_call).await {
Ok(value) => value,
Err(Errno::ECHILD) if selected.is_some() => {
let child = selected.expect("selected child checked above");
let _ = consume_child_wait(guest, child).await;
if terminal_spec.is_some_and(child_wait_can_retry_after_stale) {
continue 'select_child;
}
return Err(Errno::ECHILD.into());
}
Err(errno) => return Err(errno.into()),
};
let info_value: libc::siginfo_t = guest.memory().read_value(info)?;
let child_pid = unsafe { info_value.si_pid() };
if child_pid == 0 && selected.is_some() {
yield_once().await;
continue;
}
let consumed = child_pid != 0
&& call.options() & libc::WNOWAIT == 0
&& waitid_code_is_termination(info_value.si_code);
let result = finish_waitid_result(guest, call, value, info_value)?;
if consumed {
let _ = consume_child_wait(guest, DetPid::from_raw(child_pid)).await;
}
return Ok(result);
}
}
}
{
let blocked_mask = blocked_signal_mask();
let mut stack = guest.stack().await;
let blocked_mask_addr = stack.push(blocked_mask);
let old_mask_addr = stack.reserve::<KernelSigset>();
let action_addr = stack.reserve::<KernelSigaction>();
let _mask_guard = stack.commit()?;
let guest_signal_mask =
block_signals_for_disposition(guest, blocked_mask_addr, old_mask_addr).await?;
let inspect_signal_action = guest
.config()
.backend_requires_thread_directed_process_signals;
let poll_call = call.with_options(call.options() | libc::WNOHANG);
let mut pending_signal = None;
let result: Result<i64, Error> = loop {
let managed_spec = managed_terminal_spec;
let status = if let Some(spec) = managed_spec {
wait_for_child_lifecycle(guest, spec).await
} else {
resource_request(guest, rsrc.clone()).await
};
if pending_signal.is_none() {
pending_signal = wait_signal_disposition(
guest,
status,
&guest_signal_mask,
action_addr,
inspect_signal_action,
)
.await?;
}
let (ready, has_child) = if let Some(spec) = managed_spec {
ready_child_wait(guest, spec).await
} else {
(None, true)
};
if let Some(child) = ready {
let _ = await_exact_child_physical_exit(guest, child).await;
if let Err(error) = guest.memory().write_value(info, &empty_info) {
break Err(error.into());
}
let exact_call = call.with_which(libc::P_PID as i32).with_pid(child.as_raw());
match guest.inject_with_retry(exact_call).await {
Ok(value) => {
let info_value = match guest.memory().read_value(info) {
Ok(value) => value,
Err(error) => break Err(error.into()),
};
break finish_waitid_result(guest, call, value, info_value);
}
Err(Errno::ECHILD) => {
let _ = consume_child_wait(guest, child).await;
if managed_spec.is_some_and(child_wait_can_retry_after_stale) {
let (next_ready, _) =
ready_child_wait(guest, managed_spec.expect("managed spec"))
.await;
if stale_any_wait_must_interrupt(
pending_signal.is_some(),
next_ready,
) {
break interrupted_child_wait_result(
guest,
call,
pending_signal.expect("signal checked above"),
)
.await;
}
continue;
}
break Err(Errno::ECHILD.into());
}
Err(errno) => break Err(errno.into()),
}
}
if managed_spec.is_some() && !has_child {
break Err(Errno::ECHILD.into());
}
if let Err(error) = guest.memory().write_value(info, &empty_info) {
break Err(error.into());
}
let result = guest.inject(poll_call).await;
match result {
Ok(value) => {
let info_value: libc::siginfo_t = match guest.memory().read_value(info) {
Ok(value) => value,
Err(error) => break Err(error.into()),
};
let child_pid = unsafe { info_value.si_pid() };
match exact_wait_poll_decision(
child_pid != 0,
pending_signal.is_some(),
None,
) {
ExactWaitPollDecision::ChildReady => {
break finish_waitid_result(guest, call, value, info_value);
}
ExactWaitPollDecision::Interrupted => {
break interrupted_child_wait_result(
guest,
call,
pending_signal.expect("signal checked above"),
)
.await;
}
ExactWaitPollDecision::Retry => {}
ExactWaitPollDecision::AwaitPhysicalExit
| ExactWaitPollDecision::ReapAfterLogicalExit => unreachable!(),
}
if managed_spec.is_some() {
if !has_child {
break Ok(value);
}
continue;
}
rsrc.poll_attempt += 1;
trace!(
"Retry #{} for waitid because no child state is ready",
rsrc.poll_attempt
);
record_retry_event(guest, poll_call).await;
}
Err(Errno::ERESTARTSYS) if pending_signal.is_some() => {
break Err(Errno::EINTR.into());
}
Err(errno) => break Err(errno.into()),
}
};
restore_signals_after_disposition(guest, old_mask_addr).await?;
if result.is_ok() && call.options() & libc::WNOWAIT == 0 {
let info_value: libc::siginfo_t = guest.memory().read_value(info)?;
let child_pid = unsafe { info_value.si_pid() };
if child_pid != 0 && waitid_code_is_termination(info_value.si_code) {
let _ = consume_child_wait(guest, DetPid::from_raw(child_pid)).await;
}
}
result
}
}
pub async fn handle_sched_setaffinity<G: Guest<Self>>(
&self,
guest: &mut G,
call: syscalls::SchedSetaffinity,
) -> Result<i64, Error> {
let size_bytes = call.len() as usize;
if size_bytes == 0 {
return Err(Errno::EINVAL.into());
}
let mask = call.mask().ok_or(Errno::EFAULT)?;
let mask: Addr<u8> = mask.cast();
let mut requested = [0u8; VIRTUAL_CPUSET_BYTES];
let bytes_to_read = size_bytes.min(VIRTUAL_CPUSET_BYTES);
guest
.memory()
.read_exact(mask, &mut requested[..bytes_to_read])?;
info!(
"Suppressing sched_setaffinity mask {:?}; affinity remains virtual CPU 0",
requested
);
Ok(0)
}
pub async fn handle_sched_getaffinity<G: Guest<Self>>(
&self,
guest: &mut G,
call: syscalls::SchedGetaffinity,
) -> Result<i64, Error> {
let size_bytes: usize = call.len() as usize;
if size_bytes < VIRTUAL_CPUSET_BYTES
|| !size_bytes.is_multiple_of(std::mem::size_of::<libc::c_ulong>())
{
return Err(Errno::EINVAL.into());
}
let mut cpu_set = [0u8; VIRTUAL_CPUSET_BYTES];
cpu_set[0] = 1;
info!(
"Suppressing sched_getaffinity and returning {}-byte virtualized result, {:?}",
VIRTUAL_CPUSET_BYTES, cpu_set
);
if let Some(mask) = call.mask() {
let mask: AddrMut<u8> = mask.cast();
guest.memory().write_exact(mask, &cpu_set)?;
Ok(VIRTUAL_CPUSET_BYTES as i64)
} else {
Err(Error::Errno(Errno::EFAULT))
}
}
pub async fn handle_sched_getparam<G: Guest<Self>>(
&self,
guest: &mut G,
call: syscalls::SchedGetparam,
) -> Result<i64, Error> {
if let Some(param) = call.param() {
let p = libc::sched_param { sched_priority: 0 };
guest.memory().write_value(param, &p)?;
}
Ok(0)
}
pub async fn handle_sched_rr_get_interval<G: Guest<Self>>(
&self,
guest: &mut G,
call: syscalls::SchedRrGetInterval,
) -> Result<i64, Error> {
if let Some(tp) = call.tp() {
let t = Timespec {
tv_sec: 0,
tv_nsec: 0,
};
guest.memory().write_value(tp, &t)?;
}
Ok(0)
}
pub async fn handle_sched_getattr<G: Guest<Self>>(
&self,
guest: &mut G,
call: syscalls::SchedGetattr,
) -> Result<i64, Error> {
if call.flags() != 0 {
return Err(Errno::EINVAL.into());
}
let attr_size = std::mem::size_of::<libc::sched_attr>();
if (call.size() as usize) < attr_size {
return Err(Errno::EINVAL.into());
}
let attr = call.attr().ok_or(Errno::EINVAL)?;
let mut sa: libc::sched_attr = unsafe { std::mem::zeroed() };
sa.size = attr_size as u32;
sa.sched_policy = libc::SCHED_OTHER as u32;
let bytes = unsafe {
std::slice::from_raw_parts(&sa as *const libc::sched_attr as *const u8, attr_size)
};
let dst: AddrMut<u8> = attr.cast();
guest.memory().write_exact(dst, bytes)?;
info!(
"Emulating sched_getattr(pid={}): fixed SCHED_OTHER, nice 0, priority 0",
call.pid()
);
Ok(0)
}
pub async fn handle_sched_setattr<G: Guest<Self>>(
&self,
guest: &mut G,
call: syscalls::SchedSetattr,
) -> Result<i64, Error> {
if call.flags() != 0 || call.pid() < 0 {
return Err(Errno::EINVAL.into());
}
let attr = call.attr().ok_or(Errno::EINVAL)?;
let declared: u32 = guest
.memory()
.read_value(attr.cast())
.map_err(|_| Errno::EFAULT)?;
fn refuse_too_big<S: MemoryAccess>(mut memory: S, attr: AddrMut<libc::c_void>) -> Error {
let _ = memory.write_value(attr.cast::<u32>(), &SCHED_ATTR_KERNEL_SIZE);
Errno::E2BIG.into()
}
let size = match sched_attr_effective_size(declared) {
Ok(size) => size,
Err(()) => {
let back = call.attr().ok_or(Errno::EINVAL)?;
return Err(refuse_too_big(guest.memory(), back));
}
};
if size > SCHED_ATTR_KERNEL_SIZE {
let base: AddrMut<u8> = call.attr().ok_or(Errno::EINVAL)?.cast();
match scan_tail_is_zeroed(
&guest.memory(),
base,
SCHED_ATTR_KERNEL_SIZE as usize,
(size - SCHED_ATTR_KERNEL_SIZE) as usize,
) {
TailVerdict::AllZero => {}
TailVerdict::NotZeroed => {
let back = call.attr().ok_or(Errno::EINVAL)?;
return Err(refuse_too_big(guest.memory(), back));
}
TailVerdict::Faulted => return Err(Errno::EFAULT.into()),
}
}
let copied = std::cmp::min(size, SCHED_ATTR_KERNEL_SIZE) as usize;
let mut raw = [0u8; SCHED_ATTR_KERNEL_SIZE as usize];
let base: AddrMut<u8> = call.attr().ok_or(Errno::EINVAL)?.cast();
guest
.memory()
.read_exact(base, &mut raw[..copied])
.map_err(|_| Errno::EFAULT)?;
let field32 = |offset: usize| -> u32 {
u32::from_ne_bytes(raw[offset..offset + 4].try_into().expect("4 bytes"))
};
let field64 = |offset: usize| -> u64 {
u64::from_ne_bytes(raw[offset..offset + 8].try_into().expect("8 bytes"))
};
let fields = SchedAttrFields {
policy: field32(SCHED_ATTR_OFF_POLICY),
sched_flags: field64(SCHED_ATTR_OFF_FLAGS),
priority: field32(SCHED_ATTR_OFF_PRIORITY),
runtime: field64(SCHED_ATTR_OFF_RUNTIME),
deadline: field64(SCHED_ATTR_OFF_DEADLINE),
period: field64(SCHED_ATTR_OFF_PERIOD),
};
validate_sched_attr_before_lookup(size, &fields)?;
if call.pid() != 0 && !thread_is_live(guest, DetTid::from_raw(call.pid())).await {
return Err(Errno::ESRCH.into());
}
validate_sched_attr_after_lookup(&fields)?;
info!(
"Suppressing sched_setattr(pid={}, flags={}); Linux scheduler attributes are virtual",
call.pid(),
call.flags()
);
Ok(0)
}
pub async fn handle_ioprio_set<G: Guest<Self>>(
&self,
_guest: &mut G,
call: syscalls::IoprioSet,
) -> Result<i64, Error> {
info!(
"Suppressing ioprio_set(which={}, who={}, priority={}); I/O priority is virtual",
call.which(),
call.who(),
call.priority()
);
Ok(0)
}
pub async fn handle_ioprio_get<G: Guest<Self>>(
&self,
_guest: &mut G,
call: syscalls::IoprioGet,
) -> Result<i64, Error> {
let priority = virtual_ioprio(call.which())?;
info!(
"Emulating ioprio_get(which={}, who={}): fixed priority {}",
call.which(),
call.who(),
priority
);
Ok(priority)
}
}
struct GuestRobustEffects<'a, G, T> {
guest: &'a mut G,
dettid: DetTid,
defer_owner_death_to_backend: bool,
staged_wakes: Option<&'a mut Vec<RobustListWake>>,
tool: PhantomData<T>,
}
impl<G, T> robust_list::RobustDeathEffects for GuestRobustEffects<'_, G, T>
where
G: Guest<Detcore<T>>,
T: RecordOrReplay,
{
fn read_u64(&mut self, address: usize) -> Option<u64> {
let at = Addr::<u64>::from_raw(address)?;
self.guest.memory().read_value::<_, u64>(at).ok()
}
fn read_u32(&mut self, address: usize) -> Option<u32> {
let at = Addr::<u32>::from_raw(address)?;
self.guest.memory().read_value::<_, u32>(at).ok()
}
fn compare_and_swap(
&mut self,
address: usize,
expected: u32,
desired: u32,
) -> robust_list::FutexCasOutcome {
use robust_list::FutexCasOutcome;
let (Some(read_at), Some(write_at)) = (
Addr::<u32>::from_raw(address),
AddrMut::<u32>::from_raw(address),
) else {
return FutexCasOutcome::Faulted;
};
let observed = match self.guest.memory().read_value::<_, u32>(read_at) {
Ok(value) => value,
Err(_) => return FutexCasOutcome::Faulted,
};
if observed != expected {
return FutexCasOutcome::Changed(observed);
}
if self.defer_owner_death_to_backend {
debug!(
"[detcore, dtid {}] robust-list owner death: leaving futex word {:#x} for backend exit cleanup",
self.dettid, address,
);
return FutexCasOutcome::Deferred;
}
if self.guest.memory().write_value(write_at, &desired).is_err() {
return FutexCasOutcome::Faulted;
}
debug!(
"[detcore, dtid {}] robust-list owner death: futex word {:#x} {:#x} -> {:#x}",
self.dettid, address, expected, desired,
);
FutexCasOutcome::Stored
}
async fn wake_one(&mut self, address: usize, observed: u32) {
let futexid = self.guest.thread_state().futex_id(address, false);
if let Some(wakes) = self.staged_wakes.as_mut() {
wakes.push(RobustListWake { futex: futexid });
debug!(
"[detcore, dtid {}] staged robust-list owner-death wake until physical exit",
self.dettid,
);
return;
}
let woken = match futex_action(
self.guest,
FutexAction::WakeRequest(1),
&futexid,
observed as i32,
u32::MAX,
)
.await
{
Some(SchedValue::Value(count)) => count,
Some(SchedValue::TimeOut) | None => 0,
};
info!(
"[detcore, dtid {}] robust-list owner death woke {} waiter(s) on futex {:?}",
self.dettid, woken, futexid,
);
let _ = futex_action(
self.guest,
FutexAction::WakeFinished(0),
&futexid,
observed as i32,
u32::MAX,
)
.await;
}
}
#[cfg(test)]
mod tests {
use std::io::Read;
use std::io::Seek;
use std::os::fd::AsRawFd;
use reverie::GlobalRPC;
use reverie::GlobalTool;
use reverie::Tid;
use reverie::Tool;
use super::*;
use crate::config::Config;
use crate::tool_global::GlobalRequest;
use crate::tool_global::GlobalState;
use crate::types::MmId;
struct FailedExecStack;
struct FailedExecStackGuard;
impl Drop for FailedExecStackGuard {
fn drop(&mut self) {}
}
impl reverie::Stack for FailedExecStack {
type StackGuard = FailedExecStackGuard;
fn size(&self) -> usize {
panic!("failed exec must not use the guest stack")
}
fn capacity(&self) -> usize {
panic!("failed exec must not use the guest stack")
}
fn push<'stack, T>(&mut self, _: T) -> Addr<'stack, T> {
panic!("failed exec must not use the guest stack")
}
fn reserve<'stack, T>(&mut self) -> AddrMut<'stack, T> {
panic!("failed exec must not use the guest stack")
}
fn commit(self) -> Result<Self::StackGuard, Errno> {
panic!("failed exec must not use the guest stack")
}
}
struct FailedExecGuest<'a> {
config: &'a Config,
global: &'a GlobalState,
thread: crate::ThreadState<()>,
sender: Tid,
process: Tid,
old_mm: MmId,
errno: Errno,
injections: usize,
requests: Mutex<Vec<GlobalRequest>>,
}
#[reverie::tool]
impl GlobalRPC<GlobalState> for FailedExecGuest<'_> {
async fn send_rpc(
&self,
message: <GlobalState as GlobalTool>::Request,
) -> <GlobalState as GlobalTool>::Response {
assert_eq!(message.1, self.old_mm, "RPC follows restored exec identity");
self.requests.lock().unwrap().push(message.2.clone());
self.global.receive_rpc(self.sender, message).await
}
fn config(&self) -> &Config {
self.config
}
}
#[reverie::tool]
impl Guest<Detcore> for FailedExecGuest<'_> {
type Memory = reverie::syscalls::LocalMemory;
type Stack = FailedExecStack;
fn tid(&self) -> Tid {
self.sender
}
fn pid(&self) -> Tid {
self.process
}
fn ppid(&self) -> Option<Tid> {
None
}
fn memory(&self) -> Self::Memory {
panic!("failed exec rollback must not read guest memory")
}
fn thread_state(&self) -> &crate::ThreadState<()> {
&self.thread
}
fn thread_state_mut(&mut self) -> &mut crate::ThreadState<()> {
&mut self.thread
}
async fn regs(&mut self) -> libc::user_regs_struct {
panic!("failed exec rollback must not read registers")
}
async fn stack(&mut self) -> Self::Stack {
panic!("failed exec rollback must not use a guest stack")
}
async fn daemonize(&mut self) {
panic!("failed exec rollback must not daemonize")
}
async fn inject<S: SyscallInfo>(&mut self, call: S) -> Result<i64, Errno> {
assert_eq!(call.number(), syscalls::Sysno::execveat);
assert_eq!(
self.thread.mm_id,
self.old_mm.for_exec(self.thread.detpid.unwrap())
);
assert_eq!(self.thread.robust_list_head, None);
self.injections += 1;
Err(self.errno)
}
async fn tail_inject<S: SyscallInfo>(&mut self, _: S) -> reverie::Never {
panic!("failed exec must not retire a live guest")
}
fn set_timer(&mut self, _: reverie::TimerSchedule) -> Result<(), Error> {
panic!("failed exec rollback must not replace the timer")
}
fn set_timer_precise(&mut self, _: reverie::TimerSchedule) -> Result<(), Error> {
panic!("failed exec rollback must not replace the timer")
}
fn read_clock(&mut self) -> Result<u64, Error> {
panic!("failed exec rollback must not sample the clock")
}
}
#[tokio::test]
async fn kvm_nonleader_exec_refusal_preserves_failed_exec_rollback_and_policy() {
let process = DetPid::from_raw(17);
let worker = DetTid::from_raw(18);
for (backend_is_kvm, caller, errno, fail_closed, refused, reported) in [
(true, worker, Errno::ENOSYS, true, true, false),
(true, worker, Errno::ENOSYS, false, false, true),
(true, worker, Errno::ENOENT, true, false, false),
(true, worker, Errno::ENOEXEC, true, false, false),
(true, worker, Errno::EFAULT, true, false, false),
(true, worker, Errno::EACCES, true, false, false),
(true, worker, Errno::EOPNOTSUPP, true, false, false),
(true, process, Errno::ENOSYS, true, false, false),
(false, worker, Errno::ENOSYS, true, false, false),
] {
let mut report = tempfile::tempfile().unwrap();
let config = Config {
backend_is_kvm,
sequentialize_threads: false,
panic_on_unsupported_syscalls: fail_closed,
exit_on_unsupported_syscall: true,
shutdown_on_unsupported_syscall: false,
unsupported_syscall_report_fd: Some(report.as_raw_fd()),
..Config::default()
};
let global = GlobalState::init_global_state(&config).await;
let tool = Detcore::new(Tid::from_raw(process.as_raw()), &config);
let mut thread = crate::ThreadState::new(caller, &config, ());
thread.detpid = Some(process);
thread.mm_id = MmId::initial(process);
thread.record_robust_list_head(Some(0x12340));
thread.thread_logical_time.add_syscall_with_cost(123);
let old_time = thread.thread_logical_time.as_nanos();
let old_files = Arc::clone(&thread.file_metadata);
let old_memory = Arc::clone(&thread.memory_metadata);
let old_mm = thread.mm_id;
let mut guest = FailedExecGuest {
config: &config,
global: &global,
thread,
sender: Tid::from_raw(caller.as_raw()),
process: Tid::from_raw(process.as_raw()),
old_mm,
errno,
injections: 0,
requests: Mutex::new(Vec::new()),
};
let result = tool
.handle_execveat(&mut guest, syscalls::Execveat::new())
.await;
match result {
Err(Error::Tool(error)) if refused => assert_eq!(
error
.downcast_ref::<crate::UnsupportedSyscallError>()
.unwrap()
.0,
syscalls::Sysno::execveat
),
Err(Error::Errno(actual)) if !refused => assert_eq!(actual, errno),
result => panic!("wrong failed-exec policy: {result:?}"),
}
assert_eq!(guest.injections, 1, "backend preflight must run first");
assert_eq!(guest.thread.dettid, caller);
assert_eq!(guest.thread.detpid, Some(process));
assert_eq!(guest.thread.mm_id, old_mm);
assert_eq!(guest.thread.robust_list_head, Some(0x12340));
assert_eq!(guest.thread.thread_logical_time.as_nanos(), old_time);
assert!(Arc::ptr_eq(&guest.thread.file_metadata, &old_files));
assert!(Arc::ptr_eq(&guest.thread.memory_metadata, &old_memory));
let requests = guest.requests.lock().unwrap();
assert!(matches!(requests[0], GlobalRequest::PrepareExec(..)));
assert!(matches!(requests[1], GlobalRequest::CancelExec(..)));
assert_eq!(requests.len(), if reported { 3 } else { 2 });
if reported {
assert!(
matches!(&requests[2], GlobalRequest::ReportUnsupportedSyscall(name) if name == "execveat")
);
}
report.rewind().unwrap();
let mut aggregate = String::new();
report.read_to_string(&mut aggregate).unwrap();
assert_eq!(aggregate, if reported { "execveat\n" } else { "" });
}
}
#[test]
fn kernel_blocked_mask_preserves_libc_signal_membership() {
let mask = blocked_signal_mask();
let mut libc_mask: libc::sigset_t = unsafe { std::mem::zeroed() };
unsafe {
libc::sigfillset(&mut libc_mask);
libc::sigdelset(&mut libc_mask, reverie::PERF_EVENT_SIGNAL as i32);
}
for raw_signal in 1..=KernelSigset::BITS as i32 {
assert_eq!(
signal_is_blocked(&mask, SigWrapper(raw_signal)),
unsafe { libc::sigismember(&libc_mask, raw_signal) == 1 },
"signal {raw_signal} membership changed while converting to the kernel ABI"
);
}
}
#[test]
fn clone_without_child_cleartid_does_not_register_the_pointer() {
assert_eq!(
child_tid_clear_address(CloneFlags::CLONE_CHILD_SETTID, 0x1234),
0
);
}
#[test]
fn clone_with_child_cleartid_registers_the_pointer() {
assert_eq!(
child_tid_clear_address(CloneFlags::CLONE_CHILD_CLEARTID, 0x1234),
0x1234
);
}
#[test]
fn linux_default_dispositions_that_do_not_interrupt_child_waits() {
for signal in [libc::SIGCHLD, libc::SIGCONT, libc::SIGURG, libc::SIGWINCH] {
assert!(signal_default_disposition_does_not_interrupt_child_wait(
SigWrapper(signal)
));
}
for signal in [libc::SIGALRM, libc::SIGSTOP, libc::SIGUSR1] {
assert!(!signal_default_disposition_does_not_interrupt_child_wait(
SigWrapper(signal)
));
}
}
#[test]
fn uncatchable_signals_do_not_require_a_sigaction_query() {
assert!(signal_has_uncatchable_default_disposition(SigWrapper(
libc::SIGKILL
)));
assert!(signal_has_uncatchable_default_disposition(SigWrapper(
libc::SIGSTOP
)));
assert!(!signal_has_uncatchable_default_disposition(SigWrapper(
libc::SIGUSR1
)));
}
#[test]
fn waitid_ready_child_wins_when_scheduler_also_reports_a_signal() {
assert_eq!(
exact_wait_poll_decision(true, true, Some(ExactChildWaitState::Running)),
ExactWaitPollDecision::ChildReady
);
assert_eq!(
exact_wait_poll_decision(false, true, Some(ExactChildWaitState::LogicallyExited)),
ExactWaitPollDecision::ReapAfterLogicalExit
);
assert_eq!(
exact_wait_poll_decision(false, true, Some(ExactChildWaitState::PhysicalExitPending)),
ExactWaitPollDecision::AwaitPhysicalExit
);
assert_eq!(
exact_wait_poll_decision(false, true, Some(ExactChildWaitState::Running)),
ExactWaitPollDecision::Interrupted
);
assert_eq!(
exact_wait_poll_decision(false, false, Some(ExactChildWaitState::Running)),
ExactWaitPollDecision::Retry
);
}
#[test]
fn wait4_argument_validation_follows_linux_precedence() {
let valid_bits = [0, 1, 3, 29, 30, 31];
for bit in valid_bits {
let options = WaitPidFlag::from_bits_retain((1_u32 << bit) as libc::c_int);
assert_eq!(validate_wait4_arguments(-1, options), Ok(()), "bit {bit}");
}
for bit in (0..u32::BITS).filter(|bit| !valid_bits.contains(bit)) {
let options = WaitPidFlag::from_bits_retain((1_u32 << bit) as libc::c_int);
assert_eq!(
validate_wait4_arguments(-1, options),
Err(Errno::EINVAL),
"bit {bit}"
);
}
for pid in [libc::pid_t::MIN + 1, -2, 0, 1, libc::pid_t::MAX] {
assert_eq!(
validate_wait4_arguments(pid, WaitPidFlag::empty()),
Ok(()),
"pid {pid}"
);
}
assert_eq!(
validate_wait4_arguments(libc::pid_t::MIN, WaitPidFlag::empty()),
Err(Errno::ESRCH)
);
assert_eq!(
validate_wait4_arguments(libc::pid_t::MIN, WaitPidFlag::from_bits_retain(0x10)),
Err(Errno::EINVAL),
"invalid options must win over the INT_MIN selector"
);
}
#[test]
fn stale_any_child_preserves_interrupt_until_no_ready_child_remains() {
let next_child = DetPid::from_raw(200);
assert!(
!stale_any_wait_must_interrupt(true, Some(next_child)),
"another ready child must retain child-ready precedence"
);
assert!(
stale_any_wait_must_interrupt(true, None),
"a pending signal must interrupt before the wait parks again"
);
assert!(!stale_any_wait_must_interrupt(false, None));
}
#[test]
fn ioprio_query_reports_fixed_raw_and_effective_defaults() {
assert_eq!(virtual_ioprio(IOPRIO_WHO_PROCESS), Ok(0));
assert_eq!(
virtual_ioprio(IOPRIO_WHO_PGRP),
Ok(i64::from(IOPRIO_DEFAULT_EFFECTIVE))
);
assert_eq!(
virtual_ioprio(IOPRIO_WHO_USER),
Ok(i64::from(IOPRIO_DEFAULT_EFFECTIVE))
);
assert_eq!(virtual_ioprio(0), Err(Errno::EINVAL));
assert_eq!(virtual_ioprio(4), Err(Errno::EINVAL));
}
#[test]
fn waitid_siginfo_canonicalization_clears_only_cpu_accounting() {
let mut info: libc::siginfo_t = unsafe { std::mem::zeroed() };
info.si_signo = libc::SIGCHLD;
info.si_code = libc::CLD_EXITED;
let fields = unsafe {
&mut (*(std::ptr::addr_of_mut!(info)).cast::<WaitidSiginfoHead>())
.fields
.sigchld
};
fields.pid = 123;
fields.uid = 456;
fields.status = 7;
fields.utime = 8;
fields.stime = 9;
canonicalize_waitid_siginfo(&mut info);
assert_eq!(info.si_signo, libc::SIGCHLD);
assert_eq!(info.si_code, libc::CLD_EXITED);
assert_eq!(unsafe { info.si_pid() }, 123);
assert_eq!(unsafe { info.si_uid() }, 456);
assert_eq!(unsafe { info.si_status() }, 7);
assert_eq!(unsafe { info.si_utime() }, 0);
assert_eq!(unsafe { info.si_stime() }, 0);
}
#[test]
fn wait_status_rollup_only_accepts_process_termination() {
assert!(wait_status_is_termination(0));
assert!(wait_status_is_termination(libc::SIGTERM));
assert!(!wait_status_is_termination((libc::SIGSTOP << 8) | 0x7f));
assert!(!wait_status_is_termination(0xffff));
assert!(waitid_code_is_termination(libc::CLD_EXITED));
assert!(waitid_code_is_termination(libc::CLD_KILLED));
assert!(waitid_code_is_termination(libc::CLD_DUMPED));
assert!(!waitid_code_is_termination(libc::CLD_STOPPED));
assert!(!waitid_code_is_termination(libc::CLD_CONTINUED));
assert!(!waitid_code_is_termination(libc::CLD_TRAPPED));
}
#[test]
fn futex_timeout_units_and_modes_match_linux() {
let timeout = Timespec {
tv_sec: 2,
tv_nsec: 3,
};
assert_eq!(
parse_futex_timeout(libc::FUTEX_WAIT, timeout),
Ok(FutexTimeout::Relative(2_000_000_003))
);
assert_eq!(
parse_futex_timeout(libc::FUTEX_WAIT_BITSET, timeout),
Ok(FutexTimeout::Absolute(LogicalTime::from_nanos(
2_000_000_003
)))
);
assert_eq!(
parse_futex_timeout(libc::FUTEX_WAIT_BITSET | libc::FUTEX_PRIVATE_FLAG, timeout),
Ok(FutexTimeout::Absolute(LogicalTime::from_nanos(
2_000_000_003
)))
);
assert_eq!(
parse_futex_timeout(libc::FUTEX_WAIT | libc::FUTEX_PRIVATE_FLAG, timeout),
Ok(FutexTimeout::Relative(2_000_000_003))
);
}
#[test]
fn absolute_futex_timeout_is_rebased_to_logical_time() {
let logical_now = LogicalTime::from_secs(100);
let clock_now = LogicalTime::from_secs(5_000);
let deadline = clock_now + Duration::from_millis(100);
assert_eq!(
rebase_absolute_timeout(deadline, clock_now, logical_now),
logical_now + Duration::from_millis(100)
);
assert_eq!(
rebase_absolute_timeout(
clock_now - LogicalTime::from_nanos(1),
clock_now,
logical_now
),
logical_now
);
}
#[test]
fn absolute_futex_timeout_detects_host_and_logical_clock_domains() {
let host_monotonic_now = LogicalTime::from_secs(374_766);
let logical_now = LogicalTime::from_secs(1_640_995_199);
let delta = Duration::from_millis(100);
assert!(absolute_timeout_uses_host_clock(
host_monotonic_now + delta,
host_monotonic_now,
logical_now
));
assert!(!absolute_timeout_uses_host_clock(
logical_now + delta,
host_monotonic_now,
logical_now
));
let host_realtime_now = LogicalTime::from_secs(1_785_142_800);
assert!(absolute_timeout_uses_host_clock(
host_realtime_now + delta,
host_realtime_now,
logical_now
));
}
#[test]
fn futex_timeout_rejects_invalid_timespecs() {
assert_eq!(
parse_futex_timeout(
libc::FUTEX_WAIT,
Timespec {
tv_sec: -1,
tv_nsec: 0,
},
),
Err(Errno::EINVAL)
);
assert_eq!(
parse_futex_timeout(
libc::FUTEX_WAIT_BITSET,
Timespec {
tv_sec: 0,
tv_nsec: 1_000_000_000,
},
),
Err(Errno::EINVAL)
);
}
fn plain_attr() -> SchedAttrFields {
SchedAttrFields {
policy: 0,
sched_flags: 0,
priority: 0,
runtime: 0,
deadline: 0,
period: 0,
}
}
#[test]
fn size_zero_is_a_well_formed_ver0_request() {
assert_eq!(sched_attr_effective_size(0), Ok(SCHED_ATTR_SIZE_VER0));
}
#[test]
fn size_below_ver0_or_past_a_page_is_too_big() {
assert_eq!(sched_attr_effective_size(1), Err(()));
assert_eq!(sched_attr_effective_size(SCHED_ATTR_SIZE_VER0 - 1), Err(()));
assert_eq!(sched_attr_effective_size(SCHED_ATTR_MAX_SIZE + 1), Err(()));
}
#[test]
fn size_from_ver0_through_one_page_is_accepted_unchanged() {
for size in [
SCHED_ATTR_SIZE_VER0,
SCHED_ATTR_SIZE_VER1,
SCHED_ATTR_SIZE_VER1 + 1,
SCHED_ATTR_MAX_SIZE,
] {
assert_eq!(sched_attr_effective_size(size), Ok(size), "size {}", size);
}
}
#[test]
fn sched_ext_is_a_valid_policy_and_sched_iso_is_not() {
assert!(is_valid_sched_policy(SCHED_EXT), "SCHED_EXT is accepted");
assert!(!is_valid_sched_policy(4), "SCHED_ISO is reserved");
for policy in [0, 1, 2, 3, 5, 6] {
assert!(is_valid_sched_policy(policy), "policy {}", policy);
}
for policy in [8, 99] {
assert!(!is_valid_sched_policy(policy), "policy {}", policy);
}
}
#[test]
fn util_clamp_size_rule_is_decided_before_the_pid_lookup() {
let mut attr = plain_attr();
attr.sched_flags = SCHED_FLAG_UTIL_CLAMP_MIN;
assert_eq!(
validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER0, &attr),
Err(Errno::EINVAL)
);
assert_eq!(
validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER1, &attr),
Ok(())
);
assert_eq!(validate_sched_attr_after_lookup(&attr), Ok(()));
}
#[test]
fn a_negative_policy_is_decided_before_the_pid_lookup() {
let mut attr = plain_attr();
attr.policy = 0x8000_0000;
assert_eq!(
validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER0, &attr),
Err(Errno::EINVAL)
);
attr.sched_flags = SCHED_FLAG_KEEP_POLICY;
assert_eq!(
validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER0, &attr),
Err(Errno::EINVAL),
"KEEP_POLICY must not hide a negative policy"
);
}
#[test]
fn policy_and_flag_rules_are_decided_after_the_pid_lookup() {
let mut bad_policy = plain_attr();
bad_policy.policy = 99;
assert_eq!(
validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER0, &bad_policy),
Ok(()),
"an undefined policy must survive the near side so ESRCH can win"
);
assert_eq!(
validate_sched_attr_after_lookup(&bad_policy),
Err(Errno::EINVAL)
);
let mut bad_flag = plain_attr();
bad_flag.sched_flags = 0x80;
assert_eq!(
validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER0, &bad_flag),
Ok(()),
"an undefined sched_flags bit must survive the near side"
);
assert_eq!(
validate_sched_attr_after_lookup(&bad_flag),
Err(Errno::EINVAL)
);
}
#[test]
fn keep_policy_makes_the_policy_field_irrelevant() {
let mut attr = plain_attr();
attr.policy = 99;
assert_eq!(validate_sched_attr_after_lookup(&attr), Err(Errno::EINVAL));
attr.sched_flags = SCHED_FLAG_KEEP_POLICY;
assert_eq!(validate_sched_attr_after_lookup(&attr), Ok(()));
}
#[test]
fn defined_sched_flags_bits_are_accepted_and_undefined_ones_are_not() {
let mut attr = plain_attr();
attr.sched_flags = 0x80;
assert_eq!(validate_sched_attr_after_lookup(&attr), Err(Errno::EINVAL));
for flag in [
SCHED_FLAG_RESET_ON_FORK,
SCHED_FLAG_RECLAIM,
SCHED_FLAG_DL_OVERRUN,
SCHED_FLAG_KEEP_PARAMS,
SCHED_FLAG_KEEP_POLICY,
] {
let mut ok = plain_attr();
ok.sched_flags = flag;
assert_eq!(
validate_sched_attr_after_lookup(&ok),
Ok(()),
"flag {:#x}",
flag
);
}
let mut all = plain_attr();
all.sched_flags = SCHED_FLAG_ALL;
assert_eq!(
validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER0, &all),
Err(Errno::EINVAL)
);
assert_eq!(
validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER1, &all),
Ok(())
);
assert_eq!(validate_sched_attr_after_lookup(&all), Ok(()));
}
#[test]
fn priority_must_agree_with_the_policy() {
for policy in [0u32, 3, 5] {
let mut attr = plain_attr();
attr.policy = policy;
attr.priority = 1;
assert_eq!(
validate_sched_attr_after_lookup(&attr),
Err(Errno::EINVAL),
"policy {} with priority 1",
policy
);
attr.priority = 0;
assert_eq!(
validate_sched_attr_after_lookup(&attr),
Ok(()),
"policy {} with priority 0",
policy
);
}
for policy in [SCHED_FIFO, SCHED_RR] {
let mut attr = plain_attr();
attr.policy = policy;
attr.priority = 0;
assert_eq!(
validate_sched_attr_after_lookup(&attr),
Err(Errno::EINVAL),
"rt policy {} with priority 0",
policy
);
attr.priority = 1;
assert_eq!(validate_sched_attr_after_lookup(&attr), Ok(()));
}
}
#[test]
fn a_priority_past_max_rt_prio_is_refused() {
let mut attr = plain_attr();
attr.policy = SCHED_FIFO;
attr.priority = MAX_RT_PRIO;
assert_eq!(validate_sched_attr_after_lookup(&attr), Err(Errno::EINVAL));
attr.priority = MAX_RT_PRIO - 1;
assert_eq!(validate_sched_attr_after_lookup(&attr), Ok(()));
}
#[test]
fn deadline_parameters_are_checked() {
let mut zeroed = plain_attr();
zeroed.policy = SCHED_DEADLINE;
assert_eq!(
validate_sched_attr_after_lookup(&zeroed),
Err(Errno::EINVAL),
"all-zero deadline parameters"
);
let mut inverted = plain_attr();
inverted.policy = SCHED_DEADLINE;
inverted.runtime = 90_000_000;
inverted.deadline = 30_000_000;
inverted.period = 30_000_000;
assert_eq!(
validate_sched_attr_after_lookup(&inverted),
Err(Errno::EINVAL),
"runtime > deadline"
);
let mut sane = plain_attr();
sane.policy = SCHED_DEADLINE;
sane.runtime = 10_000_000;
sane.deadline = 30_000_000;
sane.period = 30_000_000;
assert_eq!(validate_sched_attr_after_lookup(&sane), Ok(()));
}
#[test]
fn deadline_parameter_edges_follow_checkparam_dl() {
assert!(!deadline_params_are_valid(1 << 20, 0, 0));
assert!(!deadline_params_are_valid((1 << 10) - 1, 1 << 20, 1 << 20));
assert!(deadline_params_are_valid(1 << 10, 1 << 20, 1 << 20));
assert!(!deadline_params_are_valid(1 << 20, 1 << 63, 0));
assert!(!deadline_params_are_valid(1 << 20, 1 << 20, 1 << 63));
assert!(deadline_params_are_valid(1 << 20, 1 << 21, 0));
assert!(!deadline_params_are_valid(1 << 20, 1 << 22, 1 << 21));
}
#[test]
fn the_size_written_back_on_refusal_is_the_kernel_struct_not_the_libc_mirror() {
assert_eq!(SCHED_ATTR_KERNEL_SIZE, 56);
assert_eq!(std::mem::size_of::<libc::sched_attr>(), 48);
assert_ne!(
SCHED_ATTR_KERNEL_SIZE as usize,
std::mem::size_of::<libc::sched_attr>()
);
}
struct BoundedMemory {
bytes: Vec<u8>,
readable_len: usize,
reads: std::cell::RefCell<Vec<usize>>,
}
impl reverie::syscalls::MemoryAccess for BoundedMemory {
fn read_vectored(
&self,
read_from: &[std::io::IoSlice<'_>],
write_to: &mut [std::io::IoSliceMut<'_>],
) -> Result<usize, Errno> {
let start = read_from[0].as_ptr() as usize;
let want = read_from.iter().map(|slice| slice.len()).sum::<usize>();
self.reads.borrow_mut().push(want);
if start >= self.readable_len {
return Err(Errno::EFAULT);
}
let avail = (self.readable_len - start).min(want);
let mut copied = 0;
for out in write_to.iter_mut() {
if copied == avail {
break;
}
let take = out.len().min(avail - copied);
out[..take].copy_from_slice(&self.bytes[start + copied..start + copied + take]);
copied += take;
}
Ok(copied)
}
fn write_vectored(
&mut self,
_read_from: &[std::io::IoSlice<'_>],
_write_to: &mut [std::io::IoSliceMut<'_>],
) -> Result<usize, Errno> {
unimplemented!("this fixture never writes")
}
}
fn bounded(bytes: Vec<u8>, readable_len: usize) -> BoundedMemory {
BoundedMemory {
bytes,
readable_len,
reads: std::cell::RefCell::new(Vec::new()),
}
}
fn tail_base() -> AddrMut<'static, u8> {
AddrMut::<u8>::from_raw(1).expect("nonzero base")
}
#[test]
fn a_nonzero_byte_before_a_fault_is_e2big_not_efault() {
let off = SCHED_ATTR_KERNEL_SIZE as usize;
let mut bytes = vec![0u8; off + 4096];
bytes[off + 1] = 0xAA; let readable = off + 512; let memory = bounded(bytes, readable);
assert_eq!(
scan_tail_is_zeroed(&memory, tail_base(), off, 4096),
TailVerdict::NotZeroed,
"a non-zero byte the scan reaches first must win over a later fault"
);
}
#[test]
fn a_fault_before_any_nonzero_byte_is_efault() {
let off = SCHED_ATTR_KERNEL_SIZE as usize;
let bytes = vec![0u8; off + 4096];
let memory = bounded(bytes, off + 512);
assert_eq!(
scan_tail_is_zeroed(&memory, tail_base(), off, 4096),
TailVerdict::Faulted,
"an unreadable byte reached before anything non-zero must be EFAULT"
);
}
#[test]
fn tail_reads_are_never_small_enough_to_bypass_guest_protection() {
let off = SCHED_ATTR_KERNEL_SIZE as usize;
for tail_len in 1..=16usize {
let bytes = vec![0u8; off + tail_len + 64];
let memory = bounded(bytes, off + tail_len + 64);
assert_eq!(
scan_tail_is_zeroed(&memory, tail_base(), off, tail_len),
TailVerdict::AllZero
);
let reads = memory.reads.borrow().clone();
assert!(!reads.is_empty(), "tail_len {tail_len} issued no read");
for length in reads {
assert!(
length > std::mem::size_of::<u64>(),
"tail_len {tail_len} issued a {length}-byte read, which safeptrace \
serves with PTRACE_PEEKDATA and which therefore bypasses guest \
page protections"
);
}
}
}
#[test]
fn keep_policy_validates_against_the_virtual_current_policy() {
let attr = SchedAttrFields {
policy: 99, sched_flags: SCHED_FLAG_KEEP_POLICY,
priority: 1,
runtime: 0,
deadline: 0,
period: 0,
};
assert_eq!(
validate_sched_attr_after_lookup(&attr),
Err(Errno::EINVAL),
"priority 1 under the virtual SCHED_OTHER current policy must be refused"
);
let ok = SchedAttrFields {
priority: 0,
..attr
};
assert_eq!(
validate_sched_attr_after_lookup(&ok),
Ok(()),
"KEEP_POLICY must still ignore the policy field itself"
);
}
#[test]
fn the_virtual_current_policy_is_what_sched_getattr_reports() {
assert_eq!(
VIRTUAL_CURRENT_POLICY,
libc::SCHED_OTHER as u32,
"handle_sched_getattr writes SCHED_OTHER into sched_policy for every thread; \
KEEP_POLICY must substitute that same value"
);
}
}