1use std::marker::PhantomData;
12use std::sync::Arc;
13use std::sync::Mutex;
14use std::time::Duration;
15
16use procfs::process::Process;
17use rand::Rng;
18use reverie::Error;
19use reverie::Guest;
20use reverie::Pid;
21use reverie::Stack;
22use reverie::syscalls;
23use reverie::syscalls::Addr;
24use reverie::syscalls::AddrMut;
25use reverie::syscalls::CloneFlags;
26use reverie::syscalls::Errno;
27use reverie::syscalls::MemoryAccess;
28use reverie::syscalls::Syscall;
29use reverie::syscalls::SyscallInfo;
30use reverie::syscalls::Timespec;
31use reverie::syscalls::WaitPidFlag;
32use tracing::debug;
33use tracing::info;
34use tracing::trace;
35
36use crate::config::BlockingMode;
37use crate::memory::MemoryMetadata;
38use crate::record_or_replay::RecordOrReplay;
39use crate::resources::ExternalOpId;
40use crate::resources::Permission;
41use crate::resources::ResourceID;
42use crate::resources::Resources;
43use crate::scheduler::SchedValue;
44use crate::syscalls::helpers::record_retry_event;
45use crate::syscalls::helpers::retry_nonblocking_syscall;
46use crate::syscalls::helpers::retry_nonblocking_syscall_with_timeout;
47use crate::syscalls::robust_list;
48use crate::tool_global::FutexAction;
49use crate::tool_global::ResumeStatus;
50use crate::tool_global::await_exact_child_physical_exit;
51use crate::tool_global::cancel_exec;
52use crate::tool_global::child_tid_clear_address;
53use crate::tool_global::consume_child_wait;
54use crate::tool_global::create_child_thread;
55use crate::tool_global::futex_action;
56use crate::tool_global::prepare_exec;
57use crate::tool_global::process_group;
58use crate::tool_global::ready_child_wait;
59use crate::tool_global::resource_request;
60use crate::tool_global::set_child_tid_address;
61use crate::tool_global::thread_is_live;
62use crate::tool_global::thread_observe_time;
63use crate::tool_global::wait_for_child_lifecycle;
64use crate::tool_global::yield_once;
65use crate::tool_local::Detcore;
66use crate::tool_local::PendingVfork;
67use crate::tool_local::RobustListExit;
68use crate::tool_local::RobustListWake;
69use crate::types::ChildWaitExitClass;
70use crate::types::ChildWaitSelector;
71use crate::types::ChildWaitSpec;
72use crate::types::DetPid;
73use crate::types::DetTid;
74use crate::types::ExactChildWaitState;
75use crate::types::LogicalTime;
76use crate::types::SigWrapper;
77
78const VIRTUAL_CPUSET_BYTES: usize = 16;
81
82const IOPRIO_WHO_PROCESS: libc::c_int = 1;
83const IOPRIO_WHO_PGRP: libc::c_int = 2;
84const IOPRIO_WHO_USER: libc::c_int = 3;
85const IOPRIO_CLASS_SHIFT: libc::c_int = 13;
86const IOPRIO_CLASS_BE: libc::c_int = 2;
87const IOPRIO_BE_NORM: libc::c_int = 4;
88const IOPRIO_DEFAULT_EFFECTIVE: libc::c_int =
89 (IOPRIO_CLASS_BE << IOPRIO_CLASS_SHIFT) | IOPRIO_BE_NORM;
90
91const SCHED_ATTR_SIZE_VER0: u32 = 48; const SCHED_ATTR_SIZE_VER1: u32 = 56; const SCHED_ATTR_KERNEL_SIZE: u32 = SCHED_ATTR_SIZE_VER1;
103const SCHED_ATTR_MAX_SIZE: u32 = 4096;
106
107const VIRTUAL_CURRENT_POLICY: u32 = libc::SCHED_OTHER as u32;
113
114#[derive(Debug, Eq, PartialEq)]
116enum TailVerdict {
117 AllZero,
118 NotZeroed,
121 Faulted,
123}
124
125const MIN_PROTECTION_RESPECTING_READ: usize = std::mem::size_of::<u64>() + 1;
132
133fn scan_tail_is_zeroed<M: MemoryAccess>(
149 memory: &M,
150 base: AddrMut<u8>,
151 tail_off: usize,
152 tail_len: usize,
153) -> TailVerdict {
154 const CHUNK: usize = 256;
155 let mut done = 0usize;
156 let mut buf = [0u8; CHUNK];
157 while done < tail_len {
158 let remaining = tail_len - done;
159 let want = remaining.min(CHUNK);
160 let back = MIN_PROTECTION_RESPECTING_READ
167 .saturating_sub(want)
168 .min(done + tail_off);
169 let read_len = want + back;
170 let start = tail_off + done - back;
171 let Some(addr) = base
172 .as_raw()
173 .checked_add(start)
174 .and_then(AddrMut::<u8>::from_raw)
175 else {
176 return TailVerdict::Faulted;
177 };
178 let got = memory.read(addr, &mut buf[..read_len]).unwrap_or(0);
182 let new_from = back.min(got);
184 if buf[new_from..got].iter().any(|byte| *byte != 0) {
185 return TailVerdict::NotZeroed;
186 }
187 if got < read_len {
188 return TailVerdict::Faulted;
191 }
192 done += want;
193 }
194 TailVerdict::AllZero
195}
196
197const SCHED_FIFO: u32 = 1;
199const SCHED_RR: u32 = 2;
200const SCHED_DEADLINE: u32 = 6;
201const SCHED_EXT: u32 = 7;
202const MAX_RT_PRIO: u32 = 100;
204
205const SCHED_ATTR_OFF_POLICY: usize = 4;
208const SCHED_ATTR_OFF_FLAGS: usize = 8;
209const SCHED_ATTR_OFF_PRIORITY: usize = 20;
210const SCHED_ATTR_OFF_RUNTIME: usize = 24;
211const SCHED_ATTR_OFF_DEADLINE: usize = 32;
212const SCHED_ATTR_OFF_PERIOD: usize = 40;
213
214const SCHED_FLAG_RESET_ON_FORK: u64 = 0x01;
216const SCHED_FLAG_RECLAIM: u64 = 0x02;
217const SCHED_FLAG_DL_OVERRUN: u64 = 0x04;
218const SCHED_FLAG_KEEP_POLICY: u64 = 0x08;
219const SCHED_FLAG_KEEP_PARAMS: u64 = 0x10;
220const SCHED_FLAG_UTIL_CLAMP_MIN: u64 = 0x20;
221const SCHED_FLAG_UTIL_CLAMP_MAX: u64 = 0x40;
222const SCHED_FLAG_UTIL_CLAMP: u64 = SCHED_FLAG_UTIL_CLAMP_MIN | SCHED_FLAG_UTIL_CLAMP_MAX;
223const SCHED_FLAG_ALL: u64 = SCHED_FLAG_RESET_ON_FORK
224 | SCHED_FLAG_RECLAIM
225 | SCHED_FLAG_DL_OVERRUN
226 | SCHED_FLAG_KEEP_POLICY
227 | SCHED_FLAG_KEEP_PARAMS
228 | SCHED_FLAG_UTIL_CLAMP;
229
230fn is_valid_sched_policy(policy: u32) -> bool {
236 matches!(
237 policy,
238 0 | 1 | 2 | 3
240 | 5
242 | SCHED_DEADLINE
243 | SCHED_EXT
244 )
245}
246
247fn is_rt_policy(policy: u32) -> bool {
249 policy == SCHED_FIFO || policy == SCHED_RR
250}
251
252fn deadline_params_are_valid(runtime: u64, deadline: u64, period: u64) -> bool {
260 if deadline == 0 {
262 return false;
263 }
264 if runtime < (1u64 << 10) {
267 return false;
268 }
269 if deadline & (1u64 << 63) != 0 || period & (1u64 << 63) != 0 {
271 return false;
272 }
273 let period = if period == 0 { deadline } else { period };
275 runtime <= deadline && deadline <= period
277}
278
279fn sched_attr_effective_size(declared: u32) -> Result<u32, ()> {
289 let size = if declared == 0 {
290 SCHED_ATTR_SIZE_VER0
291 } else {
292 declared
293 };
294 if !(SCHED_ATTR_SIZE_VER0..=SCHED_ATTR_MAX_SIZE).contains(&size) {
295 return Err(());
296 }
297 Ok(size)
298}
299
300#[derive(Clone, Copy, Debug)]
302struct SchedAttrFields {
303 policy: u32,
304 sched_flags: u64,
305 priority: u32,
306 runtime: u64,
307 deadline: u64,
308 period: u64,
309}
310
311fn validate_sched_attr_before_lookup(size: u32, attr: &SchedAttrFields) -> Result<(), Errno> {
321 if attr.sched_flags & SCHED_FLAG_UTIL_CLAMP != 0 && size < SCHED_ATTR_SIZE_VER1 {
324 return Err(Errno::EINVAL);
325 }
326 if (attr.policy as i32) < 0 {
330 return Err(Errno::EINVAL);
331 }
332 Ok(())
333}
334
335fn validate_sched_attr_after_lookup(attr: &SchedAttrFields) -> Result<(), Errno> {
341 let policy = if attr.sched_flags & SCHED_FLAG_KEEP_POLICY != 0 {
359 Some(VIRTUAL_CURRENT_POLICY)
360 } else {
361 if !is_valid_sched_policy(attr.policy) {
362 return Err(Errno::EINVAL);
363 }
364 Some(attr.policy)
365 };
366 if attr.sched_flags & !SCHED_FLAG_ALL != 0 {
368 return Err(Errno::EINVAL);
369 }
370 if attr.priority > MAX_RT_PRIO - 1 {
374 return Err(Errno::EINVAL);
375 }
376 if let Some(policy) = policy {
377 if is_rt_policy(policy) != (attr.priority != 0) {
378 return Err(Errno::EINVAL);
379 }
380 if policy == SCHED_DEADLINE
381 && !deadline_params_are_valid(attr.runtime, attr.deadline, attr.period)
382 {
383 return Err(Errno::EINVAL);
384 }
385 }
386 Ok(())
387}
388
389fn virtual_ioprio(which: libc::c_int) -> Result<i64, Errno> {
392 match which {
393 IOPRIO_WHO_PROCESS => Ok(0),
394 IOPRIO_WHO_PGRP | IOPRIO_WHO_USER => Ok(i64::from(IOPRIO_DEFAULT_EFFECTIVE)),
395 _ => Err(Errno::EINVAL),
396 }
397}
398
399#[repr(C)]
400#[derive(Clone, Copy)]
401struct WaitidSigchldFields {
402 pid: libc::pid_t,
403 uid: libc::uid_t,
404 status: libc::c_int,
405 utime: libc::c_long,
406 stime: libc::c_long,
407}
408
409#[repr(C)]
410union WaitidSiginfoFields {
411 _alignment: *mut libc::c_void,
412 sigchld: WaitidSigchldFields,
413}
414
415#[repr(C)]
416struct WaitidSiginfoHead {
417 _base: [libc::c_int; 3],
418 fields: WaitidSiginfoFields,
419}
420
421fn wait_status_is_termination(status: libc::c_int) -> bool {
422 libc::WIFEXITED(status) || libc::WIFSIGNALED(status)
423}
424
425fn waitid_code_is_termination(code: libc::c_int) -> bool {
426 matches!(code, libc::CLD_EXITED | libc::CLD_KILLED | libc::CLD_DUMPED)
427}
428
429fn canonicalize_waitid_siginfo(info: &mut libc::siginfo_t) {
430 debug_assert!(
431 std::mem::size_of::<WaitidSiginfoHead>() <= std::mem::size_of::<libc::siginfo_t>()
432 );
433 let sigchld = unsafe {
438 &mut (*(info as *mut libc::siginfo_t).cast::<WaitidSiginfoHead>())
439 .fields
440 .sigchld
441 };
442 sigchld.utime = 0;
443 sigchld.stime = 0;
444}
445
446fn finish_waitid_result<T, G>(
447 guest: &mut G,
448 call: syscalls::Waitid,
449 value: i64,
450 mut info_value: libc::siginfo_t,
451) -> Result<i64, Error>
452where
453 T: RecordOrReplay,
454 G: Guest<Detcore<T>>,
455{
456 let child_pid = unsafe { info_value.si_pid() };
459 if child_pid != 0 {
460 canonicalize_waitid_siginfo(&mut info_value);
461 guest.memory().write_value(
462 call.info().expect("waitid infop checked before execution"),
463 &info_value,
464 )?;
465 if call.options() & libc::WNOWAIT == 0 && waitid_code_is_termination(info_value.si_code) {
466 guest
467 .thread_state_mut()
468 .reap_child_process_cpu_time(DetPid::from_raw(child_pid));
469 }
470 if let Some(rusage) = call.rusage() {
471 let usage: libc::rusage = unsafe { std::mem::zeroed() };
473 guest.memory().write_value(rusage, &usage)?;
474 }
475 }
476 Ok(value)
477}
478
479#[derive(Debug, Eq, PartialEq)]
480enum ExactWaitPollDecision {
481 ChildReady,
482 AwaitPhysicalExit,
483 ReapAfterLogicalExit,
484 Interrupted,
485 Retry,
486}
487
488fn exact_wait_poll_decision(
489 child_ready: bool,
490 signaled: bool,
491 lifecycle: Option<ExactChildWaitState>,
492) -> ExactWaitPollDecision {
493 if child_ready {
494 ExactWaitPollDecision::ChildReady
495 } else if lifecycle == Some(ExactChildWaitState::PhysicalExitPending) {
496 ExactWaitPollDecision::AwaitPhysicalExit
497 } else if matches!(
498 lifecycle,
499 Some(ExactChildWaitState::LogicallyExited | ExactChildWaitState::PhysicallyExited)
500 ) {
501 ExactWaitPollDecision::ReapAfterLogicalExit
502 } else if signaled {
503 ExactWaitPollDecision::Interrupted
504 } else {
505 ExactWaitPollDecision::Retry
506 }
507}
508
509fn stale_any_wait_must_interrupt(signaled: bool, next_ready: Option<DetPid>) -> bool {
510 signaled && next_ready.is_none()
511}
512
513fn terminal_child_wait_spec(
514 selector: ChildWaitSelector,
515 caller: DetTid,
516 options: libc::c_int,
517) -> ChildWaitSpec {
518 let exit_class = if options & libc::__WALL != 0 {
519 ChildWaitExitClass::Any
520 } else if options & libc::__WCLONE != 0 {
521 ChildWaitExitClass::Clone
522 } else {
523 ChildWaitExitClass::Sigchld
524 };
525 ChildWaitSpec {
526 selector,
527 owner: (options & libc::__WNOTHREAD != 0).then_some(caller),
528 exit_class,
529 }
530}
531
532fn validate_wait4_arguments(pid: libc::pid_t, options: WaitPidFlag) -> Result<(), Errno> {
533 let allowed_options = WaitPidFlag::WNOHANG
534 | WaitPidFlag::WUNTRACED
535 | WaitPidFlag::WCONTINUED
536 | WaitPidFlag::__WNOTHREAD
537 | WaitPidFlag::__WCLONE
538 | WaitPidFlag::__WALL;
539
540 if options.bits() & !allowed_options.bits() != 0 {
543 return Err(Errno::EINVAL);
544 }
545 if pid == libc::pid_t::MIN {
548 return Err(Errno::ESRCH);
549 }
550 Ok(())
551}
552
553fn child_wait_can_retry_after_stale(spec: ChildWaitSpec) -> bool {
554 !matches!(spec.selector, ChildWaitSelector::Exact(_))
555}
556
557pub(super) type KernelSigset = u64;
558pub(super) const KERNEL_SIGSET_SIZE: usize = std::mem::size_of::<KernelSigset>();
559
560fn signal_is_blocked(mask: &KernelSigset, signal: SigWrapper) -> bool {
561 let raw_signal = signal.raw();
562 (1..=KernelSigset::BITS as i32).contains(&raw_signal)
563 && mask & (1_u64 << (raw_signal as u32 - 1)) != 0
564}
565
566#[repr(C)]
567#[derive(Clone, Copy)]
568pub(super) struct KernelSigaction {
569 pub(super) handler: u64,
570 pub(super) flags: u64,
571 pub(super) restorer: u64,
572 pub(super) mask: KernelSigset,
573}
574
575#[derive(Clone, Copy, Debug, Eq, PartialEq)]
576pub(super) enum WaitSignalDisposition {
577 Interrupt,
578 Restart,
579}
580
581fn signal_default_disposition_does_not_interrupt_child_wait(signal: SigWrapper) -> bool {
582 matches!(
583 signal.raw(),
584 libc::SIGCHLD | libc::SIGCONT | libc::SIGURG | libc::SIGWINCH
585 )
586}
587
588fn signal_has_uncatchable_default_disposition(signal: SigWrapper) -> bool {
589 matches!(signal.raw(), libc::SIGKILL | libc::SIGSTOP)
590}
591
592pub(super) async fn wait_signal_disposition<G, T>(
593 guest: &mut G,
594 status: ResumeStatus,
595 guest_signal_mask: &KernelSigset,
596 action_addr: AddrMut<'_, KernelSigaction>,
597 inspect_action: bool,
598) -> Result<Option<WaitSignalDisposition>, Error>
599where
600 G: Guest<Detcore<T>>,
601 T: RecordOrReplay,
602{
603 let ResumeStatus::Signaled(signals) = status else {
604 return Ok(None);
605 };
606 let Some(mut signals) = signals else {
607 return Ok(Some(WaitSignalDisposition::Interrupt));
608 };
609 signals.sort_by_key(|signal| signal.raw());
610 for signal in signals {
611 if signal_is_blocked(guest_signal_mask, signal) {
612 continue;
613 }
614 if signal_has_uncatchable_default_disposition(signal) {
615 return Ok(Some(WaitSignalDisposition::Interrupt));
616 }
617 if !inspect_action {
618 return Ok(Some(WaitSignalDisposition::Interrupt));
619 }
620 let call = syscalls::RtSigaction::new()
621 .with_signum(signal.raw())
622 .with_action(None)
623 .with_old_action(Some(action_addr.cast()))
624 .with_sigsetsize(std::mem::size_of::<u64>());
625 guest.inject_with_retry(call).await?;
626 let action: KernelSigaction = guest.memory().read_value(action_addr)?;
627 if action.handler == libc::SIG_IGN as u64
628 || action.handler == libc::SIG_DFL as u64
629 && signal_default_disposition_does_not_interrupt_child_wait(signal)
630 {
631 continue;
632 }
633 return Ok(Some(
634 if action.handler != libc::SIG_DFL as u64 && action.flags & libc::SA_RESTART as u64 != 0
635 {
636 WaitSignalDisposition::Restart
637 } else {
638 WaitSignalDisposition::Interrupt
639 },
640 ));
641 }
642 Ok(None)
643}
644
645pub(super) fn blocked_signal_mask() -> KernelSigset {
646 let mut libc_mask: libc::sigset_t = unsafe { std::mem::zeroed() };
649 unsafe {
650 libc::sigfillset(&mut libc_mask);
651 libc::sigdelset(&mut libc_mask, reverie::PERF_EVENT_SIGNAL as i32);
652 }
653 (1..=KernelSigset::BITS as i32).fold(0, |mask, raw_signal| {
654 if unsafe { libc::sigismember(&libc_mask, raw_signal) } == 1 {
655 mask | (1_u64 << (raw_signal as u32 - 1))
656 } else {
657 mask
658 }
659 })
660}
661
662pub(super) async fn block_signals_for_disposition<G, T>(
663 guest: &mut G,
664 blocked_mask_addr: Addr<'_, KernelSigset>,
665 old_mask_addr: AddrMut<'_, KernelSigset>,
666) -> Result<KernelSigset, Error>
667where
668 G: Guest<Detcore<T>>,
669 T: RecordOrReplay,
670{
671 let block_signals = syscalls::RtSigprocmask::new()
672 .with_how(libc::SIG_SETMASK)
673 .with_set(
674 (!guest
675 .config()
676 .backend_requires_thread_directed_process_signals)
677 .then_some(blocked_mask_addr.cast()),
678 )
679 .with_oldset(Some(old_mask_addr.cast()))
680 .with_sigsetsize(KERNEL_SIGSET_SIZE);
681 guest.inject_with_retry(block_signals).await?;
682 Ok(guest.memory().read_value(old_mask_addr)?)
683}
684
685pub(super) async fn restore_signals_after_disposition<G, T>(
686 guest: &mut G,
687 old_mask_addr: AddrMut<'_, KernelSigset>,
688) -> Result<(), Error>
689where
690 G: Guest<Detcore<T>>,
691 T: RecordOrReplay,
692{
693 if !guest
694 .config()
695 .backend_requires_thread_directed_process_signals
696 {
697 let old_mask: Addr<'_, KernelSigset> = old_mask_addr.into();
698 let restore_signals = syscalls::RtSigprocmask::new()
699 .with_how(libc::SIG_SETMASK)
700 .with_set(Some(old_mask.cast()))
701 .with_oldset(None)
702 .with_sigsetsize(KERNEL_SIGSET_SIZE);
703 guest.inject_with_retry(restore_signals).await?;
704 }
705 Ok(())
706}
707
708async fn interrupted_child_wait_result<G, T, S>(
709 guest: &mut G,
710 call: S,
711 disposition: WaitSignalDisposition,
712) -> Result<i64, Error>
713where
714 G: Guest<Detcore<T>>,
715 T: RecordOrReplay,
716 S: SyscallInfo,
717{
718 if !guest
719 .config()
720 .backend_requires_thread_directed_process_signals
721 {
722 return Err(Errno::ERESTARTSYS.into());
723 }
724 if disposition == WaitSignalDisposition::Interrupt {
725 return Err(Errno::EINTR.into());
726 }
727
728 guest.tail_inject(call).await
729}
730
731fn snapshot_process_group(pid: Pid) -> Result<libc::pid_t, Errno> {
732 let pgrp = Process::new(pid.as_raw())
733 .and_then(|process| process.stat())
734 .map(|stat| stat.pgrp)
735 .map_err(|_| Errno::ESRCH)?;
736 if pgrp == 0 {
737 Err(Errno::EOPNOTSUPP)
738 } else {
739 Ok(pgrp)
740 }
741}
742
743fn guest_fd_status_flags(pid: Pid, fd: libc::c_int) -> Result<libc::c_int, Errno> {
744 let path = format!("/proc/{}/fdinfo/{}", pid.as_raw(), fd);
745 let contents = std::fs::read_to_string(path).map_err(|_| Errno::EBADF)?;
746 let flags = contents
747 .lines()
748 .find_map(|line| line.strip_prefix("flags:"))
749 .map(str::trim)
750 .ok_or(Errno::EINVAL)?;
751 libc::c_int::from_str_radix(flags, 8).map_err(|_| Errno::EINVAL)
752}
753
754#[derive(Debug, Clone, Copy, PartialEq, Eq)]
755enum FutexTimeout {
756 Relative(u64),
757 Absolute(LogicalTime),
758}
759
760fn parse_futex_timeout(futex_op: i32, timeout: Timespec) -> Result<FutexTimeout, Errno> {
761 let seconds = u64::try_from(timeout.tv_sec).map_err(|_| Errno::EINVAL)?;
762 let nanoseconds = u64::try_from(timeout.tv_nsec).map_err(|_| Errno::EINVAL)?;
763 if nanoseconds >= 1_000_000_000 {
764 return Err(Errno::EINVAL);
765 }
766
767 let timeout_nanos = seconds
768 .checked_mul(1_000_000_000)
769 .and_then(|nanos| nanos.checked_add(nanoseconds))
770 .ok_or(Errno::EINVAL)?;
771 if futex_op & libc::FUTEX_CMD_MASK == libc::FUTEX_WAIT_BITSET {
778 Ok(FutexTimeout::Absolute(LogicalTime::from_nanos(
779 timeout_nanos,
780 )))
781 } else {
782 Ok(FutexTimeout::Relative(timeout_nanos))
783 }
784}
785
786fn rebase_absolute_timeout(
787 deadline: LogicalTime,
788 clock_now: LogicalTime,
789 logical_now: LogicalTime,
790) -> LogicalTime {
791 logical_now + Duration::from_nanos(deadline.as_nanos().saturating_sub(clock_now.as_nanos()))
792}
793
794fn absolute_timeout_uses_host_clock(
795 deadline: LogicalTime,
796 host_clock_now: LogicalTime,
797 logical_now: LogicalTime,
798) -> bool {
799 deadline.as_nanos().abs_diff(host_clock_now.as_nanos())
800 < deadline.as_nanos().abs_diff(logical_now.as_nanos())
801}
802
803impl<T: RecordOrReplay> Detcore<T> {
804 async fn futex_timeout_deadline<G: Guest<Self>>(
805 &self,
806 guest: &mut G,
807 futex_flags: i32,
808 timeout: Option<Addr<'_, Timespec>>,
809 ) -> Result<Option<LogicalTime>, Error> {
810 let Some(timeout) = timeout else {
811 return Ok(None);
812 };
813 let timeout = parse_futex_timeout(futex_flags, guest.memory().read_value(timeout)?)?;
814 match timeout {
815 FutexTimeout::Relative(nanos) => {
816 let now = thread_observe_time(guest).await;
817 Ok(Some(now + Duration::from_nanos(nanos)))
818 }
819 FutexTimeout::Absolute(deadline)
820 if self.cfg.virtualize_time && !self.cfg.detect_host_clock_futex_timeouts =>
821 {
822 Ok(Some(deadline))
823 }
824 FutexTimeout::Absolute(deadline) => {
825 let clockid = if futex_flags & libc::FUTEX_CLOCK_REALTIME != 0 {
826 syscalls::ClockId::CLOCK_REALTIME
827 } else {
828 syscalls::ClockId::CLOCK_MONOTONIC
829 };
830
831 let mut stack = guest.stack().await;
832 let clock_output = syscalls::TimespecMutPtr(stack.reserve());
833 let _stack_guard = stack.commit()?;
834 let clock_call = syscalls::ClockGettime::new()
835 .with_clockid(clockid)
836 .with_tp(Some(clock_output));
837 if self.cfg.virtualize_time && self.cfg.detect_host_clock_futex_timeouts {
838 guest.inject(Syscall::from(clock_call)).await?;
842 } else {
843 self.record_or_replay(guest, clock_call).await?;
844 }
845 let clock_now = match parse_futex_timeout(
846 libc::FUTEX_WAIT_BITSET,
847 guest.memory().read_value(clock_output.0)?,
848 )? {
849 FutexTimeout::Absolute(time) => time,
850 FutexTimeout::Relative(_) => unreachable!(),
851 };
852 let logical_now = thread_observe_time(guest).await;
853 if self.cfg.virtualize_time
854 && !absolute_timeout_uses_host_clock(deadline, clock_now, logical_now)
855 {
856 return Ok(Some(deadline));
857 }
858 Ok(Some(rebase_absolute_timeout(
859 deadline,
860 clock_now,
861 logical_now,
862 )))
863 }
864 }
865 }
866
867 pub async fn handle_clone_family<G: Guest<Self>>(
869 &self,
870 guest: &mut G,
871 clone_family: syscalls::family::CloneFamily,
872 ) -> Result<i64, Error> {
873 let flags = clone_family.flags(&guest.memory());
874 let exit_signal = match clone_family {
875 #[cfg(not(target_arch = "aarch64"))]
876 syscalls::family::CloneFamily::Fork(_) | syscalls::family::CloneFamily::Vfork(_) => {
877 libc::SIGCHLD
878 }
879 syscalls::family::CloneFamily::Clone(clone) => {
880 (clone.flags().bits() & 0xff) as libc::c_int
881 }
882 syscalls::family::CloneFamily::Clone3(clone) => clone
883 .args()
884 .and_then(|address| guest.memory().read_value(address).ok())
885 .map_or(0, |args: syscalls::CloneArgs| {
886 args.exit_signal as libc::c_int
887 }),
888 };
889 let ctid = child_tid_clear_address(flags, clone_family.child_tid(&guest.memory()));
890 let is_vfork = flags.contains(CloneFlags::CLONE_VFORK);
891 let parent_blocks_for_child = is_vfork
892 || (self.cfg.backend_serializes_fork_children
893 && !flags.contains(CloneFlags::CLONE_THREAD));
894 let backend_uninstrumented_thread =
895 flags.contains(CloneFlags::CLONE_THREAD) && !self.cfg.backend_dispatches_thread_tools;
896
897 let ts = guest.thread_state_mut();
898 assert_eq!(ts.clone_flags, None);
899 assert!(ts.pending_vfork.is_none());
900 ts.clone_flags = Some(flags);
901
902 let parent_dettid = ts.dettid;
903 let child_priority_entropy = if parent_blocks_for_child
904 && self.cfg.chaos
905 && self.cfg.replay_preemptions_from.is_none()
906 && self.cfg.replay_schedule_from.is_none()
907 {
908 let mut parent_chaos_prng = ts.chaos_prng.clone();
909 Some(parent_chaos_prng.next_u64())
910 } else {
911 None
912 };
913 if parent_blocks_for_child {
914 ts.pending_vfork = Some(PendingVfork {
915 parent_dettid,
916 parent_detpid: ts.detpid.expect("detpid unset"),
917 child_tid_addr: ctid,
918 flags,
919 exit_signal,
920 child_priority_entropy,
921 });
922 }
923
924 trace!("[detcore, dtid {}] parent invoking clone.", parent_dettid);
925 let blocking_child_op_id =
926 ExternalOpId::new(parent_dettid, guest.thread_state().stats.syscall_count);
927
928 if parent_blocks_for_child && self.cfg.sequentialize_threads {
932 let mut resources = Resources::new(parent_dettid);
933 resources.insert(
934 ResourceID::BlockingVfork(blocking_child_op_id),
935 Permission::RW,
936 );
937 resources.fyi(if is_vfork {
938 "clone_vfork"
939 } else {
940 "clone_serialized_child"
941 });
942 resource_request(guest, resources).await;
943 }
944
945 let maybe_res = guest.inject(Syscall::from(clone_family)).await;
946
947 if parent_blocks_for_child && self.cfg.sequentialize_threads {
948 let mut resources = Resources::new(parent_dettid);
949 if maybe_res.is_err() {
950 resources.insert(
956 ResourceID::VforkFailed(blocking_child_op_id),
957 Permission::RW,
958 );
959 resources.fyi(if is_vfork {
960 "clone_vfork_failed"
961 } else {
962 "clone_serialized_child_failed"
963 });
964 } else {
965 resources.insert(
966 ResourceID::BlockedExternalContinue(blocking_child_op_id),
967 Permission::RW,
968 );
969 resources.fyi(if is_vfork {
970 "clone_vfork"
971 } else {
972 "clone_serialized_child"
973 });
974 }
975 resource_request(guest, resources).await;
976 }
977
978 let ts = guest.thread_state_mut();
979 ts.clone_flags = None; ts.pending_vfork = None;
981
982 let res = maybe_res?;
983
984 if !flags.contains(CloneFlags::CLONE_THREAD) {
985 guest.thread_state().forget_flock_modes();
989 }
990
991 if parent_blocks_for_child
994 && self.cfg.chaos
995 && self.cfg.replay_preemptions_from.is_none()
996 && self.cfg.replay_schedule_from.is_none()
997 {
998 let _ = guest
999 .thread_state_mut()
1000 .chaos_prng_next_u64("child_priority");
1001 }
1002
1003 let child_tid = Pid::from_raw(res as i32);
1004 let child_dettid = DetTid::from_raw(child_tid.into()); trace!(
1006 "[detcore] dtid {} cloned, continuing parent + register new thread.",
1007 child_dettid
1008 );
1009
1010 if !parent_blocks_for_child && !backend_uninstrumented_thread {
1011 create_child_thread(guest, child_dettid, ctid, Some(flags), exit_signal, None).await;
1012 }
1013
1014 {
1015 let parent_pedigree = &mut guest.thread_state_mut().pedigree;
1017 let child_pedigree = parent_pedigree.fork_mut();
1018 debug!(
1019 "[dtid {}] after creating child thread (tid {}, pedigree {}) parents pedigree becomes {}",
1020 parent_dettid, child_dettid, child_pedigree, parent_pedigree,
1021 );
1022 }
1023
1024 Ok(child_dettid.as_raw() as i64)
1025 }
1026
1027 pub async fn handle_set_tid_address<G: Guest<Self>>(
1034 &self,
1035 guest: &mut G,
1036 call: syscalls::SetTidAddress,
1037 ) -> Result<i64, Error> {
1038 let address = call.tidptr().map_or(0, |pointer| pointer.as_raw());
1039 let result = self
1040 .record_or_replay(guest, Syscall::SetTidAddress(call))
1041 .await?;
1042 if guest.config().sequentialize_threads {
1043 set_child_tid_address(guest, address).await;
1044 }
1045 trace!(
1046 "[detcore, dtid {}] child TID clear address registered: {address:#x}",
1047 guest.thread_state().dettid,
1048 );
1049 Ok(result)
1050 }
1051
1052 pub async fn handle_set_robust_list<G: Guest<Self>>(
1068 &self,
1069 guest: &mut G,
1070 call: syscalls::SetRobustList,
1071 ) -> Result<i64, Error> {
1072 let head = call.head().map(AddrMut::as_raw);
1073 let len = call.len();
1074 let res = self
1075 .record_or_replay(guest, Syscall::SetRobustList(call))
1076 .await?;
1077 let recorded = match head {
1080 None => None,
1082 Some(_) if len != robust_list::ROBUST_LIST_HEAD_LEN => None,
1088 Some(addr) => Some(addr),
1089 };
1090 guest.thread_state_mut().record_robust_list_head(recorded);
1091 trace!(
1092 "[detcore, dtid {}] robust-list head registered: {:?}",
1093 guest.thread_state().dettid,
1094 recorded,
1095 );
1096 Ok(res)
1097 }
1098
1099 async fn run_robust_list_owner_death<G: Guest<Self>>(&self, guest: &mut G) {
1117 if !self.cfg.backend_runs_exit_robust_list
1118 || !self.cfg.sequentialize_threads
1119 || self.cfg.debug_futex_mode != BlockingMode::Precise
1120 {
1121 return;
1122 }
1123 let Some(head) = guest.thread_state().robust_list_head else {
1124 return;
1125 };
1126 let dettid = guest.thread_state().dettid;
1127 let owner_tid = dettid.as_raw() as u32;
1140
1141 let mut effects = GuestRobustEffects::<'_, G, T> {
1142 guest,
1143 dettid,
1144 defer_owner_death_to_backend: true,
1145 staged_wakes: None,
1146 tool: PhantomData,
1147 };
1148 let outcome = robust_list::exit_robust_list(&mut effects, head, owner_tid).await;
1149 if outcome.head_unreadable {
1150 trace!(
1151 "[detcore, dtid {}] unreadable robust-list head at {:#x}; no owner-death wakeups",
1152 dettid, head,
1153 );
1154 } else if outcome.aborted || outcome.next_faulted || outcome.truncated {
1155 trace!(
1156 "[detcore, dtid {}] robust-list walk stopped early after {} entr(ies): {:?}",
1157 dettid, outcome.entries_visited, outcome,
1158 );
1159 }
1160 }
1161
1162 pub(crate) async fn stage_thread_group_robust_list_wakes<G: Guest<Self>>(
1167 &self,
1168 guest: &mut G,
1169 reason: RobustListExit,
1170 ) {
1171 if !self.cfg.backend_runs_exit_robust_list
1172 || !self.cfg.sequentialize_threads
1173 || self.cfg.debug_futex_mode != BlockingMode::Precise
1174 {
1175 return;
1176 }
1177
1178 let heads = guest.thread_state().robust_list_heads();
1179 let mut staged = Vec::with_capacity(heads.len());
1180 for (owner, head) in heads {
1181 let mut wakes = Vec::new();
1182 let mut effects = GuestRobustEffects::<'_, G, T> {
1183 guest,
1184 dettid: owner,
1185 defer_owner_death_to_backend: true,
1186 staged_wakes: Some(&mut wakes),
1187 tool: PhantomData,
1188 };
1189 let outcome =
1190 robust_list::exit_robust_list(&mut effects, head, owner.as_raw() as u32).await;
1191 if outcome.head_unreadable || outcome.aborted {
1192 trace!(
1193 "[detcore, dtid {}] could not stage complete robust-list owner-death effects: {:?}",
1194 owner, outcome,
1195 );
1196 }
1197 staged.push((owner, wakes));
1198 }
1199 guest.thread_state().stage_robust_list_wakes(reason, staged);
1200 }
1201
1202 pub async fn handle_exit<G: Guest<Self>>(
1204 &self,
1205 guest: &mut G,
1206 call: syscalls::Exit,
1207 ) -> Result<i64, Error> {
1208 let request = guest.thread_state().mk_request(
1209 ResourceID::Exit {
1210 group: false,
1211 process: guest.thread_state().detpid.expect("detpid unset"),
1212 mm: guest.thread_state().mm_id,
1213 },
1214 Permission::RW,
1215 );
1216 resource_request(guest, request).await;
1217 self.run_robust_list_owner_death(guest).await;
1218 guest.tail_inject(call).await
1220 }
1221
1222 pub async fn handle_exit_group<G: Guest<Self>>(
1224 &self,
1225 guest: &mut G,
1226 call: syscalls::ExitGroup,
1227 ) -> Result<i64, Error> {
1228 let request = guest.thread_state().mk_request(
1229 ResourceID::Exit {
1230 group: true,
1231 process: guest.thread_state().detpid.expect("detpid unset"),
1232 mm: guest.thread_state().mm_id,
1233 },
1234 Permission::RW,
1235 );
1236 resource_request(guest, request).await;
1237 self.stage_thread_group_robust_list_wakes(guest, RobustListExit::ExitGroup)
1238 .await;
1239 guest.tail_inject(call).await
1241 }
1242
1243 pub async fn handle_futex<G: Guest<Self>>(
1245 &self,
1246 guest: &mut G,
1247 call: syscalls::Futex,
1248 ) -> Result<i64, Error> {
1249 let dettid = guest.thread_state().dettid;
1250 let ptr = match call.uaddr() {
1251 None => {
1252 return Ok(guest.inject(call).await?);
1254 }
1255 Some(x) => x,
1256 };
1257 let init_val = guest.memory().read_value(ptr)?;
1258 trace!(
1259 "[detcore, dtid {}] futex op with memory address containing value {}",
1260 &dettid, init_val
1261 );
1262
1263 if !self.cfg.sequentialize_threads {
1264 Ok(guest.inject(call).await?)
1265 } else {
1266 match self.cfg.debug_futex_mode {
1267 BlockingMode::Precise => self.handle_futex_blocking(guest, call, init_val).await,
1268 BlockingMode::Polling => self.handle_futex_polling(guest, call, init_val).await,
1269 BlockingMode::External => self.record_or_replay_blocking(guest, call.into()).await,
1270 }
1271 }
1272 }
1273
1274 pub async fn handle_futex_blocking<G: Guest<Self>>(
1278 &self,
1279 guest: &mut G,
1280 call: syscalls::Futex,
1281 init_val: i32,
1282 ) -> Result<i64, Error> {
1283 let ptr = call.uaddr().unwrap();
1284 let futexid = guest.thread_state().futex_id(
1285 AddrMut::as_raw(ptr),
1286 call.futex_op() & libc::FUTEX_PRIVATE_FLAG != 0,
1287 );
1288 let futex_op = call.futex_op() & libc::FUTEX_CMD_MASK;
1289 let bitset = match futex_op {
1290 libc::FUTEX_WAKE_BITSET | libc::FUTEX_WAIT_BITSET => call.val3() as u32,
1291 _ => u32::MAX,
1292 };
1293 if bitset == 0 {
1294 return Err(Error::Errno(Errno::EINVAL));
1295 }
1296 let dettid = guest.thread_state().dettid;
1297 match futex_op {
1298 libc::FUTEX_WAKE | libc::FUTEX_WAKE_BITSET => {
1299 let num = match futex_action(
1300 guest,
1301 FutexAction::WakeRequest(call.val()),
1302 &futexid,
1303 init_val,
1304 bitset,
1305 )
1306 .await
1307 .expect("futex wake must return value")
1308 {
1309 SchedValue::Value(num) => num,
1310 SchedValue::TimeOut => panic!("impossible, futex wake doesn't have a timeout"),
1311 };
1312 match guest.memory().read_value(ptr) {
1315 Ok(observed) => trace!(
1316 "[detcore, dtid {}] emulated futex wake committed, memory value is {}, expected {}",
1317 &dettid,
1318 observed,
1319 call.val(),
1320 ),
1321 Err(error) => trace!(
1322 "[detcore, dtid {}] skipped post-wake futex memory diagnostic: {}",
1323 &dettid, error,
1324 ),
1325 }
1326 let _ = futex_action(
1327 guest,
1328 FutexAction::WakeFinished(0),
1329 &futexid,
1330 init_val,
1331 bitset,
1332 )
1333 .await;
1334 Ok(num as i64)
1335 }
1336 libc::FUTEX_WAIT | libc::FUTEX_WAIT_BITSET => {
1337 if init_val != call.val() {
1338 info!(
1339 "[detcore, dtid {}] Futex wait running immediately because it will fizzle ({} != {}).",
1340 &dettid,
1341 init_val,
1342 call.val()
1343 );
1344 Err(Error::Errno(Errno::EAGAIN))
1345 } else {
1346 let maybe_timeout_lt = self
1347 .futex_timeout_deadline(guest, call.futex_op(), call.timeout())
1348 .await?;
1349 let ans = futex_action(
1350 guest,
1351 FutexAction::WaitRequest(maybe_timeout_lt),
1352 &futexid,
1353 init_val,
1354 bitset,
1355 )
1356 .await;
1357 let res = if ans != Some(SchedValue::TimeOut) {
1358 let expected = call.val();
1359 match guest.memory().read_value(ptr) {
1362 Ok(observed) => {
1363 trace!(
1364 "[detcore, dtid {}] after (emulated) futex wait, memory value is {}, expected {}",
1365 &dettid, observed, expected,
1366 );
1367 if expected == observed {
1368 debug!(
1369 "WARNING: fishy that the futex value did not change before wakeup. Weird application-level protocol.\n"
1370 );
1371 }
1372 }
1373 Err(error) => trace!(
1374 "[detcore, dtid {}] skipped post-wait futex memory diagnostic: {}",
1375 &dettid, error,
1376 ),
1377 }
1378 Ok(0)
1379 } else {
1380 trace!("[detcore, dtid {}] futex wait timed out", &dettid);
1381 Err(Error::Errno(Errno::ETIMEDOUT))
1382 };
1383 futex_action(guest, FutexAction::WaitFinished, &futexid, init_val, bitset)
1384 .await;
1385 res
1386 }
1387 }
1388 libc::FUTEX_FD => {
1389 panic!("[detcore] refusing to execute FUTEX_FD, which was removed in Linux 2.6.26.")
1390 }
1391 other => {
1392 panic!("[detcore] futex op not handled yet: {}", other);
1393 }
1394 }
1395 }
1396
1397 pub async fn handle_futex_polling<G: Guest<Self>>(
1400 &self,
1401 guest: &mut G,
1402 call: syscalls::Futex,
1403 init_val: i32,
1404 ) -> Result<i64, Error> {
1405 fn make_futex_wake_request(dettid: DetTid) -> Resources {
1406 let mut rsrc = Resources::new(dettid);
1407 rsrc.fyi("futex_wake");
1408 rsrc
1409 }
1410
1411 fn make_futex_wait_request(dettid: DetTid) -> Resources {
1412 let mut rsrc = Resources::new(dettid);
1413 rsrc.insert(ResourceID::InternalIOPolling, Permission::W);
1414 rsrc.fyi("futex_wait");
1415 rsrc
1416 }
1417
1418 let dettid = guest.thread_state().dettid;
1419 let futex_op = call.futex_op() & libc::FUTEX_CMD_MASK;
1420 match futex_op {
1421 libc::FUTEX_WAKE | libc::FUTEX_WAKE_BITSET => {
1422 let rsrc = make_futex_wake_request(dettid);
1423 resource_request(guest, rsrc.clone()).await; let res = guest.inject(call).await;
1425 Ok(res?)
1429 }
1430 libc::FUTEX_WAIT | libc::FUTEX_WAIT_BITSET => {
1431 if init_val != call.val() {
1432 info!(
1433 "[detcore, dtid {}] Futex wait running immediately because it will fizzle ({} != {}).",
1434 dettid,
1435 init_val,
1436 call.val()
1437 );
1438 let res = guest.inject(call).await;
1439 Ok(res?)
1440 } else {
1441 let rsrc = make_futex_wait_request(dettid);
1442 let deadline = self
1443 .futex_timeout_deadline(guest, call.futex_op(), call.timeout())
1444 .await?;
1445 let res =
1446 retry_nonblocking_syscall_with_timeout(guest, call, rsrc, deadline).await?;
1447 trace!(
1448 "[detcore, dtid {}] after futex wait, memory value is {}",
1449 &dettid,
1450 guest.memory().read_value(call.uaddr().unwrap()).unwrap()
1451 );
1452 Ok(res)
1453 }
1454 }
1455 libc::FUTEX_FD => {
1456 panic!("[detcore] refusing to execute FUTEX_FD, which was removed in Linux 2.6.26.")
1457 }
1458 other => {
1459 panic!("[detcore] futex op not handled yet: {}", other);
1460 }
1461 }
1462 }
1463
1464 pub async fn handle_execveat<G: Guest<Self>>(
1466 &self,
1467 guest: &mut G,
1468 call: syscalls::Execveat,
1469 ) -> Result<i64, Error> {
1470 let (old_metadata, old_memory_metadata, table_is_shared, dettid, detpid, old_mm_id) = {
1471 let thread_state = guest.thread_state();
1472 (
1473 Arc::clone(&thread_state.file_metadata),
1474 Arc::clone(&thread_state.memory_metadata),
1475 Arc::strong_count(&thread_state.file_metadata) > 1,
1476 thread_state.dettid,
1477 thread_state.detpid.expect("detpid unset"),
1478 thread_state.mm_id,
1479 )
1480 };
1481 let (new_metadata, closed_open_files, exec_fd_blocking) = {
1482 let metadata = old_metadata.lock().unwrap();
1483 let new_metadata = metadata.for_exec(dettid);
1484 (
1485 new_metadata.clone(),
1486 metadata.open_files_closed_on_exec(table_is_shared),
1487 new_metadata.exec_blocking_overrides(),
1488 )
1489 };
1490 let preserve_exec_fd_status = guest.thread_state().discover_live_file_metadata;
1491
1492 prepare_exec(
1493 guest,
1494 old_mm_id,
1495 if preserve_exec_fd_status {
1496 exec_fd_blocking
1497 } else {
1498 Default::default()
1499 },
1500 )
1501 .await;
1502
1503 let mut released_ports = Vec::new();
1504 for open_file_id in closed_open_files {
1505 if let Some(port) = self.release_port_for_open_file(guest, open_file_id).await {
1506 released_ports.push((open_file_id, port));
1507 }
1508 }
1509
1510 let old_robust_list_head;
1513
1514 {
1515 let thread_state = guest.thread_state_mut();
1516 thread_state.file_metadata = Arc::new(Mutex::new(new_metadata));
1517 thread_state.memory_metadata = Arc::new(Mutex::new(MemoryMetadata::new()));
1518 thread_state.mm_id = old_mm_id.for_exec(detpid);
1519 old_robust_list_head = thread_state.take_robust_list_for_exec();
1520 }
1521
1522 let errno = self.record_or_replay(guest, call).await.unwrap_err();
1524
1525 {
1526 let thread_state = guest.thread_state_mut();
1527 thread_state.file_metadata = old_metadata;
1528 thread_state.memory_metadata = old_memory_metadata;
1529 thread_state.mm_id = old_mm_id;
1530 thread_state.restore_robust_list_after_failed_exec(old_robust_list_head);
1531 }
1532
1533 cancel_exec(guest).await;
1534 for (open_file_id, port) in released_ports {
1535 self.restore_port_for_open_file(guest, open_file_id, port)
1536 .await;
1537 }
1538
1539 if self.cfg.backend_is_kvm && dettid != detpid && errno == Errno::ENOSYS {
1543 tracing::error!(
1544 "[detcore, dtid {dettid}] KVM nonleader exec is unsupported; \
1545 the replacement image did not run"
1546 );
1547 if !self.cfg.panic_on_unsupported_syscalls {
1548 crate::tool_global::report_unsupported_syscall(guest, call.number()).await;
1549 }
1550 return self
1551 .refuse_unserviceable_operation(guest, call.number(), errno)
1552 .await;
1553 }
1554
1555 Err(errno.into())
1556 }
1557
1558 pub async fn handle_sched_yield<G: Guest<Self>>(
1562 &self,
1563 guest: &mut G,
1564 call: syscalls::SchedYield,
1565 ) -> Result<i64, Error> {
1566 if self.cfg.sequentialize_threads {
1567 if self.cfg.chaos && self.cfg.max_timeslice.is_none() {
1581 let change_time = guest.thread_state().thread_logical_time.as_nanos();
1582 let request = Self::random_priority_changepoint_request(guest, change_time);
1583 resource_request(guest, request).await;
1584 } else if !self.cfg.chaos && self.cfg.replay_preemptions_from.is_some() {
1585 if self.cfg.max_timeslice.is_some() {
1586 guest
1587 .thread_state_mut()
1588 .reset_timeslice_for_explicit_yield();
1589 }
1590 let request = Self::sched_yield_request(guest);
1591 resource_request(guest, request).await;
1592 } else if self.cfg.chaos || self.cfg.replay_schedule_from.is_some() {
1593 let request = Self::yield_request(guest);
1594 resource_request(guest, request).await;
1595 } else {
1596 self.end_timeslice_for_sched_yield(guest).await;
1597 }
1598 trace!("sched_yield yielded to the scheduler; NOT performing actual syscall");
1599 Ok(0)
1600 } else {
1601 Ok(self.record_or_replay(guest, call).await?)
1602 }
1603 }
1604
1605 pub async fn handle_wait4<G: Guest<Self>>(
1609 &self,
1610 guest: &mut G,
1611 call: syscalls::Wait4,
1612 ) -> Result<i64, Error> {
1613 let dettid = guest.thread_state().dettid;
1614 let mut rsrc = Resources::new(dettid);
1615 rsrc.insert(ResourceID::InternalIOPolling, Permission::W);
1616 rsrc.fyi("wait4");
1617
1618 validate_wait4_arguments(call.pid(), call.options())?;
1619
1620 let parent = guest.thread_state().detpid.expect("detpid unset");
1621 let selector = if call.options().intersects(
1626 WaitPidFlag::WUNTRACED
1627 | WaitPidFlag::WCONTINUED
1628 | WaitPidFlag::__WCLONE
1629 | WaitPidFlag::__WALL,
1630 ) {
1631 None
1632 } else {
1633 match call.pid() {
1634 pid if pid > 0 => Some(ChildWaitSelector::Exact(DetPid::from_raw(pid))),
1635 -1 => Some(ChildWaitSelector::Any),
1636 0 => process_group(guest, parent)
1637 .await
1638 .map(ChildWaitSelector::ProcessGroup),
1639 pid if pid < -1 => Some(ChildWaitSelector::ProcessGroup(DetPid::from_raw(-pid))),
1640 _ => unreachable!(),
1641 }
1642 };
1643 let spec = selector
1644 .map(|selector| terminal_child_wait_spec(selector, dettid, call.options().bits()));
1645 let complete_lineage = guest.config().backend_tracks_process_children;
1646 let managed_spec = if let Some(spec) = spec {
1647 let (_, has_child) = ready_child_wait(guest, spec).await;
1648 (has_child || complete_lineage).then_some(spec)
1649 } else {
1650 None
1651 };
1652
1653 let value = if call.options().contains(WaitPidFlag::WNOHANG) {
1654 resource_request(guest, rsrc.clone()).await;
1655 info!(
1656 "[dtid {}] Executing non-blocking wait4 in one shot.",
1657 dettid
1658 );
1659 if let Some(spec) = spec {
1660 'select_child: loop {
1661 let (ready, has_child) = ready_child_wait(guest, spec).await;
1662 let Some(child) = ready else {
1663 break if has_child {
1664 0
1665 } else if complete_lineage {
1666 return Err(Errno::ECHILD.into());
1667 } else {
1668 guest.inject_with_retry(call).await?
1669 };
1670 };
1671 let _ = await_exact_child_physical_exit(guest, child).await;
1672 let exact_call = call.with_pid(child.as_raw());
1673 loop {
1674 match guest.inject_with_retry(exact_call).await {
1675 Ok(value) if value != 0 => break 'select_child value,
1676 Ok(_) => yield_once().await,
1677 Err(Errno::ECHILD) => {
1678 let _ = consume_child_wait(guest, child).await;
1679 if child_wait_can_retry_after_stale(spec) {
1680 continue 'select_child;
1681 }
1682 return Err(Errno::ECHILD.into());
1683 }
1684 Err(errno) => return Err(errno.into()),
1685 }
1686 }
1687 }
1688 } else {
1689 guest.inject_with_retry(call).await?
1690 }
1691 } else if let Some(spec) = managed_spec {
1692 {
1693 let blocked_mask = blocked_signal_mask();
1699 let mut stack = guest.stack().await;
1700 let blocked_mask_addr = stack.push(blocked_mask);
1701 let old_mask_addr = stack.reserve::<KernelSigset>();
1702 let action_addr = stack.reserve::<KernelSigaction>();
1703 let _mask_guard = stack.commit()?;
1704 let guest_signal_mask =
1705 block_signals_for_disposition(guest, blocked_mask_addr, old_mask_addr).await?;
1706 let inspect_signal_action = guest
1707 .config()
1708 .backend_requires_thread_directed_process_signals;
1709
1710 let poll_call = call.with_options(call.options() | WaitPidFlag::WNOHANG);
1711 let mut pending_signal = None;
1712 let result: Result<i64, Error> = loop {
1713 let status = wait_for_child_lifecycle(guest, spec).await;
1714 if pending_signal.is_none() {
1715 pending_signal = wait_signal_disposition(
1716 guest,
1717 status,
1718 &guest_signal_mask,
1719 action_addr,
1720 inspect_signal_action,
1721 )
1722 .await?;
1723 }
1724 let (ready, has_child) = ready_child_wait(guest, spec).await;
1725 if let Some(child) = ready {
1726 let _ = await_exact_child_physical_exit(guest, child).await;
1727 match guest.inject_with_retry(call.with_pid(child.as_raw())).await {
1728 Ok(value) => break Ok(value),
1729 Err(Errno::ECHILD) => {
1730 let _ = consume_child_wait(guest, child).await;
1731 if child_wait_can_retry_after_stale(spec) {
1732 let (next_ready, _) = ready_child_wait(guest, spec).await;
1733 if stale_any_wait_must_interrupt(
1734 pending_signal.is_some(),
1735 next_ready,
1736 ) {
1737 break interrupted_child_wait_result(
1738 guest,
1739 call,
1740 pending_signal.expect("signal checked above"),
1741 )
1742 .await;
1743 }
1744 continue;
1745 }
1746 break Err(Errno::ECHILD.into());
1747 }
1748 Err(errno) => break Err(errno.into()),
1749 }
1750 }
1751 if !has_child {
1752 break Err(Errno::ECHILD.into());
1753 }
1754 match guest.inject(poll_call).await {
1755 Ok(value) => {
1756 if value > 0 {
1757 break Ok(value);
1758 }
1759 if let Some(disposition) = pending_signal {
1760 break interrupted_child_wait_result(guest, call, disposition)
1761 .await;
1762 }
1763 }
1764 Err(errno) => break Err(errno.into()),
1765 }
1766 };
1767
1768 restore_signals_after_disposition(guest, old_mask_addr).await?;
1769 result?
1770 }
1771 } else {
1772 retry_nonblocking_syscall(guest, call, rsrc, None).await?
1775 };
1776 let consumed_termination = if value <= 0 {
1777 false
1778 } else if let Some(status) = call.wstatus() {
1779 wait_status_is_termination(guest.memory().read_value(status)?)
1780 } else {
1781 guest
1782 .thread_state()
1783 .has_exited_child_process_cpu_time(DetPid::from_raw(value as i32))
1784 };
1785 if consumed_termination {
1786 guest
1787 .thread_state_mut()
1788 .reap_child_process_cpu_time(DetPid::from_raw(value as i32));
1789 let _ = consume_child_wait(guest, DetPid::from_raw(value as i32)).await;
1790 }
1791 if value > 0
1792 && let Some(rusage) = call.rusage()
1793 {
1794 let usage: libc::rusage = unsafe { std::mem::zeroed() };
1796 guest.memory().write_value(rusage, &usage)?;
1797 }
1798 Ok(value)
1799 }
1800
1801 pub async fn handle_waitid<G: Guest<Self>>(
1807 &self,
1808 guest: &mut G,
1809 mut call: syscalls::Waitid,
1810 ) -> Result<i64, Error> {
1811 let dettid = guest.thread_state().dettid;
1812 let mut rsrc = Resources::new(dettid);
1813 rsrc.insert(ResourceID::InternalIOPolling, Permission::W);
1814 rsrc.fyi("waitid");
1815
1816 let event_options = libc::WEXITED | libc::WSTOPPED | libc::WCONTINUED;
1817 let allowed_options = event_options
1818 | libc::WNOHANG
1819 | libc::WNOWAIT
1820 | libc::__WNOTHREAD
1821 | libc::__WALL
1822 | libc::__WCLONE;
1823 if call.options() & event_options == 0 || call.options() & !allowed_options != 0 {
1824 return Err(Errno::EINVAL.into());
1825 }
1826
1827 if call.info().is_none() {
1832 return Err(Errno::EFAULT.into());
1833 }
1834
1835 let terminal_events_only = call.options() & libc::WEXITED != 0
1838 && call.options() & (libc::WSTOPPED | libc::WCONTINUED | libc::__WCLONE | libc::__WALL)
1839 == 0;
1840
1841 if !terminal_events_only && call.which() == libc::P_PGID as i32 && call.pid() == 0 {
1844 call = call.with_pid(snapshot_process_group(guest.pid())?);
1845 }
1846
1847 let pidfd_nonblocking =
1851 if call.which() == libc::P_PIDFD as i32 && call.options() & libc::WNOHANG == 0 {
1852 resource_request(guest, rsrc.clone()).await;
1853 guest_fd_status_flags(guest.pid(), call.pid())? & libc::O_NONBLOCK != 0
1854 } else {
1855 false
1856 };
1857 if call.which() == libc::P_PIDFD as i32
1858 && call.options() & libc::WNOHANG == 0
1859 && !pidfd_nonblocking
1860 {
1861 return Err(Errno::EOPNOTSUPP.into());
1865 }
1866 let info = call.info().expect("waitid infop checked above");
1867
1868 let empty_info: libc::siginfo_t = unsafe { std::mem::zeroed() };
1877
1878 let terminal_spec = if terminal_events_only {
1882 let selector = match call.which() {
1883 which if which == libc::P_PID as i32 => {
1884 Some(ChildWaitSelector::Exact(DetPid::from_raw(call.pid())))
1885 }
1886 which if which == libc::P_ALL as i32 => Some(ChildWaitSelector::Any),
1887 which if which == libc::P_PGID as i32 => {
1888 let group = if call.pid() == 0 {
1889 process_group(guest, guest.thread_state().detpid.expect("detpid unset"))
1890 .await
1891 } else {
1892 Some(DetPid::from_raw(call.pid()))
1893 };
1894 group.map(ChildWaitSelector::ProcessGroup)
1895 }
1896 _ => None,
1897 };
1898 selector.map(|selector| terminal_child_wait_spec(selector, dettid, call.options()))
1899 } else {
1900 None
1901 };
1902 let complete_lineage = guest.config().backend_tracks_process_children;
1903 let managed_terminal_spec = if let Some(spec) = terminal_spec {
1904 let (_, has_child) = ready_child_wait(guest, spec).await;
1905 (has_child || complete_lineage).then_some(spec)
1906 } else {
1907 None
1908 };
1909
1910 if call.options() & libc::WNOHANG != 0 || pidfd_nonblocking {
1911 if !pidfd_nonblocking {
1912 resource_request(guest, rsrc).await;
1913 }
1914 info!(
1915 "[dtid {}] Executing non-blocking waitid in one shot.",
1916 dettid
1917 );
1918 'select_child: loop {
1919 let selected = if let Some(spec) = terminal_spec {
1920 let (ready, has_child) = ready_child_wait(guest, spec).await;
1921 if ready.is_none() && has_child {
1922 guest.memory().write_value(info, &empty_info)?;
1923 return Ok(0);
1924 }
1925 if ready.is_none() && complete_lineage {
1926 return Err(Errno::ECHILD.into());
1927 }
1928 ready
1929 } else {
1930 None
1931 };
1932 if let Some(child) = selected {
1933 let _ = await_exact_child_physical_exit(guest, child).await;
1934 }
1935 let effective_call = selected.map_or(call, |child| {
1936 call.with_which(libc::P_PID as i32).with_pid(child.as_raw())
1937 });
1938 loop {
1939 guest.memory().write_value(info, &empty_info)?;
1940 let value = match guest.inject_with_retry(effective_call).await {
1941 Ok(value) => value,
1942 Err(Errno::ECHILD) if selected.is_some() => {
1943 let child = selected.expect("selected child checked above");
1944 let _ = consume_child_wait(guest, child).await;
1945 if terminal_spec.is_some_and(child_wait_can_retry_after_stale) {
1946 continue 'select_child;
1947 }
1948 return Err(Errno::ECHILD.into());
1949 }
1950 Err(errno) => return Err(errno.into()),
1951 };
1952 let info_value: libc::siginfo_t = guest.memory().read_value(info)?;
1953 let child_pid = unsafe { info_value.si_pid() };
1954 if child_pid == 0 && selected.is_some() {
1955 yield_once().await;
1956 continue;
1957 }
1958 let consumed = child_pid != 0
1959 && call.options() & libc::WNOWAIT == 0
1960 && waitid_code_is_termination(info_value.si_code);
1961 let result = finish_waitid_result(guest, call, value, info_value)?;
1962 if consumed {
1963 let _ = consume_child_wait(guest, DetPid::from_raw(child_pid)).await;
1964 }
1965 return Ok(result);
1966 }
1967 }
1968 }
1969
1970 {
1971 let blocked_mask = blocked_signal_mask();
1979 let mut stack = guest.stack().await;
1980 let blocked_mask_addr = stack.push(blocked_mask);
1981 let old_mask_addr = stack.reserve::<KernelSigset>();
1982 let action_addr = stack.reserve::<KernelSigaction>();
1983 let _mask_guard = stack.commit()?;
1984 let guest_signal_mask =
1985 block_signals_for_disposition(guest, blocked_mask_addr, old_mask_addr).await?;
1986 let inspect_signal_action = guest
1987 .config()
1988 .backend_requires_thread_directed_process_signals;
1989
1990 let poll_call = call.with_options(call.options() | libc::WNOHANG);
1991 let mut pending_signal = None;
1992 let result: Result<i64, Error> = loop {
1993 let managed_spec = managed_terminal_spec;
2004 let status = if let Some(spec) = managed_spec {
2008 wait_for_child_lifecycle(guest, spec).await
2009 } else {
2010 resource_request(guest, rsrc.clone()).await
2011 };
2012 if pending_signal.is_none() {
2013 pending_signal = wait_signal_disposition(
2014 guest,
2015 status,
2016 &guest_signal_mask,
2017 action_addr,
2018 inspect_signal_action,
2019 )
2020 .await?;
2021 }
2022 let (ready, has_child) = if let Some(spec) = managed_spec {
2023 ready_child_wait(guest, spec).await
2024 } else {
2025 (None, true)
2026 };
2027 if let Some(child) = ready {
2028 let _ = await_exact_child_physical_exit(guest, child).await;
2029 if let Err(error) = guest.memory().write_value(info, &empty_info) {
2030 break Err(error.into());
2031 }
2032 let exact_call = call.with_which(libc::P_PID as i32).with_pid(child.as_raw());
2033 match guest.inject_with_retry(exact_call).await {
2034 Ok(value) => {
2035 let info_value = match guest.memory().read_value(info) {
2036 Ok(value) => value,
2037 Err(error) => break Err(error.into()),
2038 };
2039 break finish_waitid_result(guest, call, value, info_value);
2040 }
2041 Err(Errno::ECHILD) => {
2042 let _ = consume_child_wait(guest, child).await;
2043 if managed_spec.is_some_and(child_wait_can_retry_after_stale) {
2044 let (next_ready, _) =
2045 ready_child_wait(guest, managed_spec.expect("managed spec"))
2046 .await;
2047 if stale_any_wait_must_interrupt(
2048 pending_signal.is_some(),
2049 next_ready,
2050 ) {
2051 break interrupted_child_wait_result(
2052 guest,
2053 call,
2054 pending_signal.expect("signal checked above"),
2055 )
2056 .await;
2057 }
2058 continue;
2059 }
2060 break Err(Errno::ECHILD.into());
2061 }
2062 Err(errno) => break Err(errno.into()),
2063 }
2064 }
2065 if managed_spec.is_some() && !has_child {
2066 break Err(Errno::ECHILD.into());
2067 }
2068
2069 if let Err(error) = guest.memory().write_value(info, &empty_info) {
2070 break Err(error.into());
2071 }
2072 let result = guest.inject(poll_call).await;
2073 match result {
2074 Ok(value) => {
2075 let info_value: libc::siginfo_t = match guest.memory().read_value(info) {
2076 Ok(value) => value,
2077 Err(error) => break Err(error.into()),
2078 };
2079 let child_pid = unsafe { info_value.si_pid() };
2082 match exact_wait_poll_decision(
2083 child_pid != 0,
2084 pending_signal.is_some(),
2085 None,
2086 ) {
2087 ExactWaitPollDecision::ChildReady => {
2088 break finish_waitid_result(guest, call, value, info_value);
2089 }
2090 ExactWaitPollDecision::Interrupted => {
2091 break interrupted_child_wait_result(
2092 guest,
2093 call,
2094 pending_signal.expect("signal checked above"),
2095 )
2096 .await;
2097 }
2098 ExactWaitPollDecision::Retry => {}
2099 ExactWaitPollDecision::AwaitPhysicalExit
2100 | ExactWaitPollDecision::ReapAfterLogicalExit => unreachable!(),
2101 }
2102 if managed_spec.is_some() {
2103 if !has_child {
2104 break Ok(value);
2105 }
2106 continue;
2107 }
2108 rsrc.poll_attempt += 1;
2109 trace!(
2110 "Retry #{} for waitid because no child state is ready",
2111 rsrc.poll_attempt
2112 );
2113 record_retry_event(guest, poll_call).await;
2114 }
2115 Err(Errno::ERESTARTSYS) if pending_signal.is_some() => {
2116 break Err(Errno::EINTR.into());
2117 }
2118 Err(errno) => break Err(errno.into()),
2119 }
2120 };
2121
2122 restore_signals_after_disposition(guest, old_mask_addr).await?;
2123 if result.is_ok() && call.options() & libc::WNOWAIT == 0 {
2124 let info_value: libc::siginfo_t = guest.memory().read_value(info)?;
2125 let child_pid = unsafe { info_value.si_pid() };
2126 if child_pid != 0 && waitid_code_is_termination(info_value.si_code) {
2127 let _ = consume_child_wait(guest, DetPid::from_raw(child_pid)).await;
2128 }
2129 }
2130 result
2131 }
2132 }
2133
2134 pub async fn handle_sched_setaffinity<G: Guest<Self>>(
2138 &self,
2139 guest: &mut G,
2140 call: syscalls::SchedSetaffinity,
2141 ) -> Result<i64, Error> {
2142 let size_bytes = call.len() as usize;
2143 if size_bytes == 0 {
2144 return Err(Errno::EINVAL.into());
2145 }
2146
2147 let mask = call.mask().ok_or(Errno::EFAULT)?;
2148 let mask: Addr<u8> = mask.cast();
2149 let mut requested = [0u8; VIRTUAL_CPUSET_BYTES];
2150 let bytes_to_read = size_bytes.min(VIRTUAL_CPUSET_BYTES);
2151 guest
2152 .memory()
2153 .read_exact(mask, &mut requested[..bytes_to_read])?;
2154 info!(
2155 "Suppressing sched_setaffinity mask {:?}; affinity remains virtual CPU 0",
2156 requested
2157 );
2158 Ok(0)
2159 }
2160
2161 pub async fn handle_sched_getaffinity<G: Guest<Self>>(
2165 &self,
2166 guest: &mut G,
2167 call: syscalls::SchedGetaffinity,
2168 ) -> Result<i64, Error> {
2169 let size_bytes: usize = call.len() as usize;
2170 if size_bytes < VIRTUAL_CPUSET_BYTES
2171 || !size_bytes.is_multiple_of(std::mem::size_of::<libc::c_ulong>())
2172 {
2173 return Err(Errno::EINVAL.into());
2174 }
2175
2176 let mut cpu_set = [0u8; VIRTUAL_CPUSET_BYTES];
2180 cpu_set[0] = 1;
2181
2182 info!(
2183 "Suppressing sched_getaffinity and returning {}-byte virtualized result, {:?}",
2184 VIRTUAL_CPUSET_BYTES, cpu_set
2185 );
2186 if let Some(mask) = call.mask() {
2187 let mask: AddrMut<u8> = mask.cast();
2188 guest.memory().write_exact(mask, &cpu_set)?;
2189 Ok(VIRTUAL_CPUSET_BYTES as i64)
2194 } else {
2195 Err(Error::Errno(Errno::EFAULT))
2196 }
2197 }
2198
2199 pub async fn handle_sched_getparam<G: Guest<Self>>(
2207 &self,
2208 guest: &mut G,
2209 call: syscalls::SchedGetparam,
2210 ) -> Result<i64, Error> {
2211 if let Some(param) = call.param() {
2212 let p = libc::sched_param { sched_priority: 0 };
2213 guest.memory().write_value(param, &p)?;
2214 }
2215 Ok(0)
2216 }
2217
2218 pub async fn handle_sched_rr_get_interval<G: Guest<Self>>(
2225 &self,
2226 guest: &mut G,
2227 call: syscalls::SchedRrGetInterval,
2228 ) -> Result<i64, Error> {
2229 if let Some(tp) = call.tp() {
2230 let t = Timespec {
2231 tv_sec: 0,
2232 tv_nsec: 0,
2233 };
2234 guest.memory().write_value(tp, &t)?;
2235 }
2236 Ok(0)
2237 }
2238
2239 pub async fn handle_sched_getattr<G: Guest<Self>>(
2247 &self,
2248 guest: &mut G,
2249 call: syscalls::SchedGetattr,
2250 ) -> Result<i64, Error> {
2251 if call.flags() != 0 {
2253 return Err(Errno::EINVAL.into());
2254 }
2255 let attr_size = std::mem::size_of::<libc::sched_attr>();
2257 if (call.size() as usize) < attr_size {
2258 return Err(Errno::EINVAL.into());
2259 }
2260 let attr = call.attr().ok_or(Errno::EINVAL)?;
2261 let mut sa: libc::sched_attr = unsafe { std::mem::zeroed() };
2264 sa.size = attr_size as u32;
2265 sa.sched_policy = libc::SCHED_OTHER as u32;
2266 let bytes = unsafe {
2269 std::slice::from_raw_parts(&sa as *const libc::sched_attr as *const u8, attr_size)
2270 };
2271 let dst: AddrMut<u8> = attr.cast();
2272 guest.memory().write_exact(dst, bytes)?;
2273 info!(
2274 "Emulating sched_getattr(pid={}): fixed SCHED_OTHER, nice 0, priority 0",
2275 call.pid()
2276 );
2277 Ok(0)
2278 }
2279
2280 pub async fn handle_sched_setattr<G: Guest<Self>>(
2342 &self,
2343 guest: &mut G,
2344 call: syscalls::SchedSetattr,
2345 ) -> Result<i64, Error> {
2346 if call.flags() != 0 || call.pid() < 0 {
2351 return Err(Errno::EINVAL.into());
2352 }
2353 let attr = call.attr().ok_or(Errno::EINVAL)?;
2354
2355 let declared: u32 = guest
2364 .memory()
2365 .read_value(attr.cast())
2366 .map_err(|_| Errno::EFAULT)?;
2367
2368 fn refuse_too_big<S: MemoryAccess>(mut memory: S, attr: AddrMut<libc::c_void>) -> Error {
2374 let _ = memory.write_value(attr.cast::<u32>(), &SCHED_ATTR_KERNEL_SIZE);
2375 Errno::E2BIG.into()
2376 }
2377
2378 let size = match sched_attr_effective_size(declared) {
2379 Ok(size) => size,
2380 Err(()) => {
2381 let back = call.attr().ok_or(Errno::EINVAL)?;
2382 return Err(refuse_too_big(guest.memory(), back));
2383 }
2384 };
2385
2386 if size > SCHED_ATTR_KERNEL_SIZE {
2392 let base: AddrMut<u8> = call.attr().ok_or(Errno::EINVAL)?.cast();
2393 match scan_tail_is_zeroed(
2394 &guest.memory(),
2395 base,
2396 SCHED_ATTR_KERNEL_SIZE as usize,
2397 (size - SCHED_ATTR_KERNEL_SIZE) as usize,
2398 ) {
2399 TailVerdict::AllZero => {}
2400 TailVerdict::NotZeroed => {
2401 let back = call.attr().ok_or(Errno::EINVAL)?;
2402 return Err(refuse_too_big(guest.memory(), back));
2403 }
2404 TailVerdict::Faulted => return Err(Errno::EFAULT.into()),
2405 }
2406 }
2407
2408 let copied = std::cmp::min(size, SCHED_ATTR_KERNEL_SIZE) as usize;
2412 let mut raw = [0u8; SCHED_ATTR_KERNEL_SIZE as usize];
2413 let base: AddrMut<u8> = call.attr().ok_or(Errno::EINVAL)?.cast();
2414 guest
2415 .memory()
2416 .read_exact(base, &mut raw[..copied])
2417 .map_err(|_| Errno::EFAULT)?;
2418 let field32 = |offset: usize| -> u32 {
2419 u32::from_ne_bytes(raw[offset..offset + 4].try_into().expect("4 bytes"))
2420 };
2421 let field64 = |offset: usize| -> u64 {
2422 u64::from_ne_bytes(raw[offset..offset + 8].try_into().expect("8 bytes"))
2423 };
2424 let fields = SchedAttrFields {
2425 policy: field32(SCHED_ATTR_OFF_POLICY),
2426 sched_flags: field64(SCHED_ATTR_OFF_FLAGS),
2427 priority: field32(SCHED_ATTR_OFF_PRIORITY),
2428 runtime: field64(SCHED_ATTR_OFF_RUNTIME),
2429 deadline: field64(SCHED_ATTR_OFF_DEADLINE),
2430 period: field64(SCHED_ATTR_OFF_PERIOD),
2431 };
2432
2433 validate_sched_attr_before_lookup(size, &fields)?;
2435
2436 if call.pid() != 0 && !thread_is_live(guest, DetTid::from_raw(call.pid())).await {
2447 return Err(Errno::ESRCH.into());
2448 }
2449
2450 validate_sched_attr_after_lookup(&fields)?;
2452
2453 info!(
2454 "Suppressing sched_setattr(pid={}, flags={}); Linux scheduler attributes are virtual",
2455 call.pid(),
2456 call.flags()
2457 );
2458 Ok(0)
2459 }
2460
2461 pub async fn handle_ioprio_set<G: Guest<Self>>(
2469 &self,
2470 _guest: &mut G,
2471 call: syscalls::IoprioSet,
2472 ) -> Result<i64, Error> {
2473 info!(
2474 "Suppressing ioprio_set(which={}, who={}, priority={}); I/O priority is virtual",
2475 call.which(),
2476 call.who(),
2477 call.priority()
2478 );
2479 Ok(0)
2480 }
2481
2482 pub async fn handle_ioprio_get<G: Guest<Self>>(
2489 &self,
2490 _guest: &mut G,
2491 call: syscalls::IoprioGet,
2492 ) -> Result<i64, Error> {
2493 let priority = virtual_ioprio(call.which())?;
2494
2495 info!(
2496 "Emulating ioprio_get(which={}, who={}): fixed priority {}",
2497 call.which(),
2498 call.who(),
2499 priority
2500 );
2501 Ok(priority)
2502 }
2503}
2504
2505struct GuestRobustEffects<'a, G, T> {
2511 guest: &'a mut G,
2512 dettid: DetTid,
2513 defer_owner_death_to_backend: bool,
2514 staged_wakes: Option<&'a mut Vec<RobustListWake>>,
2515 tool: PhantomData<T>,
2516}
2517
2518impl<G, T> robust_list::RobustDeathEffects for GuestRobustEffects<'_, G, T>
2519where
2520 G: Guest<Detcore<T>>,
2521 T: RecordOrReplay,
2522{
2523 fn read_u64(&mut self, address: usize) -> Option<u64> {
2524 let at = Addr::<u64>::from_raw(address)?;
2525 self.guest.memory().read_value::<_, u64>(at).ok()
2526 }
2527
2528 fn read_u32(&mut self, address: usize) -> Option<u32> {
2529 let at = Addr::<u32>::from_raw(address)?;
2530 self.guest.memory().read_value::<_, u32>(at).ok()
2531 }
2532
2533 fn compare_and_swap(
2534 &mut self,
2535 address: usize,
2536 expected: u32,
2537 desired: u32,
2538 ) -> robust_list::FutexCasOutcome {
2539 use robust_list::FutexCasOutcome;
2540
2541 let (Some(read_at), Some(write_at)) = (
2542 Addr::<u32>::from_raw(address),
2543 AddrMut::<u32>::from_raw(address),
2544 ) else {
2545 return FutexCasOutcome::Faulted;
2546 };
2547 let observed = match self.guest.memory().read_value::<_, u32>(read_at) {
2557 Ok(value) => value,
2558 Err(_) => return FutexCasOutcome::Faulted,
2559 };
2560 if observed != expected {
2561 return FutexCasOutcome::Changed(observed);
2562 }
2563 if self.defer_owner_death_to_backend {
2564 debug!(
2571 "[detcore, dtid {}] robust-list owner death: leaving futex word {:#x} for backend exit cleanup",
2572 self.dettid, address,
2573 );
2574 return FutexCasOutcome::Deferred;
2575 }
2576 if self.guest.memory().write_value(write_at, &desired).is_err() {
2580 return FutexCasOutcome::Faulted;
2581 }
2582 debug!(
2587 "[detcore, dtid {}] robust-list owner death: futex word {:#x} {:#x} -> {:#x}",
2588 self.dettid, address, expected, desired,
2589 );
2590 FutexCasOutcome::Stored
2591 }
2592
2593 async fn wake_one(&mut self, address: usize, observed: u32) {
2594 let futexid = self.guest.thread_state().futex_id(address, false);
2597 if let Some(wakes) = self.staged_wakes.as_mut() {
2598 wakes.push(RobustListWake { futex: futexid });
2599 debug!(
2600 "[detcore, dtid {}] staged robust-list owner-death wake until physical exit",
2601 self.dettid,
2602 );
2603 return;
2604 }
2605 let woken = match futex_action(
2606 self.guest,
2607 FutexAction::WakeRequest(1),
2608 &futexid,
2609 observed as i32,
2610 u32::MAX,
2611 )
2612 .await
2613 {
2614 Some(SchedValue::Value(count)) => count,
2615 Some(SchedValue::TimeOut) | None => 0,
2617 };
2618 info!(
2622 "[detcore, dtid {}] robust-list owner death woke {} waiter(s) on futex {:?}",
2623 self.dettid, woken, futexid,
2624 );
2625 let _ = futex_action(
2626 self.guest,
2627 FutexAction::WakeFinished(0),
2628 &futexid,
2629 observed as i32,
2630 u32::MAX,
2631 )
2632 .await;
2633 }
2634}
2635
2636#[cfg(test)]
2637mod tests {
2638 use std::io::Read;
2639 use std::io::Seek;
2640 use std::os::fd::AsRawFd;
2641
2642 use reverie::GlobalRPC;
2643 use reverie::GlobalTool;
2644 use reverie::Tid;
2645 use reverie::Tool;
2646
2647 use super::*;
2648 use crate::config::Config;
2649 use crate::tool_global::GlobalRequest;
2650 use crate::tool_global::GlobalState;
2651 use crate::types::MmId;
2652
2653 struct FailedExecStack;
2654 struct FailedExecStackGuard;
2655
2656 impl Drop for FailedExecStackGuard {
2657 fn drop(&mut self) {}
2658 }
2659
2660 impl reverie::Stack for FailedExecStack {
2661 type StackGuard = FailedExecStackGuard;
2662
2663 fn size(&self) -> usize {
2664 panic!("failed exec must not use the guest stack")
2665 }
2666 fn capacity(&self) -> usize {
2667 panic!("failed exec must not use the guest stack")
2668 }
2669 fn push<'stack, T>(&mut self, _: T) -> Addr<'stack, T> {
2670 panic!("failed exec must not use the guest stack")
2671 }
2672 fn reserve<'stack, T>(&mut self) -> AddrMut<'stack, T> {
2673 panic!("failed exec must not use the guest stack")
2674 }
2675 fn commit(self) -> Result<Self::StackGuard, Errno> {
2676 panic!("failed exec must not use the guest stack")
2677 }
2678 }
2679
2680 struct FailedExecGuest<'a> {
2683 config: &'a Config,
2684 global: &'a GlobalState,
2685 thread: crate::ThreadState<()>,
2686 sender: Tid,
2687 process: Tid,
2688 old_mm: MmId,
2689 errno: Errno,
2690 injections: usize,
2691 requests: Mutex<Vec<GlobalRequest>>,
2692 }
2693
2694 #[reverie::tool]
2695 impl GlobalRPC<GlobalState> for FailedExecGuest<'_> {
2696 async fn send_rpc(
2697 &self,
2698 message: <GlobalState as GlobalTool>::Request,
2699 ) -> <GlobalState as GlobalTool>::Response {
2700 assert_eq!(message.1, self.old_mm, "RPC follows restored exec identity");
2701 self.requests.lock().unwrap().push(message.2.clone());
2702 self.global.receive_rpc(self.sender, message).await
2703 }
2704 fn config(&self) -> &Config {
2705 self.config
2706 }
2707 }
2708
2709 #[reverie::tool]
2710 impl Guest<Detcore> for FailedExecGuest<'_> {
2711 type Memory = reverie::syscalls::LocalMemory;
2712 type Stack = FailedExecStack;
2713
2714 fn tid(&self) -> Tid {
2715 self.sender
2716 }
2717 fn pid(&self) -> Tid {
2718 self.process
2719 }
2720 fn ppid(&self) -> Option<Tid> {
2721 None
2722 }
2723 fn memory(&self) -> Self::Memory {
2724 panic!("failed exec rollback must not read guest memory")
2725 }
2726 fn thread_state(&self) -> &crate::ThreadState<()> {
2727 &self.thread
2728 }
2729 fn thread_state_mut(&mut self) -> &mut crate::ThreadState<()> {
2730 &mut self.thread
2731 }
2732 async fn regs(&mut self) -> libc::user_regs_struct {
2733 panic!("failed exec rollback must not read registers")
2734 }
2735 async fn stack(&mut self) -> Self::Stack {
2736 panic!("failed exec rollback must not use a guest stack")
2737 }
2738 async fn daemonize(&mut self) {
2739 panic!("failed exec rollback must not daemonize")
2740 }
2741 async fn inject<S: SyscallInfo>(&mut self, call: S) -> Result<i64, Errno> {
2742 assert_eq!(call.number(), syscalls::Sysno::execveat);
2743 assert_eq!(
2744 self.thread.mm_id,
2745 self.old_mm.for_exec(self.thread.detpid.unwrap())
2746 );
2747 assert_eq!(self.thread.robust_list_head, None);
2748 self.injections += 1;
2749 Err(self.errno)
2750 }
2751 async fn tail_inject<S: SyscallInfo>(&mut self, _: S) -> reverie::Never {
2752 panic!("failed exec must not retire a live guest")
2753 }
2754 fn set_timer(&mut self, _: reverie::TimerSchedule) -> Result<(), Error> {
2755 panic!("failed exec rollback must not replace the timer")
2756 }
2757 fn set_timer_precise(&mut self, _: reverie::TimerSchedule) -> Result<(), Error> {
2758 panic!("failed exec rollback must not replace the timer")
2759 }
2760 fn read_clock(&mut self) -> Result<u64, Error> {
2761 panic!("failed exec rollback must not sample the clock")
2762 }
2763 }
2764
2765 #[tokio::test]
2766 async fn kvm_nonleader_exec_refusal_preserves_failed_exec_rollback_and_policy() {
2767 let process = DetPid::from_raw(17);
2768 let worker = DetTid::from_raw(18);
2769 for (backend_is_kvm, caller, errno, fail_closed, refused, reported) in [
2770 (true, worker, Errno::ENOSYS, true, true, false),
2771 (true, worker, Errno::ENOSYS, false, false, true),
2772 (true, worker, Errno::ENOENT, true, false, false),
2773 (true, worker, Errno::ENOEXEC, true, false, false),
2774 (true, worker, Errno::EFAULT, true, false, false),
2775 (true, worker, Errno::EACCES, true, false, false),
2776 (true, worker, Errno::EOPNOTSUPP, true, false, false),
2777 (true, process, Errno::ENOSYS, true, false, false),
2778 (false, worker, Errno::ENOSYS, true, false, false),
2779 ] {
2780 let mut report = tempfile::tempfile().unwrap();
2781 let config = Config {
2782 backend_is_kvm,
2783 sequentialize_threads: false,
2784 panic_on_unsupported_syscalls: fail_closed,
2785 exit_on_unsupported_syscall: true,
2786 shutdown_on_unsupported_syscall: false,
2787 unsupported_syscall_report_fd: Some(report.as_raw_fd()),
2788 ..Config::default()
2789 };
2790 let global = GlobalState::init_global_state(&config).await;
2791 let tool = Detcore::new(Tid::from_raw(process.as_raw()), &config);
2792 let mut thread = crate::ThreadState::new(caller, &config, ());
2793 thread.detpid = Some(process);
2794 thread.mm_id = MmId::initial(process);
2795 thread.record_robust_list_head(Some(0x12340));
2796 thread.thread_logical_time.add_syscall_with_cost(123);
2797 let old_time = thread.thread_logical_time.as_nanos();
2798 let old_files = Arc::clone(&thread.file_metadata);
2799 let old_memory = Arc::clone(&thread.memory_metadata);
2800 let old_mm = thread.mm_id;
2801 let mut guest = FailedExecGuest {
2802 config: &config,
2803 global: &global,
2804 thread,
2805 sender: Tid::from_raw(caller.as_raw()),
2806 process: Tid::from_raw(process.as_raw()),
2807 old_mm,
2808 errno,
2809 injections: 0,
2810 requests: Mutex::new(Vec::new()),
2811 };
2812 let result = tool
2813 .handle_execveat(&mut guest, syscalls::Execveat::new())
2814 .await;
2815 match result {
2816 Err(Error::Tool(error)) if refused => assert_eq!(
2817 error
2818 .downcast_ref::<crate::UnsupportedSyscallError>()
2819 .unwrap()
2820 .0,
2821 syscalls::Sysno::execveat
2822 ),
2823 Err(Error::Errno(actual)) if !refused => assert_eq!(actual, errno),
2824 result => panic!("wrong failed-exec policy: {result:?}"),
2825 }
2826 assert_eq!(guest.injections, 1, "backend preflight must run first");
2827 assert_eq!(guest.thread.dettid, caller);
2828 assert_eq!(guest.thread.detpid, Some(process));
2829 assert_eq!(guest.thread.mm_id, old_mm);
2830 assert_eq!(guest.thread.robust_list_head, Some(0x12340));
2831 assert_eq!(guest.thread.thread_logical_time.as_nanos(), old_time);
2832 assert!(Arc::ptr_eq(&guest.thread.file_metadata, &old_files));
2833 assert!(Arc::ptr_eq(&guest.thread.memory_metadata, &old_memory));
2834 let requests = guest.requests.lock().unwrap();
2835 assert!(matches!(requests[0], GlobalRequest::PrepareExec(..)));
2836 assert!(matches!(requests[1], GlobalRequest::CancelExec(..)));
2837 assert_eq!(requests.len(), if reported { 3 } else { 2 });
2838 if reported {
2839 assert!(
2840 matches!(&requests[2], GlobalRequest::ReportUnsupportedSyscall(name) if name == "execveat")
2841 );
2842 }
2843 report.rewind().unwrap();
2844 let mut aggregate = String::new();
2845 report.read_to_string(&mut aggregate).unwrap();
2846 assert_eq!(aggregate, if reported { "execveat\n" } else { "" });
2847 }
2848 }
2849
2850 #[test]
2851 fn kernel_blocked_mask_preserves_libc_signal_membership() {
2852 let mask = blocked_signal_mask();
2853 let mut libc_mask: libc::sigset_t = unsafe { std::mem::zeroed() };
2854 unsafe {
2855 libc::sigfillset(&mut libc_mask);
2856 libc::sigdelset(&mut libc_mask, reverie::PERF_EVENT_SIGNAL as i32);
2857 }
2858 for raw_signal in 1..=KernelSigset::BITS as i32 {
2859 assert_eq!(
2860 signal_is_blocked(&mask, SigWrapper(raw_signal)),
2861 unsafe { libc::sigismember(&libc_mask, raw_signal) == 1 },
2862 "signal {raw_signal} membership changed while converting to the kernel ABI"
2863 );
2864 }
2865 }
2866
2867 #[test]
2868 fn clone_without_child_cleartid_does_not_register_the_pointer() {
2869 assert_eq!(
2870 child_tid_clear_address(CloneFlags::CLONE_CHILD_SETTID, 0x1234),
2871 0
2872 );
2873 }
2874
2875 #[test]
2876 fn clone_with_child_cleartid_registers_the_pointer() {
2877 assert_eq!(
2878 child_tid_clear_address(CloneFlags::CLONE_CHILD_CLEARTID, 0x1234),
2879 0x1234
2880 );
2881 }
2882
2883 #[test]
2884 fn linux_default_dispositions_that_do_not_interrupt_child_waits() {
2885 for signal in [libc::SIGCHLD, libc::SIGCONT, libc::SIGURG, libc::SIGWINCH] {
2886 assert!(signal_default_disposition_does_not_interrupt_child_wait(
2887 SigWrapper(signal)
2888 ));
2889 }
2890 for signal in [libc::SIGALRM, libc::SIGSTOP, libc::SIGUSR1] {
2891 assert!(!signal_default_disposition_does_not_interrupt_child_wait(
2892 SigWrapper(signal)
2893 ));
2894 }
2895 }
2896
2897 #[test]
2898 fn uncatchable_signals_do_not_require_a_sigaction_query() {
2899 assert!(signal_has_uncatchable_default_disposition(SigWrapper(
2900 libc::SIGKILL
2901 )));
2902 assert!(signal_has_uncatchable_default_disposition(SigWrapper(
2903 libc::SIGSTOP
2904 )));
2905 assert!(!signal_has_uncatchable_default_disposition(SigWrapper(
2906 libc::SIGUSR1
2907 )));
2908 }
2909
2910 #[test]
2911 fn waitid_ready_child_wins_when_scheduler_also_reports_a_signal() {
2912 assert_eq!(
2913 exact_wait_poll_decision(true, true, Some(ExactChildWaitState::Running)),
2914 ExactWaitPollDecision::ChildReady
2915 );
2916 assert_eq!(
2917 exact_wait_poll_decision(false, true, Some(ExactChildWaitState::LogicallyExited)),
2918 ExactWaitPollDecision::ReapAfterLogicalExit
2919 );
2920 assert_eq!(
2921 exact_wait_poll_decision(false, true, Some(ExactChildWaitState::PhysicalExitPending)),
2922 ExactWaitPollDecision::AwaitPhysicalExit
2923 );
2924 assert_eq!(
2925 exact_wait_poll_decision(false, true, Some(ExactChildWaitState::Running)),
2926 ExactWaitPollDecision::Interrupted
2927 );
2928 assert_eq!(
2929 exact_wait_poll_decision(false, false, Some(ExactChildWaitState::Running)),
2930 ExactWaitPollDecision::Retry
2931 );
2932 }
2933
2934 #[test]
2935 fn wait4_argument_validation_follows_linux_precedence() {
2936 let valid_bits = [0, 1, 3, 29, 30, 31];
2937 for bit in valid_bits {
2938 let options = WaitPidFlag::from_bits_retain((1_u32 << bit) as libc::c_int);
2939 assert_eq!(validate_wait4_arguments(-1, options), Ok(()), "bit {bit}");
2940 }
2941
2942 for bit in (0..u32::BITS).filter(|bit| !valid_bits.contains(bit)) {
2943 let options = WaitPidFlag::from_bits_retain((1_u32 << bit) as libc::c_int);
2944 assert_eq!(
2945 validate_wait4_arguments(-1, options),
2946 Err(Errno::EINVAL),
2947 "bit {bit}"
2948 );
2949 }
2950
2951 for pid in [libc::pid_t::MIN + 1, -2, 0, 1, libc::pid_t::MAX] {
2954 assert_eq!(
2955 validate_wait4_arguments(pid, WaitPidFlag::empty()),
2956 Ok(()),
2957 "pid {pid}"
2958 );
2959 }
2960 assert_eq!(
2961 validate_wait4_arguments(libc::pid_t::MIN, WaitPidFlag::empty()),
2962 Err(Errno::ESRCH)
2963 );
2964 assert_eq!(
2965 validate_wait4_arguments(libc::pid_t::MIN, WaitPidFlag::from_bits_retain(0x10)),
2966 Err(Errno::EINVAL),
2967 "invalid options must win over the INT_MIN selector"
2968 );
2969 }
2970
2971 #[test]
2972 fn stale_any_child_preserves_interrupt_until_no_ready_child_remains() {
2973 let next_child = DetPid::from_raw(200);
2974
2975 assert!(
2976 !stale_any_wait_must_interrupt(true, Some(next_child)),
2977 "another ready child must retain child-ready precedence"
2978 );
2979 assert!(
2980 stale_any_wait_must_interrupt(true, None),
2981 "a pending signal must interrupt before the wait parks again"
2982 );
2983 assert!(!stale_any_wait_must_interrupt(false, None));
2984 }
2985
2986 #[test]
2987 fn ioprio_query_reports_fixed_raw_and_effective_defaults() {
2988 assert_eq!(virtual_ioprio(IOPRIO_WHO_PROCESS), Ok(0));
2989 assert_eq!(
2990 virtual_ioprio(IOPRIO_WHO_PGRP),
2991 Ok(i64::from(IOPRIO_DEFAULT_EFFECTIVE))
2992 );
2993 assert_eq!(
2994 virtual_ioprio(IOPRIO_WHO_USER),
2995 Ok(i64::from(IOPRIO_DEFAULT_EFFECTIVE))
2996 );
2997 assert_eq!(virtual_ioprio(0), Err(Errno::EINVAL));
2998 assert_eq!(virtual_ioprio(4), Err(Errno::EINVAL));
2999 }
3000
3001 #[test]
3002 fn waitid_siginfo_canonicalization_clears_only_cpu_accounting() {
3003 let mut info: libc::siginfo_t = unsafe { std::mem::zeroed() };
3004 info.si_signo = libc::SIGCHLD;
3005 info.si_code = libc::CLD_EXITED;
3006 let fields = unsafe {
3009 &mut (*(std::ptr::addr_of_mut!(info)).cast::<WaitidSiginfoHead>())
3010 .fields
3011 .sigchld
3012 };
3013 fields.pid = 123;
3014 fields.uid = 456;
3015 fields.status = 7;
3016 fields.utime = 8;
3017 fields.stime = 9;
3018
3019 canonicalize_waitid_siginfo(&mut info);
3020
3021 assert_eq!(info.si_signo, libc::SIGCHLD);
3022 assert_eq!(info.si_code, libc::CLD_EXITED);
3023 assert_eq!(unsafe { info.si_pid() }, 123);
3024 assert_eq!(unsafe { info.si_uid() }, 456);
3025 assert_eq!(unsafe { info.si_status() }, 7);
3026 assert_eq!(unsafe { info.si_utime() }, 0);
3027 assert_eq!(unsafe { info.si_stime() }, 0);
3028 }
3029
3030 #[test]
3031 fn wait_status_rollup_only_accepts_process_termination() {
3032 assert!(wait_status_is_termination(0));
3033 assert!(wait_status_is_termination(libc::SIGTERM));
3034 assert!(!wait_status_is_termination((libc::SIGSTOP << 8) | 0x7f));
3035 assert!(!wait_status_is_termination(0xffff));
3036
3037 assert!(waitid_code_is_termination(libc::CLD_EXITED));
3038 assert!(waitid_code_is_termination(libc::CLD_KILLED));
3039 assert!(waitid_code_is_termination(libc::CLD_DUMPED));
3040 assert!(!waitid_code_is_termination(libc::CLD_STOPPED));
3041 assert!(!waitid_code_is_termination(libc::CLD_CONTINUED));
3042 assert!(!waitid_code_is_termination(libc::CLD_TRAPPED));
3043 }
3044
3045 #[test]
3046 fn futex_timeout_units_and_modes_match_linux() {
3047 let timeout = Timespec {
3048 tv_sec: 2,
3049 tv_nsec: 3,
3050 };
3051 assert_eq!(
3052 parse_futex_timeout(libc::FUTEX_WAIT, timeout),
3053 Ok(FutexTimeout::Relative(2_000_000_003))
3054 );
3055 assert_eq!(
3056 parse_futex_timeout(libc::FUTEX_WAIT_BITSET, timeout),
3057 Ok(FutexTimeout::Absolute(LogicalTime::from_nanos(
3058 2_000_000_003
3059 )))
3060 );
3061 assert_eq!(
3065 parse_futex_timeout(libc::FUTEX_WAIT_BITSET | libc::FUTEX_PRIVATE_FLAG, timeout),
3066 Ok(FutexTimeout::Absolute(LogicalTime::from_nanos(
3067 2_000_000_003
3068 )))
3069 );
3070 assert_eq!(
3071 parse_futex_timeout(libc::FUTEX_WAIT | libc::FUTEX_PRIVATE_FLAG, timeout),
3072 Ok(FutexTimeout::Relative(2_000_000_003))
3073 );
3074 }
3075
3076 #[test]
3077 fn absolute_futex_timeout_is_rebased_to_logical_time() {
3078 let logical_now = LogicalTime::from_secs(100);
3079 let clock_now = LogicalTime::from_secs(5_000);
3080 let deadline = clock_now + Duration::from_millis(100);
3081 assert_eq!(
3082 rebase_absolute_timeout(deadline, clock_now, logical_now),
3083 logical_now + Duration::from_millis(100)
3084 );
3085 assert_eq!(
3086 rebase_absolute_timeout(
3087 clock_now - LogicalTime::from_nanos(1),
3088 clock_now,
3089 logical_now
3090 ),
3091 logical_now
3092 );
3093 }
3094
3095 #[test]
3096 fn absolute_futex_timeout_detects_host_and_logical_clock_domains() {
3097 let host_monotonic_now = LogicalTime::from_secs(374_766);
3098 let logical_now = LogicalTime::from_secs(1_640_995_199);
3099 let delta = Duration::from_millis(100);
3100
3101 assert!(absolute_timeout_uses_host_clock(
3102 host_monotonic_now + delta,
3103 host_monotonic_now,
3104 logical_now
3105 ));
3106 assert!(!absolute_timeout_uses_host_clock(
3107 logical_now + delta,
3108 host_monotonic_now,
3109 logical_now
3110 ));
3111
3112 let host_realtime_now = LogicalTime::from_secs(1_785_142_800);
3113 assert!(absolute_timeout_uses_host_clock(
3114 host_realtime_now + delta,
3115 host_realtime_now,
3116 logical_now
3117 ));
3118 }
3119
3120 #[test]
3121 fn futex_timeout_rejects_invalid_timespecs() {
3122 assert_eq!(
3123 parse_futex_timeout(
3124 libc::FUTEX_WAIT,
3125 Timespec {
3126 tv_sec: -1,
3127 tv_nsec: 0,
3128 },
3129 ),
3130 Err(Errno::EINVAL)
3131 );
3132 assert_eq!(
3133 parse_futex_timeout(
3134 libc::FUTEX_WAIT_BITSET,
3135 Timespec {
3136 tv_sec: 0,
3137 tv_nsec: 1_000_000_000,
3138 },
3139 ),
3140 Err(Errno::EINVAL)
3141 );
3142 }
3143
3144 fn plain_attr() -> SchedAttrFields {
3151 SchedAttrFields {
3152 policy: 0,
3153 sched_flags: 0,
3154 priority: 0,
3155 runtime: 0,
3156 deadline: 0,
3157 period: 0,
3158 }
3159 }
3160
3161 #[test]
3162 fn size_zero_is_a_well_formed_ver0_request() {
3163 assert_eq!(sched_attr_effective_size(0), Ok(SCHED_ATTR_SIZE_VER0));
3166 }
3167
3168 #[test]
3169 fn size_below_ver0_or_past_a_page_is_too_big() {
3170 assert_eq!(sched_attr_effective_size(1), Err(()));
3172 assert_eq!(sched_attr_effective_size(SCHED_ATTR_SIZE_VER0 - 1), Err(()));
3173 assert_eq!(sched_attr_effective_size(SCHED_ATTR_MAX_SIZE + 1), Err(()));
3174 }
3175
3176 #[test]
3177 fn size_from_ver0_through_one_page_is_accepted_unchanged() {
3178 for size in [
3180 SCHED_ATTR_SIZE_VER0,
3181 SCHED_ATTR_SIZE_VER1,
3182 SCHED_ATTR_SIZE_VER1 + 1,
3183 SCHED_ATTR_MAX_SIZE,
3184 ] {
3185 assert_eq!(sched_attr_effective_size(size), Ok(size), "size {}", size);
3186 }
3187 }
3188
3189 #[test]
3190 fn sched_ext_is_a_valid_policy_and_sched_iso_is_not() {
3191 assert!(is_valid_sched_policy(SCHED_EXT), "SCHED_EXT is accepted");
3194 assert!(!is_valid_sched_policy(4), "SCHED_ISO is reserved");
3195 for policy in [0, 1, 2, 3, 5, 6] {
3196 assert!(is_valid_sched_policy(policy), "policy {}", policy);
3197 }
3198 for policy in [8, 99] {
3200 assert!(!is_valid_sched_policy(policy), "policy {}", policy);
3201 }
3202 }
3203
3204 #[test]
3216 fn util_clamp_size_rule_is_decided_before_the_pid_lookup() {
3217 let mut attr = plain_attr();
3219 attr.sched_flags = SCHED_FLAG_UTIL_CLAMP_MIN;
3220 assert_eq!(
3221 validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER0, &attr),
3222 Err(Errno::EINVAL)
3223 );
3224 assert_eq!(
3225 validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER1, &attr),
3226 Ok(())
3227 );
3228 assert_eq!(validate_sched_attr_after_lookup(&attr), Ok(()));
3230 }
3231
3232 #[test]
3233 fn a_negative_policy_is_decided_before_the_pid_lookup() {
3234 let mut attr = plain_attr();
3237 attr.policy = 0x8000_0000;
3238 assert_eq!(
3239 validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER0, &attr),
3240 Err(Errno::EINVAL)
3241 );
3242 attr.sched_flags = SCHED_FLAG_KEEP_POLICY;
3243 assert_eq!(
3244 validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER0, &attr),
3245 Err(Errno::EINVAL),
3246 "KEEP_POLICY must not hide a negative policy"
3247 );
3248 }
3249
3250 #[test]
3251 fn policy_and_flag_rules_are_decided_after_the_pid_lookup() {
3252 let mut bad_policy = plain_attr();
3255 bad_policy.policy = 99;
3256 assert_eq!(
3257 validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER0, &bad_policy),
3258 Ok(()),
3259 "an undefined policy must survive the near side so ESRCH can win"
3260 );
3261 assert_eq!(
3262 validate_sched_attr_after_lookup(&bad_policy),
3263 Err(Errno::EINVAL)
3264 );
3265
3266 let mut bad_flag = plain_attr();
3267 bad_flag.sched_flags = 0x80;
3268 assert_eq!(
3269 validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER0, &bad_flag),
3270 Ok(()),
3271 "an undefined sched_flags bit must survive the near side"
3272 );
3273 assert_eq!(
3274 validate_sched_attr_after_lookup(&bad_flag),
3275 Err(Errno::EINVAL)
3276 );
3277 }
3278
3279 #[test]
3280 fn keep_policy_makes_the_policy_field_irrelevant() {
3281 let mut attr = plain_attr();
3286 attr.policy = 99;
3287 assert_eq!(validate_sched_attr_after_lookup(&attr), Err(Errno::EINVAL));
3288 attr.sched_flags = SCHED_FLAG_KEEP_POLICY;
3289 assert_eq!(validate_sched_attr_after_lookup(&attr), Ok(()));
3290 }
3291
3292 #[test]
3293 fn defined_sched_flags_bits_are_accepted_and_undefined_ones_are_not() {
3294 let mut attr = plain_attr();
3298 attr.sched_flags = 0x80;
3299 assert_eq!(validate_sched_attr_after_lookup(&attr), Err(Errno::EINVAL));
3300 for flag in [
3301 SCHED_FLAG_RESET_ON_FORK,
3302 SCHED_FLAG_RECLAIM,
3303 SCHED_FLAG_DL_OVERRUN,
3304 SCHED_FLAG_KEEP_PARAMS,
3305 SCHED_FLAG_KEEP_POLICY,
3306 ] {
3307 let mut ok = plain_attr();
3308 ok.sched_flags = flag;
3309 assert_eq!(
3310 validate_sched_attr_after_lookup(&ok),
3311 Ok(()),
3312 "flag {:#x}",
3313 flag
3314 );
3315 }
3316 let mut all = plain_attr();
3320 all.sched_flags = SCHED_FLAG_ALL;
3321 assert_eq!(
3322 validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER0, &all),
3323 Err(Errno::EINVAL)
3324 );
3325 assert_eq!(
3326 validate_sched_attr_before_lookup(SCHED_ATTR_SIZE_VER1, &all),
3327 Ok(())
3328 );
3329 assert_eq!(validate_sched_attr_after_lookup(&all), Ok(()));
3330 }
3331
3332 #[test]
3333 fn priority_must_agree_with_the_policy() {
3334 for policy in [0u32, 3, 5] {
3338 let mut attr = plain_attr();
3339 attr.policy = policy;
3340 attr.priority = 1;
3341 assert_eq!(
3342 validate_sched_attr_after_lookup(&attr),
3343 Err(Errno::EINVAL),
3344 "policy {} with priority 1",
3345 policy
3346 );
3347 attr.priority = 0;
3348 assert_eq!(
3349 validate_sched_attr_after_lookup(&attr),
3350 Ok(()),
3351 "policy {} with priority 0",
3352 policy
3353 );
3354 }
3355 for policy in [SCHED_FIFO, SCHED_RR] {
3356 let mut attr = plain_attr();
3357 attr.policy = policy;
3358 attr.priority = 0;
3359 assert_eq!(
3360 validate_sched_attr_after_lookup(&attr),
3361 Err(Errno::EINVAL),
3362 "rt policy {} with priority 0",
3363 policy
3364 );
3365 attr.priority = 1;
3369 assert_eq!(validate_sched_attr_after_lookup(&attr), Ok(()));
3370 }
3371 }
3372
3373 #[test]
3374 fn a_priority_past_max_rt_prio_is_refused() {
3375 let mut attr = plain_attr();
3378 attr.policy = SCHED_FIFO;
3379 attr.priority = MAX_RT_PRIO;
3380 assert_eq!(validate_sched_attr_after_lookup(&attr), Err(Errno::EINVAL));
3381 attr.priority = MAX_RT_PRIO - 1;
3382 assert_eq!(validate_sched_attr_after_lookup(&attr), Ok(()));
3383 }
3384
3385 #[test]
3386 fn deadline_parameters_are_checked() {
3387 let mut zeroed = plain_attr();
3390 zeroed.policy = SCHED_DEADLINE;
3391 assert_eq!(
3392 validate_sched_attr_after_lookup(&zeroed),
3393 Err(Errno::EINVAL),
3394 "all-zero deadline parameters"
3395 );
3396
3397 let mut inverted = plain_attr();
3398 inverted.policy = SCHED_DEADLINE;
3399 inverted.runtime = 90_000_000;
3400 inverted.deadline = 30_000_000;
3401 inverted.period = 30_000_000;
3402 assert_eq!(
3403 validate_sched_attr_after_lookup(&inverted),
3404 Err(Errno::EINVAL),
3405 "runtime > deadline"
3406 );
3407
3408 let mut sane = plain_attr();
3411 sane.policy = SCHED_DEADLINE;
3412 sane.runtime = 10_000_000;
3413 sane.deadline = 30_000_000;
3414 sane.period = 30_000_000;
3415 assert_eq!(validate_sched_attr_after_lookup(&sane), Ok(()));
3416 }
3417
3418 #[test]
3419 fn deadline_parameter_edges_follow_checkparam_dl() {
3420 assert!(!deadline_params_are_valid(1 << 20, 0, 0));
3422 assert!(!deadline_params_are_valid((1 << 10) - 1, 1 << 20, 1 << 20));
3424 assert!(deadline_params_are_valid(1 << 10, 1 << 20, 1 << 20));
3425 assert!(!deadline_params_are_valid(1 << 20, 1 << 63, 0));
3427 assert!(!deadline_params_are_valid(1 << 20, 1 << 20, 1 << 63));
3428 assert!(deadline_params_are_valid(1 << 20, 1 << 21, 0));
3431 assert!(!deadline_params_are_valid(1 << 20, 1 << 22, 1 << 21));
3433 }
3434
3435 #[test]
3436 fn the_size_written_back_on_refusal_is_the_kernel_struct_not_the_libc_mirror() {
3437 assert_eq!(SCHED_ATTR_KERNEL_SIZE, 56);
3441 assert_eq!(std::mem::size_of::<libc::sched_attr>(), 48);
3442 assert_ne!(
3443 SCHED_ATTR_KERNEL_SIZE as usize,
3444 std::mem::size_of::<libc::sched_attr>()
3445 );
3446 }
3447
3448 struct BoundedMemory {
3451 bytes: Vec<u8>,
3452 readable_len: usize,
3453 reads: std::cell::RefCell<Vec<usize>>,
3454 }
3455
3456 impl reverie::syscalls::MemoryAccess for BoundedMemory {
3457 fn read_vectored(
3458 &self,
3459 read_from: &[std::io::IoSlice<'_>],
3460 write_to: &mut [std::io::IoSliceMut<'_>],
3461 ) -> Result<usize, Errno> {
3462 let start = read_from[0].as_ptr() as usize;
3463 let want = read_from.iter().map(|slice| slice.len()).sum::<usize>();
3464 self.reads.borrow_mut().push(want);
3465 if start >= self.readable_len {
3466 return Err(Errno::EFAULT);
3467 }
3468 let avail = (self.readable_len - start).min(want);
3469 let mut copied = 0;
3470 for out in write_to.iter_mut() {
3471 if copied == avail {
3472 break;
3473 }
3474 let take = out.len().min(avail - copied);
3475 out[..take].copy_from_slice(&self.bytes[start + copied..start + copied + take]);
3476 copied += take;
3477 }
3478 Ok(copied)
3479 }
3480
3481 fn write_vectored(
3482 &mut self,
3483 _read_from: &[std::io::IoSlice<'_>],
3484 _write_to: &mut [std::io::IoSliceMut<'_>],
3485 ) -> Result<usize, Errno> {
3486 unimplemented!("this fixture never writes")
3487 }
3488 }
3489
3490 fn bounded(bytes: Vec<u8>, readable_len: usize) -> BoundedMemory {
3491 BoundedMemory {
3492 bytes,
3493 readable_len,
3494 reads: std::cell::RefCell::new(Vec::new()),
3495 }
3496 }
3497
3498 fn tail_base() -> AddrMut<'static, u8> {
3499 AddrMut::<u8>::from_raw(1).expect("nonzero base")
3500 }
3501
3502 #[test]
3510 fn a_nonzero_byte_before_a_fault_is_e2big_not_efault() {
3511 let off = SCHED_ATTR_KERNEL_SIZE as usize;
3512 let mut bytes = vec![0u8; off + 4096];
3513 bytes[off + 1] = 0xAA; let readable = off + 512; let memory = bounded(bytes, readable);
3516 assert_eq!(
3517 scan_tail_is_zeroed(&memory, tail_base(), off, 4096),
3518 TailVerdict::NotZeroed,
3519 "a non-zero byte the scan reaches first must win over a later fault"
3520 );
3521 }
3522
3523 #[test]
3526 fn a_fault_before_any_nonzero_byte_is_efault() {
3527 let off = SCHED_ATTR_KERNEL_SIZE as usize;
3528 let bytes = vec![0u8; off + 4096];
3529 let memory = bounded(bytes, off + 512);
3530 assert_eq!(
3531 scan_tail_is_zeroed(&memory, tail_base(), off, 4096),
3532 TailVerdict::Faulted,
3533 "an unreadable byte reached before anything non-zero must be EFAULT"
3534 );
3535 }
3536
3537 #[test]
3546 fn tail_reads_are_never_small_enough_to_bypass_guest_protection() {
3547 let off = SCHED_ATTR_KERNEL_SIZE as usize;
3548 for tail_len in 1..=16usize {
3549 let bytes = vec![0u8; off + tail_len + 64];
3550 let memory = bounded(bytes, off + tail_len + 64);
3551 assert_eq!(
3552 scan_tail_is_zeroed(&memory, tail_base(), off, tail_len),
3553 TailVerdict::AllZero
3554 );
3555 let reads = memory.reads.borrow().clone();
3556 assert!(!reads.is_empty(), "tail_len {tail_len} issued no read");
3557 for length in reads {
3558 assert!(
3559 length > std::mem::size_of::<u64>(),
3560 "tail_len {tail_len} issued a {length}-byte read, which safeptrace \
3561 serves with PTRACE_PEEKDATA and which therefore bypasses guest \
3562 page protections"
3563 );
3564 }
3565 }
3566 }
3567
3568 #[test]
3575 fn keep_policy_validates_against_the_virtual_current_policy() {
3576 let attr = SchedAttrFields {
3577 policy: 99, sched_flags: SCHED_FLAG_KEEP_POLICY,
3579 priority: 1,
3580 runtime: 0,
3581 deadline: 0,
3582 period: 0,
3583 };
3584 assert_eq!(
3585 validate_sched_attr_after_lookup(&attr),
3586 Err(Errno::EINVAL),
3587 "priority 1 under the virtual SCHED_OTHER current policy must be refused"
3588 );
3589
3590 let ok = SchedAttrFields {
3593 priority: 0,
3594 ..attr
3595 };
3596 assert_eq!(
3597 validate_sched_attr_after_lookup(&ok),
3598 Ok(()),
3599 "KEEP_POLICY must still ignore the policy field itself"
3600 );
3601 }
3602
3603 #[test]
3606 fn the_virtual_current_policy_is_what_sched_getattr_reports() {
3607 assert_eq!(
3608 VIRTUAL_CURRENT_POLICY,
3609 libc::SCHED_OTHER as u32,
3610 "handle_sched_getattr writes SCHED_OTHER into sched_policy for every thread; \
3611 KEEP_POLICY must substitute that same value"
3612 );
3613 }
3614}