Skip to main content

detcore/syscalls/
signal.rs

1/*
2 * Copyright (c) Meta Platforms, Inc. and affiliates.
3 * All rights reserved.
4 *
5 * This source code is licensed under the BSD-style license found in the
6 * LICENSE file in the root directory of this source tree.
7 */
8
9//! System calls dealing with signals.
10
11use std::time::Duration;
12
13use detcore_model::schedule::SigWrapper;
14use nix::sys::signal::Signal;
15use reverie::Errno;
16use reverie::Error;
17use reverie::Guest;
18use reverie::Stack;
19use reverie::syscalls;
20use reverie::syscalls::Addr;
21use reverie::syscalls::AddrMut;
22use reverie::syscalls::MemoryAccess;
23use reverie::syscalls::Timespec;
24use tracing::info;
25
26use crate::Detcore;
27use crate::record_or_replay::RecordOrReplay;
28use crate::resources::Permission;
29use crate::resources::ResourceID;
30use crate::resources::Resources;
31use crate::syscalls::helpers::retry_nonblocking_syscall_with_timeout;
32use crate::syscalls::threads::KERNEL_SIGSET_SIZE;
33use crate::syscalls::threads::KernelSigaction;
34use crate::syscalls::threads::KernelSigset;
35use crate::tool_global::ResumeStatus;
36use crate::tool_global::alarm_remaining;
37use crate::tool_global::notify_signal_pending;
38use crate::tool_global::register_alarm;
39use crate::tool_global::resolve_kill_targets;
40use crate::tool_global::resource_request;
41use crate::tool_global::thread_observe_time;
42use crate::types::DetPid;
43use crate::types::DetTid;
44use crate::types::LogicalTime;
45
46/// Signals hermit takes from the guest's namespace, and what a guest loses.
47///
48/// ⚠️ ENUMERATED BY MEASUREMENT, NOT BY READING (2026-08-25). Every signal 1..64
49/// was run under two delivery paths -- self-directed `raise` and a sibling
50/// thread's `pthread_kill` -- with a native run as the control for each. Exactly
51/// TWO differ from native, and they fail in different ways at different points:
52///
53///   SIGSTKFLT (16)  `rt_sigaction` is NO-OPED below, so the handler is never
54///                   installed and the default disposition terminates the guest:
55///                   observed exit 144 (= 128 + 16).
56///   SIGTRAP    (5)  `rt_sigaction` passes through and the handler IS installed,
57///                   but ptrace consumes every SIGTRAP (syscall stops, seccomp
58///                   stops, breakpoints), so the handler never runs. ⚠️ THE GUEST
59///                   THEN EXITS 0 WITH NO DIAGNOSTIC AT ALL -- a clean pass that
60///                   behaved differently from native, which no cell can catch.
61///
62/// Everything else in 1..31 matches native exactly; 9/19 and 32/33 refuse
63/// `sigaction` natively too and are not hermit's. Realtime 34..64 fail by a
64/// different mechanism (they cannot be represented at the reverie ptrace
65/// boundary) and are tracked separately -- they are not appropriation.
66const APPROPRIATED_SIGNALS: [(i32, &str); 2] = [
67    (
68        libc::SIGTRAP,
69        "ptrace consumes every SIGTRAP (syscall/seccomp stops, breakpoints)",
70    ),
71    (
72        libc::SIGSTKFLT,
73        "reverie uses it as PERF_EVENT_SIGNAL, the PMU preemption timer",
74    ),
75];
76
77/// Say, once per installation, that a guest handler will never run.
78///
79/// ⚠️ WHY A DIAGNOSTIC AND NOT A REFUSAL. Returning `EINVAL` from `sigaction`
80/// was considered and deliberately rejected for SIGSTKFLT -- see the comment at
81/// the no-op below: the Go runtime registers that handler, and refusing would
82/// break every Go guest at startup. That reasoning generalises: installing a
83/// handler defensively is common, actually raising these signals is rare, so
84/// refusal breaks MORE programs than the current behaviour. The contract
85/// question of what a guest is owed here is genuinely open.
86///
87/// What is NOT open is that hermit currently says NOTHING. This line commits to
88/// no policy, breaks no conforming program, and turns a silent wrong answer into
89/// a visible one -- the same reasoning as the `HERMIT_INTERNAL_FAILURE` marker.
90/// The decision, separated from the reporting so a test can exercise THIS and
91/// not a copy of it. A unit test that re-implements a predicate keeps passing
92/// when the real one is gutted -- measured in this project's own pipe work,
93/// where deleting the production wiring left every unit test green.
94fn appropriated_reason(signum: i32, handler: u64) -> Option<&'static str> {
95    // SIG_DFL (0) and SIG_IGN (1) lose nothing: the guest is not asking to be
96    // called back, so there is no expectation to disappoint.
97    if handler <= 1 {
98        return None;
99    }
100    APPROPRIATED_SIGNALS
101        .iter()
102        .find(|(s, _)| *s == signum)
103        .map(|(_, why)| *why)
104}
105
106fn warn_appropriated_signal(signum: i32, handler: u64) {
107    if let Some(why) = appropriated_reason(signum, handler) {
108        tracing::warn!(
109            "HERMIT_APPROPRIATED_SIGNAL signum={signum} effect=handler-installed-but-never-invoked reason={why}"
110        );
111    }
112}
113
114// NB: the kernel uses an eight-byte signal mask in its raw signal syscalls on
115// x86_64. `libc::sigset_t` is the 128-byte userspace wrapper type and must not
116// be used to access these buffers. See:
117// https://elixir.bootlin.com/linux/latest/source/include/uapi/asm-generic/signal.h#L75
118fn validate_kernel_sigset_size(sigsetsize: usize) -> Result<(), Errno> {
119    if sigsetsize == KERNEL_SIGSET_SIZE {
120        Ok(())
121    } else {
122        Err(Errno::EINVAL)
123    }
124}
125
126fn without_perf_event_signal(mask: KernelSigset) -> KernelSigset {
127    let bit = (reverie::PERF_EVENT_SIGNAL as u32) - 1;
128    mask & !(1_u64 << bit)
129}
130
131/// Read one raw kernel signal mask while preserving the kernel's user-access check.
132///
133/// `safeptrace::Stopped::read` deliberately uses `PTRACE_PEEKDATA` for reads of
134/// eight bytes or less. That operation can read a `PROT_NONE` page, unlike the
135/// kernel's `copy_from_user`, so using `MemoryAccess::read_value` alone would
136/// turn an `EFAULT` from a raw signal syscall into success. An invalid `how`
137/// value makes `rt_sigprocmask` copy exactly the kernel-sized input and then
138/// return `EINVAL` without changing the mask. Use that as a permission probe,
139/// then read the already-validated word while the guest is stopped.
140pub(super) async fn read_kernel_sigset<G, T>(
141    guest: &mut G,
142    address: Addr<'_, libc::sigset_t>,
143) -> Result<KernelSigset, Error>
144where
145    G: Guest<Detcore<T>>,
146    T: RecordOrReplay,
147{
148    let validation = syscalls::RtSigprocmask::new()
149        .with_how(-1)
150        .with_set(Some(address))
151        .with_oldset(None)
152        .with_sigsetsize(KERNEL_SIGSET_SIZE);
153    match guest.inject(validation).await {
154        Err(Errno::EINVAL) => {}
155        Err(errno) => return Err(errno.into()),
156        Ok(_) => {
157            // Both Linux and the KVM syscall implementation reject an unknown
158            // operation. Success means the backend did not validate the probe.
159            return Err(Errno::EIO.into());
160        }
161    }
162    Ok(guest.memory().read_value(address.cast())?)
163}
164
165/// Validate the entire action through the kernel before copying it privately.
166/// A partial `read_value` can fall back to ptrace for its final eight bytes,
167/// which would read through a protected page. SIGKILL accepts no new action,
168/// but both Linux and KVM copy the complete input before rejecting it.
169async fn read_kernel_sigaction<G, T>(
170    guest: &mut G,
171    address: Addr<'_, libc::sigaction>,
172) -> Result<KernelSigaction, Error>
173where
174    G: Guest<Detcore<T>>,
175    T: RecordOrReplay,
176{
177    let validation = syscalls::RtSigaction::new()
178        .with_signum(libc::SIGKILL)
179        .with_action(Some(address))
180        .with_old_action(None)
181        .with_sigsetsize(KERNEL_SIGSET_SIZE);
182    match guest.inject(validation).await {
183        Err(Errno::EINVAL) => {}
184        Err(errno) => return Err(errno.into()),
185        Ok(_) => return Err(Errno::EIO.into()),
186    }
187    Ok(guest.memory().read_value(address.cast())?)
188}
189
190// AUTONOMOUS-BOT-IMPLEMENTED
191// TODO-HUMAN-REVIEW(#663)
192fn timeval_to_logical_time(value: libc::timeval) -> Result<LogicalTime, Errno> {
193    let seconds = u64::try_from(value.tv_sec).map_err(|_| Errno::EINVAL)?;
194    let micros = u64::try_from(value.tv_usec).map_err(|_| Errno::EINVAL)?;
195    if micros >= 1_000_000 {
196        return Err(Errno::EINVAL);
197    }
198    let nanos = seconds
199        .checked_mul(1_000_000_000)
200        .and_then(|nanos| nanos.checked_add(micros * 1_000))
201        .ok_or(Errno::EINVAL)?;
202    Ok(LogicalTime::from_nanos(nanos))
203}
204
205// AUTONOMOUS-BOT-IMPLEMENTED
206// TODO-HUMAN-REVIEW(#663)
207fn logical_time_to_timeval(value: LogicalTime) -> libc::timeval {
208    libc::timeval {
209        tv_sec: value.as_secs() as libc::time_t,
210        tv_usec: value.subsec_micros() as libc::suseconds_t,
211    }
212}
213
214// AUTONOMOUS-BOT-IMPLEMENTED
215// TODO-HUMAN-REVIEW(#663)
216fn logical_time_to_alarm_seconds(value: LogicalTime) -> i64 {
217    value.as_nanos().div_ceil(1_000_000_000) as i64
218}
219
220// AUTONOMOUS-BOT-IMPLEMENTED
221// TODO-HUMAN-REVIEW(#663)
222fn deterministic_kill_target(targets: &[DetTid], sig: libc::c_int) -> Result<DetTid, Errno> {
223    match targets {
224        [] => Err(Errno::ESRCH),
225        [target] => Ok(*target),
226        [target, ..] if sig == 0 => Ok(*target),
227        _ => Err(Errno::ENOSYS),
228    }
229}
230
231// AUTONOMOUS-BOT-IMPLEMENTED
232// TODO-HUMAN-REVIEW(PR-1119): Review unmaskable process-group SIGKILL forwarding.
233fn can_forward_process_group_signal(
234    pid: libc::pid_t,
235    sig: libc::c_int,
236    backend_requires_pid_translation: bool,
237) -> bool {
238    pid < -1 && sig == libc::SIGKILL && !backend_requires_pid_translation
239}
240
241/// Whether one of Linux's three ordinary signal syscalls names the calling
242/// task exactly and asks for the one signal that cannot return successfully.
243///
244/// Keep process-group and broadcast spellings out of this predicate. Even when
245/// the caller belongs to the named group, KVM can refuse a group containing a
246/// second process; reserving an exit before that refusal would strand the
247/// scheduler's terminal barrier.
248fn self_sigkill_targets_current_task(
249    signal: libc::c_int,
250    target_process: Option<DetPid>,
251    target_thread: Option<DetTid>,
252    current_process: DetPid,
253    current_thread: DetTid,
254) -> bool {
255    signal == libc::SIGKILL
256        && (target_process.is_some() || target_thread.is_some())
257        && target_process.is_none_or(|target| target == current_process)
258        && target_thread.is_none_or(|target| target == current_thread)
259}
260
261impl<T: RecordOrReplay> Detcore<T> {
262    // AUTONOMOUS-BOT-IMPLEMENTED
263    // TODO-HUMAN-REVIEW(#663)
264    /// We send the alarms to the global scheduler to handle.
265    pub async fn handle_alarm<G: Guest<Self>>(
266        &self,
267        guest: &mut G,
268        call: syscalls::Alarm,
269    ) -> Result<i64, Error> {
270        if guest.config().sequentialize_threads {
271            let remaining = register_alarm(
272                guest,
273                LogicalTime::from_secs(call.seconds() as u64),
274                LogicalTime::ZERO,
275                Signal::SIGALRM,
276            )
277            .await;
278            Ok(logical_time_to_alarm_seconds(remaining.0))
279        } else {
280            info!(
281                "[dtid {}] Running without scheduler, so letting alarm call through...",
282                guest.thread_state().dettid
283            );
284            Ok(guest.inject(call).await?)
285        }
286    }
287
288    // AUTONOMOUS-BOT-IMPLEMENTED
289    // TODO-HUMAN-REVIEW(#663)
290    // TODO-HUMAN-REVIEW(#869)
291    /// Schedule a one-shot or periodic real-time interval timer on Detcore logical time.
292    pub async fn handle_setitimer<G: Guest<Self>>(
293        &self,
294        guest: &mut G,
295        call: syscalls::Setitimer,
296    ) -> Result<i64, Error> {
297        if !guest.config().sequentialize_threads {
298            info!(
299                "[dtid {}] Running without scheduler, so letting setitimer call through...",
300                guest.thread_state().dettid
301            );
302            return Ok(guest.inject(call).await?);
303        }
304        if call.which() != libc::ITIMER_REAL {
305            return Err(Error::Errno(Errno::ENOSYS));
306        }
307
308        let value = call.value().ok_or(Errno::EFAULT)?;
309        let timer: libc::itimerval = guest.memory().read_value(value)?;
310        let interval = timeval_to_logical_time(timer.it_interval)?;
311        let duration = timeval_to_logical_time(timer.it_value)?;
312        let (remaining, old_interval) =
313            register_alarm(guest, duration, interval, Signal::SIGALRM).await;
314        if let Some(old_value) = call.ovalue() {
315            let old_timer = libc::itimerval {
316                it_interval: logical_time_to_timeval(old_interval),
317                it_value: logical_time_to_timeval(remaining),
318            };
319            guest.memory().write_value(old_value, &old_timer)?;
320        }
321        Ok(0)
322    }
323
324    // AUTONOMOUS-BOT-IMPLEMENTED
325    // TODO-HUMAN-REVIEW(PR-892)
326    /// Return interval-timer state from Detcore's logical scheduler.
327    pub async fn handle_getitimer<G: Guest<Self>>(
328        &self,
329        guest: &mut G,
330        call: syscalls::Getitimer,
331    ) -> Result<i64, Error> {
332        if !guest.config().sequentialize_threads {
333            info!(
334                "[dtid {}] Running without scheduler, so letting getitimer call through...",
335                guest.thread_state().dettid
336            );
337            return Ok(guest.inject(call).await?);
338        }
339
340        let snapshot = match call.which() {
341            libc::ITIMER_REAL => alarm_remaining(guest).await,
342            libc::ITIMER_VIRTUAL | libc::ITIMER_PROF => {
343                crate::scheduler::real_timer::ItimerSnapshot::default()
344            }
345            _ => return Err(Errno::EINVAL.into()),
346        };
347        let value = call.value().ok_or(Errno::EFAULT)?;
348        let timer = libc::itimerval {
349            it_interval: logical_time_to_timeval(snapshot.interval),
350            it_value: logical_time_to_timeval(snapshot.remaining),
351        };
352        guest.memory().write_value(value, &timer)?;
353        Ok(0)
354    }
355
356    /// A pause is really just an unbounded sleep.
357    pub async fn handle_pause<G: Guest<Self>>(
358        &self,
359        guest: &mut G,
360        call: syscalls::Pause,
361    ) -> Result<i64, Error> {
362        if guest.config().sequentialize_threads {
363            // `pause` has no deadline: it returns only when a signal is delivered.
364            // `LogicalTime::INDEFINITE` records that, and the scheduler refuses to
365            // fast-forward virtual time onto it (see `step2d_handle_empty_queue`),
366            // so the `Normal` arm below stays unreachable.
367            let req = Self::sleep_request_abs(guest, LogicalTime::INDEFINITE).await;
368            match crate::tool_global::parked_wait_request(
369                guest,
370                req,
371                crate::scheduler::parked::ParkedWaitPolicy::PauseNoHandlerRestart,
372            )
373            .await
374            {
375                ResumeStatus::Normal => {
376                    panic!(
377                        "Internal violation: pause should never return from the scheduler except by interruption!"
378                    )
379                }
380                ResumeStatus::Signaled(_) => Err(reverie::Error::Errno(Errno::EINTR)),
381            }
382        } else {
383            info!(
384                "[dtid {}] Running without scheduler, so letting pause call through...",
385                guest.thread_state().dettid
386            );
387            Ok(guest.inject(call).await?)
388        }
389    }
390
391    /// Run rt_sigsuspend without holding the deterministic scheduler turn.
392    ///
393    /// The kernel must perform the temporary mask swap atomically and restore the
394    /// original mask after signal delivery, so execute the real blocking syscall
395    /// while marking this thread as blocked outside the runnable set.
396    pub async fn handle_rt_sigsuspend<G: Guest<Self>>(
397        &self,
398        guest: &mut G,
399        call: syscalls::RtSigsuspend,
400    ) -> Result<i64, Error> {
401        // Invalid arguments return immediately from the kernel and therefore are
402        // not signal-only waits.
403        validate_kernel_sigset_size(call.sigsetsize())?;
404        let Some(mask_addr) = call.mask() else {
405            return Err(Errno::EFAULT.into());
406        };
407
408        let temporary_mask = read_kernel_sigset(guest, mask_addr).await?;
409        let mut stack = guest.stack().await;
410        let pending_addr = stack.push(0_u64);
411        let pending_guard = stack.commit()?;
412        let pending_out = AddrMut::<libc::sigset_t>::from_raw(pending_addr.as_raw())
413            .expect("stack address must be non-null");
414        let pending_call = syscalls::RtSigpending::new()
415            .with_set(Some(pending_out))
416            .with_sigsetsize(KERNEL_SIGSET_SIZE);
417        guest.inject_with_retry(pending_call).await?;
418        let pending: u64 = guest.memory().read_value(pending_addr)?;
419        drop(pending_guard);
420
421        if pending & !temporary_mask != 0 {
422            // The kernel will consume an already-pending signal as soon as it
423            // atomically installs the temporary mask. Keep this immediate case
424            // out of the terminal-wait classification; the real syscall still
425            // performs delivery and restores the old mask.
426            self.record_or_replay_blocking(guest, call.into()).await
427        } else {
428            self.record_or_replay_rt_sigsuspend(guest, call).await
429        }
430    }
431
432    /// rt_sigaction
433    pub async fn handle_rt_sigaction<G: Guest<Self>>(
434        &self,
435        guest: &mut G,
436        call: syscalls::RtSigaction,
437    ) -> Result<i64, Error> {
438        // Linux rejects an invalid size before inspecting either user pointer.
439        // Preserve that EINVAL-before-EFAULT ordering even for the signal that
440        // Hermit reserves for deterministic preemption.
441        validate_kernel_sigset_size(call.sigsetsize())?;
442
443        // Copy the complete kernel object before inspecting the signal number or
444        // writing old_action. This preserves Linux's action-before-old_action
445        // pointer ordering, including when both arguments alias.
446        let kernel_action = match call.action() {
447            Some(action) => Some(read_kernel_sigaction(guest, action).await?),
448            None => None,
449        };
450
451        // Both appropriated signals are reported here, at the one point where the
452        // guest states its expectation. SIGTRAP falls through to the ordinary
453        // path below (its handler really is installed; ptrace just eats the
454        // signal), so this must run before the SIGSTKFLT early return.
455        if let Some(action) = kernel_action {
456            warn_appropriated_signal(call.signum(), action.handler);
457        }
458
459        // PERF_EVENT_SIGNAL is reserved.
460        if call.signum() == reverie::PERF_EVENT_SIGNAL as i32 {
461            // The go runtime attempts to register this (unused) signal handler.  We will never
462            // deliver signals of this kind to the guest, so we just turn this action into a noop
463            // rather than returning `Err(Errno::EINVAL.into())`. Preserve that established
464            // policy while still honoring the raw syscall's pointer accesses: the virtual old
465            // disposition is the default action, never Reverie's private preemption handler.
466            if call.old_action().is_some() {
467                // SIGKILL's disposition cannot be changed by userspace, so its
468                // query copies the same four zero words as our virtual default.
469                // Let the kernel copy them: a short write_value can finish with
470                // PTRACE_POKEDATA and overwrite a protected final word. This also
471                // preserves the kernel's partial-copy effects on EFAULT without
472                // exposing Reverie's private preemption action.
473                return Ok(guest
474                    .inject(call.with_signum(libc::SIGKILL).with_action(None))
475                    .await?);
476            }
477            return Ok(0);
478        }
479        Ok(if let Some(kernel_action) = kernel_action {
480            // The kernel treats `action` as input. Sanitize a private copy instead
481            // of writing through the guest pointer: the input may be read-only,
482            // and changing it would make the syscall wrapper observable.
483            let mut kernel_action = kernel_action;
484            kernel_action.mask = without_perf_event_signal(kernel_action.mask);
485            let mut stack = guest.stack().await;
486            let sanitized_action = stack.push(kernel_action);
487            let _stack_guard = stack.commit()?;
488            guest
489                .inject(call.with_action(Some(sanitized_action.cast())))
490                .await?
491        } else {
492            guest.inject(call).await?
493        })
494    }
495
496    // AUTONOMOUS-BOT-IMPLEMENTED
497    // TODO-HUMAN-REVIEW(#1046): Review retrying interrupted signal-mask injections.
498    /// rt_sigprocmask
499    pub async fn handle_rt_sigprocmask<G: Guest<Self>>(
500        &self,
501        guest: &mut G,
502        call: syscalls::RtSigprocmask,
503    ) -> Result<i64, Error> {
504        // The kernel checks sigsetsize before copying from either user pointer.
505        validate_kernel_sigset_size(call.sigsetsize())?;
506
507        if call.how() != libc::SIG_BLOCK && call.how() != libc::SIG_SETMASK {
508            Ok(guest.inject_with_retry(call).await?)
509        } else if let Some(set) = call.set() {
510            let set_mask = read_kernel_sigset(guest, set).await?;
511            let mut stack = guest.stack().await;
512            let new_set = stack.push(without_perf_event_signal(set_mask));
513            let _stack_guard = stack.commit()?;
514            let modified_call = syscalls::RtSigprocmask::new()
515                .with_how(call.how())
516                .with_set(Some(new_set.cast()))
517                .with_oldset(call.oldset())
518                .with_sigsetsize(call.sigsetsize());
519            // Keep returning to the handler so post_handler_hook can run, but
520            // do not expose a tracer preemption as ERESTARTSYS to the guest.
521            Ok(guest.inject_with_retry(modified_call).await?)
522        } else {
523            Ok(guest.inject_with_retry(call).await?)
524        }
525    }
526
527    /// rt_sigtimedwait system call
528    ///
529    /// This is handled by the scheduler and not passed to the record/replay layer,
530    /// because currently signals are not recorded.
531    pub async fn handle_rt_sigtimedwait<G: Guest<Self>>(
532        &self,
533        guest: &mut G,
534        call: syscalls::RtSigtimedwait,
535    ) -> Result<i64, Error> {
536        // Linux rejects an invalid mask width before inspecting its pointers.
537        validate_kernel_sigset_size(call.sigsetsize())?;
538        // Linux keeps this entry snapshot for the whole wait. The retry helper
539        // currently injects from the caller's pointer again, so another thread
540        // can change the set between retries. Fixing that requires sharing one
541        // scratch-stack guard with the helper's zero-timeout object; do not hide
542        // that remaining difference by treating this validation read as a snapshot.
543
544        let dettid = guest.thread_state().dettid;
545
546        let maybe_timeout = if let Some(timeout) = call.timeout() {
547            let ts: Timespec = guest.memory().read_value(timeout)?;
548            let ns_delta =
549                Duration::from_secs(ts.tv_sec as u64) + Duration::from_nanos(ts.tv_nsec as u64);
550            let base_time = thread_observe_time(guest).await;
551            let target_time = base_time + ns_delta;
552            Some(target_time)
553        } else {
554            None
555        };
556        let mut rsrc = Resources::new(dettid);
557        rsrc.insert(ResourceID::InternalIOPolling, Permission::W);
558        rsrc.fyi("rt_sigtimedwait");
559        retry_nonblocking_syscall_with_timeout(guest, call, rsrc, maybe_timeout).await
560    }
561
562    /// Fence an exact self-SIGKILL at the scheduler-selected syscall turn.
563    ///
564    /// KVM commits this syscall as an immediate, nonreturning process exit. It
565    /// therefore has no later return boundary at which the ordinary pending
566    /// signal path could acquire a delivery permit. Without this preflight, the
567    /// backend's terminal child callback arrives with no generation-bound
568    /// reservation and must fail closed. `ResourceID::Exit` is the same fence
569    /// used by `exit_group`; the KVM-only configuration guard avoids changing
570    /// ptrace, DBT, or non-sequential execution.
571    async fn reserve_kvm_self_sigkill_exit<G: Guest<Self>>(
572        &self,
573        guest: &mut G,
574        signal: libc::c_int,
575        target_process: Option<DetPid>,
576        target_thread: Option<DetTid>,
577    ) -> bool {
578        if !self.cfg.kvm_shared_dequeue_timers {
579            return false;
580        }
581        let (current_thread, mm) = {
582            let state = guest.thread_state();
583            (state.dettid, state.mm_id)
584        };
585        if !self_sigkill_targets_current_task(
586            signal,
587            target_process,
588            target_thread,
589            self.detpid,
590            current_thread,
591        ) {
592            return false;
593        }
594        let request = guest.thread_state().mk_request(
595            ResourceID::Exit {
596                group: true,
597                process: self.detpid,
598                mm,
599            },
600            Permission::RW,
601        );
602        resource_request(guest, request).await;
603        true
604    }
605
606    // AUTONOMOUS-BOT-IMPLEMENTED
607    // TODO-HUMAN-REVIEW(#663)
608    // TODO-HUMAN-REVIEW(PR-1058): Review process-pending signal preservation.
609    // TODO-HUMAN-REVIEW(PR-1119): Review unmaskable process-group SIGKILL forwarding.
610    /// Resolve signal-zero existence checks in the fixed PID namespace, then route an
611    /// unambiguous positive-PID process signal through the backend. Backends that can execute
612    /// with guest PIDs preserve process-directed delivery; DBT translates it to the sole live
613    /// thread because its native process uses a host PID. An unmaskable SIGKILL to a specific
614    /// process group is also safe to preserve on backends whose guests use real namespace PIDs;
615    /// other process-group and broadcast delivery remains refused until Detcore models eligible
616    /// signal masks.
617    pub async fn handle_kill<G: Guest<Self>>(
618        &self,
619        guest: &mut G,
620        call: syscalls::Kill,
621    ) -> Result<i64, Error> {
622        if !guest.config().sequentialize_threads {
623            return Ok(self.record_or_replay(guest, call).await?);
624        }
625
626        if call.sig() == 0 {
627            return Ok(self.record_or_replay(guest, call).await?);
628        }
629
630        let tgid = call.pid();
631        if can_forward_process_group_signal(
632            tgid,
633            call.sig(),
634            guest
635                .config()
636                .backend_requires_thread_directed_process_signals,
637        ) {
638            return Ok(self.record_or_replay(guest, call).await?);
639        }
640        if tgid <= 0 {
641            return Err(Errno::ENOSYS.into());
642        }
643        // Exact self-SIGKILL is group-fatal, so it has no recipient-selection
644        // ambiguity even when the process has several live threads. Reserve it
645        // before the generic process-signal path rejects that thread set.
646        if self
647            .reserve_kvm_self_sigkill_exit(guest, call.sig(), Some(DetPid::from_raw(tgid)), None)
648            .await
649        {
650            return Ok(self.record_or_replay(guest, call).await?);
651        }
652        let targets = resolve_kill_targets(guest, DetPid::from_raw(tgid)).await;
653        let tid = deterministic_kill_target(&targets, call.sig())?;
654        let value = if !guest
655            .config()
656            .backend_requires_thread_directed_process_signals
657        {
658            self.record_or_replay(guest, call).await?
659        } else {
660            let targeted = syscalls::Tgkill::new()
661                .with_tgid(tgid)
662                .with_tid(tid.as_raw())
663                .with_sig(call.sig());
664            self.record_or_replay(guest, targeted).await?
665        };
666        self.notify_cross_task_signal(guest, tid, call.sig(), Some(DetPid::from_raw(tgid)))
667            .await;
668        Ok(value)
669    }
670
671    // AUTONOMOUS-BOT-IMPLEMENTED
672    // TODO-HUMAN-REVIEW(#663)
673    /// Send a thread-directed signal through the kernel. Guest PID/TID values are
674    /// stable in the fresh PID namespace and delivery is scheduler-serialized.
675    pub async fn handle_tgkill<G: Guest<Self>>(
676        &self,
677        guest: &mut G,
678        call: syscalls::Tgkill,
679    ) -> Result<i64, Error> {
680        let _reserved = self
681            .reserve_kvm_self_sigkill_exit(
682                guest,
683                call.sig(),
684                Some(DetPid::from_raw(call.tgid())),
685                Some(DetTid::from_raw(call.tid())),
686            )
687            .await;
688        let value = self.record_or_replay(guest, call).await?;
689        // `pthread_kill` lowers to `tgkill`, so this is the ordinary way one
690        // guest THREAD signals a sibling. Like `kill`, a successful cross-task
691        // send must tell the scheduler, or a target parked in a child wait is
692        // never woken and the wait hangs.
693        self.notify_cross_task_signal(guest, DetTid::from_raw(call.tid()), call.sig(), None)
694            .await;
695        Ok(value)
696    }
697
698    // AUTONOMOUS-BOT-IMPLEMENTED
699    // TODO-HUMAN-REVIEW(#812)
700    /// Send a thread-directed signal through the older two-argument `tkill`.
701    /// Like its `tgkill` sibling, the target thread is addressed by a guest TID
702    /// that is stable in the fresh PID namespace and delivery is
703    /// scheduler-serialized, so forwarding the kernel call is deterministic.
704    pub async fn handle_tkill<G: Guest<Self>>(
705        &self,
706        guest: &mut G,
707        call: syscalls::Tkill,
708    ) -> Result<i64, Error> {
709        let _reserved = self
710            .reserve_kvm_self_sigkill_exit(
711                guest,
712                call.sig(),
713                None,
714                Some(DetTid::from_raw(call.tid())),
715            )
716            .await;
717        let value = self.record_or_replay(guest, call).await?;
718        // Same wakeup obligation as `tgkill`; `tkill` is the older two-argument
719        // spelling of the same thread-directed send.
720        self.notify_cross_task_signal(guest, DetTid::from_raw(call.tid()), call.sig(), None)
721            .await;
722        Ok(value)
723    }
724
725    /// Tell the scheduler that a successful thread-directed send left a signal
726    /// physically pending for another task.
727    ///
728    /// Shared by `kill`, `tgkill` and `tkill` so the three cannot drift: a fix
729    /// applied to one spelling of "signal another task" must apply to all of
730    /// them, which is exactly the gap that let a `pthread_kill` from a sibling
731    /// thread hang a `waitid` that a `kill` from a sibling process could
732    /// interrupt. Self-directed signals are excluded: the sender is running, so
733    /// it is not parked waiting to be woken.
734    async fn notify_cross_task_signal<G: Guest<Self>>(
735        &self,
736        guest: &mut G,
737        target: DetTid,
738        raw_signal: i32,
739        target_process: Option<DetPid>,
740    ) {
741        // ⚠️ NO `Signal::try_from` GATE. It used to stand here, and because
742        // `nix`'s `Signal` models only 1..=31 it rejected EVERY realtime signal:
743        // measured in-tree, the gate admitted exactly 1..=31 and zero of the 31
744        // realtime signals. The notification was skipped silently, so a target
745        // parked on `ResourceID::WaitChild` was never woken and the wait hung.
746        // `SigWrapper` now carries the raw number, so every deliverable signal
747        // can be represented and none is dropped for being unnameable. Signal
748        // zero is only an existence/permission probe and queues no signal.
749        if should_notify_cross_task_signal(guest.thread_state().dettid, target, raw_signal) {
750            notify_signal_pending(guest, target, SigWrapper(raw_signal), target_process).await;
751        }
752    }
753
754    // AUTONOMOUS-BOT-IMPLEMENTED
755    // TODO-HUMAN-REVIEW(#812)
756    /// Queue a thread-directed signal with an accompanying `siginfo_t`. Like
757    /// `tgkill`, the target is a specific thread named by stable guest TGID/TID
758    /// and delivery is scheduler-serialized; the guest-supplied siginfo is
759    /// deterministic input, so forwarding the kernel call is deterministic.
760    pub async fn handle_rt_tgsigqueueinfo<G: Guest<Self>>(
761        &self,
762        guest: &mut G,
763        call: syscalls::RtTgsigqueueinfo,
764    ) -> Result<i64, Error> {
765        let value = self.record_or_replay(guest, call).await?;
766        // Thread-directed like `tgkill`, so it carries the same wakeup
767        // obligation. `sigqueue`/`pthread_sigqueue` reach a parked sibling
768        // through here.
769        self.notify_cross_task_signal(guest, DetTid::from_raw(call.tid()), call.sig(), None)
770            .await;
771        Ok(value)
772    }
773
774    // AUTONOMOUS-BOT-IMPLEMENTED
775    // TODO-HUMAN-REVIEW(#812)
776    // TODO-HUMAN-REVIEW(PR-1058): Review queued process-signal preservation.
777    /// Queue a process-directed signal with an accompanying `siginfo_t`. Mirrors `handle_kill`:
778    /// preserve process-directed delivery when the backend accepts guest PIDs, otherwise route an
779    /// unambiguous positive-PID target to its sole live thread via `rt_tgsigqueueinfo`. Ambiguous
780    /// multithreaded process-directed delivery is refused until Detcore models eligible masks.
781    pub async fn handle_rt_sigqueueinfo<G: Guest<Self>>(
782        &self,
783        guest: &mut G,
784        call: syscalls::RtSigqueueinfo,
785    ) -> Result<i64, Error> {
786        if !guest.config().sequentialize_threads {
787            return Ok(self.record_or_replay(guest, call).await?);
788        }
789
790        if call.sig() == 0 {
791            return Ok(self.record_or_replay(guest, call).await?);
792        }
793
794        let tgid = call.tgid();
795        if tgid <= 0 {
796            return Err(Errno::ENOSYS.into());
797        }
798        let targets = resolve_kill_targets(guest, DetPid::from_raw(tgid)).await;
799        let tid = deterministic_kill_target(&targets, call.sig())?;
800        let value = if !guest
801            .config()
802            .backend_requires_thread_directed_process_signals
803        {
804            self.record_or_replay(guest, call).await?
805        } else {
806            let targeted = syscalls::RtTgsigqueueinfo::new()
807                .with_tgid(tgid)
808                .with_tid(tid.as_raw())
809                .with_sig(call.sig())
810                .with_siginfo(call.siginfo());
811            self.record_or_replay(guest, targeted).await?
812        };
813        self.notify_cross_task_signal(guest, tid, call.sig(), Some(DetPid::from_raw(tgid)))
814            .await;
815        Ok(value)
816    }
817
818    // AUTONOMOUS-BOT-IMPLEMENTED
819    // TODO-HUMAN-REVIEW(#663)
820    /// Read the kernel pending-signal mask after Detcore has serialized all signal
821    /// generation and delivery events that can change it.
822    pub async fn handle_rt_sigpending<G: Guest<Self>>(
823        &self,
824        guest: &mut G,
825        call: syscalls::RtSigpending,
826    ) -> Result<i64, Error> {
827        Ok(self.record_or_replay(guest, call).await?)
828    }
829}
830
831fn should_notify_cross_task_signal(sender: DetTid, target: DetTid, raw_signal: i32) -> bool {
832    raw_signal != 0 && target != sender
833}
834
835#[cfg(test)]
836mod tests {
837    use super::*;
838
839    fn timeval(seconds: libc::time_t, micros: libc::suseconds_t) -> libc::timeval {
840        libc::timeval {
841            tv_sec: seconds,
842            tv_usec: micros,
843        }
844    }
845
846    #[test]
847    fn raw_kernel_signal_masks_are_exactly_one_u64() {
848        assert_eq!(KERNEL_SIGSET_SIZE, 8);
849        assert_eq!(std::mem::size_of::<KernelSigset>(), 8);
850        assert!(std::mem::size_of::<libc::sigset_t>() > KERNEL_SIGSET_SIZE);
851    }
852
853    #[test]
854    fn raw_signal_size_validation_rejects_before_pointer_processing() {
855        assert_eq!(validate_kernel_sigset_size(7), Err(Errno::EINVAL));
856        assert_eq!(validate_kernel_sigset_size(8), Ok(()));
857        assert_eq!(validate_kernel_sigset_size(16), Err(Errno::EINVAL));
858    }
859
860    #[test]
861    fn reserved_signal_is_removed_from_the_kernel_sized_mask_only() {
862        let reserved = 1_u64 << (reverie::PERF_EVENT_SIGNAL as u32 - 1);
863        let usr1 = 1_u64 << (libc::SIGUSR1 as u32 - 1);
864        assert_eq!(without_perf_event_signal(reserved | usr1), usr1);
865        assert_eq!(without_perf_event_signal(usr1), usr1);
866    }
867
868    #[test]
869    fn timeval_conversion_preserves_subsecond_precision() {
870        let logical_time =
871            timeval_to_logical_time(timeval(2, 345_678)).expect("valid timeval should convert");
872        assert_eq!(
873            logical_time,
874            LogicalTime::from_nanos(2_345_678_000),
875            "timeval conversion should preserve microsecond precision"
876        );
877
878        let round_trip = logical_time_to_timeval(logical_time);
879        assert_eq!(round_trip.tv_sec, 2, "round trip should preserve seconds");
880        assert_eq!(
881            round_trip.tv_usec, 345_678,
882            "round trip should preserve microseconds"
883        );
884    }
885
886    #[test]
887    fn timeval_conversion_rejects_invalid_values() {
888        for invalid in [
889            timeval(-1, 0),
890            timeval(0, -1),
891            timeval(0, 1_000_000),
892            timeval(libc::time_t::MAX, 0),
893        ] {
894            assert_eq!(
895                timeval_to_logical_time(invalid),
896                Err(Errno::EINVAL),
897                "invalid timeval should return EINVAL"
898            );
899        }
900    }
901
902    #[test]
903    fn alarm_remaining_seconds_round_up() {
904        assert_eq!(logical_time_to_alarm_seconds(LogicalTime::ZERO), 0);
905        assert_eq!(logical_time_to_alarm_seconds(LogicalTime::from_nanos(1)), 1);
906        assert_eq!(
907            logical_time_to_alarm_seconds(LogicalTime::from_nanos(999_999_999)),
908            1
909        );
910        assert_eq!(
911            logical_time_to_alarm_seconds(LogicalTime::from_nanos(1_000_000_000)),
912            1
913        );
914        assert_eq!(
915            logical_time_to_alarm_seconds(LogicalTime::from_nanos(1_000_000_001)),
916            2
917        );
918    }
919
920    #[test]
921    fn kill_targets_only_unambiguous_process_delivery() {
922        let first = DetTid::from_raw(42);
923        let second = DetTid::from_raw(43);
924        assert_eq!(
925            deterministic_kill_target(&[], libc::SIGUSR1),
926            Err(Errno::ESRCH)
927        );
928        assert_eq!(
929            deterministic_kill_target(&[first], libc::SIGUSR1),
930            Ok(first)
931        );
932        assert_eq!(
933            deterministic_kill_target(&[first, second], libc::SIGUSR1),
934            Err(Errno::ENOSYS)
935        );
936        assert_eq!(deterministic_kill_target(&[first, second], 0), Ok(first));
937
938        let process = DetPid::from_raw(41);
939        assert!(self_sigkill_targets_current_task(
940            libc::SIGKILL,
941            Some(process),
942            None,
943            process,
944            first,
945        ));
946        assert!(self_sigkill_targets_current_task(
947            libc::SIGKILL,
948            None,
949            Some(first),
950            process,
951            first,
952        ));
953        assert!(self_sigkill_targets_current_task(
954            libc::SIGKILL,
955            Some(process),
956            Some(first),
957            process,
958            first,
959        ));
960        for (signal, target_process, target_thread) in [
961            (libc::SIGTERM, Some(process), Some(first)),
962            (libc::SIGKILL, Some(DetPid::from_raw(40)), Some(first)),
963            (libc::SIGKILL, Some(process), Some(second)),
964            (libc::SIGKILL, None, None),
965        ] {
966            assert!(!self_sigkill_targets_current_task(
967                signal,
968                target_process,
969                target_thread,
970                process,
971                first,
972            ));
973        }
974    }
975
976    #[test]
977    fn process_group_forwarding_is_limited_to_unmaskable_sigkill() {
978        assert!(can_forward_process_group_signal(-42, libc::SIGKILL, false));
979        assert!(!can_forward_process_group_signal(-42, libc::SIGTERM, false));
980        assert!(!can_forward_process_group_signal(0, libc::SIGKILL, false));
981        assert!(!can_forward_process_group_signal(-1, libc::SIGKILL, false));
982        assert!(!can_forward_process_group_signal(-42, libc::SIGKILL, true));
983    }
984
985    #[test]
986    fn signal_zero_never_notifies_a_target() {
987        let sender = DetTid::from_raw(42);
988        let target = DetTid::from_raw(43);
989        assert!(!should_notify_cross_task_signal(sender, target, 0));
990        assert!(should_notify_cross_task_signal(
991            sender,
992            target,
993            libc::SIGUSR1
994        ));
995        assert!(!should_notify_cross_task_signal(
996            sender,
997            sender,
998            libc::SIGUSR1
999        ));
1000    }
1001}
1002
1003#[cfg(test)]
1004mod appropriated_signal_tests {
1005    use super::*;
1006
1007    /// The set is closed, and it is closed BY MEASUREMENT: every signal 1..64 was
1008    /// run under two delivery paths against a native control, and exactly these
1009    /// two differ. If a third is ever appropriated it must be added here, because
1010    /// the diagnostic is the only thing that makes the loss visible.
1011    #[test]
1012    fn the_appropriated_set_is_exactly_sigtrap_and_sigstkflt() {
1013        let signums: Vec<i32> = APPROPRIATED_SIGNALS.iter().map(|(s, _)| *s).collect();
1014        assert_eq!(signums, vec![libc::SIGTRAP, libc::SIGSTKFLT]);
1015        // SIGSTKFLT is appropriated because reverie uses it as the PMU timer, so
1016        // the two must not drift apart.
1017        assert_eq!(libc::SIGSTKFLT, reverie::PERF_EVENT_SIGNAL as i32);
1018    }
1019
1020    /// ⚠️ SIG_DFL and SIG_IGN LOSE NOTHING. A guest that is not asking to be
1021    /// called back has no expectation to disappoint, and warning there would make
1022    /// the marker noise instead of signal -- Go registers SIGSTKFLT routinely.
1023    #[test]
1024    fn only_a_real_handler_is_reported() {
1025        for signum in [libc::SIGTRAP, libc::SIGSTKFLT] {
1026            assert!(!reports(signum, 0), "SIG_DFL must not warn");
1027            assert!(!reports(signum, 1), "SIG_IGN must not warn");
1028            assert!(reports(signum, 0x4000_1234), "a real handler must warn");
1029        }
1030    }
1031
1032    /// An ordinary signal is delivered normally and must never be reported.
1033    #[test]
1034    fn an_unappropriated_signal_is_never_reported() {
1035        for signum in [libc::SIGUSR1, libc::SIGTERM, libc::SIGINT, 10, 30] {
1036            assert!(
1037                !reports(signum, 0x4000_1234),
1038                "signal {signum} is not appropriated"
1039            );
1040        }
1041    }
1042
1043    /// Calls the REAL predicate. Deliberately not a re-implementation: a
1044    /// mirrored copy would keep passing if `appropriated_reason` were gutted.
1045    fn reports(signum: i32, handler: u64) -> bool {
1046        appropriated_reason(signum, handler).is_some()
1047    }
1048}